mirror of
https://github.com/google-gemini/gemini-cli.git
synced 2026-08-12 01:46:27 -07:00
Compare commits
76 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 4238b0b2b5 | |||
| 1583322bb9 | |||
| 5024443c72 | |||
| 58ba19945a | |||
| 659c7aacd9 | |||
| 188e255bf5 | |||
| eef19f25c3 | |||
| cf22ac7e86 | |||
| 493113457b | |||
| cd5ac173cf | |||
| 1b53dfea2b | |||
| d419cb6b67 | |||
| afebb8702e | |||
| 6cb9f2e061 | |||
| 8cb94fe645 | |||
| d9b600b1c9 | |||
| 66708e3c4c | |||
| 2139b121bc | |||
| d5c9a97dc0 | |||
| 49d6b32f98 | |||
| 9f5f032b8e | |||
| 761f604c16 | |||
| 63c5b74770 | |||
| 348fc35f17 | |||
| 56f9688b30 | |||
| 6863148728 | |||
| bde504f250 | |||
| b6b41f79eb | |||
| 8b60087673 | |||
| ac42fb0a24 | |||
| f47d6c6f7a | |||
| d55e366f6a | |||
| dc859e8e48 | |||
| 4bb7e93c45 | |||
| 55a31ef909 | |||
| 3499c84f7b | |||
| d29268d360 | |||
| fccc043bd4 | |||
| c5622fec27 | |||
| e07280eb4e | |||
| bef6119500 | |||
| b94c9775b1 | |||
| 3818efbbfb | |||
| e2a5375d10 | |||
| 69b51f8fa2 | |||
| 3c1bb8c35d | |||
| d76d2d0742 | |||
| a96259c9e5 | |||
| 87f785192c | |||
| 1c21640f97 | |||
| 455d721a0c | |||
| 9681621c6b | |||
| f743ab5790 | |||
| c776c665b0 | |||
| acae7124bd | |||
| 69e0c2659e | |||
| aea9e5e8ed | |||
| 3ff5ba20fc | |||
| 1ae8ba6496 | |||
| fa975395bc | |||
| a345621404 | |||
| f5e58ff18b | |||
| 42ee2b74c7 | |||
| f354eebaf4 | |||
| a4c91ce191 | |||
| 9a023bbba0 | |||
| a8b115caa1 | |||
| 172ff92c34 | |||
| 70f4d573f2 | |||
| 132f38c3a4 | |||
| b31b755bbf | |||
| c988cbb1e3 | |||
| b7c61c9e3d | |||
| 27a3da3e88 | |||
| 15a9429b69 | |||
| 892b35fcfb |
@@ -172,7 +172,7 @@ runs:
|
||||
--workspace="${INPUTS_CORE_PACKAGE_NAME}" \
|
||||
--tag staging-tmp
|
||||
if [[ "${INPUTS_DRY_RUN}" == "false" ]]; then
|
||||
npm dist-tag rm ${INPUTS_CORE_PACKAGE_NAME} staging-tmp
|
||||
npm dist-tag rm ${INPUTS_CORE_PACKAGE_NAME} staging-tmp || echo "Warning: Failed to remove staging-tmp tag (this is normal on registries that forbid tag deletion, like Wombat)"
|
||||
fi
|
||||
|
||||
- name: '🔗 Install latest core package'
|
||||
@@ -251,7 +251,7 @@ runs:
|
||||
${PUBLISH_TARGET} \
|
||||
--tag staging-tmp
|
||||
if [[ "${INPUTS_DRY_RUN}" == "false" ]]; then
|
||||
npm dist-tag rm ${INPUTS_CLI_PACKAGE_NAME} staging-tmp
|
||||
npm dist-tag rm ${INPUTS_CLI_PACKAGE_NAME} staging-tmp || echo "Warning: Failed to remove staging-tmp tag (this is normal on registries that forbid tag deletion, like Wombat)"
|
||||
fi
|
||||
|
||||
- name: 'Get a2a-server Token'
|
||||
@@ -278,9 +278,9 @@ runs:
|
||||
--dry-run="${INPUTS_DRY_RUN}" \
|
||||
--workspace="${INPUTS_A2A_PACKAGE_NAME}" \
|
||||
--tag staging-tmp
|
||||
if [[ "${INPUTS_DRY_RUN}" == "false" ]]; then
|
||||
npm dist-tag rm ${INPUTS_A2A_PACKAGE_NAME} staging-tmp
|
||||
fi
|
||||
if [[ "${INPUTS_DRY_RUN}" == "false" ]]; then
|
||||
npm dist-tag rm ${INPUTS_A2A_PACKAGE_NAME} staging-tmp || echo "Warning: Failed to remove staging-tmp tag (this is normal on registries that forbid tag deletion, like Wombat)"
|
||||
fi
|
||||
|
||||
- name: '🏷️ Tag release'
|
||||
uses: './.github/actions/tag-npm-release'
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
name: 'Testing: Tools (Python)'
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- 'main'
|
||||
- 'release/**'
|
||||
paths:
|
||||
- 'tools/**'
|
||||
pull_request:
|
||||
branches:
|
||||
- 'main'
|
||||
- 'release/**'
|
||||
paths:
|
||||
- 'tools/**'
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: 'bash'
|
||||
|
||||
jobs:
|
||||
python-tests:
|
||||
name: 'Python Tests'
|
||||
runs-on: 'ubuntu-latest'
|
||||
steps:
|
||||
- name: 'Checkout'
|
||||
uses: 'actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683' # ratchet:actions/checkout@v4
|
||||
|
||||
- name: 'Set up Python'
|
||||
uses: 'actions/setup-python@8d9ed9ac5c53483de85588cdf95a591a75ab9f55' # ratchet:actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.13'
|
||||
cache: 'pip'
|
||||
cache-dependency-path: 'tools/caretaker-agent/cloudrun/triage-worker/requirements.txt'
|
||||
|
||||
- name: 'Install dependencies'
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
if [ -f tools/caretaker-agent/cloudrun/triage-worker/requirements.txt ]; then
|
||||
python -m pip install -r tools/caretaker-agent/cloudrun/triage-worker/requirements.txt
|
||||
fi
|
||||
|
||||
- name: 'Run unittest suite'
|
||||
run: |
|
||||
PYTHONPATH=tools/caretaker-agent/cloudrun/triage-worker python -m unittest discover -s tools/caretaker-agent/cloudrun/triage-worker/tests -t tools/caretaker-agent/cloudrun/triage-worker
|
||||
+3
-3
@@ -421,9 +421,9 @@ To debug the CLI's React-based UI, you can use React DevTools.
|
||||
|
||||
On macOS, `gemini` uses Seatbelt (`sandbox-exec`) under a `permissive-open`
|
||||
profile (see `packages/cli/src/utils/sandbox-macos-permissive-open.sb`) that
|
||||
restricts writes to the project folder but otherwise allows all other operations
|
||||
and outbound network traffic ("open") by default. You can switch to a
|
||||
`strict-open` profile (see
|
||||
denies operations by default, confining writes to the project folder while
|
||||
allowing broad file reads and outbound network traffic ("open") by default. You
|
||||
can switch to a `strict-open` profile (see
|
||||
`packages/cli/src/utils/sandbox-macos-strict-open.sb`) that restricts both reads
|
||||
and writes to the working directory while allowing outbound network traffic by
|
||||
setting `SEATBELT_PROFILE=strict-open` in your environment or `.env` file.
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
# Behavioral Evaluations & EDK Guide
|
||||
|
||||
This guide introduces the **Eval Development Kit (EDK)** and details how to
|
||||
write, validate, run, and report on **behavioral evaluations** in the Gemini CLI
|
||||
codebase.
|
||||
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
Behavioral evaluations are automated tests designed to assert on the
|
||||
**behavior** of the Gemini CLI agent (e.g., verifying which tools are called,
|
||||
checking call ordering, or avoiding destructive commands) rather than checking
|
||||
the final prose output.
|
||||
|
||||
Evaluating agent behavior is critical because:
|
||||
|
||||
1. Model responses are non-deterministic, making exact prose matching highly
|
||||
fragile.
|
||||
2. We must ensure the model utilizes the most efficient tools (e.g., batching
|
||||
files via `read_many_files` instead of sequential `read_file` calls).
|
||||
3. We must enforce safety boundaries (e.g., preventing execution of raw shell
|
||||
commands when safe alternatives exist).
|
||||
|
||||
All behavioral evaluations are stored under the `evals/` directory.
|
||||
|
||||
---
|
||||
|
||||
## EDK Developer Commands
|
||||
|
||||
The EDK provides CLI tools under `scripts/` to help contributors audit, check,
|
||||
and monitor evals.
|
||||
|
||||
### 1. `npm run eval:inventory`
|
||||
|
||||
Scans all eval files under `evals/`, statically parses them, and provides a
|
||||
structured overview of what exists in the repository.
|
||||
|
||||
- **Usage:**
|
||||
```bash
|
||||
npm run eval:inventory
|
||||
```
|
||||
- **JSON Output:** For CI integration or inventory indexing, generate a
|
||||
machine-readable JSON report:
|
||||
```bash
|
||||
npm run eval:inventory -- --json
|
||||
```
|
||||
- **Custom Root:** Run against another directory or repository:
|
||||
```bash
|
||||
npm run eval:inventory -- --root /path/to/other/repo
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 2. `npm run eval:validate`
|
||||
|
||||
A lint-like checker that validates eval source files against standard structural
|
||||
guidelines and best practices.
|
||||
|
||||
- **Usage:**
|
||||
```bash
|
||||
npm run eval:validate
|
||||
```
|
||||
- **Custom Scopes:** Validate a specific file:
|
||||
```bash
|
||||
npm run eval:validate -- evals/my-test.eval.ts
|
||||
```
|
||||
|
||||
#### Validation Rules & Severities
|
||||
|
||||
| Rule ID | Severity | Description |
|
||||
| :------------------- | :---------- | :--------------------------------------------------------------------------------------------------------------------- |
|
||||
| `file-naming` | **Error** | File must match `*.eval.ts` or `*.eval.tsx` naming conventions. |
|
||||
| `valid-policy` | **Error** | Policy must be one of `ALWAYS_PASSES`, `USUALLY_PASSES`, or `USUALLY_FAILS`. |
|
||||
| `suite-metadata` | **Error** | Both `suiteName` and `suiteType` must be present as static string literals. |
|
||||
| `prompt-presence` | **Error** | Every eval case must have a non-empty `prompt` string. |
|
||||
| `case-name-static` | **Error** | The case name must be a static string literal, not computed dynamically. |
|
||||
| `invalid-tool-refs` | **Error** | All tools referenced in assertions must match known built-in or legacy tools. |
|
||||
| `positive-assertion` | **Error** | Evaluation cases must assert on at least one tool call (e.g., check `waitForToolCall` has been invoked). |
|
||||
| `workspace-setup` | **Error** | Workspace behaviors (like file-system edits/reads) must set up a `files` object. |
|
||||
| `new-evals-policy` | **Warning** | New evals must not use `ALWAYS_PASSES` policy initially (they should be promoted after nightly data proves stability). |
|
||||
|
||||
Warnings (`new-evals-policy`) will be logged with `⚠` and will **not** cause
|
||||
the CLI process to exit with status `1`. Errors (`✗`) will block CI builds and
|
||||
return exit status `1`.
|
||||
|
||||
---
|
||||
|
||||
### 3. `npm run eval:report`
|
||||
|
||||
Aggregates local vitest `report.json` artifacts, maps them against inventory
|
||||
policies, and summarizes the pass rates per model.
|
||||
|
||||
- **Usage:**
|
||||
```bash
|
||||
npm run eval:report
|
||||
```
|
||||
By default, it scans `evals/logs/` recursively for `report.json` files.
|
||||
- **Specifying Directory:**
|
||||
```bash
|
||||
npm run eval:report -- /path/to/logs
|
||||
```
|
||||
- **JSON Output:**
|
||||
```bash
|
||||
npm run eval:report -- --json
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Contributor Workflow
|
||||
|
||||
When writing a new behavioral evaluation, adhere to this workflow to ensure
|
||||
high-quality, non-flaky test runs.
|
||||
|
||||
### Step-by-Step Guide
|
||||
|
||||
1. **Identify the Target Behavior**: Determine which tool calls need
|
||||
verification (e.g., `web_fetch` must be called).
|
||||
2. **Author the Eval File**: Create your file under `evals/<name>.eval.ts`
|
||||
naming it properly.
|
||||
3. **Configure Workspace Files**: If the eval reads or edits files, define them
|
||||
inside the `files` metadata field.
|
||||
4. **Assert Behavior, Not Prose**: Ensure the `assert` block checks tool
|
||||
interactions using `rig.waitForToolCall` or similar. Do not check final
|
||||
prose.
|
||||
5. **Run Locally**:
|
||||
```bash
|
||||
RUN_EVALS=true npx vitest run evals/my-test.eval.ts
|
||||
```
|
||||
6. **Deflake**: Run your eval at least 3 times locally to verify it does not
|
||||
fail due to model variance.
|
||||
7. **Run Validation**: Run `npm run eval:validate` to ensure no linting errors
|
||||
are present.
|
||||
|
||||
### Acceptance Criteria Checklist
|
||||
|
||||
- [ ] **Naming**: File ends with `.eval.ts` or `.eval.tsx`.
|
||||
- [ ] **Policy**: New evals start as `USUALLY_PASSES`.
|
||||
- [ ] **Metadata**: Static `suiteName` and `suiteType` (e.g. `'behavioral'`) are
|
||||
specified.
|
||||
- [ ] **Assertions**: Uses `rig.waitForToolCall` or asserts tool arguments
|
||||
explicitly.
|
||||
- [ ] **Clean workspace**: Does not write to files outside `rig.testDir`.
|
||||
|
||||
### Common Anti-Patterns to Avoid
|
||||
|
||||
- **Restricting core tools**: Never override `settings.tools.core` to limit
|
||||
tools. Evals must run against the default toolset.
|
||||
- **Checking model prose**: Avoid `expect(result).toContain('something')` since
|
||||
model wording is non-deterministic.
|
||||
- **Integration-only testing**: Evals that only write files without checking
|
||||
realistic model prompts are integration tests and belong under
|
||||
`integration-tests/`.
|
||||
|
||||
---
|
||||
|
||||
## CI & Dashboard Integration
|
||||
|
||||
You can easily automate behavioral evaluations or compile dashboard data using
|
||||
EDK's JSON reporters.
|
||||
|
||||
### CI Validation Block
|
||||
|
||||
Add a step in your PR checks or GitHub workflows to automatically lint new evals
|
||||
and block pull requests containing validation errors:
|
||||
|
||||
```yaml
|
||||
- name: Run Eval Validator
|
||||
run: npm run eval:validate
|
||||
```
|
||||
|
||||
### Publishing to a Dashboard
|
||||
|
||||
To record nightly performance metrics across multiple models:
|
||||
|
||||
1. Configure your workflow to run evaluations with the JSON reporter:
|
||||
```bash
|
||||
cross-env GEMINI_MODEL=gemini-2.5-pro npx vitest run --config evals/vitest.config.ts --reporter=json --outputFile="evals/logs/eval-logs-gemini-2.5-pro/report.json"
|
||||
```
|
||||
2. Aggregate all test runs using the reporting tool:
|
||||
```bash
|
||||
npm run eval:report -- evals/logs --json > aggregated_report.json
|
||||
```
|
||||
3. Upload `aggregated_report.json` to your dashboard storage backend to
|
||||
visualize pass rates over time.
|
||||
@@ -18,6 +18,74 @@ on GitHub.
|
||||
| [Preview](preview.md) | Experimental features ready for early feedback. |
|
||||
| [Stable](latest.md) | Stable, recommended for general use. |
|
||||
|
||||
## Announcements: v0.54.0 - 2026-08-06
|
||||
|
||||
- **PR Automation & Antigravity Agent:** Integrated the Antigravity agent runner
|
||||
with dual-locking Firestore concurrency controls to secure the PR generator
|
||||
([#28434](https://github.com/google-gemini/gemini-cli/pull/28434),
|
||||
[#28432](https://github.com/google-gemini/gemini-cli/pull/28432) by
|
||||
@joneba-google).
|
||||
- **Caretaker Triaging & Security:** Enhanced caretaker triage to post comments
|
||||
prior to auto-closing issues and sanitized issue titles under untrusted
|
||||
context ([#28411](https://github.com/google-gemini/gemini-cli/pull/28411),
|
||||
[#28352](https://github.com/google-gemini/gemini-cli/pull/28352) by @chadd28).
|
||||
- **Security and Session Robustness:** Prevented cleartext credential leakage by
|
||||
enforcing HTTPS, rotated session IDs on model fallbacks, and skipped merged
|
||||
function-response turns in active loops
|
||||
([#28517](https://github.com/google-gemini/gemini-cli/pull/28517) by
|
||||
@amelidev, [#28565](https://github.com/google-gemini/gemini-cli/pull/28565) by
|
||||
@adamfweidman).
|
||||
|
||||
## Announcements: v0.53.0 - 2026-07-28
|
||||
|
||||
- **Caretaker Triage Orchestration:** Implemented an LLM triage orchestrator and
|
||||
container build setup
|
||||
([#28345](https://github.com/google-gemini/gemini-cli/pull/28345) by
|
||||
@chadd28).
|
||||
- **Eval Coverage Reporting:** Introduced a new command for generating
|
||||
evaluation coverage reports
|
||||
([#28169](https://github.com/google-gemini/gemini-cli/pull/28169) by @ved015).
|
||||
- **Security & Loop Mitigations:** Enforced workspace trust and task isolation
|
||||
in the A2A server, aligned macOS Seatbelt profiles with the deny-default
|
||||
model, and mitigated infinite ReAct/prompt injection loops
|
||||
([#28470](https://github.com/google-gemini/gemini-cli/pull/28470) by
|
||||
@luisfelipe-alt,
|
||||
[#28424](https://github.com/google-gemini/gemini-cli/pull/28424) by
|
||||
@ompatel-aiml).
|
||||
|
||||
## Announcements: v0.52.0 - 2026-07-22
|
||||
|
||||
- **Caretaker Triage & Egress Services:** Implemented the core triage worker
|
||||
foundational modules, main worker execution loops, and egress action
|
||||
publishers alongside the octokit GitHub Action handler for egress services
|
||||
([#28163](https://github.com/google-gemini/gemini-cli/pull/28163),
|
||||
[#28306](https://github.com/google-gemini/gemini-cli/pull/28306) by @chadd28).
|
||||
- **Core Tool Enhancements:** Bypassed LLM correction for JSON and IPYNB files
|
||||
in `write_file` and `replace` tools, and simplified plan mode write policy to
|
||||
support relative paths
|
||||
([#28223](https://github.com/google-gemini/gemini-cli/pull/28223) by
|
||||
@amelidev, [#28398](https://github.com/google-gemini/gemini-cli/pull/28398) by
|
||||
@DavidAPierce).
|
||||
- **Auth & Privacy Improvements:** Displayed clear error messages when user
|
||||
account has no Code Assist tier, and bumped `google-auth-library` to version
|
||||
10.9.0 ([#28304](https://github.com/google-gemini/gemini-cli/pull/28304) by
|
||||
@ompatel-aiml,
|
||||
[#28385](https://github.com/google-gemini/gemini-cli/pull/28385) by
|
||||
@jerrylin3321).
|
||||
|
||||
## Announcements: v0.50.0 - 2026-07-08
|
||||
|
||||
- **Tool Registry Discovery:** Introduced tool registry discovery capabilities
|
||||
to automatically detect and register available tools
|
||||
([#28113](https://github.com/google-gemini/gemini-cli/pull/28113) by @ved015).
|
||||
- **Release Verification & CI Stability:** Enhanced release verification by
|
||||
ignoring scripts during verification, preventing workspace binary shadowing,
|
||||
and safeguarding against bad NPM releases
|
||||
([#28116](https://github.com/google-gemini/gemini-cli/pull/28116) by
|
||||
@rmedranollamas,
|
||||
[#28132](https://github.com/google-gemini/gemini-cli/pull/28132) by
|
||||
@galdawave).
|
||||
|
||||
## Announcements: v0.45.0 - 2026-06-03
|
||||
|
||||
- **Context Simplification:** Completed major architectural work to simplify the
|
||||
|
||||
+58
-47
@@ -1,6 +1,6 @@
|
||||
# Latest stable release: v0.45.0
|
||||
# Latest stable release: v0.54.0
|
||||
|
||||
Released: June 03, 2026
|
||||
Released: August 6, 2026
|
||||
|
||||
For most users, our latest stable release is the recommended release. Install
|
||||
the latest stable version with:
|
||||
@@ -11,55 +11,66 @@ npm install -g @google/gemini-cli
|
||||
|
||||
## Highlights
|
||||
|
||||
- **Context Manager Simplification:** Completed a significant refactoring of the
|
||||
context management system to improve reliability and architectural clarity.
|
||||
- **A2A Usage Metadata:** Enhanced the Agent-to-Agent protocol to expose usage
|
||||
metadata, enabling more transparent resource monitoring.
|
||||
- **Terminal & PTY Robustness:** Resolved several critical issues related to
|
||||
terminal interactions, including Termux relaunch loops and PTY resize errors.
|
||||
- **Routing Optimizations:** Updated default auto-routing and bypassed
|
||||
classifiers for specific tool responses to prevent orphaned function errors.
|
||||
- **Tool Execution Control:** Forced the `update_topic` tool to execute
|
||||
sequentially, ensuring consistent narrative flow in agent interactions.
|
||||
- **PR Generation & Antigravity Agent:** Implemented Firestore concurrency
|
||||
dual-locking mechanisms in the database and introduced the Antigravity agent
|
||||
runner with comprehensive prompt templates.
|
||||
- **Caretaker Triaging & Issue Security:** Improved the caretaker triage loop to
|
||||
post a descriptive comment prior to auto-closing issues, and sanitized issue
|
||||
titles within an untrusted context to ensure secure processing.
|
||||
- **Enhanced Authentication & Security:** Enforced strict HTTPS validation for
|
||||
GoogleCredentialsAuthProvider to block cleartext leakage, and implemented tag
|
||||
length validation for the file keychain system.
|
||||
- **Model Fallback & History Filtering:** Resolved stateful API errors by
|
||||
rotating session IDs on model fallback, optimized conversation history
|
||||
retrieval by filtering out thought parts when context management is disabled,
|
||||
and correctly skipped merged function responses when tracking active loops.
|
||||
|
||||
## What's Changed
|
||||
|
||||
- chore(release): bump version to 0.45.0-nightly.20260521.g854f811be by
|
||||
- Changelog for v0.53.0-preview.0 by @gemini-cli-robot in
|
||||
[#28507](https://github.com/google-gemini/gemini-cli/pull/28507)
|
||||
- Changelog for v0.52.0 by @gemini-cli-robot in
|
||||
[#28508](https://github.com/google-gemini/gemini-cli/pull/28508)
|
||||
- chore(release): bump version to 0.54.0-nightly.20260722.gf743ab579 by
|
||||
@gemini-cli-robot in
|
||||
[#27362](https://github.com/google-gemini/gemini-cli/pull/27362)
|
||||
- fix(cli): prevent Termux relaunch and resize remount loops by @saymanq in
|
||||
[#27110](https://github.com/google-gemini/gemini-cli/pull/27110)
|
||||
- Feat/a2a expose usage metadata by @jvargassanchez-dot in
|
||||
[#27288](https://github.com/google-gemini/gemini-cli/pull/27288)
|
||||
- feat(context): Complete simplification work. by @joshualitt in
|
||||
[#27345](https://github.com/google-gemini/gemini-cli/pull/27345)
|
||||
- fix(core): force update_topic tool to execute sequentially by
|
||||
@jvargassanchez-dot in
|
||||
[#27357](https://github.com/google-gemini/gemini-cli/pull/27357)
|
||||
- Changelog for v0.44.0-preview.0 by @gemini-cli-robot in
|
||||
[#27360](https://github.com/google-gemini/gemini-cli/pull/27360)
|
||||
- Changelog for v0.43.0 by @gemini-cli-robot in
|
||||
[#27361](https://github.com/google-gemini/gemini-cli/pull/27361)
|
||||
- Revert "fix(core): prevent SIGHUP kills in PTY environments" by @bbiggs in
|
||||
[#27401](https://github.com/google-gemini/gemini-cli/pull/27401)
|
||||
- fix(cli): filter internal session context from history during resumption by
|
||||
@rmedranollamas in
|
||||
[#27391](https://github.com/google-gemini/gemini-cli/pull/27391)
|
||||
- Update default auto routing by @DavidAPierce in
|
||||
[#27071](https://github.com/google-gemini/gemini-cli/pull/27071)
|
||||
- fix(core): bypass routing classifiers to prevent orphaned function response
|
||||
errors by @danielweis in
|
||||
[#27389](https://github.com/google-gemini/gemini-cli/pull/27389)
|
||||
- fix(core): suppress PTY resize EBADF errors by @scidomino in
|
||||
[#27461](https://github.com/google-gemini/gemini-cli/pull/27461)
|
||||
- fix(core): prevent blacklist bypass in mcp list by @ompatel-aiml in
|
||||
[#27377](https://github.com/google-gemini/gemini-cli/pull/27377)
|
||||
- fix(cli): ignore unmapped vim normal keys by @MukundaKatta in
|
||||
[#27102](https://github.com/google-gemini/gemini-cli/pull/27102)
|
||||
- fix(patch): cherry-pick bd53951 to release/v0.45.0-preview.0-pr-27496 to patch
|
||||
version v0.45.0-preview.0 and create version 0.45.0-preview.1 by
|
||||
[#28510](https://github.com/google-gemini/gemini-cli/pull/28510)
|
||||
- fix(caretaker): sanitize and wrap issue title in untrusted_context by @chadd28
|
||||
in [#28352](https://github.com/google-gemini/gemini-cli/pull/28352)
|
||||
- chore(caretaker): update vitest to v3.2.4 and add package-lock.json files by
|
||||
@chadd28 in [#28409](https://github.com/google-gemini/gemini-cli/pull/28409)
|
||||
- fix(core): rotate session ID on model fallback to prevent stateful API errors
|
||||
by @amelidev in
|
||||
[#28469](https://github.com/google-gemini/gemini-cli/pull/28469)
|
||||
- feat(caretaker-triage): post comment before auto-closing issues by @chadd28 in
|
||||
[#28411](https://github.com/google-gemini/gemini-cli/pull/28411)
|
||||
- fix(core): enforce HTTPS for GoogleCredentialsAuthProvider to prevent
|
||||
cleartext leakage by @amelidev in
|
||||
[#28517](https://github.com/google-gemini/gemini-cli/pull/28517)
|
||||
- fix(core): filter out thought parts from getHistoryTurns when context
|
||||
management is disabled by @DavidAPierce in
|
||||
[#28509](https://github.com/google-gemini/gemini-cli/pull/28509)
|
||||
- fix(a2a-server): normalize CRLF line endings to LF in getProposedContent by
|
||||
@luisfelipe-alt in
|
||||
[#28531](https://github.com/google-gemini/gemini-cli/pull/28531)
|
||||
- fix(core): enforce explicit tag length and validation in file keychain by
|
||||
@luisfelipe-alt in
|
||||
[#28523](https://github.com/google-gemini/gemini-cli/pull/28523)
|
||||
- chore/release: bump version to 0.54.0-nightly.20260728.gbef611950 by
|
||||
@gemini-cli-robot in
|
||||
[#27535](https://github.com/google-gemini/gemini-cli/pull/27535)
|
||||
[#28552](https://github.com/google-gemini/gemini-cli/pull/28552)
|
||||
- feat(pr-generator-db): implement Firestore concurrency dual-locking and test
|
||||
ingestion utilities by @joneba-google in
|
||||
[#28432](https://github.com/google-gemini/gemini-cli/pull/28432)
|
||||
- feat(pr-generator-agent): implement Antigravity agent runner and prompt
|
||||
templates … by @joneba-google in
|
||||
[#28434](https://github.com/google-gemini/gemini-cli/pull/28434)
|
||||
- fix(core): skip merged function-response turns when finding the active loop by
|
||||
@adamfweidman in
|
||||
[#28565](https://github.com/google-gemini/gemini-cli/pull/28565)
|
||||
- fix(patch): cherry-pick f47d6c6 to release/v0.54.0-preview.0-pr-28566 to patch
|
||||
version v0.54.0-preview.0 and create version 0.54.0-preview.1 by
|
||||
@gemini-cli-robot in
|
||||
[#28609](https://github.com/google-gemini/gemini-cli/pull/28609)
|
||||
|
||||
**Full Changelog**:
|
||||
https://github.com/google-gemini/gemini-cli/compare/v0.44.1...v0.45.0
|
||||
https://github.com/google-gemini/gemini-cli/compare/v0.53.1...v0.54.0
|
||||
|
||||
+92
-57
@@ -1,6 +1,6 @@
|
||||
# Preview release: v0.50.0-preview.1
|
||||
# Preview release: v0.55.0-preview.1
|
||||
|
||||
Released: June 25, 2026
|
||||
Released: August 06, 2026
|
||||
|
||||
Our preview release includes the latest, new, and experimental features. This
|
||||
release may not be as stable as our [latest weekly release](latest.md).
|
||||
@@ -13,66 +13,101 @@ npm install -g @google/gemini-cli@preview
|
||||
|
||||
## Highlights
|
||||
|
||||
- **GDC Service Identity Support**: Added support for GDC air-gapped Service
|
||||
Identity after a major auth library update.
|
||||
- **Standardised Tool Outputs**: Standardised tool output formatting to ensure
|
||||
consistency and readability across different CLI commands.
|
||||
- **Static Evaluation Analyzer**: Introduced a new static evaluation source
|
||||
analyzer to improve development and testing.
|
||||
- **Vulnerability Prevention**: Hardened CLI security by preventing path
|
||||
traversal vulnerabilities during the installation of Skills.
|
||||
- **Configuration & Error Hardening**: Migrated the `coreTools` configuration
|
||||
setting to `tools.core` and ensured zero-quota limits fail fast to prevent
|
||||
infinite retry loops.
|
||||
- **Antigravity Agent & PR Generator:** Integrated the Antigravity agent runner,
|
||||
Firestore dual-locking for concurrency, prompt templates, and ingestion
|
||||
testing utilities.
|
||||
- **Caretaker Triage & Issue Management:** Enhanced the issue triage workflow by
|
||||
automatically posting a comment before closing issues, and sanitizing and
|
||||
wrapping issue titles in `untrusted_context`.
|
||||
- **Core API & Session Stability:** Enforced HTTPS for
|
||||
GoogleCredentialsAuthProvider to prevent cleartext leakage, rotated session
|
||||
IDs on model fallback to prevent stateful API errors, and refined chat history
|
||||
by filtering out thought parts when context management is disabled.
|
||||
|
||||
## What's Changed
|
||||
|
||||
- fix/verify release npm ci ignore scripts by @rmedranollamas in
|
||||
[#28116](https://github.com/google-gemini/gemini-cli/pull/28116)
|
||||
- fix(ci): prevent workspace binary shadowing in release verification by @galz10
|
||||
in [#28132](https://github.com/google-gemini/gemini-cli/pull/28132)
|
||||
- Feat/tool registry discovery by @ved015 in
|
||||
[#28113](https://github.com/google-gemini/gemini-cli/pull/28113)
|
||||
- fix(ci): prevent bad NPM releases and promote job crashes by @galz10 in
|
||||
[#28147](https://github.com/google-gemini/gemini-cli/pull/28147)
|
||||
- chore(release): bump version to 0.48.0-nightly.20260609.g3a13b8eeb by
|
||||
- chore(release): bump version to 0.55.0-nightly.20260728.gd29268d36 by
|
||||
@gemini-cli-robot in
|
||||
[#27779](https://github.com/google-gemini/gemini-cli/pull/27779)
|
||||
- ci(dependabot): enable cooldown period for npm packages by @ruomengz in
|
||||
[#27743](https://github.com/google-gemini/gemini-cli/pull/27743)
|
||||
- refactor(core): standardize tool output formatting by @galz10 in
|
||||
[#27772](https://github.com/google-gemini/gemini-cli/pull/27772)
|
||||
- ci: update workflow logging and policy configurations by @galz10 in
|
||||
[#27853](https://github.com/google-gemini/gemini-cli/pull/27853)
|
||||
- fix(core): Ensure zero-quota limits fail fast to prevent retry loop hang by
|
||||
[#28569](https://github.com/google-gemini/gemini-cli/pull/28569)
|
||||
- Changelog for v0.54.0-preview.0 by @gemini-cli-robot in
|
||||
[#28567](https://github.com/google-gemini/gemini-cli/pull/28567)
|
||||
- Changelog for v0.53.0 by @gemini-cli-robot in
|
||||
[#28568](https://github.com/google-gemini/gemini-cli/pull/28568)
|
||||
- chore/release: bump version to 0.55.0-nightly.20260729.g3499c84f7 by
|
||||
@gemini-cli-robot in
|
||||
[#28573](https://github.com/google-gemini/gemini-cli/pull/28573)
|
||||
- fix(core): classify capacity exhaustion as terminal to prevent retry hangs by
|
||||
@luisfelipe-alt in
|
||||
[#27698](https://github.com/google-gemini/gemini-cli/pull/27698)
|
||||
- fix(core): handle multi-line escaped quotes in stripShellWrapper by
|
||||
@sanchezcoraspe in
|
||||
[#27467](https://github.com/google-gemini/gemini-cli/pull/27467)
|
||||
- fix(cli): prevent path traversal vulnerabilities during skill install… by
|
||||
@ompatel-aiml in
|
||||
[#27767](https://github.com/google-gemini/gemini-cli/pull/27767)
|
||||
- Fix/pending tools and trust overrides by @jvargassanchez-dot in
|
||||
[#27854](https://github.com/google-gemini/gemini-cli/pull/27854)
|
||||
- ci: use internal environment for scheduled nightly releases (#27865) by
|
||||
@rmedranollamas in
|
||||
[#27939](https://github.com/google-gemini/gemini-cli/pull/27939)
|
||||
- feat(core): Support GDC air-gapped Service Identity after auth library update
|
||||
by @sidhantgoyal-droid in
|
||||
[#27956](https://github.com/google-gemini/gemini-cli/pull/27956)
|
||||
- fix(cli): handle tmux false positive background detection by @amelidev in
|
||||
[#27572](https://github.com/google-gemini/gemini-cli/pull/27572)
|
||||
- Add static eval source analyzer by @ved015 in
|
||||
[#27631](https://github.com/google-gemini/gemini-cli/pull/27631)
|
||||
- fix(config): migrate coreTools setting to tools.core by @galz10 in
|
||||
[#27947](https://github.com/google-gemini/gemini-cli/pull/27947)
|
||||
- fix(core-tools): resolve defensive path resolution for at-reference files by
|
||||
[#28599](https://github.com/google-gemini/gemini-cli/pull/28599)
|
||||
- fix(core,cli): propagate InvalidStreamError details to UI for specific empty
|
||||
response guidance by @DavidAPierce in
|
||||
[#28566](https://github.com/google-gemini/gemini-cli/pull/28566)
|
||||
- fix(cli): fall back to embedded macOS seatbelt profiles if missing by
|
||||
@amelidev in [#28551](https://github.com/google-gemini/gemini-cli/pull/28551)
|
||||
- feat(pr-generator-core): add environment config parser, command executor,
|
||||
GitHub R… by @joneba-google in
|
||||
[#28435](https://github.com/google-gemini/gemini-cli/pull/28435)
|
||||
- feat(pr-generator-orchestrator): implement iterative bug-fixing state machine
|
||||
and container worker entrypoint by @joneba-google in
|
||||
[#28433](https://github.com/google-gemini/gemini-cli/pull/28433)
|
||||
- feat(pr-generator-infra): configure Cloud Run job, Workflows definition, and
|
||||
Dockerfile by @joneba-google in
|
||||
[#28431](https://github.com/google-gemini/gemini-cli/pull/28431)
|
||||
- fix(release): handle npm dist-tag deletion failures on registries that forbid
|
||||
it by @DavidAPierce in
|
||||
[#28694](https://github.com/google-gemini/gemini-cli/pull/28694)
|
||||
- fix(core): stop a new user message fusing into an unanswered tool response by
|
||||
@adamfweidman in
|
||||
[#28700](https://github.com/google-gemini/gemini-cli/pull/28700)
|
||||
- fix(core,cli): repair /compress session reload and quota-fallback tool
|
||||
response loss by @adamfweidman in
|
||||
[#28672](https://github.com/google-gemini/gemini-cli/pull/28672)
|
||||
- fix(core): preserve functionCall thoughtSignature when stripping thought parts
|
||||
by @sarbojitrana in
|
||||
[#28607](https://github.com/google-gemini/gemini-cli/pull/28607)
|
||||
- fix(core): unwrap and parse nested gaxios streaming errors from cause message
|
||||
by @luisfelipe-alt in
|
||||
[#28689](https://github.com/google-gemini/gemini-cli/pull/28689)
|
||||
- Changelog for v0.53.0-preview.0 by @gemini-cli-robot in
|
||||
[#28507](https://github.com/google-gemini/gemini-cli/pull/28507)
|
||||
- Changelog for v0.52.0 by @gemini-cli-robot in
|
||||
[#28508](https://github.com/google-gemini/gemini-cli/pull/28508)
|
||||
- chore(release): bump version to 0.54.0-nightly.20260722.gf743ab579 by
|
||||
@gemini-cli-robot in
|
||||
[#28510](https://github.com/google-gemini/gemini-cli/pull/28510)
|
||||
- fix(caretaker): sanitize and wrap issue title in untrusted_context by @chadd28
|
||||
in [#28352](https://github.com/google-gemini/gemini-cli/pull/28352)
|
||||
- chore(caretaker): update vitest to v3.2.4 and add package-lock.json files by
|
||||
@chadd28 in [#28409](https://github.com/google-gemini/gemini-cli/pull/28409)
|
||||
- fix(core): rotate session ID on model fallback to prevent stateful API errors
|
||||
by @amelidev in
|
||||
[#28469](https://github.com/google-gemini/gemini-cli/pull/28469)
|
||||
- feat(caretaker-triage): post comment before auto-closing issues by @chadd28 in
|
||||
[#28411](https://github.com/google-gemini/gemini-cli/pull/28411)
|
||||
- fix(core): enforce HTTPS for GoogleCredentialsAuthProvider to prevent
|
||||
cleartext leakage by @amelidev in
|
||||
[#28517](https://github.com/google-gemini/gemini-cli/pull/28517)
|
||||
- fix(core): filter out thought parts from getHistoryTurns when context
|
||||
management is disabled by @DavidAPierce in
|
||||
[#28509](https://github.com/google-gemini/gemini-cli/pull/28509)
|
||||
- fix(a2a-server): normalize CRLF line endings to LF in getProposedContent by
|
||||
@luisfelipe-alt in
|
||||
[#27943](https://github.com/google-gemini/gemini-cli/pull/27943)
|
||||
- Revert "fix(core-tools): resolve defensive path resolution for at-reference
|
||||
files" by @galz10 in
|
||||
[#27992](https://github.com/google-gemini/gemini-cli/pull/27992)
|
||||
[#28531](https://github.com/google-gemini/gemini-cli/pull/28531)
|
||||
- fix(core): enforce explicit tag length and validation in file keychain by
|
||||
@luisfelipe-alt in
|
||||
[#28523](https://github.com/google-gemini/gemini-cli/pull/28523)
|
||||
- chore/release: bump version to 0.54.0-nightly.20260728.gbef611950 by
|
||||
@gemini-cli-robot in
|
||||
[#28552](https://github.com/google-gemini/gemini-cli/pull/28552)
|
||||
- feat(pr-generator-db): implement Firestore concurrency dual-locking and test
|
||||
ingestion utilities by @joneba-google in
|
||||
[#28432](https://github.com/google-gemini/gemini-cli/pull/28432)
|
||||
- feat(pr-generator-agent): implement Antigravity agent runner and prompt
|
||||
templates … by @joneba-google in
|
||||
[#28434](https://github.com/google-gemini/gemini-cli/pull/28434)
|
||||
- fix(core): skip merged function-response turns when finding the active loop by
|
||||
@adamfweidman in
|
||||
[#28565](https://github.com/google-gemini/gemini-cli/pull/28565)
|
||||
|
||||
**Full Changelog**:
|
||||
https://github.com/google-gemini/gemini-cli/compare/v0.47.0-preview.0...v0.50.0-preview.1
|
||||
https://github.com/google-gemini/gemini-cli/compare/v0.53.0-preview.0...v0.55.0-preview.1
|
||||
|
||||
+3
-2
@@ -87,8 +87,9 @@ preferred container solution.
|
||||
|
||||
Lightweight, built-in sandboxing using `sandbox-exec`.
|
||||
|
||||
**Default profile**: `permissive-open` - restricts writes outside project
|
||||
directory but allows most other operations.
|
||||
**Default profile**: `permissive-open` - denies operations by default; confines
|
||||
writes to the project directory while allowing broad file reads and network
|
||||
access.
|
||||
|
||||
Built-in profiles (set via `SEATBELT_PROFILE` env var):
|
||||
|
||||
|
||||
@@ -2736,10 +2736,10 @@ the `advanced.excludedEnvVars` setting in your `settings.json` file.
|
||||
- Run the CLI once with this set to generate the file.
|
||||
- **`SEATBELT_PROFILE`** (macOS specific):
|
||||
- Switches the Seatbelt (`sandbox-exec`) profile on macOS.
|
||||
- `permissive-open`: (Default) Restricts writes to the project folder (and a
|
||||
few other folders, see
|
||||
`packages/cli/src/utils/sandbox-macos-permissive-open.sb`) but allows other
|
||||
operations.
|
||||
- `permissive-open`: (Default) Denies operations by default, confining writes
|
||||
to the project folder (and a few other folders, see
|
||||
`packages/cli/src/utils/sandbox-macos-permissive-open.sb`) while allowing
|
||||
broad file reads and network access.
|
||||
- `restrictive-open`: Declines operations by default, allows network.
|
||||
- `strict-open`: Restricts both reads and writes to the working directory,
|
||||
allows network.
|
||||
|
||||
@@ -217,6 +217,10 @@
|
||||
{
|
||||
"label": "Development",
|
||||
"items": [
|
||||
{
|
||||
"label": "Behavioral evaluations",
|
||||
"slug": "docs/behavioral-evals"
|
||||
},
|
||||
{ "label": "Contribution guide", "slug": "docs/contributing" },
|
||||
{ "label": "Integration testing", "slug": "docs/integration-tests" },
|
||||
{
|
||||
|
||||
@@ -8,12 +8,10 @@ import { describe, expect } from 'vitest';
|
||||
import { evalTest, assertModelHasOutput } from './test-helper.js';
|
||||
|
||||
describe('Hierarchical Memory', () => {
|
||||
const conflictResolutionTest =
|
||||
'Agent follows hierarchy for contradictory instructions';
|
||||
evalTest('ALWAYS_PASSES', {
|
||||
suiteName: 'default',
|
||||
suiteType: 'behavioral',
|
||||
name: conflictResolutionTest,
|
||||
name: 'Agent follows hierarchy for contradictory instructions',
|
||||
params: {
|
||||
settings: {
|
||||
security: {
|
||||
@@ -47,11 +45,10 @@ What is my favorite fruit? Tell me just the name of the fruit.`,
|
||||
},
|
||||
});
|
||||
|
||||
const provenanceAwarenessTest = 'Agent is aware of memory provenance';
|
||||
evalTest('USUALLY_PASSES', {
|
||||
suiteName: 'default',
|
||||
suiteType: 'behavioral',
|
||||
name: provenanceAwarenessTest,
|
||||
name: 'Agent is aware of memory provenance',
|
||||
params: {
|
||||
settings: {
|
||||
security: {
|
||||
@@ -88,11 +85,10 @@ Provide the answer as an XML block like this:
|
||||
},
|
||||
});
|
||||
|
||||
const extensionVsGlobalTest = 'Extension memory wins over Global memory';
|
||||
evalTest('ALWAYS_PASSES', {
|
||||
suiteName: 'default',
|
||||
suiteType: 'behavioral',
|
||||
name: extensionVsGlobalTest,
|
||||
name: 'Extension memory wins over Global memory',
|
||||
params: {
|
||||
settings: {
|
||||
security: {
|
||||
|
||||
@@ -74,12 +74,10 @@ async function waitForSessionScratchpad(
|
||||
}
|
||||
|
||||
describe('memory persistence', () => {
|
||||
const proactiveMemoryFromLongSession =
|
||||
'Agent saves preference from earlier in conversation history';
|
||||
evalTest('USUALLY_PASSES', {
|
||||
suiteName: 'default',
|
||||
suiteType: 'behavioral',
|
||||
name: proactiveMemoryFromLongSession,
|
||||
name: 'Agent saves preference from earlier in conversation history',
|
||||
messages: [
|
||||
{
|
||||
id: 'msg-1',
|
||||
@@ -195,12 +193,10 @@ describe('memory persistence', () => {
|
||||
},
|
||||
});
|
||||
|
||||
const memoryRoutesTeamConventionsToProjectGemini =
|
||||
'Agent routes team-shared project conventions to ./GEMINI.md';
|
||||
evalTest('USUALLY_PASSES', {
|
||||
suiteName: 'default',
|
||||
suiteType: 'behavioral',
|
||||
name: memoryRoutesTeamConventionsToProjectGemini,
|
||||
name: 'Agent routes team-shared project conventions to ./GEMINI.md',
|
||||
messages: [
|
||||
{
|
||||
id: 'msg-1',
|
||||
@@ -303,12 +299,10 @@ describe('memory persistence', () => {
|
||||
},
|
||||
});
|
||||
|
||||
const memorySessionScratchpad =
|
||||
'Session summary persists memory scratchpad for memory-saving sessions';
|
||||
evalTest('USUALLY_PASSES', {
|
||||
suiteName: 'default',
|
||||
suiteType: 'behavioral',
|
||||
name: memorySessionScratchpad,
|
||||
name: 'Session summary persists memory scratchpad for memory-saving sessions',
|
||||
sessionId: 'memory-scratchpad-eval',
|
||||
messages: [
|
||||
{
|
||||
@@ -395,12 +389,10 @@ describe('memory persistence', () => {
|
||||
},
|
||||
});
|
||||
|
||||
const memoryRoutesUserProject =
|
||||
'Agent routes personal-to-user project notes to user-project memory';
|
||||
evalTest('USUALLY_PASSES', {
|
||||
suiteName: 'default',
|
||||
suiteType: 'behavioral',
|
||||
name: memoryRoutesUserProject,
|
||||
name: 'Agent routes personal-to-user project notes to user-project memory',
|
||||
prompt: `Please remember my personal local dev setup for THIS project's Postgres database. This is private to my machine — do NOT commit it to the repo.
|
||||
|
||||
Connection details:
|
||||
@@ -486,12 +478,10 @@ Quirks to remember:
|
||||
},
|
||||
});
|
||||
|
||||
const memoryRoutesCrossProjectToGlobal =
|
||||
'Agent routes cross-project personal preferences to ~/.gemini/GEMINI.md';
|
||||
evalTest('USUALLY_PASSES', {
|
||||
suiteName: 'default',
|
||||
suiteType: 'behavioral',
|
||||
name: memoryRoutesCrossProjectToGlobal,
|
||||
name: 'Agent routes cross-project personal preferences to ~/.gemini/GEMINI.md',
|
||||
prompt:
|
||||
'Please remember this about me in general: across all my projects I always prefer Prettier with single quotes and trailing commas, and I always prefer tabs over spaces for indentation. These are my personal coding-style defaults that follow me into every workspace.',
|
||||
assert: async (rig, result) => {
|
||||
|
||||
@@ -21,7 +21,7 @@ function snapshotEvalTest(policy: EvalPolicy, evalCase: ComponentEvalCase) {
|
||||
describe('snapshot_fidelity', () => {
|
||||
snapshotEvalTest('ALWAYS_PASSES', {
|
||||
suiteName: 'default',
|
||||
suiteType: 'behavioral',
|
||||
suiteType: 'component-level',
|
||||
name: 'SnapshotGenerator strictly retains specific empirical facts',
|
||||
assert: async (config) => {
|
||||
// 1. Construct a highly specific mock transcript containing 3 empirical facts we can test for:
|
||||
|
||||
@@ -35,7 +35,7 @@ describe('Interactive file system', () => {
|
||||
const run = await rig.runInteractive();
|
||||
|
||||
// Step 1: Read the file
|
||||
const readPrompt = `Read the version from ${fileName}`;
|
||||
const readPrompt = `Read the version from ${fileName} using the read_file tool`;
|
||||
await run.type(readPrompt);
|
||||
await run.type('\r');
|
||||
|
||||
|
||||
Generated
+91
-1442
File diff suppressed because it is too large
Load Diff
+8
-2
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@google/gemini-cli",
|
||||
"version": "0.51.0-nightly.20260625.g3fbf93e26",
|
||||
"version": "0.56.0-nightly.20260806.g761f604c1",
|
||||
"engines": {
|
||||
"node": ">=20.0.0"
|
||||
},
|
||||
@@ -14,7 +14,7 @@
|
||||
"url": "git+https://github.com/google-gemini/gemini-cli.git"
|
||||
},
|
||||
"config": {
|
||||
"sandboxImageUri": "us-docker.pkg.dev/gemini-code-dev/gemini-cli/sandbox:0.51.0-nightly.20260625.g3fbf93e26"
|
||||
"sandboxImageUri": "us-docker.pkg.dev/gemini-code-dev/gemini-cli/sandbox:0.56.0-nightly.20260806.g761f604c1"
|
||||
},
|
||||
"scripts": {
|
||||
"start": "cross-env NODE_ENV=development node scripts/start.js",
|
||||
@@ -32,8 +32,11 @@
|
||||
"schema:settings": "tsx ./scripts/generate-settings-schema.ts",
|
||||
"docs:settings": "tsx ./scripts/generate-settings-doc.ts",
|
||||
"docs:keybindings": "tsx ./scripts/generate-keybindings-doc.ts",
|
||||
"eval:validate": "tsx ./scripts/eval-validate-cli.ts",
|
||||
"eval:inventory": "tsx ./scripts/eval-inventory-cli.ts",
|
||||
"eval:inventory:json": "tsx ./scripts/eval-inventory-cli.ts --json",
|
||||
"eval:report": "tsx ./scripts/eval-report-cli.ts",
|
||||
"eval:coverage": "tsx ./scripts/eval-coverage-cli.ts",
|
||||
"build": "node scripts/build.js",
|
||||
"build-and-start": "npm run build && npm run start --",
|
||||
"build:vscode": "node scripts/build_vscode_companion.js",
|
||||
@@ -95,8 +98,10 @@
|
||||
],
|
||||
"devDependencies": {
|
||||
"@agentclientprotocol/sdk": "0.16.1",
|
||||
"@modelcontextprotocol/sdk": "1.23.0",
|
||||
"read-package-up": "11.0.0",
|
||||
"@octokit/rest": "22.0.0",
|
||||
"@types/express": "5.0.3",
|
||||
"@types/marked": "5.0.2",
|
||||
"@types/mime-types": "3.0.1",
|
||||
"@types/minimatch": "5.1.2",
|
||||
@@ -121,6 +126,7 @@
|
||||
"eslint-plugin-import": "2.32.0",
|
||||
"eslint-plugin-react": "7.37.5",
|
||||
"eslint-plugin-react-hooks": "5.2.0",
|
||||
"express": "5.1.0",
|
||||
"glob": "12.0.0",
|
||||
"globals": "16.0.0",
|
||||
"google-artifactregistry-auth": "3.4.0",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@google/gemini-cli-a2a-server",
|
||||
"version": "0.51.0-nightly.20260625.g3fbf93e26",
|
||||
"version": "0.56.0-nightly.20260806.g761f604c1",
|
||||
"description": "Gemini CLI A2A Server",
|
||||
"repository": {
|
||||
"type": "git",
|
||||
|
||||
@@ -14,7 +14,24 @@ import type {
|
||||
import { EventEmitter } from 'node:events';
|
||||
import { requestStorage } from '../http/requestStorage.js';
|
||||
|
||||
vi.mock('../utils/path_utils.js', () => ({
|
||||
validateWorkspacePath: vi
|
||||
.fn()
|
||||
.mockImplementation(async (path?: string) => path || process.cwd()),
|
||||
}));
|
||||
|
||||
// Mocks for constructor dependencies
|
||||
vi.mock('@google/gemini-cli-core', () => ({
|
||||
GeminiEventType: {
|
||||
PRIMARY_TURN_STARTED: 'PRIMARY_TURN_STARTED',
|
||||
SECONDARY_TURN_STARTED: 'SECONDARY_TURN_STARTED',
|
||||
},
|
||||
SimpleExtensionLoader: vi.fn(),
|
||||
checkPathTrust: vi.fn().mockReturnValue({ isTrusted: false }),
|
||||
isHeadlessMode: vi.fn().mockReturnValue(true),
|
||||
resolveToRealPath: vi.fn().mockImplementation((p) => p),
|
||||
}));
|
||||
|
||||
vi.mock('../config/config.js', () => ({
|
||||
loadConfig: vi.fn().mockReturnValue({
|
||||
getSessionId: () => 'test-session',
|
||||
@@ -24,6 +41,10 @@ vi.mock('../config/config.js', () => ({
|
||||
loadEnvironment: vi.fn(),
|
||||
setIsTrusted: vi.fn().mockReturnValue(false),
|
||||
setTargetDir: vi.fn().mockReturnValue('/tmp'),
|
||||
envStorage: {
|
||||
run: (env: Record<string, string>, cb: () => unknown) => cb(),
|
||||
},
|
||||
cwdSymbol: Symbol('cwd'),
|
||||
}));
|
||||
|
||||
vi.mock('../config/settings.js', () => ({
|
||||
@@ -300,4 +321,125 @@ describe('CoderAgentExecutor', () => {
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
it('cancelTask should abort the active execution loop', async () => {
|
||||
const abortSpy = vi.spyOn(AbortController.prototype, 'abort');
|
||||
const taskId = 'test-task-to-cancel';
|
||||
const contextId = 'test-context';
|
||||
|
||||
const mockSocket = new EventEmitter();
|
||||
(requestStorage.getStore as Mock).mockReturnValue({
|
||||
req: { socket: mockSocket },
|
||||
});
|
||||
|
||||
const requestContext = {
|
||||
userMessage: {
|
||||
messageId: 'msg-1',
|
||||
taskId,
|
||||
contextId,
|
||||
parts: [{ kind: 'text', text: 'a long running prompt' }],
|
||||
metadata: {
|
||||
coderAgent: { kind: 'agent-settings', workspacePath: '/tmp' },
|
||||
},
|
||||
},
|
||||
} as unknown as RequestContext;
|
||||
|
||||
// Don't await this, let it run in the background.
|
||||
let primaryError: Error | null = null;
|
||||
const primaryPromise = executor.execute(requestContext, mockEventBus);
|
||||
primaryPromise.catch((err) => {
|
||||
primaryError = err as Error;
|
||||
});
|
||||
|
||||
// Poll until the task is registered in the executor to avoid flaky timeouts in slow CI environments.
|
||||
let attempts = 0;
|
||||
while (!executor.getTask(taskId)) {
|
||||
if (primaryError) {
|
||||
throw new Error(`Primary execution failed early: ${primaryError}`);
|
||||
}
|
||||
if (attempts++ > 100) {
|
||||
// 100 * 5ms = 500ms timeout
|
||||
throw new Error('Timed out waiting for task to be registered');
|
||||
}
|
||||
await new Promise((resolve) => setTimeout(resolve, 5));
|
||||
}
|
||||
|
||||
const wrapper = executor.getTask(taskId);
|
||||
expect(wrapper).toBeDefined();
|
||||
const setTaskStateSpy = vi
|
||||
.spyOn(wrapper!.task, 'setTaskStateAndPublishUpdate')
|
||||
.mockImplementation((newState) => {
|
||||
// Make the mock realistic: actually update the state when called.
|
||||
wrapper!.task.taskState = newState;
|
||||
});
|
||||
|
||||
// Now, cancel the task.
|
||||
await executor.cancelTask(taskId, mockEventBus);
|
||||
|
||||
// Verify that the abort method on the controller was called and state was updated.
|
||||
expect(abortSpy).toHaveBeenCalledOnce();
|
||||
expect(setTaskStateSpy).toHaveBeenCalledWith(
|
||||
'canceled',
|
||||
expect.any(Object),
|
||||
'Task canceled by user request.',
|
||||
undefined,
|
||||
true,
|
||||
);
|
||||
|
||||
// Clean up the test by allowing the promise to resolve.
|
||||
// The abort call should have unblocked the acceptUserMessage generator.
|
||||
await primaryPromise;
|
||||
|
||||
// Verify task is evicted from cache
|
||||
expect(executor.getTask(taskId)).toBeUndefined();
|
||||
|
||||
abortSpy.mockRestore();
|
||||
});
|
||||
|
||||
it('cancelTask should explicitly save task state to TaskStore and evict task during active aborts', async () => {
|
||||
const taskId = 'test-task-active-abort-save';
|
||||
const contextId = 'test-context';
|
||||
|
||||
const mockSocket = new EventEmitter();
|
||||
(requestStorage.getStore as Mock).mockReturnValue({
|
||||
req: { socket: mockSocket },
|
||||
});
|
||||
|
||||
const requestContext = {
|
||||
userMessage: {
|
||||
messageId: 'msg-1',
|
||||
taskId,
|
||||
contextId,
|
||||
parts: [{ kind: 'text', text: 'a long running prompt' }],
|
||||
metadata: {
|
||||
coderAgent: { kind: 'agent-settings', workspacePath: '/tmp' },
|
||||
},
|
||||
},
|
||||
} as unknown as RequestContext;
|
||||
|
||||
const primaryPromise = executor.execute(requestContext, mockEventBus);
|
||||
|
||||
// Wait for task to be registered
|
||||
let attempts = 0;
|
||||
while (!executor.getTask(taskId)) {
|
||||
if (attempts++ > 100) {
|
||||
throw new Error('Timed out waiting for task to be registered');
|
||||
}
|
||||
await new Promise((resolve) => setTimeout(resolve, 5));
|
||||
}
|
||||
|
||||
const wrapper = executor.getTask(taskId)!;
|
||||
const saveSpy = vi.spyOn(mockTaskStore, 'save');
|
||||
|
||||
// Now, cancel the task.
|
||||
await executor.cancelTask(taskId, mockEventBus);
|
||||
|
||||
// Verify that the task state was saved to TaskStore during cancelTask
|
||||
expect(saveSpy).toHaveBeenCalled();
|
||||
expect(wrapper.task.dispose).toHaveBeenCalled();
|
||||
expect(executor.getTask(taskId)).toBeUndefined();
|
||||
|
||||
// Clean up the test by allowing the promise to resolve.
|
||||
await primaryPromise;
|
||||
});
|
||||
});
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -752,4 +752,107 @@ describe('Task', () => {
|
||||
expect(changed3).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe('getProposedContent (CRLF Line Ending Normalization)', () => {
|
||||
it('should successfully replace LF-based strings in CRLF-based files', async () => {
|
||||
const fs = await import('node:fs');
|
||||
const path = await import('node:path');
|
||||
const os = await import('node:os');
|
||||
|
||||
const mockConfig = createMockConfig({
|
||||
getTargetDir: () => os.tmpdir(),
|
||||
validatePathAccess: () => null,
|
||||
});
|
||||
const mockEventBus: ExecutionEventBus = {
|
||||
publish: vi.fn(),
|
||||
on: vi.fn(),
|
||||
off: vi.fn(),
|
||||
once: vi.fn(),
|
||||
removeAllListeners: vi.fn(),
|
||||
finished: vi.fn(),
|
||||
};
|
||||
|
||||
// @ts-expect-error - Calling private constructor
|
||||
const task = new Task(
|
||||
'task-id',
|
||||
'context-id',
|
||||
mockConfig as Config,
|
||||
mockEventBus,
|
||||
);
|
||||
|
||||
const tempFile = path.resolve(os.tmpdir(), 'crlf_test_file.txt');
|
||||
const crlfContent = 'line1\r\nline2\r\nline3\r\n';
|
||||
fs.writeFileSync(tempFile, crlfContent, 'utf8');
|
||||
|
||||
try {
|
||||
const oldString = 'line2\n';
|
||||
const newString = 'line2-optimized\n';
|
||||
|
||||
const result = await task['getProposedContent'](
|
||||
tempFile,
|
||||
oldString,
|
||||
newString,
|
||||
);
|
||||
|
||||
expect(result).toContain('line2-optimized');
|
||||
expect(result).toContain('\r\n'); // It should preserve the original CRLF line endings
|
||||
} finally {
|
||||
if (fs.existsSync(tempFile)) {
|
||||
fs.unlinkSync(tempFile);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
it('should successfully replace CRLF-based strings in CRLF-based files by normalizing all to LF', async () => {
|
||||
const fs = await import('node:fs');
|
||||
const path = await import('node:path');
|
||||
const os = await import('node:os');
|
||||
|
||||
const mockConfig = createMockConfig({
|
||||
getTargetDir: () => os.tmpdir(),
|
||||
validatePathAccess: () => null,
|
||||
});
|
||||
const mockEventBus: ExecutionEventBus = {
|
||||
publish: vi.fn(),
|
||||
on: vi.fn(),
|
||||
off: vi.fn(),
|
||||
once: vi.fn(),
|
||||
removeAllListeners: vi.fn(),
|
||||
finished: vi.fn(),
|
||||
};
|
||||
|
||||
// @ts-expect-error - Calling private constructor
|
||||
const task = new Task(
|
||||
'task-id',
|
||||
'context-id',
|
||||
mockConfig as Config,
|
||||
mockEventBus,
|
||||
);
|
||||
|
||||
const tempFile = path.resolve(
|
||||
os.tmpdir(),
|
||||
'crlf_test_file_crlf_inputs.txt',
|
||||
);
|
||||
const crlfContent = 'line1\r\nline2\r\nline3\r\n';
|
||||
fs.writeFileSync(tempFile, crlfContent, 'utf8');
|
||||
|
||||
try {
|
||||
const oldString = 'line2\r\n';
|
||||
const newString = 'line2-optimized\r\n';
|
||||
|
||||
const result = await task['getProposedContent'](
|
||||
tempFile,
|
||||
oldString,
|
||||
newString,
|
||||
);
|
||||
|
||||
expect(result).toContain('line2-optimized');
|
||||
expect(result).toContain('\r\n'); // It should preserve the original CRLF line endings
|
||||
} finally {
|
||||
if (fs.existsSync(tempFile)) {
|
||||
fs.unlinkSync(tempFile);
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -131,9 +131,11 @@ export class Task {
|
||||
this.autoExecute = autoExecute;
|
||||
this.config.setFallbackModelHandler(
|
||||
// For a2a-server, we want to automatically switch to the fallback model
|
||||
// for future requests without retrying the current one. The 'stop'
|
||||
// intent achieves this.
|
||||
async () => 'stop',
|
||||
// for future requests without retrying the current one.
|
||||
async (failedModel, fallbackModel) => {
|
||||
this.config.activateFallbackMode(fallbackModel, failedModel);
|
||||
return 'stop';
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
@@ -666,13 +668,18 @@ export class Task {
|
||||
}
|
||||
|
||||
try {
|
||||
const currentContent = await fs.readFile(resolvedPath, 'utf8');
|
||||
return this._applyReplacement(
|
||||
const rawContent = await fs.readFile(resolvedPath, 'utf8');
|
||||
const hasCrlf = rawContent.includes('\r\n');
|
||||
const currentContent = rawContent.replace(/\r\n/g, '\n');
|
||||
const normalizedOldString = old_string.replace(/\r\n/g, '\n');
|
||||
const normalizedNewString = new_string.replace(/\r\n/g, '\n');
|
||||
const proposedContent = this._applyReplacement(
|
||||
currentContent,
|
||||
old_string,
|
||||
new_string,
|
||||
old_string === '' && currentContent === '',
|
||||
normalizedOldString,
|
||||
normalizedNewString,
|
||||
normalizedOldString === '' && currentContent === '',
|
||||
);
|
||||
return hasCrlf ? proposedContent.replace(/\n/g, '\r\n') : proposedContent;
|
||||
} catch (err) {
|
||||
if (!isNodeError(err) || err.code !== 'ENOENT') throw err;
|
||||
return '';
|
||||
@@ -1092,19 +1099,21 @@ export class Task {
|
||||
logger.info(
|
||||
`[Task] Adding ${completedTools.length} tool responses to history without generating a new response.`,
|
||||
);
|
||||
const responsesToAdd = completedTools.flatMap(
|
||||
(toolCall) => toolCall.response.responseParts,
|
||||
);
|
||||
|
||||
for (const response of responsesToAdd) {
|
||||
let parts: genAiPart[];
|
||||
if (Array.isArray(response)) {
|
||||
parts = response;
|
||||
} else if (typeof response === 'string') {
|
||||
parts = [{ text: response }];
|
||||
} else {
|
||||
parts = [response];
|
||||
const parts: genAiPart[] = [];
|
||||
for (const toolCall of completedTools) {
|
||||
const response = toolCall.response?.responseParts;
|
||||
if (!response) {
|
||||
continue;
|
||||
}
|
||||
if (Array.isArray(response)) {
|
||||
parts.push(...response);
|
||||
} else if (typeof response === 'string') {
|
||||
parts.push({ text: response });
|
||||
} else {
|
||||
parts.push(response);
|
||||
}
|
||||
}
|
||||
if (parts.length > 0) {
|
||||
// eslint-disable-next-line @typescript-eslint/no-floating-promises
|
||||
this.geminiClient.addHistory({
|
||||
role: 'user',
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as dotenv from 'dotenv';
|
||||
import { AsyncLocalStorage } from 'node:async_hooks';
|
||||
|
||||
import {
|
||||
AuthType,
|
||||
@@ -17,6 +18,7 @@ import {
|
||||
startupProfiler,
|
||||
PREVIEW_GEMINI_MODEL,
|
||||
homedir,
|
||||
tmpdir,
|
||||
GitService,
|
||||
fetchAdminControlsOnce,
|
||||
getCodeAssistServer,
|
||||
@@ -28,28 +30,246 @@ import {
|
||||
type TelemetryTarget,
|
||||
type ConfigParameters,
|
||||
type ExtensionLoader,
|
||||
resolveToRealPath,
|
||||
} from '@google/gemini-cli-core';
|
||||
|
||||
import { logger } from '../utils/logger.js';
|
||||
import type { Settings } from './settings.js';
|
||||
import { type AgentSettings, CoderAgentEvent } from '../types.js';
|
||||
|
||||
const INITIAL_FOLDER_TRUST = process.env['GEMINI_FOLDER_TRUST'];
|
||||
export const envStorage = new AsyncLocalStorage<TaskEnv>();
|
||||
|
||||
const deletedKeysSymbol = Symbol('deletedKeys');
|
||||
export const cwdSymbol = Symbol('cwd');
|
||||
|
||||
export interface TaskEnv extends Record<string, string | undefined> {
|
||||
[deletedKeysSymbol]?: Set<string>;
|
||||
[cwdSymbol]?: string;
|
||||
}
|
||||
|
||||
// Set up a Proxy on process.env to intercept reads and writes, isolating environment variables per task
|
||||
const originalEnv = process.env;
|
||||
const envProxy = new Proxy(originalEnv, {
|
||||
get(target, prop) {
|
||||
if (typeof prop === 'string') {
|
||||
const taskEnv = envStorage.getStore();
|
||||
if (taskEnv) {
|
||||
const deleted = taskEnv[deletedKeysSymbol];
|
||||
if (deleted?.has(prop)) {
|
||||
return undefined;
|
||||
}
|
||||
if (Object.prototype.hasOwnProperty.call(taskEnv, prop)) {
|
||||
return taskEnv[prop];
|
||||
}
|
||||
}
|
||||
return target[prop];
|
||||
}
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any, @typescript-eslint/no-unsafe-type-assertion
|
||||
return target[prop as any];
|
||||
},
|
||||
has(target, prop) {
|
||||
if (typeof prop === 'string') {
|
||||
const taskEnv = envStorage.getStore();
|
||||
if (taskEnv) {
|
||||
const deleted = taskEnv[deletedKeysSymbol];
|
||||
if (deleted?.has(prop)) {
|
||||
return false;
|
||||
}
|
||||
if (Object.prototype.hasOwnProperty.call(taskEnv, prop)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return prop in target;
|
||||
}
|
||||
return prop in target;
|
||||
},
|
||||
set(target, prop, value) {
|
||||
if (typeof prop === 'string') {
|
||||
if (
|
||||
prop === '__proto__' ||
|
||||
prop === 'constructor' ||
|
||||
prop === 'prototype'
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
const taskEnv = envStorage.getStore();
|
||||
if (taskEnv) {
|
||||
taskEnv[deletedKeysSymbol]?.delete(prop);
|
||||
taskEnv[prop] = String(value);
|
||||
return true;
|
||||
}
|
||||
target[prop] = String(value);
|
||||
return true;
|
||||
}
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any, @typescript-eslint/no-unsafe-type-assertion, @typescript-eslint/no-unsafe-assignment
|
||||
target[prop as any] = value;
|
||||
return true;
|
||||
},
|
||||
deleteProperty(target, prop) {
|
||||
if (typeof prop === 'string') {
|
||||
if (
|
||||
prop === '__proto__' ||
|
||||
prop === 'constructor' ||
|
||||
prop === 'prototype'
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
const taskEnv = envStorage.getStore();
|
||||
if (taskEnv) {
|
||||
delete taskEnv[prop];
|
||||
(taskEnv[deletedKeysSymbol] ??= new Set()).add(prop);
|
||||
return true;
|
||||
}
|
||||
delete target[prop];
|
||||
return true;
|
||||
}
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any, @typescript-eslint/no-unsafe-type-assertion
|
||||
return delete target[prop as any];
|
||||
},
|
||||
ownKeys(target) {
|
||||
const taskEnv = envStorage.getStore();
|
||||
if (taskEnv) {
|
||||
const keys = new Set<string | symbol>([
|
||||
...Object.getOwnPropertyNames(target),
|
||||
...Object.getOwnPropertySymbols(target),
|
||||
...Object.keys(taskEnv),
|
||||
]);
|
||||
taskEnv[deletedKeysSymbol]?.forEach((key) => {
|
||||
keys.delete(key);
|
||||
});
|
||||
return Array.from(keys);
|
||||
}
|
||||
return [
|
||||
...Object.getOwnPropertyNames(target),
|
||||
...Object.getOwnPropertySymbols(target),
|
||||
];
|
||||
},
|
||||
getOwnPropertyDescriptor(target, prop) {
|
||||
const taskEnv = envStorage.getStore();
|
||||
if (taskEnv && typeof prop === 'string') {
|
||||
const deleted = taskEnv[deletedKeysSymbol];
|
||||
if (deleted?.has(prop)) {
|
||||
return undefined;
|
||||
}
|
||||
if (Object.prototype.hasOwnProperty.call(taskEnv, prop)) {
|
||||
return {
|
||||
value: taskEnv[prop],
|
||||
writable: true,
|
||||
enumerable: true,
|
||||
configurable: true,
|
||||
};
|
||||
}
|
||||
}
|
||||
return Object.getOwnPropertyDescriptor(target, prop);
|
||||
},
|
||||
defineProperty(target, prop, descriptor) {
|
||||
if (typeof prop === 'string') {
|
||||
if (
|
||||
prop === '__proto__' ||
|
||||
prop === 'constructor' ||
|
||||
prop === 'prototype'
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
const taskEnv = envStorage.getStore();
|
||||
if (taskEnv) {
|
||||
taskEnv[deletedKeysSymbol]?.delete(prop);
|
||||
taskEnv[prop] =
|
||||
descriptor.value !== undefined ? String(descriptor.value) : undefined;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any, @typescript-eslint/no-unsafe-type-assertion
|
||||
Object.defineProperty(target, prop as any, descriptor);
|
||||
return true;
|
||||
},
|
||||
});
|
||||
|
||||
Object.defineProperty(process, 'env', {
|
||||
value: envProxy,
|
||||
writable: false,
|
||||
configurable: true,
|
||||
});
|
||||
|
||||
// NOTE: Monkey-patching process.cwd and process.chdir via AsyncLocalStorage is a robust way
|
||||
// to simulate workspace isolation in a concurrent server. However, please be aware of a critical
|
||||
// limitation: Node.js native C++ APIs (such as fs.readFileSync, fs.writeFile, etc.) and child
|
||||
// process spawning APIs (like child_process.spawn) resolve relative paths using the OS-level
|
||||
// working directory of the process, NOT the JS-level process.cwd() function.
|
||||
// To prevent cross-task interference, all file paths in the core package must be resolved to
|
||||
// absolute paths using path.resolve/path.join relative to config.getTargetDir() or config.getCwd()
|
||||
// before being passed to native APIs.
|
||||
const originalCwd = process.cwd;
|
||||
process.cwd = function () {
|
||||
const taskEnv = envStorage.getStore();
|
||||
if (taskEnv && taskEnv[cwdSymbol]) {
|
||||
return taskEnv[cwdSymbol];
|
||||
}
|
||||
return originalCwd.call(process);
|
||||
};
|
||||
|
||||
const originalChdir = process.chdir;
|
||||
process.chdir = function (directory: string) {
|
||||
const taskEnv = envStorage.getStore();
|
||||
if (taskEnv) {
|
||||
const resolved = path.resolve(process.cwd(), directory);
|
||||
try {
|
||||
const stats = fs.statSync(resolved);
|
||||
if (!stats.isDirectory()) {
|
||||
const err = new Error(
|
||||
"ENOTDIR: not a directory, chdir '" + resolved + "'",
|
||||
);
|
||||
(err as NodeJS.ErrnoException).code = 'ENOTDIR';
|
||||
throw err;
|
||||
}
|
||||
} catch (err: unknown) {
|
||||
if (
|
||||
err &&
|
||||
typeof err === 'object' &&
|
||||
'code' in err &&
|
||||
err.code === 'ENOENT'
|
||||
) {
|
||||
const chdirErr = new Error(
|
||||
"ENOENT: no such file or directory, chdir '" + resolved + "'",
|
||||
);
|
||||
(chdirErr as NodeJS.ErrnoException).code = 'ENOENT';
|
||||
throw chdirErr;
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
taskEnv[cwdSymbol] = resolved;
|
||||
return;
|
||||
}
|
||||
return originalChdir.call(process, directory);
|
||||
};
|
||||
|
||||
export function getEnv(key: string): string | undefined {
|
||||
return process.env[key];
|
||||
}
|
||||
|
||||
export async function loadConfig(
|
||||
settings: Settings,
|
||||
extensionLoader: ExtensionLoader,
|
||||
taskId: string,
|
||||
trusted: boolean = false,
|
||||
workspaceDir: string = process.cwd(),
|
||||
): Promise<Config> {
|
||||
const workspaceDir = process.cwd();
|
||||
const workspaceEnv = await loadEnvironment(trusted, workspaceDir);
|
||||
// eslint-disable-next-line @typescript-eslint/no-unsafe-type-assertion
|
||||
const envVars: Record<string, string> = { ...process.env } as Record<
|
||||
string,
|
||||
string
|
||||
>;
|
||||
Object.assign(envVars, workspaceEnv);
|
||||
|
||||
const getEnvLocal = (key: string) => envVars[key];
|
||||
|
||||
const folderTrust =
|
||||
settings.folderTrust === true ||
|
||||
process.env['GEMINI_FOLDER_TRUST'] === 'true';
|
||||
getEnvLocal('GEMINI_FOLDER_TRUST') === 'true';
|
||||
|
||||
let checkpointing = process.env['CHECKPOINTING']
|
||||
? process.env['CHECKPOINTING'] === 'true'
|
||||
let checkpointing = getEnvLocal('CHECKPOINTING')
|
||||
? getEnvLocal('CHECKPOINTING') === 'true'
|
||||
: settings.checkpointing?.enabled;
|
||||
|
||||
if (checkpointing) {
|
||||
@@ -62,7 +282,7 @@ export async function loadConfig(
|
||||
}
|
||||
|
||||
const approvalMode =
|
||||
process.env['GEMINI_YOLO_MODE'] === 'true'
|
||||
getEnvLocal('GEMINI_YOLO_MODE') === 'true'
|
||||
? ApprovalMode.YOLO
|
||||
: ApprovalMode.DEFAULT;
|
||||
|
||||
@@ -91,8 +311,9 @@ export async function loadConfig(
|
||||
embeddingModel: DEFAULT_GEMINI_EMBEDDING_MODEL,
|
||||
sandbox: undefined, // Sandbox might not be relevant for a server-side agent
|
||||
targetDir: workspaceDir, // Or a specific directory the agent operates on
|
||||
debugMode: process.env['DEBUG'] === 'true' || false,
|
||||
debugMode: getEnvLocal('DEBUG') === 'true' || false,
|
||||
question: '', // Not used in server mode directly like CLI
|
||||
env: envVars,
|
||||
|
||||
coreTools: settings.tools?.core || undefined,
|
||||
excludeTools: settings.tools?.exclude || undefined,
|
||||
@@ -107,7 +328,7 @@ export async function loadConfig(
|
||||
// eslint-disable-next-line @typescript-eslint/no-unsafe-type-assertion
|
||||
target: settings.telemetry?.target as TelemetryTarget,
|
||||
otlpEndpoint:
|
||||
process.env['OTEL_EXPORTER_OTLP_ENDPOINT'] ??
|
||||
getEnvLocal('OTEL_EXPORTER_OTLP_ENDPOINT') ??
|
||||
settings.telemetry?.otlpEndpoint,
|
||||
logPrompts: settings.telemetry?.logPrompts,
|
||||
},
|
||||
@@ -119,8 +340,8 @@ export async function loadConfig(
|
||||
settings.fileFiltering?.enableRecursiveFileSearch,
|
||||
customIgnoreFilePaths: [
|
||||
...(settings.fileFiltering?.customIgnoreFilePaths || []),
|
||||
...(process.env['CUSTOM_IGNORE_FILE_PATHS']
|
||||
? process.env['CUSTOM_IGNORE_FILE_PATHS'].split(path.delimiter)
|
||||
...(getEnvLocal('CUSTOM_IGNORE_FILE_PATHS')
|
||||
? getEnvLocal('CUSTOM_IGNORE_FILE_PATHS').split(path.delimiter)
|
||||
: []),
|
||||
],
|
||||
},
|
||||
@@ -179,7 +400,7 @@ export async function loadConfig(
|
||||
await config.waitForMcpInit();
|
||||
startupProfiler.flush(config);
|
||||
|
||||
await refreshAuthentication(config, 'Config');
|
||||
await refreshAuthentication(config, 'Config', envVars);
|
||||
|
||||
return config;
|
||||
}
|
||||
@@ -187,16 +408,19 @@ export async function loadConfig(
|
||||
export function setIsTrusted(
|
||||
agentSettings: AgentSettings | undefined,
|
||||
): boolean {
|
||||
if (INITIAL_FOLDER_TRUST !== undefined) {
|
||||
return INITIAL_FOLDER_TRUST === 'true';
|
||||
const folderTrustEnv = getEnv('GEMINI_FOLDER_TRUST');
|
||||
if (folderTrustEnv !== undefined) {
|
||||
return folderTrustEnv === 'true';
|
||||
}
|
||||
return !!agentSettings?.isTrusted;
|
||||
}
|
||||
|
||||
export function setTargetDir(agentSettings: AgentSettings | undefined): string {
|
||||
export async function setTargetDir(
|
||||
agentSettings: AgentSettings | undefined,
|
||||
): Promise<string> {
|
||||
const originalCWD = process.cwd();
|
||||
const targetDir =
|
||||
process.env['CODER_AGENT_WORKSPACE_PATH'] ??
|
||||
getEnv('CODER_AGENT_WORKSPACE_PATH') ??
|
||||
(agentSettings?.kind === CoderAgentEvent.StateAgentSettingsEvent
|
||||
? agentSettings.workspacePath
|
||||
: undefined);
|
||||
@@ -210,58 +434,170 @@ export function setTargetDir(agentSettings: AgentSettings | undefined): string {
|
||||
);
|
||||
|
||||
try {
|
||||
const resolvedPath = path.resolve(targetDir);
|
||||
process.chdir(resolvedPath);
|
||||
let resolvedPath: string;
|
||||
try {
|
||||
resolvedPath = resolveToRealPath(targetDir);
|
||||
} catch (err: unknown) {
|
||||
if (
|
||||
err &&
|
||||
typeof err === 'object' &&
|
||||
'code' in err &&
|
||||
err.code === 'ENOENT'
|
||||
) {
|
||||
const parentDir = path.dirname(path.resolve(targetDir));
|
||||
resolvedPath = path.join(
|
||||
resolveToRealPath(parentDir),
|
||||
path.basename(targetDir),
|
||||
);
|
||||
} else {
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
const isTestEnv =
|
||||
process.env['VITEST'] === 'true' ||
|
||||
process.env['NODE_ENV'] === 'test' ||
|
||||
process.argv.some((arg) => arg.includes('vitest')) ||
|
||||
resolvedPath.startsWith(resolveToRealPath(tmpdir()));
|
||||
|
||||
const allowedRoot = resolveToRealPath(
|
||||
getEnv('CODER_AGENT_ALLOWED_ROOT') ||
|
||||
(isTestEnv ? path.parse(resolvedPath).root : homedir()),
|
||||
);
|
||||
const relative = path.relative(allowedRoot, resolvedPath);
|
||||
if (relative.startsWith('..') || path.isAbsolute(relative)) {
|
||||
throw new Error(
|
||||
`Workspace path ${resolvedPath} is outside the allowed root directory`,
|
||||
);
|
||||
}
|
||||
|
||||
let stats: fs.Stats;
|
||||
try {
|
||||
stats = await fs.promises.stat(resolvedPath);
|
||||
} catch (err: unknown) {
|
||||
if (
|
||||
err &&
|
||||
typeof err === 'object' &&
|
||||
'code' in err &&
|
||||
err.code === 'ENOENT'
|
||||
) {
|
||||
if (isTestEnv) {
|
||||
await fs.promises.mkdir(resolvedPath, { recursive: true });
|
||||
stats = await fs.promises.stat(resolvedPath);
|
||||
} else {
|
||||
throw new Error(`Workspace path ${resolvedPath} does not exist`);
|
||||
}
|
||||
} else {
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
if (!stats.isDirectory()) {
|
||||
throw new Error(`Workspace path ${resolvedPath} is not a directory`);
|
||||
}
|
||||
|
||||
return resolvedPath;
|
||||
} catch (e) {
|
||||
logger.error(
|
||||
`[CoderAgentExecutor] Error resolving workspace path: ${e}, returning original os.cwd()`,
|
||||
);
|
||||
return originalCWD;
|
||||
logger.error(`[CoderAgentExecutor] Error resolving workspace path: ${e}`);
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
export function loadEnvironment(): void {
|
||||
const envFilePath = findEnvFile(process.cwd());
|
||||
export async function loadEnvironment(
|
||||
isTrusted: boolean = false,
|
||||
workspacePath: string = process.cwd(),
|
||||
): Promise<Record<string, string>> {
|
||||
// For untrusted workspaces, we completely bypass workspace-level .env loading
|
||||
// and only load environment variables from the user's trusted home directory.
|
||||
let envFilePath: string | null = null;
|
||||
if (isTrusted) {
|
||||
envFilePath = await findEnvFile(workspacePath);
|
||||
} else {
|
||||
const homeGeminiEnvPath = path.join(homedir(), GEMINI_DIR, '.env');
|
||||
try {
|
||||
await fs.promises.access(homeGeminiEnvPath);
|
||||
envFilePath = homeGeminiEnvPath;
|
||||
} catch {
|
||||
const homeEnvPath = path.join(homedir(), '.env');
|
||||
try {
|
||||
await fs.promises.access(homeEnvPath);
|
||||
envFilePath = homeEnvPath;
|
||||
} catch {
|
||||
// Ignore
|
||||
}
|
||||
}
|
||||
}
|
||||
const envVars: Record<string, string> = {};
|
||||
if (envFilePath) {
|
||||
dotenv.config({ path: envFilePath, override: true });
|
||||
try {
|
||||
const content = await fs.promises.readFile(envFilePath, 'utf-8');
|
||||
const parsed = dotenv.parse(content);
|
||||
for (const key in parsed) {
|
||||
if (
|
||||
Object.prototype.hasOwnProperty.call(parsed, key) &&
|
||||
key !== '__proto__' &&
|
||||
key !== 'constructor' &&
|
||||
key !== 'prototype'
|
||||
) {
|
||||
envVars[key] = parsed[key];
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// Ignore errors
|
||||
}
|
||||
}
|
||||
return envVars;
|
||||
}
|
||||
|
||||
function findEnvFile(startDir: string): string | null {
|
||||
async function findEnvFile(startDir: string): Promise<string | null> {
|
||||
let currentDir = path.resolve(startDir);
|
||||
while (true) {
|
||||
// prefer gemini-specific .env under GEMINI_DIR
|
||||
const geminiEnvPath = path.join(currentDir, GEMINI_DIR, '.env');
|
||||
if (fs.existsSync(geminiEnvPath)) {
|
||||
try {
|
||||
await fs.promises.access(geminiEnvPath);
|
||||
return geminiEnvPath;
|
||||
} catch {
|
||||
// Ignore
|
||||
}
|
||||
const envPath = path.join(currentDir, '.env');
|
||||
if (fs.existsSync(envPath)) {
|
||||
try {
|
||||
await fs.promises.access(envPath);
|
||||
return envPath;
|
||||
} catch {
|
||||
// Ignore
|
||||
}
|
||||
const parentDir = path.dirname(currentDir);
|
||||
if (parentDir === currentDir || !parentDir) {
|
||||
// check .env under home as fallback, again preferring gemini-specific .env
|
||||
const homeGeminiEnvPath = path.join(process.cwd(), GEMINI_DIR, '.env');
|
||||
if (fs.existsSync(homeGeminiEnvPath)) {
|
||||
return homeGeminiEnvPath;
|
||||
}
|
||||
const homeEnvPath = path.join(homedir(), '.env');
|
||||
if (fs.existsSync(homeEnvPath)) {
|
||||
return homeEnvPath;
|
||||
}
|
||||
return null;
|
||||
break;
|
||||
}
|
||||
currentDir = parentDir;
|
||||
}
|
||||
// check .env under home as fallback, again preferring gemini-specific .env
|
||||
const homeGeminiEnvPath = path.join(homedir(), GEMINI_DIR, '.env');
|
||||
try {
|
||||
await fs.promises.access(homeGeminiEnvPath);
|
||||
return homeGeminiEnvPath;
|
||||
} catch {
|
||||
// Ignore
|
||||
}
|
||||
const homeEnvPath = path.join(homedir(), '.env');
|
||||
try {
|
||||
await fs.promises.access(homeEnvPath);
|
||||
return homeEnvPath;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
async function refreshAuthentication(
|
||||
config: Config,
|
||||
logPrefix: string,
|
||||
envVars: Record<string, string>,
|
||||
): Promise<void> {
|
||||
if (process.env['USE_CCPA']) {
|
||||
const getEnvLocal = (key: string) => envVars[key];
|
||||
|
||||
if (getEnvLocal('USE_CCPA')) {
|
||||
logger.info(`[${logPrefix}] Using CCPA Auth:`);
|
||||
|
||||
logger.info(`[${logPrefix}] Attempting COMPUTE_ADC first.`);
|
||||
@@ -276,7 +612,7 @@ async function refreshAuthentication(
|
||||
);
|
||||
|
||||
const useComputeAdc =
|
||||
process.env['GEMINI_CLI_USE_COMPUTE_ADC'] === 'true';
|
||||
getEnvLocal('GEMINI_CLI_USE_COMPUTE_ADC') === 'true';
|
||||
const isHeadless = isHeadlessMode();
|
||||
|
||||
if (isHeadless || useComputeAdc) {
|
||||
@@ -305,11 +641,14 @@ async function refreshAuthentication(
|
||||
}
|
||||
|
||||
logger.info(
|
||||
`[${logPrefix}] GOOGLE_CLOUD_PROJECT: ${process.env['GOOGLE_CLOUD_PROJECT']}`,
|
||||
`[${logPrefix}] GOOGLE_CLOUD_PROJECT: ${getEnvLocal('GOOGLE_CLOUD_PROJECT')}`,
|
||||
);
|
||||
} else if (process.env['GEMINI_API_KEY']) {
|
||||
} else if (getEnvLocal('GEMINI_API_KEY')) {
|
||||
logger.info(`[${logPrefix}] Using Gemini API Key`);
|
||||
await config.refreshAuth(AuthType.USE_GEMINI);
|
||||
await config.refreshAuth(
|
||||
AuthType.USE_GEMINI,
|
||||
getEnvLocal('GEMINI_API_KEY'),
|
||||
);
|
||||
} else {
|
||||
const errorMessage = `[${logPrefix}] Unable to set GeneratorConfig. Please provide a GEMINI_API_KEY or set USE_CCPA.`;
|
||||
logger.error(errorMessage);
|
||||
|
||||
@@ -36,9 +36,12 @@ interface ExtensionConfig {
|
||||
excludeTools?: string[];
|
||||
}
|
||||
|
||||
export function loadExtensions(workspaceDir: string): GeminiCLIExtension[] {
|
||||
export function loadExtensions(
|
||||
workspaceDir: string,
|
||||
isTrusted: boolean = false,
|
||||
): GeminiCLIExtension[] {
|
||||
const allExtensions = [
|
||||
...loadExtensionsFromDir(workspaceDir),
|
||||
...(isTrusted ? loadExtensionsFromDir(workspaceDir) : []),
|
||||
...loadExtensionsFromDir(homedir()),
|
||||
];
|
||||
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
/**
|
||||
* @license
|
||||
* Copyright 2026 Google LLC
|
||||
* SPDX-License-Identifier: Apache-2.0
|
||||
*/
|
||||
|
||||
import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
|
||||
let mockHomeDir = '';
|
||||
|
||||
vi.mock('@google/gemini-cli-core', async (importOriginal) => {
|
||||
const original =
|
||||
await importOriginal<typeof import('@google/gemini-cli-core')>();
|
||||
return {
|
||||
...original,
|
||||
homedir: () => mockHomeDir,
|
||||
};
|
||||
});
|
||||
|
||||
import { loadEnvironment } from './config.js';
|
||||
|
||||
describe('Vulnerability Mitigation: b-519269096', () => {
|
||||
let tempWorkspaceDir: string;
|
||||
|
||||
beforeEach(() => {
|
||||
// Create a temporary home directory securely using mkdtempSync to ensure hermeticity
|
||||
mockHomeDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gemini-mock-home-'));
|
||||
|
||||
// Create a temporary workspace directory representing an untrusted repo
|
||||
tempWorkspaceDir = fs.mkdtempSync(
|
||||
path.join(os.tmpdir(), 'gemini-exploit-workspace-'),
|
||||
);
|
||||
const geminiDir = path.join(tempWorkspaceDir, '.gemini');
|
||||
fs.mkdirSync(geminiDir, { recursive: true });
|
||||
|
||||
// Mock process.cwd to return the untrusted workspace
|
||||
vi.spyOn(process, 'cwd').mockReturnValue(tempWorkspaceDir);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.unstubAllEnvs();
|
||||
vi.restoreAllMocks();
|
||||
fs.rmSync(tempWorkspaceDir, { recursive: true, force: true });
|
||||
fs.rmSync(mockHomeDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it('should ignore GEMINI_CLI_TRUST_WORKSPACE and GEMINI_YOLO_MODE in untrusted workspaces', async () => {
|
||||
const geminiDir = path.join(tempWorkspaceDir, '.gemini');
|
||||
fs.writeFileSync(
|
||||
path.join(geminiDir, '.env'),
|
||||
'GEMINI_CLI_TRUST_WORKSPACE=true\nGEMINI_YOLO_MODE=true\n',
|
||||
);
|
||||
|
||||
// Ensure initially not set
|
||||
vi.stubEnv('GEMINI_CLI_TRUST_WORKSPACE', '');
|
||||
vi.stubEnv('GEMINI_YOLO_MODE', '');
|
||||
|
||||
// Act: load environment with isTrusted = false
|
||||
const envVars = await loadEnvironment(false);
|
||||
|
||||
// Assert: In a SECURE system, these variables should NOT be loaded
|
||||
expect(envVars['GEMINI_CLI_TRUST_WORKSPACE']).toBeUndefined();
|
||||
expect(envVars['GEMINI_YOLO_MODE']).toBeUndefined();
|
||||
expect(process.env['GEMINI_CLI_TRUST_WORKSPACE']).toBeFalsy();
|
||||
expect(process.env['GEMINI_YOLO_MODE']).toBeFalsy();
|
||||
});
|
||||
|
||||
it('should not load any variables from untrusted workspaces', async () => {
|
||||
fs.writeFileSync(
|
||||
path.join(tempWorkspaceDir, '.env'),
|
||||
'GEMINI_API_KEY=safe-key-123;rm -rf /\nGOOGLE_CLOUD_PROJECT=my-project\n',
|
||||
);
|
||||
|
||||
// Ensure initially not set
|
||||
vi.stubEnv('GEMINI_API_KEY', '');
|
||||
vi.stubEnv('GOOGLE_CLOUD_PROJECT', '');
|
||||
|
||||
// Act: load environment with isTrusted = false
|
||||
const envVars = await loadEnvironment(false);
|
||||
|
||||
// Assert: No variables should be loaded from the untrusted workspace
|
||||
expect(envVars['GEMINI_API_KEY']).toBeUndefined();
|
||||
expect(envVars['GOOGLE_CLOUD_PROJECT']).toBeUndefined();
|
||||
expect(process.env['GEMINI_API_KEY']).toBeFalsy();
|
||||
expect(process.env['GOOGLE_CLOUD_PROJECT']).toBeFalsy();
|
||||
});
|
||||
|
||||
it('should load all variables in trusted workspaces with isolation', async () => {
|
||||
const geminiDir = path.join(tempWorkspaceDir, '.gemini');
|
||||
fs.writeFileSync(
|
||||
path.join(geminiDir, '.env'),
|
||||
'GEMINI_CLI_TRUST_WORKSPACE=true\nGEMINI_YOLO_MODE=true\n',
|
||||
);
|
||||
|
||||
// Ensure initially not set
|
||||
vi.stubEnv('GEMINI_CLI_TRUST_WORKSPACE', '');
|
||||
vi.stubEnv('GEMINI_YOLO_MODE', '');
|
||||
|
||||
// Load environment variables
|
||||
const envVars = await loadEnvironment(true);
|
||||
|
||||
// Assert: In a trusted workspace, variables should be loaded
|
||||
expect(envVars['GEMINI_CLI_TRUST_WORKSPACE']).toBe('true');
|
||||
expect(envVars['GEMINI_YOLO_MODE']).toBe('true');
|
||||
|
||||
// Assert: Global process.env should NOT be polluted
|
||||
expect(process.env['GEMINI_CLI_TRUST_WORKSPACE']).toBeFalsy();
|
||||
expect(process.env['GEMINI_YOLO_MODE']).toBeFalsy();
|
||||
});
|
||||
});
|
||||
@@ -197,8 +197,7 @@ async function handleExecuteCommand(
|
||||
export async function createApp() {
|
||||
try {
|
||||
// Load the server configuration once on startup.
|
||||
const workspaceRoot = setTargetDir(undefined);
|
||||
loadEnvironment();
|
||||
const workspaceRoot = await setTargetDir(undefined);
|
||||
|
||||
// Use a temporary settings load to check if folder trust is enabled.
|
||||
// This is similar to how the CLI handles the initial trust check.
|
||||
@@ -209,13 +208,35 @@ export async function createApp() {
|
||||
isHeadless: isHeadlessMode(),
|
||||
});
|
||||
|
||||
// Change the global working directory to the workspace root during startup
|
||||
process.chdir(workspaceRoot);
|
||||
|
||||
// Load environment globally for the server startup
|
||||
const globalEnv = await loadEnvironment(isTrusted ?? false, workspaceRoot);
|
||||
// Only assign safe server-config variables to process.env to prevent credential leakage
|
||||
const allowedServerKeys = [
|
||||
'CODER_AGENT_PORT',
|
||||
'CODER_AGENT_WORKSPACE_PATH',
|
||||
'GCS_BUCKET_NAME',
|
||||
'LOG_LEVEL',
|
||||
'GOOGLE_APPLICATION_CREDENTIALS',
|
||||
'GOOGLE_CLOUD_PROJECT',
|
||||
'GEMINI_CLI_USE_COMPUTE_ADC',
|
||||
];
|
||||
for (const key of allowedServerKeys) {
|
||||
if (globalEnv[key] !== undefined) {
|
||||
process.env[key] = globalEnv[key];
|
||||
}
|
||||
}
|
||||
|
||||
const settings = loadSettings(workspaceRoot, isTrusted ?? false);
|
||||
const extensions = loadExtensions(workspaceRoot);
|
||||
const extensions = loadExtensions(workspaceRoot, isTrusted ?? false);
|
||||
const config = await loadConfig(
|
||||
settings,
|
||||
new SimpleExtensionLoader(extensions),
|
||||
'a2a-server',
|
||||
isTrusted ?? false,
|
||||
workspaceRoot,
|
||||
);
|
||||
|
||||
let git: GitService | undefined;
|
||||
|
||||
@@ -258,7 +258,7 @@ export class GCSTaskStore implements TaskStore {
|
||||
}
|
||||
const agentSettings = persistedState._agentSettings;
|
||||
|
||||
const workDir = setTargetDir(agentSettings);
|
||||
const workDir = await setTargetDir(agentSettings);
|
||||
await fse.ensureDir(workDir);
|
||||
const workspaceFile = this.storage
|
||||
.bucket(this.bucketName)
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
/**
|
||||
* @license
|
||||
* Copyright 2026 Google LLC
|
||||
* SPDX-License-Identifier: Apache-2.0
|
||||
*/
|
||||
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import { resolveToRealPath, isSubpath } from '@google/gemini-cli-core';
|
||||
|
||||
/**
|
||||
* Validates a workspace path to prevent path traversal attacks.
|
||||
*
|
||||
* @param workspacePath The path to validate.
|
||||
* @param allowedRoot The root directory the path must be within. Defaults to CWD.
|
||||
* @returns The resolved, safe path.
|
||||
* @throws An error if the path is invalid or outside the allowed root.
|
||||
*/
|
||||
export async function validateWorkspacePath(
|
||||
workspacePath?: string,
|
||||
allowedRoot: string = process.cwd(),
|
||||
): Promise<string> {
|
||||
const trimmedPath = workspacePath?.trim();
|
||||
if (!trimmedPath) {
|
||||
return resolveToRealPath(allowedRoot);
|
||||
}
|
||||
|
||||
if (trimmedPath.includes('\0')) {
|
||||
throw new Error('Security violation: Null byte detected in path.');
|
||||
}
|
||||
|
||||
try {
|
||||
const canonicalAllowedRoot = resolveToRealPath(allowedRoot);
|
||||
const resolvedWorkspacePath = path.resolve(
|
||||
canonicalAllowedRoot,
|
||||
trimmedPath,
|
||||
);
|
||||
const canonicalWorkspacePath = resolveToRealPath(resolvedWorkspacePath);
|
||||
|
||||
// Check if the resolved path is within the allowed root directory
|
||||
if (
|
||||
canonicalWorkspacePath !== canonicalAllowedRoot &&
|
||||
!isSubpath(canonicalAllowedRoot, canonicalWorkspacePath)
|
||||
) {
|
||||
throw new Error(
|
||||
`Security violation: The path "${trimmedPath}" is outside the allowed root directory.`,
|
||||
);
|
||||
}
|
||||
|
||||
const stats = await fs.promises.stat(canonicalWorkspacePath);
|
||||
if (!stats.isDirectory()) {
|
||||
throw new Error(`The path "${trimmedPath}" is not a directory.`);
|
||||
}
|
||||
|
||||
return canonicalWorkspacePath;
|
||||
} catch (e) {
|
||||
if (e instanceof Error && 'code' in e && e.code === 'ENOENT') {
|
||||
throw new Error(`The path "${trimmedPath}" does not exist.`);
|
||||
}
|
||||
throw e; // Re-throw other errors
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@google/gemini-cli",
|
||||
"version": "0.51.0-nightly.20260625.g3fbf93e26",
|
||||
"version": "0.56.0-nightly.20260806.g761f604c1",
|
||||
"description": "Gemini CLI",
|
||||
"license": "Apache-2.0",
|
||||
"repository": {
|
||||
@@ -27,7 +27,7 @@
|
||||
"dist"
|
||||
],
|
||||
"config": {
|
||||
"sandboxImageUri": "us-docker.pkg.dev/gemini-code-dev/gemini-cli/sandbox:0.51.0-nightly.20260625.g3fbf93e26"
|
||||
"sandboxImageUri": "us-docker.pkg.dev/gemini-code-dev/gemini-cli/sandbox:0.56.0-nightly.20260806.g761f604c1"
|
||||
},
|
||||
"dependencies": {
|
||||
"@agentclientprotocol/sdk": "0.16.1",
|
||||
|
||||
@@ -319,6 +319,41 @@ describe('Session', () => {
|
||||
expect(result).toMatchObject({ stopReason: 'end_turn' });
|
||||
});
|
||||
|
||||
it.each([
|
||||
{ type: 'MAX_TOKENS_EXCEEDED', reason: 'MAX_TOKENS' },
|
||||
{ type: 'SAFETY_BLOCKED', reason: 'SAFETY' },
|
||||
{ type: 'RECITATION_BLOCKED', reason: 'RECITATION' },
|
||||
{ type: 'OTHER_BLOCKED', reason: 'OTHER' },
|
||||
{ type: 'THINKING_ONLY_RESPONSE', reason: 'STOP' },
|
||||
])(
|
||||
'should gracefully handle InvalidStreamError with type $type in ACP session',
|
||||
async ({ type, reason }) => {
|
||||
const error = new InvalidStreamError(
|
||||
`Stream failed with ${reason}`,
|
||||
type as InvalidStreamError['type'],
|
||||
);
|
||||
mockSendMessageStream.mockImplementation(() => {
|
||||
async function* errorGen(): AsyncGenerator<
|
||||
ServerGeminiStreamEvent,
|
||||
void,
|
||||
unknown
|
||||
> {
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
yield* [] as any;
|
||||
throw error;
|
||||
}
|
||||
return errorGen();
|
||||
});
|
||||
|
||||
const result = await session.prompt({
|
||||
sessionId: 'session-1',
|
||||
prompt: [{ type: 'text', text: 'Hi' }],
|
||||
});
|
||||
|
||||
expect(result).toMatchObject({ stopReason: 'end_turn' });
|
||||
},
|
||||
);
|
||||
|
||||
it('should handle /memory command', async () => {
|
||||
const handleCommandSpy = vi
|
||||
.spyOn(
|
||||
|
||||
@@ -510,7 +510,12 @@ export class Session {
|
||||
(error.type === 'NO_RESPONSE_TEXT' ||
|
||||
error.type === 'NO_FINISH_REASON' ||
|
||||
error.type === 'MALFORMED_FUNCTION_CALL' ||
|
||||
error.type === 'UNEXPECTED_TOOL_CALL'))
|
||||
error.type === 'UNEXPECTED_TOOL_CALL' ||
|
||||
error.type === 'MAX_TOKENS_EXCEEDED' ||
|
||||
error.type === 'SAFETY_BLOCKED' ||
|
||||
error.type === 'RECITATION_BLOCKED' ||
|
||||
error.type === 'OTHER_BLOCKED' ||
|
||||
error.type === 'THINKING_ONLY_RESPONSE'))
|
||||
) {
|
||||
// The stream ended with an empty response or malformed tool call.
|
||||
// Treat this as a graceful end to the model's turn rather than a crash.
|
||||
|
||||
@@ -103,7 +103,10 @@ vi.mock('../utils.js', () => ({
|
||||
|
||||
describe('extensions install command', () => {
|
||||
it('should fail if no source is provided', () => {
|
||||
const validationParser = yargs([]).command(installCommand).fail(false);
|
||||
const validationParser = yargs([])
|
||||
.locale('en')
|
||||
.command(installCommand)
|
||||
.fail(false);
|
||||
expect(() => validationParser.parse('install')).toThrow(
|
||||
'Not enough non-option arguments: got 0, need at least 1',
|
||||
);
|
||||
|
||||
@@ -27,7 +27,10 @@ vi.mock('../utils.js', () => ({
|
||||
|
||||
describe('extensions validate command', () => {
|
||||
it('should fail if no path is provided', () => {
|
||||
const validationParser = yargs([]).command(validateCommand).fail(false);
|
||||
const validationParser = yargs([])
|
||||
.locale('en')
|
||||
.command(validateCommand)
|
||||
.fail(false);
|
||||
expect(() => validationParser.parse('validate')).toThrow(
|
||||
'Not enough non-option arguments: got 0, need at least 1',
|
||||
);
|
||||
|
||||
@@ -17,7 +17,7 @@ describe('mcp command', () => {
|
||||
});
|
||||
|
||||
it('should show help when no subcommand is provided', async () => {
|
||||
const yargsInstance = yargs();
|
||||
const yargsInstance = yargs().locale('en');
|
||||
(mcpCommand.builder as (y: Argv) => Argv)(yargsInstance);
|
||||
|
||||
const parser = yargsInstance.command(mcpCommand).help();
|
||||
|
||||
@@ -22,6 +22,7 @@ import {
|
||||
CoreEvent,
|
||||
CoreToolCallStatus,
|
||||
JsonStreamEventType,
|
||||
TRUE_EMPTY_RESPONSE_MESSAGE,
|
||||
} from '@google/gemini-cli-core';
|
||||
import type { Part } from '@google/genai';
|
||||
import { runNonInteractive } from './nonInteractiveCli.js';
|
||||
@@ -78,6 +79,7 @@ vi.mock('@google/gemini-cli-core', async (importOriginal) => {
|
||||
ChatRecordingService: MockChatRecordingService,
|
||||
uiTelemetryService: {
|
||||
getMetrics: vi.fn(),
|
||||
recordSemanticValidationError: vi.fn(),
|
||||
},
|
||||
coreEvents: mockCoreEvents,
|
||||
createWorkingStdio: vi.fn(() => ({
|
||||
@@ -110,6 +112,7 @@ describe('runNonInteractive', () => {
|
||||
sendMessageStream: Mock;
|
||||
resumeChat: Mock;
|
||||
getChatRecordingService: Mock;
|
||||
getCurrentSequenceModel: Mock;
|
||||
};
|
||||
const MOCK_SESSION_METRICS: SessionMetrics = {
|
||||
models: {},
|
||||
@@ -165,6 +168,7 @@ describe('runNonInteractive', () => {
|
||||
recordMessageTokens: vi.fn(),
|
||||
recordToolCalls: vi.fn(),
|
||||
})),
|
||||
getCurrentSequenceModel: vi.fn().mockReturnValue('gemini-2.5-flash'),
|
||||
};
|
||||
|
||||
mockConfig = {
|
||||
@@ -193,6 +197,7 @@ describe('runNonInteractive', () => {
|
||||
getRawOutput: vi.fn().mockReturnValue(false),
|
||||
getAcceptRawOutputRisk: vi.fn().mockReturnValue(false),
|
||||
getAgentSessionNoninteractiveEnabled: vi.fn().mockReturnValue(false),
|
||||
getUsageStatisticsEnabled: vi.fn().mockReturnValue(false),
|
||||
} as unknown as Config;
|
||||
|
||||
mockSettings = {
|
||||
@@ -1820,7 +1825,6 @@ describe('runNonInteractive', () => {
|
||||
};
|
||||
// @ts-expect-error - Mocking internal structure
|
||||
mockGeminiClient.getChat = vi.fn().mockReturnValue(mockChat);
|
||||
// @ts-expect-error - Mocking internal structure
|
||||
mockGeminiClient.getCurrentSequenceModel = vi
|
||||
.fn()
|
||||
.mockReturnValue('model-1');
|
||||
@@ -2298,7 +2302,13 @@ describe('runNonInteractive', () => {
|
||||
|
||||
it('should handle InvalidStream event gracefully in TEXT mode', async () => {
|
||||
const events: ServerGeminiStreamEvent[] = [
|
||||
{ type: GeminiEventType.InvalidStream },
|
||||
{
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_RESPONSE_TEXT',
|
||||
message: 'Empty response',
|
||||
},
|
||||
},
|
||||
];
|
||||
mockGeminiClient.sendMessageStream.mockReturnValue(
|
||||
createStreamFromEvents(events),
|
||||
@@ -2312,7 +2322,7 @@ describe('runNonInteractive', () => {
|
||||
});
|
||||
|
||||
expect(processStderrSpy).toHaveBeenCalledWith(
|
||||
'[ERROR] Invalid stream: The model returned an empty response or malformed tool call.\n',
|
||||
`[ERROR] ${TRUE_EMPTY_RESPONSE_MESSAGE}\n`,
|
||||
);
|
||||
expect(mockGeminiClient.sendMessageStream).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
@@ -2325,7 +2335,13 @@ describe('runNonInteractive', () => {
|
||||
OutputFormat.STREAM_JSON,
|
||||
);
|
||||
const events: ServerGeminiStreamEvent[] = [
|
||||
{ type: GeminiEventType.InvalidStream },
|
||||
{
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_RESPONSE_TEXT',
|
||||
message: 'Empty response',
|
||||
},
|
||||
},
|
||||
];
|
||||
mockGeminiClient.sendMessageStream.mockReturnValue(
|
||||
createStreamFromEvents(events),
|
||||
@@ -2341,9 +2357,7 @@ describe('runNonInteractive', () => {
|
||||
const output = getWrittenOutput();
|
||||
expect(output).toContain('"type":"error"');
|
||||
expect(output).toContain('"severity":"error"');
|
||||
expect(output).toContain(
|
||||
'Invalid stream: The model returned an empty response or malformed tool call.',
|
||||
);
|
||||
expect(output).toContain(TRUE_EMPTY_RESPONSE_MESSAGE);
|
||||
expect(mockGeminiClient.sendMessageStream).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
@@ -2355,7 +2369,13 @@ describe('runNonInteractive', () => {
|
||||
OutputFormat.JSON,
|
||||
);
|
||||
const events: ServerGeminiStreamEvent[] = [
|
||||
{ type: GeminiEventType.InvalidStream },
|
||||
{
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_RESPONSE_TEXT',
|
||||
message: 'Empty response',
|
||||
},
|
||||
},
|
||||
];
|
||||
mockGeminiClient.sendMessageStream.mockReturnValue(
|
||||
createStreamFromEvents(events),
|
||||
@@ -2371,8 +2391,33 @@ describe('runNonInteractive', () => {
|
||||
const output = getWrittenOutput();
|
||||
expect(output).toContain('"error": {');
|
||||
expect(output).toContain('"type": "INVALID_STREAM"');
|
||||
expect(output).toContain(
|
||||
'Invalid stream: The model returned an empty response or malformed tool call.',
|
||||
expect(output).toContain(TRUE_EMPTY_RESPONSE_MESSAGE);
|
||||
expect(mockGeminiClient.sendMessageStream).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('should handle non-NO_RESPONSE_TEXT InvalidStream event gracefully and use message from eventValue', async () => {
|
||||
const events: ServerGeminiStreamEvent[] = [
|
||||
{
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'MALFORMED_FUNCTION_CALL',
|
||||
message: 'Custom malformed function call message',
|
||||
},
|
||||
},
|
||||
];
|
||||
mockGeminiClient.sendMessageStream.mockReturnValue(
|
||||
createStreamFromEvents(events),
|
||||
);
|
||||
|
||||
await runNonInteractive({
|
||||
config: mockConfig,
|
||||
settings: mockSettings,
|
||||
input: 'test invalid stream malformed',
|
||||
prompt_id: 'prompt-id-invalid-malformed',
|
||||
});
|
||||
|
||||
expect(processStderrSpy).toHaveBeenCalledWith(
|
||||
'[ERROR] Custom malformed function call message\n',
|
||||
);
|
||||
expect(mockGeminiClient.sendMessageStream).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
@@ -30,6 +30,12 @@ import {
|
||||
ToolErrorType,
|
||||
Scheduler,
|
||||
ROOT_SCHEDULER_ID,
|
||||
THINKING_ONLY_COMPRESS_SUGGESTION,
|
||||
MAX_TOKENS_EXCEEDED_SUGGESTION,
|
||||
SAFETY_BLOCKED_MESSAGE,
|
||||
RECITATION_BLOCKED_MESSAGE,
|
||||
OTHER_BLOCKED_MESSAGE,
|
||||
TRUE_EMPTY_RESPONSE_MESSAGE,
|
||||
} from '@google/gemini-cli-core';
|
||||
|
||||
import type { Content, Part } from '@google/genai';
|
||||
@@ -433,8 +439,31 @@ export async function runNonInteractive(
|
||||
}
|
||||
warnings.push(blockMessage);
|
||||
} else if (event.type === GeminiEventType.InvalidStream) {
|
||||
invalidStreamError =
|
||||
'Invalid stream: The model returned an empty response or malformed tool call.';
|
||||
const eventValue = event.value;
|
||||
if (eventValue?.type === 'NO_RESPONSE_TEXT') {
|
||||
invalidStreamError = TRUE_EMPTY_RESPONSE_MESSAGE;
|
||||
} else if (eventValue?.type === 'THINKING_ONLY_RESPONSE') {
|
||||
invalidStreamError = THINKING_ONLY_COMPRESS_SUGGESTION;
|
||||
} else if (eventValue?.type === 'MAX_TOKENS_EXCEEDED') {
|
||||
invalidStreamError = MAX_TOKENS_EXCEEDED_SUGGESTION;
|
||||
} else if (eventValue?.type === 'SAFETY_BLOCKED') {
|
||||
invalidStreamError = SAFETY_BLOCKED_MESSAGE;
|
||||
} else if (eventValue?.type === 'RECITATION_BLOCKED') {
|
||||
invalidStreamError = RECITATION_BLOCKED_MESSAGE;
|
||||
} else if (eventValue?.type === 'OTHER_BLOCKED') {
|
||||
invalidStreamError = OTHER_BLOCKED_MESSAGE;
|
||||
} else {
|
||||
invalidStreamError =
|
||||
eventValue?.message?.trim() ||
|
||||
'Invalid stream: The model returned an empty response or malformed tool call.';
|
||||
}
|
||||
|
||||
// Log semantic error telemetry without double-counting requests
|
||||
uiTelemetryService.recordSemanticValidationError(
|
||||
geminiClient.getCurrentSequenceModel() ?? config.getModel(),
|
||||
eventValue?.type || 'INVALID_STREAM',
|
||||
);
|
||||
|
||||
if (streamFormatter) {
|
||||
streamFormatter.emitEvent({
|
||||
type: JsonStreamEventType.ERROR,
|
||||
|
||||
@@ -22,6 +22,7 @@ import {
|
||||
CoreEvent,
|
||||
CoreToolCallStatus,
|
||||
JsonStreamEventType,
|
||||
TRUE_EMPTY_RESPONSE_MESSAGE,
|
||||
} from '@google/gemini-cli-core';
|
||||
import type { Part } from '@google/genai';
|
||||
import { runNonInteractive } from './nonInteractiveCliAgentSession.js';
|
||||
@@ -78,6 +79,7 @@ vi.mock('@google/gemini-cli-core', async (importOriginal) => {
|
||||
ChatRecordingService: MockChatRecordingService,
|
||||
uiTelemetryService: {
|
||||
getMetrics: vi.fn(),
|
||||
recordSemanticValidationError: vi.fn(),
|
||||
},
|
||||
LegacyAgentSession: original.LegacyAgentSession,
|
||||
geminiPartsToContentParts: original.geminiPartsToContentParts,
|
||||
@@ -199,6 +201,7 @@ describe('runNonInteractive', () => {
|
||||
getRawOutput: vi.fn().mockReturnValue(false),
|
||||
getAcceptRawOutputRisk: vi.fn().mockReturnValue(false),
|
||||
getAgentSessionNoninteractiveEnabled: vi.fn().mockReturnValue(false),
|
||||
getUsageStatisticsEnabled: vi.fn().mockReturnValue(false),
|
||||
} as unknown as Config;
|
||||
|
||||
mockSettings = {
|
||||
@@ -2457,6 +2460,126 @@ describe('runNonInteractive', () => {
|
||||
const output = JSON.parse(getWrittenOutput());
|
||||
expect(output.warnings).toBeUndefined();
|
||||
});
|
||||
|
||||
it('should handle InvalidStream event gracefully in TEXT mode', async () => {
|
||||
const events: ServerGeminiStreamEvent[] = [
|
||||
{
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_RESPONSE_TEXT',
|
||||
message: 'Empty response',
|
||||
},
|
||||
},
|
||||
];
|
||||
mockGeminiClient.sendMessageStream.mockReturnValue(
|
||||
createStreamFromEvents(events),
|
||||
);
|
||||
|
||||
await runNonInteractive({
|
||||
config: mockConfig,
|
||||
settings: mockSettings,
|
||||
input: 'test invalid stream',
|
||||
prompt_id: 'prompt-id-invalid',
|
||||
});
|
||||
|
||||
expect(processStderrSpy).toHaveBeenCalledWith(
|
||||
`[ERROR] ${TRUE_EMPTY_RESPONSE_MESSAGE}\n`,
|
||||
);
|
||||
expect(mockGeminiClient.sendMessageStream).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('should handle InvalidStream event gracefully in STREAM_JSON mode', async () => {
|
||||
vi.spyOn(uiTelemetryService, 'getMetrics').mockReturnValue(
|
||||
MOCK_SESSION_METRICS,
|
||||
);
|
||||
vi.spyOn(mockConfig, 'getOutputFormat').mockReturnValue(
|
||||
OutputFormat.STREAM_JSON,
|
||||
);
|
||||
const events: ServerGeminiStreamEvent[] = [
|
||||
{
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_RESPONSE_TEXT',
|
||||
message: 'Empty response',
|
||||
},
|
||||
},
|
||||
];
|
||||
mockGeminiClient.sendMessageStream.mockReturnValue(
|
||||
createStreamFromEvents(events),
|
||||
);
|
||||
|
||||
await runNonInteractive({
|
||||
config: mockConfig,
|
||||
settings: mockSettings,
|
||||
input: 'test invalid stream',
|
||||
prompt_id: 'prompt-id-invalid',
|
||||
});
|
||||
|
||||
const output = getWrittenOutput();
|
||||
expect(output).toContain('"type":"error"');
|
||||
expect(output).toContain('"severity":"error"');
|
||||
expect(output).toContain(TRUE_EMPTY_RESPONSE_MESSAGE);
|
||||
expect(mockGeminiClient.sendMessageStream).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('should handle InvalidStream event gracefully in JSON mode', async () => {
|
||||
vi.spyOn(uiTelemetryService, 'getMetrics').mockReturnValue(
|
||||
MOCK_SESSION_METRICS,
|
||||
);
|
||||
vi.spyOn(mockConfig, 'getOutputFormat').mockReturnValue(
|
||||
OutputFormat.JSON,
|
||||
);
|
||||
const events: ServerGeminiStreamEvent[] = [
|
||||
{
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_RESPONSE_TEXT',
|
||||
message: 'Empty response',
|
||||
},
|
||||
},
|
||||
];
|
||||
mockGeminiClient.sendMessageStream.mockReturnValue(
|
||||
createStreamFromEvents(events),
|
||||
);
|
||||
|
||||
await runNonInteractive({
|
||||
config: mockConfig,
|
||||
settings: mockSettings,
|
||||
input: 'test invalid stream',
|
||||
prompt_id: 'prompt-id-invalid',
|
||||
});
|
||||
|
||||
const output = getWrittenOutput();
|
||||
expect(output).toContain('"error": {');
|
||||
expect(output).toContain('"type": "INVALID_STREAM"');
|
||||
expect(output).toContain(TRUE_EMPTY_RESPONSE_MESSAGE);
|
||||
expect(mockGeminiClient.sendMessageStream).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('should handle non-NO_RESPONSE_TEXT InvalidStream event gracefully and use message from eventValue', async () => {
|
||||
const events: ServerGeminiStreamEvent[] = [
|
||||
{
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'MALFORMED_FUNCTION_CALL',
|
||||
message: 'Malformed call',
|
||||
},
|
||||
},
|
||||
];
|
||||
mockGeminiClient.sendMessageStream.mockReturnValue(
|
||||
createStreamFromEvents(events),
|
||||
);
|
||||
|
||||
await runNonInteractive({
|
||||
config: mockConfig,
|
||||
settings: mockSettings,
|
||||
input: 'test invalid stream',
|
||||
prompt_id: 'prompt-id-invalid',
|
||||
});
|
||||
|
||||
expect(processStderrSpy).toHaveBeenCalledWith('[ERROR] Malformed call\n');
|
||||
expect(mockGeminiClient.sendMessageStream).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Output Sanitization', () => {
|
||||
|
||||
@@ -39,6 +39,12 @@ import {
|
||||
geminiPartsToContentParts,
|
||||
displayContentToString,
|
||||
debugLogger,
|
||||
THINKING_ONLY_COMPRESS_SUGGESTION,
|
||||
MAX_TOKENS_EXCEEDED_SUGGESTION,
|
||||
SAFETY_BLOCKED_MESSAGE,
|
||||
RECITATION_BLOCKED_MESSAGE,
|
||||
OTHER_BLOCKED_MESSAGE,
|
||||
TRUE_EMPTY_RESPONSE_MESSAGE,
|
||||
} from '@google/gemini-cli-core';
|
||||
|
||||
import type { Part } from '@google/genai';
|
||||
@@ -332,14 +338,17 @@ export async function runNonInteractive({
|
||||
return text ? text : undefined;
|
||||
};
|
||||
|
||||
const emitFinalSuccessResult = (): void => {
|
||||
const emitFinalResult = (errorPayload?: {
|
||||
type: string;
|
||||
message: string;
|
||||
}): void => {
|
||||
if (streamFormatter) {
|
||||
const metrics = uiTelemetryService.getMetrics();
|
||||
const durationMs = Date.now() - startTime;
|
||||
streamFormatter.emitEvent({
|
||||
type: JsonStreamEventType.RESULT,
|
||||
timestamp: new Date().toISOString(),
|
||||
status: 'success',
|
||||
status: errorPayload ? 'error' : 'success',
|
||||
stats: streamFormatter.convertToStreamStats(metrics, durationMs),
|
||||
});
|
||||
} else if (config.getOutputFormat() === OutputFormat.JSON) {
|
||||
@@ -350,7 +359,7 @@ export async function runNonInteractive({
|
||||
config.getSessionId(),
|
||||
responseText,
|
||||
stats,
|
||||
undefined,
|
||||
errorPayload,
|
||||
warnings,
|
||||
),
|
||||
);
|
||||
@@ -545,6 +554,52 @@ export async function runNonInteractive({
|
||||
break;
|
||||
}
|
||||
case 'error': {
|
||||
if (event._meta?.['code'] === 'INVALID_STREAM') {
|
||||
const errorTypeVal = event._meta?.['errorType'];
|
||||
const errorType =
|
||||
typeof errorTypeVal === 'string' ? errorTypeVal : undefined;
|
||||
|
||||
let errorMessage = event.message;
|
||||
if (errorType === 'NO_RESPONSE_TEXT') {
|
||||
errorMessage = TRUE_EMPTY_RESPONSE_MESSAGE;
|
||||
} else if (errorType === 'THINKING_ONLY_RESPONSE') {
|
||||
errorMessage = THINKING_ONLY_COMPRESS_SUGGESTION;
|
||||
} else if (errorType === 'MAX_TOKENS_EXCEEDED') {
|
||||
errorMessage = MAX_TOKENS_EXCEEDED_SUGGESTION;
|
||||
} else if (errorType === 'SAFETY_BLOCKED') {
|
||||
errorMessage = SAFETY_BLOCKED_MESSAGE;
|
||||
} else if (errorType === 'RECITATION_BLOCKED') {
|
||||
errorMessage = RECITATION_BLOCKED_MESSAGE;
|
||||
} else if (errorType === 'OTHER_BLOCKED') {
|
||||
errorMessage = OTHER_BLOCKED_MESSAGE;
|
||||
}
|
||||
|
||||
if (streamFormatter) {
|
||||
streamFormatter.emitEvent({
|
||||
type: JsonStreamEventType.ERROR,
|
||||
timestamp: new Date().toISOString(),
|
||||
severity: 'error',
|
||||
message: errorMessage,
|
||||
});
|
||||
} else if (config.getOutputFormat() === OutputFormat.TEXT) {
|
||||
process.stderr.write(`[ERROR] ${errorMessage}\n`);
|
||||
}
|
||||
|
||||
// Log semantic error telemetry without double-counting requests
|
||||
uiTelemetryService.recordSemanticValidationError(
|
||||
geminiClient.getCurrentSequenceModel() ?? config.getModel(),
|
||||
errorType || 'INVALID_STREAM',
|
||||
);
|
||||
|
||||
// If it's a fatal stream error, we should terminate and output final results
|
||||
emitFinalResult({
|
||||
type: 'INVALID_STREAM',
|
||||
message: errorMessage,
|
||||
});
|
||||
streamEnded = true;
|
||||
break;
|
||||
}
|
||||
|
||||
if (event.fatal) {
|
||||
throw reconstructFatalError(event);
|
||||
}
|
||||
@@ -613,7 +668,7 @@ export async function runNonInteractive({
|
||||
process.stderr.write(`Agent execution stopped: ${stopMessage}\n`);
|
||||
}
|
||||
|
||||
emitFinalSuccessResult();
|
||||
emitFinalResult();
|
||||
streamEnded = true;
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -86,6 +86,7 @@ export const DialogManager = ({
|
||||
message={quotaState.proQuotaRequest.message}
|
||||
isTerminalQuotaError={quotaState.proQuotaRequest.isTerminalQuotaError}
|
||||
isModelNotFoundError={!!quotaState.proQuotaRequest.isModelNotFoundError}
|
||||
isCapacityExceeded={!!quotaState.proQuotaRequest.isCapacityExceeded}
|
||||
authType={quotaState.proQuotaRequest.authType}
|
||||
tierName={config?.getUserTierName()}
|
||||
onChoice={uiActions.handleProQuotaChoice}
|
||||
|
||||
@@ -271,6 +271,40 @@ describe('ProQuotaDialog', () => {
|
||||
);
|
||||
unmount();
|
||||
});
|
||||
|
||||
it('should render keep trying, switch, and stop options even if isTerminalQuotaError is true when isCapacityExceeded is true', async () => {
|
||||
const { unmount } = await render(
|
||||
<ProQuotaDialog
|
||||
failedModel="gemini-2.5-pro"
|
||||
fallbackModel="gemini-2.5-flash"
|
||||
message="capacity error"
|
||||
isTerminalQuotaError={true}
|
||||
isCapacityExceeded={true}
|
||||
isModelNotFoundError={false}
|
||||
onChoice={mockOnChoice}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(RadioButtonSelect).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
items: [
|
||||
{
|
||||
label: 'Keep trying',
|
||||
value: 'retry_once',
|
||||
key: 'retry_once',
|
||||
},
|
||||
{
|
||||
label: 'Switch to gemini-2.5-flash',
|
||||
value: 'retry_always',
|
||||
key: 'retry_always',
|
||||
},
|
||||
{ label: 'Stop', value: 'retry_later', key: 'retry_later' },
|
||||
],
|
||||
}),
|
||||
undefined,
|
||||
);
|
||||
unmount();
|
||||
});
|
||||
});
|
||||
|
||||
describe('when it is a model not found error', () => {
|
||||
|
||||
@@ -17,6 +17,7 @@ interface ProQuotaDialogProps {
|
||||
message: string;
|
||||
isTerminalQuotaError: boolean;
|
||||
isModelNotFoundError?: boolean;
|
||||
isCapacityExceeded?: boolean;
|
||||
authType?: AuthType;
|
||||
tierName?: string;
|
||||
onChoice: (
|
||||
@@ -30,6 +31,7 @@ export function ProQuotaDialog({
|
||||
message,
|
||||
isTerminalQuotaError,
|
||||
isModelNotFoundError,
|
||||
isCapacityExceeded,
|
||||
authType,
|
||||
tierName,
|
||||
onChoice,
|
||||
@@ -49,6 +51,24 @@ export function ProQuotaDialog({
|
||||
key: 'retry_later',
|
||||
},
|
||||
];
|
||||
} else if (isCapacityExceeded) {
|
||||
items = [
|
||||
{
|
||||
label: 'Keep trying',
|
||||
value: 'retry_once' as const,
|
||||
key: 'retry_once',
|
||||
},
|
||||
{
|
||||
label: `Switch to ${fallbackModel}`,
|
||||
value: 'retry_always' as const,
|
||||
key: 'retry_always',
|
||||
},
|
||||
{
|
||||
label: 'Stop',
|
||||
value: 'retry_later' as const,
|
||||
key: 'retry_later',
|
||||
},
|
||||
];
|
||||
} else if (isModelNotFoundError || isTerminalQuotaError) {
|
||||
const isUltra = isUltraTier(tierName);
|
||||
|
||||
@@ -75,7 +95,7 @@ export function ProQuotaDialog({
|
||||
},
|
||||
];
|
||||
} else {
|
||||
// capacity error
|
||||
// capacity error or generic fallback
|
||||
items = [
|
||||
{
|
||||
label: 'Keep trying',
|
||||
|
||||
@@ -36,6 +36,18 @@ function areModelMetricsEqual(a: ModelMetrics, b: ModelMetrics): boolean {
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
const errorsA = a.api.errorsByType || {};
|
||||
const errorsB = b.api.errorsByType || {};
|
||||
const keysA = Object.keys(errorsA);
|
||||
const keysB = Object.keys(errorsB);
|
||||
if (keysA.length !== keysB.length) {
|
||||
return false;
|
||||
}
|
||||
for (const key of keysA) {
|
||||
if (errorsA[key] !== errorsB[key]) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (
|
||||
a.tokens.input !== b.tokens.input ||
|
||||
a.tokens.prompt !== b.tokens.prompt ||
|
||||
|
||||
@@ -40,6 +40,7 @@ export interface ProQuotaDialogRequest {
|
||||
message: string;
|
||||
isTerminalQuotaError: boolean;
|
||||
isModelNotFoundError?: boolean;
|
||||
isCapacityExceeded?: boolean;
|
||||
authType?: AuthType;
|
||||
resolve: (intent: FallbackIntent) => void;
|
||||
}
|
||||
|
||||
@@ -54,6 +54,7 @@ import {
|
||||
GeminiCliOperation,
|
||||
getPlanModeExitMessage,
|
||||
UPDATE_TOPIC_TOOL_NAME,
|
||||
TRUE_EMPTY_RESPONSE_MESSAGE,
|
||||
} from '@google/gemini-cli-core';
|
||||
import type { Part, PartListUnion } from '@google/genai';
|
||||
import type { UseHistoryManagerReturn } from './useHistoryManager.js';
|
||||
@@ -1045,6 +1046,107 @@ describe('useGeminiStream', () => {
|
||||
});
|
||||
});
|
||||
|
||||
it('should record tool responses in history when the model was switched due to a quota error', async () => {
|
||||
// Regression test: returning early on a quota-triggered model switch
|
||||
// without recording the responses leaves the already-recorded
|
||||
// functionCall unpaired, which corrupts all subsequent requests.
|
||||
const responseParts: Part[] = [
|
||||
{
|
||||
functionResponse: {
|
||||
name: 'testTool',
|
||||
id: 'call1',
|
||||
response: { output: 'tool result' },
|
||||
},
|
||||
},
|
||||
];
|
||||
const completedToolCalls: TrackedToolCall[] = [
|
||||
{
|
||||
request: {
|
||||
callId: 'call1',
|
||||
name: 'testTool',
|
||||
args: {},
|
||||
isClientInitiated: false,
|
||||
prompt_id: 'prompt-id-quota',
|
||||
},
|
||||
status: CoreToolCallStatus.Success,
|
||||
responseSubmittedToGemini: false,
|
||||
response: {
|
||||
callId: 'call1',
|
||||
responseParts,
|
||||
errorType: undefined,
|
||||
},
|
||||
tool: { displayName: 'MockTool' },
|
||||
invocation: {
|
||||
getDescription: () => `Mock description`,
|
||||
} as unknown as AnyToolInvocation,
|
||||
} as TrackedCompletedToolCall,
|
||||
];
|
||||
|
||||
const client = new MockedGeminiClientClass(mockConfig);
|
||||
const mockConsumeUserHint = vi.fn(() => 'switch to the nprd database');
|
||||
|
||||
let capturedOnComplete:
|
||||
| ((completedTools: TrackedToolCall[]) => Promise<void>)
|
||||
| null = null;
|
||||
|
||||
mockUseToolScheduler.mockImplementation((onComplete) => {
|
||||
capturedOnComplete = onComplete;
|
||||
return [
|
||||
[],
|
||||
mockScheduleToolCalls,
|
||||
mockMarkToolsAsSubmitted,
|
||||
vi.fn(),
|
||||
mockCancelAllToolCalls,
|
||||
0,
|
||||
];
|
||||
});
|
||||
|
||||
await renderHookWithProviders(() =>
|
||||
useGeminiStream(
|
||||
client,
|
||||
[],
|
||||
mockAddItem,
|
||||
mockConfig,
|
||||
mockLoadedSettings,
|
||||
mockOnDebugMessage,
|
||||
mockHandleSlashCommand,
|
||||
false,
|
||||
() => 'vscode' as EditorType,
|
||||
() => {},
|
||||
() => Promise.resolve(),
|
||||
true, // modelSwitchedFromQuotaError
|
||||
() => {},
|
||||
() => {},
|
||||
() => {},
|
||||
80,
|
||||
24,
|
||||
false,
|
||||
mockConsumeUserHint,
|
||||
),
|
||||
);
|
||||
|
||||
await act(async () => {
|
||||
if (capturedOnComplete) {
|
||||
await new Promise((resolve) => setTimeout(resolve, 0));
|
||||
await capturedOnComplete(completedToolCalls);
|
||||
}
|
||||
});
|
||||
|
||||
await waitFor(() => {
|
||||
expect(mockMarkToolsAsSubmitted).toHaveBeenCalledWith(['call1']);
|
||||
// The tool response must be paired with its functionCall in history,
|
||||
// with no steering-hint text ahead of it...
|
||||
expect(client.addHistory).toHaveBeenCalledWith({
|
||||
role: 'user',
|
||||
parts: responseParts,
|
||||
});
|
||||
// ...the turn must NOT auto-continue on the fallback model...
|
||||
expect(mockSendMessageStream).not.toHaveBeenCalled();
|
||||
// ...and the pending hint is left for the next real submit.
|
||||
expect(mockConsumeUserHint).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
|
||||
it('should NOT stop responding when only update_topic is called', async () => {
|
||||
const topicToolCalls: TrackedToolCall[] = [
|
||||
{
|
||||
@@ -1772,6 +1874,120 @@ describe('useGeminiStream', () => {
|
||||
expect(mockCancelAllToolCalls).toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('should transition to Idle state when cancelled while a tool call is in progress and completes', async () => {
|
||||
const toolCalls: TrackedToolCall[] = [
|
||||
{
|
||||
request: { callId: 'call1', name: 'tool1', args: {} },
|
||||
status: CoreToolCallStatus.Executing,
|
||||
responseSubmittedToGemini: false,
|
||||
tool: {
|
||||
name: 'tool1',
|
||||
description: 'desc1',
|
||||
build: vi.fn().mockImplementation((_) => ({
|
||||
getDescription: () => `Mock description`,
|
||||
})),
|
||||
} as any,
|
||||
invocation: {
|
||||
getDescription: () => `Mock description`,
|
||||
},
|
||||
startTime: Date.now(),
|
||||
liveOutput: '...',
|
||||
} as TrackedExecutingToolCall,
|
||||
];
|
||||
|
||||
const { result } = await renderTestHook(toolCalls);
|
||||
|
||||
// State is `Responding` because a tool is running
|
||||
expect(result.current.streamingState).toBe(StreamingState.Responding);
|
||||
|
||||
// Try to cancel
|
||||
simulateEscapeKeyPress();
|
||||
|
||||
// Trigger the onComplete callback with the cancelled tool call
|
||||
await act(async () => {
|
||||
if (capturedOnComplete) {
|
||||
await capturedOnComplete([
|
||||
{
|
||||
...toolCalls[0],
|
||||
status: CoreToolCallStatus.Cancelled,
|
||||
response: {
|
||||
callId: 'call1',
|
||||
responseParts: [],
|
||||
},
|
||||
} as any,
|
||||
]);
|
||||
}
|
||||
});
|
||||
|
||||
// The final state should be idle because the cancelled tool call was marked as submitted
|
||||
expect(result.current.streamingState).toBe(StreamingState.Idle);
|
||||
});
|
||||
|
||||
it('should append cancelled tool responses to history when cancelled while a tool call is in progress and completes with response parts', async () => {
|
||||
const toolCalls: TrackedToolCall[] = [
|
||||
{
|
||||
request: { callId: 'call1', name: 'tool1', args: {} },
|
||||
status: CoreToolCallStatus.Executing,
|
||||
responseSubmittedToGemini: false,
|
||||
tool: {
|
||||
name: 'tool1',
|
||||
description: 'desc1',
|
||||
build: vi.fn().mockImplementation((_) => ({
|
||||
getDescription: () => `Mock description`,
|
||||
})),
|
||||
} as any,
|
||||
invocation: {
|
||||
getDescription: () => `Mock description`,
|
||||
},
|
||||
startTime: Date.now(),
|
||||
liveOutput: '...',
|
||||
} as TrackedExecutingToolCall,
|
||||
];
|
||||
|
||||
const { result, client } = await renderTestHook(toolCalls);
|
||||
|
||||
// State is `Responding` because a tool is running
|
||||
expect(result.current.streamingState).toBe(StreamingState.Responding);
|
||||
|
||||
// Try to cancel
|
||||
simulateEscapeKeyPress();
|
||||
|
||||
const expectedResponseParts = [
|
||||
{
|
||||
functionResponse: {
|
||||
name: 'tool1',
|
||||
id: 'call1',
|
||||
response: { error: 'cancelled' },
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
// Trigger the onComplete callback with the cancelled tool call having non-empty response parts
|
||||
await act(async () => {
|
||||
if (capturedOnComplete) {
|
||||
await capturedOnComplete([
|
||||
{
|
||||
...toolCalls[0],
|
||||
status: CoreToolCallStatus.Cancelled,
|
||||
response: {
|
||||
callId: 'call1',
|
||||
responseParts: expectedResponseParts,
|
||||
},
|
||||
} as any,
|
||||
]);
|
||||
}
|
||||
});
|
||||
|
||||
// Assert that addHistory was called with the combined response parts
|
||||
expect(client.addHistory).toHaveBeenCalledWith({
|
||||
role: 'user',
|
||||
parts: expectedResponseParts,
|
||||
});
|
||||
|
||||
// The final state should be idle because the cancelled tool call was marked as submitted
|
||||
expect(result.current.streamingState).toBe(StreamingState.Idle);
|
||||
});
|
||||
|
||||
it('should cancel a request when a tool is awaiting confirmation', async () => {
|
||||
const mockOnConfirm = vi.fn().mockResolvedValue(undefined);
|
||||
const toolCalls: TrackedToolCall[] = [
|
||||
@@ -2306,6 +2522,68 @@ describe('useGeminiStream', () => {
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
it('should use TRUE_EMPTY_RESPONSE_MESSAGE when receiving an invalid stream event of type NO_RESPONSE_TEXT', async () => {
|
||||
mockSendMessageStream.mockClear();
|
||||
mockSendMessageStream.mockReturnValue(
|
||||
(async function* () {
|
||||
yield {
|
||||
type: ServerGeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_RESPONSE_TEXT',
|
||||
message: 'empty response text',
|
||||
},
|
||||
};
|
||||
})(),
|
||||
);
|
||||
|
||||
const { result } = await renderTestHook();
|
||||
|
||||
await act(async () => {
|
||||
await result.current.submitQuery('test query');
|
||||
});
|
||||
|
||||
await waitFor(() => {
|
||||
expect(mockAddItem).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
type: MessageType.ERROR,
|
||||
text: TRUE_EMPTY_RESPONSE_MESSAGE,
|
||||
}),
|
||||
expect.any(Number),
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
it('should use the event message when receiving a non-NO_RESPONSE_TEXT invalid stream event', async () => {
|
||||
mockSendMessageStream.mockClear();
|
||||
mockSendMessageStream.mockReturnValue(
|
||||
(async function* () {
|
||||
yield {
|
||||
type: ServerGeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'MALFORMED_FUNCTION_CALL',
|
||||
message: 'Custom malformed function call message',
|
||||
},
|
||||
};
|
||||
})(),
|
||||
);
|
||||
|
||||
const { result } = await renderTestHook();
|
||||
|
||||
await act(async () => {
|
||||
await result.current.submitQuery('test query');
|
||||
});
|
||||
|
||||
await waitFor(() => {
|
||||
expect(mockAddItem).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
type: MessageType.ERROR,
|
||||
text: 'Custom malformed function call message',
|
||||
}),
|
||||
expect.any(Number),
|
||||
);
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('handleApprovalModeChange', () => {
|
||||
|
||||
@@ -14,6 +14,7 @@ import {
|
||||
GitService,
|
||||
UnauthorizedError,
|
||||
UserPromptEvent,
|
||||
uiTelemetryService,
|
||||
DEFAULT_GEMINI_FLASH_MODEL,
|
||||
logConversationFinishedEvent,
|
||||
ConversationFinishedEvent,
|
||||
@@ -45,6 +46,12 @@ import {
|
||||
buildToolVisibilityContext,
|
||||
UPDATE_TOPIC_TOOL_NAME,
|
||||
UPDATE_TOPIC_DISPLAY_NAME,
|
||||
THINKING_ONLY_COMPRESS_SUGGESTION,
|
||||
MAX_TOKENS_EXCEEDED_SUGGESTION,
|
||||
SAFETY_BLOCKED_MESSAGE,
|
||||
RECITATION_BLOCKED_MESSAGE,
|
||||
OTHER_BLOCKED_MESSAGE,
|
||||
TRUE_EMPTY_RESPONSE_MESSAGE,
|
||||
} from '@google/gemini-cli-core';
|
||||
import type {
|
||||
Config,
|
||||
@@ -54,6 +61,7 @@ import type {
|
||||
ServerGeminiContentEvent as ContentEvent,
|
||||
ServerGeminiFinishedEvent,
|
||||
ServerGeminiStreamEvent as GeminiEvent,
|
||||
ServerGeminiInvalidStreamEvent,
|
||||
ThoughtSummary,
|
||||
ToolCallRequestInfo,
|
||||
ToolCallResponseInfo,
|
||||
@@ -1229,6 +1237,61 @@ export const useGeminiStream = (
|
||||
],
|
||||
);
|
||||
|
||||
const handleInvalidStreamEvent = useCallback(
|
||||
(
|
||||
eventValue: ServerGeminiInvalidStreamEvent['value'],
|
||||
userMessageTimestamp: number,
|
||||
) => {
|
||||
if (pendingHistoryItemRef.current) {
|
||||
addItem(pendingHistoryItemRef.current, userMessageTimestamp);
|
||||
setPendingHistoryItem(null);
|
||||
}
|
||||
maybeAddSuppressedToolErrorNote(userMessageTimestamp);
|
||||
|
||||
let text =
|
||||
eventValue?.message?.trim() || 'Invalid stream received from model';
|
||||
if (eventValue?.type === 'NO_RESPONSE_TEXT') {
|
||||
text = TRUE_EMPTY_RESPONSE_MESSAGE;
|
||||
} else if (eventValue?.type === 'THINKING_ONLY_RESPONSE') {
|
||||
text = THINKING_ONLY_COMPRESS_SUGGESTION;
|
||||
} else if (eventValue?.type === 'MAX_TOKENS_EXCEEDED') {
|
||||
text = MAX_TOKENS_EXCEEDED_SUGGESTION;
|
||||
} else if (eventValue?.type === 'SAFETY_BLOCKED') {
|
||||
text = SAFETY_BLOCKED_MESSAGE;
|
||||
} else if (eventValue?.type === 'RECITATION_BLOCKED') {
|
||||
text = RECITATION_BLOCKED_MESSAGE;
|
||||
} else if (eventValue?.type === 'OTHER_BLOCKED') {
|
||||
text = OTHER_BLOCKED_MESSAGE;
|
||||
}
|
||||
|
||||
// Log semantic error telemetry without double-counting requests
|
||||
uiTelemetryService.recordSemanticValidationError(
|
||||
geminiClient.getCurrentSequenceModel() ?? config.getModel(),
|
||||
eventValue?.type || 'INVALID_STREAM',
|
||||
);
|
||||
|
||||
addItem(
|
||||
{
|
||||
type: MessageType.ERROR,
|
||||
text,
|
||||
},
|
||||
userMessageTimestamp,
|
||||
);
|
||||
maybeAddLowVerbosityFailureNote(userMessageTimestamp);
|
||||
setThought(null); // Reset thought when there's an error
|
||||
},
|
||||
[
|
||||
addItem,
|
||||
pendingHistoryItemRef,
|
||||
setPendingHistoryItem,
|
||||
setThought,
|
||||
maybeAddSuppressedToolErrorNote,
|
||||
maybeAddLowVerbosityFailureNote,
|
||||
config,
|
||||
geminiClient,
|
||||
],
|
||||
);
|
||||
|
||||
const handleCitationEvent = useCallback(
|
||||
(text: string, userMessageTimestamp: number) => {
|
||||
if (!showCitations(settings)) {
|
||||
@@ -1541,8 +1604,10 @@ export const useGeminiStream = (
|
||||
loopDetectedRef.current = true;
|
||||
break;
|
||||
case ServerGeminiEventType.Retry:
|
||||
// Handled transparently by the backend stream retries.
|
||||
break;
|
||||
case ServerGeminiEventType.InvalidStream:
|
||||
// Will add the missing logic later
|
||||
handleInvalidStreamEvent(event.value, userMessageTimestamp);
|
||||
break;
|
||||
default: {
|
||||
// enforces exhaustive switch-case
|
||||
@@ -1575,6 +1640,7 @@ export const useGeminiStream = (
|
||||
handleChatModelEvent,
|
||||
handleAgentExecutionStoppedEvent,
|
||||
handleAgentExecutionBlockedEvent,
|
||||
handleInvalidStreamEvent,
|
||||
addItem,
|
||||
pendingHistoryItemRef,
|
||||
setPendingHistoryItem,
|
||||
@@ -1886,6 +1952,30 @@ export const useGeminiStream = (
|
||||
},
|
||||
);
|
||||
|
||||
if (turnCancelledRef.current) {
|
||||
setIsResponding(false);
|
||||
const geminiTools = completedAndReadyToSubmitTools.filter(
|
||||
(t) => !t.request.isClientInitiated,
|
||||
);
|
||||
if (geminiClient && geminiTools.length > 0) {
|
||||
const combinedParts = geminiTools.flatMap(
|
||||
(toolCall) => toolCall.response.responseParts,
|
||||
);
|
||||
if (combinedParts.length > 0) {
|
||||
// eslint-disable-next-line @typescript-eslint/no-floating-promises
|
||||
geminiClient.addHistory({
|
||||
role: 'user',
|
||||
parts: combinedParts,
|
||||
});
|
||||
}
|
||||
}
|
||||
const callIdsToMarkAsSubmitted = toolCalls.map(
|
||||
(toolCall) => toolCall.request.callId,
|
||||
);
|
||||
markToolsAsSubmitted(callIdsToMarkAsSubmitted);
|
||||
return;
|
||||
}
|
||||
|
||||
// Finalize any client-initiated tools as soon as they are done.
|
||||
const clientTools = completedAndReadyToSubmitTools.filter(
|
||||
(t) => t.request.isClientInitiated,
|
||||
@@ -2020,6 +2110,27 @@ export const useGeminiStream = (
|
||||
(toolCall) => toolCall.response.responseParts,
|
||||
);
|
||||
|
||||
const callIdsToMarkAsSubmitted = geminiTools.map(
|
||||
(toolCall) => toolCall.request.callId,
|
||||
);
|
||||
|
||||
markToolsAsSubmitted(callIdsToMarkAsSubmitted);
|
||||
|
||||
// Don't continue if model was switched due to quota error, but still
|
||||
// record the responses: the matching functionCall is already in history,
|
||||
// and leaving it unpaired corrupts every subsequent request. Any pending
|
||||
// steering hint is deliberately left unconsumed so it rides along with
|
||||
// the next query the user actually submits.
|
||||
if (modelSwitchedFromQuotaError) {
|
||||
if (geminiClient && responsesToSend.length > 0) {
|
||||
await geminiClient.addHistory({
|
||||
role: 'user',
|
||||
parts: responsesToSend,
|
||||
});
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if (consumeUserHint) {
|
||||
const userHint = consumeUserHint();
|
||||
if (userHint && userHint.trim().length > 0) {
|
||||
@@ -2030,21 +2141,10 @@ export const useGeminiStream = (
|
||||
}
|
||||
}
|
||||
|
||||
const callIdsToMarkAsSubmitted = geminiTools.map(
|
||||
(toolCall) => toolCall.request.callId,
|
||||
);
|
||||
|
||||
const prompt_ids = geminiTools.map(
|
||||
(toolCall) => toolCall.request.prompt_id,
|
||||
);
|
||||
|
||||
markToolsAsSubmitted(callIdsToMarkAsSubmitted);
|
||||
|
||||
// Don't continue if model was switched due to quota error
|
||||
if (modelSwitchedFromQuotaError) {
|
||||
return;
|
||||
}
|
||||
|
||||
// eslint-disable-next-line @typescript-eslint/no-floating-promises
|
||||
submitQuery(
|
||||
responsesToSend,
|
||||
@@ -2066,6 +2166,7 @@ export const useGeminiStream = (
|
||||
maybeAddSuppressedToolErrorNote,
|
||||
maybeAddLowVerbosityFailureNote,
|
||||
setIsResponding,
|
||||
toolCalls,
|
||||
],
|
||||
);
|
||||
|
||||
|
||||
@@ -49,7 +49,7 @@ describe('usePrivacySettings', () => {
|
||||
};
|
||||
};
|
||||
|
||||
it('should throw error when content generator is not a CodeAssistServer', async () => {
|
||||
it('should report tier unavailable when OAuth is not being used', async () => {
|
||||
vi.mocked(getCodeAssistServer).mockReturnValue(undefined);
|
||||
|
||||
const { result } = await act(async () => renderPrivacySettingsHook());
|
||||
@@ -58,7 +58,8 @@ describe('usePrivacySettings', () => {
|
||||
expect(result.current.privacyState.isLoading).toBe(false);
|
||||
});
|
||||
|
||||
expect(result.current.privacyState.error).toBe('Oauth not being used');
|
||||
expect(result.current.privacyState.isTierUnavailable).toBe(true);
|
||||
expect(result.current.privacyState.error).toBeUndefined();
|
||||
});
|
||||
|
||||
it('should handle paid tier users correctly', async () => {
|
||||
@@ -79,7 +80,7 @@ describe('usePrivacySettings', () => {
|
||||
expect(result.current.privacyState.dataCollectionOptIn).toBeUndefined();
|
||||
});
|
||||
|
||||
it('should throw error when CodeAssistServer has no projectId', async () => {
|
||||
it('should report tier unavailable when CodeAssistServer has no projectId', async () => {
|
||||
vi.mocked(getCodeAssistServer).mockReturnValue({
|
||||
userTier: UserTierId.FREE,
|
||||
} as unknown as CodeAssistServer);
|
||||
@@ -90,9 +91,63 @@ describe('usePrivacySettings', () => {
|
||||
expect(result.current.privacyState.isLoading).toBe(false);
|
||||
});
|
||||
|
||||
expect(result.current.privacyState.error).toBe(
|
||||
'CodeAssist server is missing a project ID',
|
||||
);
|
||||
expect(result.current.privacyState.isTierUnavailable).toBe(true);
|
||||
expect(result.current.privacyState.error).toBeUndefined();
|
||||
});
|
||||
|
||||
it('should report tier unavailable when the user has no tier', async () => {
|
||||
vi.mocked(getCodeAssistServer).mockReturnValue({
|
||||
projectId: 'test-project-id',
|
||||
userTier: undefined,
|
||||
} as unknown as CodeAssistServer);
|
||||
|
||||
const { result } = await act(async () => renderPrivacySettingsHook());
|
||||
|
||||
await waitFor(() => {
|
||||
expect(result.current.privacyState.isLoading).toBe(false);
|
||||
});
|
||||
|
||||
expect(result.current.privacyState.isTierUnavailable).toBe(true);
|
||||
expect(result.current.privacyState.isFreeTier).toBeUndefined();
|
||||
expect(result.current.privacyState.error).toBeUndefined();
|
||||
});
|
||||
|
||||
it('should report tier unavailable when the backend reports no current tier', async () => {
|
||||
vi.mocked(getCodeAssistServer).mockReturnValue({
|
||||
projectId: 'test-project-id',
|
||||
userTier: UserTierId.FREE,
|
||||
getCodeAssistGlobalUserSetting: vi
|
||||
.fn()
|
||||
.mockRejectedValue(new Error('User does not have a current tier')),
|
||||
} as unknown as CodeAssistServer);
|
||||
|
||||
const { result } = await act(async () => renderPrivacySettingsHook());
|
||||
|
||||
await waitFor(() => {
|
||||
expect(result.current.privacyState.isLoading).toBe(false);
|
||||
});
|
||||
|
||||
expect(result.current.privacyState.isTierUnavailable).toBe(true);
|
||||
expect(result.current.privacyState.error).toBeUndefined();
|
||||
});
|
||||
|
||||
it('should surface unexpected errors while loading opt-in settings', async () => {
|
||||
vi.mocked(getCodeAssistServer).mockReturnValue({
|
||||
projectId: 'test-project-id',
|
||||
userTier: UserTierId.FREE,
|
||||
getCodeAssistGlobalUserSetting: vi
|
||||
.fn()
|
||||
.mockRejectedValue(new Error('network unavailable')),
|
||||
} as unknown as CodeAssistServer);
|
||||
|
||||
const { result } = await act(async () => renderPrivacySettingsHook());
|
||||
|
||||
await waitFor(() => {
|
||||
expect(result.current.privacyState.isLoading).toBe(false);
|
||||
});
|
||||
|
||||
expect(result.current.privacyState.error).toBe('network unavailable');
|
||||
expect(result.current.privacyState.isTierUnavailable).toBeUndefined();
|
||||
});
|
||||
|
||||
it('should update data collection opt-in setting', async () => {
|
||||
|
||||
@@ -18,8 +18,22 @@ export interface PrivacyState {
|
||||
error?: string;
|
||||
isFreeTier?: boolean;
|
||||
dataCollectionOptIn?: boolean;
|
||||
/**
|
||||
* True when the signed-in account has no consumer Code Assist tier, so the
|
||||
* data-collection opt-in isn't applicable (e.g. Workspace/enterprise accounts,
|
||||
* or an OAuth login without a Google Cloud project). This is an expected state
|
||||
* rendered as a friendly, actionable notice rather than a raw backend `error`.
|
||||
*/
|
||||
isTierUnavailable?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Signals that the current account can't be mapped to a consumer Code Assist
|
||||
* tier, so the privacy opt-in can't be shown. Handled by rendering a friendly
|
||||
* notice instead of surfacing a raw backend error.
|
||||
*/
|
||||
class TierUnavailableError extends Error {}
|
||||
|
||||
export const usePrivacySettings = (config: Config) => {
|
||||
const [privacyState, setPrivacyState] = useState<PrivacyState>({
|
||||
isLoading: true,
|
||||
@@ -34,7 +48,13 @@ export const usePrivacySettings = (config: Config) => {
|
||||
const server = getCodeAssistServerOrFail(config);
|
||||
const tier = server.userTier;
|
||||
if (tier === undefined) {
|
||||
throw new Error('Could not determine user tier.');
|
||||
// The account has no resolved Code Assist tier (e.g. Workspace or an
|
||||
// incomplete OAuth). Show a friendly notice instead of a raw error.
|
||||
setPrivacyState({
|
||||
isLoading: false,
|
||||
isTierUnavailable: true,
|
||||
});
|
||||
return;
|
||||
}
|
||||
if (tier !== UserTierId.FREE) {
|
||||
// We don't need to fetch opt-out info since non-free tier
|
||||
@@ -53,6 +73,13 @@ export const usePrivacySettings = (config: Config) => {
|
||||
dataCollectionOptIn: optIn,
|
||||
});
|
||||
} catch (e) {
|
||||
if (isTierUnavailableError(e)) {
|
||||
setPrivacyState({
|
||||
isLoading: false,
|
||||
isTierUnavailable: true,
|
||||
});
|
||||
return;
|
||||
}
|
||||
setPrivacyState({
|
||||
isLoading: false,
|
||||
error: e instanceof Error ? e.message : String(e),
|
||||
@@ -74,6 +101,13 @@ export const usePrivacySettings = (config: Config) => {
|
||||
dataCollectionOptIn: updatedOptIn,
|
||||
});
|
||||
} catch (e) {
|
||||
if (isTierUnavailableError(e)) {
|
||||
setPrivacyState({
|
||||
isLoading: false,
|
||||
isTierUnavailable: true,
|
||||
});
|
||||
return;
|
||||
}
|
||||
setPrivacyState({
|
||||
isLoading: false,
|
||||
error: e instanceof Error ? e.message : String(e),
|
||||
@@ -92,13 +126,30 @@ export const usePrivacySettings = (config: Config) => {
|
||||
function getCodeAssistServerOrFail(config: Config): CodeAssistServer {
|
||||
const server = getCodeAssistServer(config);
|
||||
if (server === undefined) {
|
||||
throw new Error('Oauth not being used');
|
||||
throw new TierUnavailableError('Oauth not being used');
|
||||
} else if (server.projectId === undefined) {
|
||||
throw new Error('CodeAssist server is missing a project ID');
|
||||
throw new TierUnavailableError('CodeAssist server is missing a project ID');
|
||||
}
|
||||
return server;
|
||||
}
|
||||
|
||||
/**
|
||||
* Determines whether an error means the account simply has no consumer Code
|
||||
* Assist tier, as opposed to an unexpected failure. Covers the local
|
||||
* {@link TierUnavailableError} as well as the Code Assist backend error (e.g.
|
||||
* "User does not have a current tier") returned for Workspace/enterprise
|
||||
* accounts.
|
||||
*/
|
||||
function isTierUnavailableError(error: unknown): boolean {
|
||||
if (error instanceof TierUnavailableError) {
|
||||
return true;
|
||||
}
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
// Match the specific Code Assist backend message rather than a broad substring
|
||||
// so an unrelated error that merely mentions "tier" isn't masked as a benign notice.
|
||||
return /does not have a current tier/i.test(message);
|
||||
}
|
||||
|
||||
async function getRemoteDataCollectionOptIn(
|
||||
server: CodeAssistServer,
|
||||
): Promise<boolean> {
|
||||
|
||||
@@ -222,6 +222,126 @@ describe('useQuotaAndFallback', () => {
|
||||
await promise!;
|
||||
});
|
||||
|
||||
it('should auto-retry terminal quota capacity failures in low verbosity mode', async () => {
|
||||
const { result } = await renderHook(() =>
|
||||
useQuotaAndFallback({
|
||||
config: mockConfig,
|
||||
historyManager: mockHistoryManager,
|
||||
userTier: UserTierId.FREE,
|
||||
setModelSwitchedFromQuotaError: mockSetModelSwitchedFromQuotaError,
|
||||
onShowAuthSelection: mockOnShowAuthSelection,
|
||||
paidTier: null,
|
||||
settings: mockSettings,
|
||||
errorVerbosity: 'low',
|
||||
}),
|
||||
);
|
||||
|
||||
const handler = setFallbackHandlerSpy.mock
|
||||
.calls[0][0] as FallbackModelHandler;
|
||||
const intent = await handler(
|
||||
'gemini-pro',
|
||||
'gemini-flash',
|
||||
new TerminalQuotaError(
|
||||
'pro capacity exhausted',
|
||||
mockGoogleApiError,
|
||||
undefined,
|
||||
'MODEL_CAPACITY_EXHAUSTED',
|
||||
),
|
||||
);
|
||||
|
||||
expect(intent).toBe('retry_once');
|
||||
expect(result.current.proQuotaRequest).toBeNull();
|
||||
});
|
||||
|
||||
it('should auto-retry capacity failures matched by regex on message in low verbosity mode', async () => {
|
||||
const { result } = await renderHook(() =>
|
||||
useQuotaAndFallback({
|
||||
config: mockConfig,
|
||||
historyManager: mockHistoryManager,
|
||||
userTier: UserTierId.FREE,
|
||||
setModelSwitchedFromQuotaError: mockSetModelSwitchedFromQuotaError,
|
||||
onShowAuthSelection: mockOnShowAuthSelection,
|
||||
paidTier: null,
|
||||
settings: mockSettings,
|
||||
errorVerbosity: 'low',
|
||||
}),
|
||||
);
|
||||
|
||||
const handler = setFallbackHandlerSpy.mock
|
||||
.calls[0][0] as FallbackModelHandler;
|
||||
const intent = await handler(
|
||||
'gemini-pro',
|
||||
'gemini-flash',
|
||||
new Error('you have exhausted your capacity limit'),
|
||||
);
|
||||
|
||||
expect(intent).toBe('retry_once');
|
||||
expect(result.current.proQuotaRequest).toBeNull();
|
||||
});
|
||||
|
||||
it('should auto-retry capacity failures thrown as raw string error in low verbosity mode', async () => {
|
||||
const { result } = await renderHook(() =>
|
||||
useQuotaAndFallback({
|
||||
config: mockConfig,
|
||||
historyManager: mockHistoryManager,
|
||||
userTier: UserTierId.FREE,
|
||||
setModelSwitchedFromQuotaError: mockSetModelSwitchedFromQuotaError,
|
||||
onShowAuthSelection: mockOnShowAuthSelection,
|
||||
paidTier: null,
|
||||
settings: mockSettings,
|
||||
errorVerbosity: 'low',
|
||||
}),
|
||||
);
|
||||
|
||||
const handler = setFallbackHandlerSpy.mock
|
||||
.calls[0][0] as FallbackModelHandler;
|
||||
const intent = await handler(
|
||||
'gemini-pro',
|
||||
'gemini-flash',
|
||||
'MODEL_CAPACITY_EXHAUSTED',
|
||||
);
|
||||
|
||||
expect(intent).toBe('retry_once');
|
||||
expect(result.current.proQuotaRequest).toBeNull();
|
||||
});
|
||||
|
||||
it('should show high demand message for MODEL_CAPACITY_EXHAUSTED', async () => {
|
||||
const { result } = await renderHook(() =>
|
||||
useQuotaAndFallback({
|
||||
config: mockConfig,
|
||||
historyManager: mockHistoryManager,
|
||||
userTier: UserTierId.FREE,
|
||||
setModelSwitchedFromQuotaError: mockSetModelSwitchedFromQuotaError,
|
||||
onShowAuthSelection: mockOnShowAuthSelection,
|
||||
paidTier: null,
|
||||
settings: mockSettings,
|
||||
}),
|
||||
);
|
||||
|
||||
const handler = setFallbackHandlerSpy.mock
|
||||
.calls[0][0] as FallbackModelHandler;
|
||||
|
||||
const error = new TerminalQuotaError(
|
||||
'pro capacity exhausted',
|
||||
mockGoogleApiError,
|
||||
undefined,
|
||||
'MODEL_CAPACITY_EXHAUSTED',
|
||||
);
|
||||
|
||||
act(() => {
|
||||
void handler('gemini-pro', 'gemini-flash', error);
|
||||
});
|
||||
|
||||
expect(result.current.proQuotaRequest).not.toBeNull();
|
||||
expect(result.current.proQuotaRequest?.isCapacityExceeded).toBe(true);
|
||||
expect(result.current.proQuotaRequest?.message).toContain(
|
||||
'We are currently experiencing high demand',
|
||||
);
|
||||
expect(result.current.proQuotaRequest?.message).not.toContain(
|
||||
'Usage limit reached',
|
||||
);
|
||||
});
|
||||
|
||||
describe('Interactive Fallback', () => {
|
||||
it('should set an interactive request for a terminal quota error', async () => {
|
||||
const { result } = await renderHook(() =>
|
||||
@@ -1040,7 +1160,7 @@ Your admin might have disabled the access. Contact them to enable the Preview Re
|
||||
);
|
||||
});
|
||||
|
||||
it('should show a special message when falling back from the preview model, but do not show periodical check message for flash model fallback', async () => {
|
||||
it('should show a special message when falling back from the preview model, but not show the periodical check message for flash model fallbacks', async () => {
|
||||
const { result } = await renderHook(() =>
|
||||
useQuotaAndFallback({
|
||||
config: mockConfig,
|
||||
|
||||
@@ -45,6 +45,9 @@ interface UseQuotaAndFallbackArgs {
|
||||
errorVerbosity?: 'low' | 'full';
|
||||
}
|
||||
|
||||
const isObject = (val: unknown): val is Record<string, unknown> =>
|
||||
typeof val === 'object' && val !== null;
|
||||
|
||||
export function useQuotaAndFallback({
|
||||
config,
|
||||
historyManager,
|
||||
@@ -79,6 +82,28 @@ export function useQuotaAndFallback({
|
||||
let message: string;
|
||||
let isTerminalQuotaError = false;
|
||||
let isModelNotFoundError = false;
|
||||
|
||||
const errorObj = isObject(error) ? error : null;
|
||||
|
||||
const errorReasonValue = errorObj?.['reason'];
|
||||
const errorReason =
|
||||
typeof errorReasonValue === 'string' ? errorReasonValue : undefined;
|
||||
|
||||
const errorMessageValue = errorObj?.['message'];
|
||||
const errorMessage =
|
||||
typeof errorMessageValue === 'string' ? errorMessageValue : undefined;
|
||||
|
||||
const isCapacityExceeded =
|
||||
errorReason === 'MODEL_CAPACITY_EXHAUSTED' ||
|
||||
errorReason === 'MODEL_CAPACITY_EXCEEDED' ||
|
||||
(typeof errorMessage === 'string' &&
|
||||
/exhausted your capacity|capacity exceeded|MODEL_CAPACITY_EXHAUSTED/i.test(
|
||||
errorMessage,
|
||||
)) ||
|
||||
(typeof error === 'string' &&
|
||||
/exhausted your capacity|capacity exceeded|MODEL_CAPACITY_EXHAUSTED/i.test(
|
||||
error,
|
||||
));
|
||||
const usageLimitReachedModel = isProModel(failedModel)
|
||||
? 'all Pro models'
|
||||
: failedModel;
|
||||
@@ -121,18 +146,30 @@ export function useQuotaAndFallback({
|
||||
}
|
||||
|
||||
// Default: Show existing ProQuotaDialog (for overageStrategy: 'never' or non-G1 users)
|
||||
const messageLines = [
|
||||
`Usage limit reached for ${usageLimitReachedModel}.`,
|
||||
error.retryDelayMs
|
||||
? `Access resets at ${getResetTimeMessage(error.retryDelayMs)}.`
|
||||
: null,
|
||||
`/stats model for usage details`,
|
||||
`/model to switch models.`,
|
||||
contentGeneratorConfig?.authType === AuthType.LOGIN_WITH_GOOGLE
|
||||
? `/auth to switch to API key.`
|
||||
: null,
|
||||
].filter(Boolean);
|
||||
message = messageLines.join('\n');
|
||||
if (isCapacityExceeded) {
|
||||
const messageLines = [
|
||||
`We are currently experiencing high demand for ${usageLimitReachedModel}.`,
|
||||
'We apologize and appreciate your patience.',
|
||||
error.retryDelayMs
|
||||
? `Access resets at ${getResetTimeMessage(error.retryDelayMs)}.`
|
||||
: null,
|
||||
`/model to switch models.`,
|
||||
].filter(Boolean);
|
||||
message = messageLines.join('\n');
|
||||
} else {
|
||||
const messageLines = [
|
||||
`Usage limit reached for ${usageLimitReachedModel}.`,
|
||||
error.retryDelayMs
|
||||
? `Access resets at ${getResetTimeMessage(error.retryDelayMs)}.`
|
||||
: null,
|
||||
`/stats model for usage details`,
|
||||
`/model to switch models.`,
|
||||
contentGeneratorConfig?.authType === AuthType.LOGIN_WITH_GOOGLE
|
||||
? `/auth to switch to API key.`
|
||||
: null,
|
||||
].filter(Boolean);
|
||||
message = messageLines.join('\n');
|
||||
}
|
||||
} else if (error instanceof ModelNotFoundError) {
|
||||
isModelNotFoundError = true;
|
||||
if (
|
||||
@@ -174,7 +211,7 @@ export function useQuotaAndFallback({
|
||||
// without interrupting with a dialog.
|
||||
if (
|
||||
errorVerbosity === 'low' &&
|
||||
!isTerminalQuotaError &&
|
||||
(!isTerminalQuotaError || isCapacityExceeded) &&
|
||||
!isModelNotFoundError
|
||||
) {
|
||||
return 'retry_once';
|
||||
@@ -197,6 +234,7 @@ export function useQuotaAndFallback({
|
||||
message,
|
||||
isTerminalQuotaError,
|
||||
isModelNotFoundError,
|
||||
isCapacityExceeded,
|
||||
authType: contentGeneratorConfig?.authType,
|
||||
});
|
||||
},
|
||||
|
||||
@@ -71,6 +71,11 @@ describe('CloudFreePrivacyNotice', () => {
|
||||
mockState: { isFreeTier: false },
|
||||
expectedText: 'Gemini Code Assist Privacy Notice',
|
||||
},
|
||||
{
|
||||
stateName: 'tier unavailable state',
|
||||
mockState: { isFreeTier: undefined, isTierUnavailable: true },
|
||||
expectedText: 'GOOGLE_CLOUD_PROJECT',
|
||||
},
|
||||
{
|
||||
stateName: 'free tier state',
|
||||
mockState: { isFreeTier: true },
|
||||
@@ -101,6 +106,11 @@ describe('CloudFreePrivacyNotice', () => {
|
||||
mockState: { isFreeTier: false },
|
||||
shouldExit: true,
|
||||
},
|
||||
{
|
||||
stateName: 'tier unavailable state',
|
||||
mockState: { isFreeTier: undefined, isTierUnavailable: true },
|
||||
shouldExit: true,
|
||||
},
|
||||
{
|
||||
stateName: 'free tier state (no selection)',
|
||||
mockState: { isFreeTier: true },
|
||||
|
||||
@@ -27,7 +27,9 @@ export const CloudFreePrivacyNotice = ({
|
||||
useKeypress(
|
||||
(key) => {
|
||||
if (
|
||||
(privacyState.error || privacyState.isFreeTier === false) &&
|
||||
(privacyState.error ||
|
||||
privacyState.isFreeTier === false ||
|
||||
privacyState.isTierUnavailable) &&
|
||||
key.name === 'escape'
|
||||
) {
|
||||
onExit();
|
||||
@@ -53,6 +55,35 @@ export const CloudFreePrivacyNotice = ({
|
||||
);
|
||||
}
|
||||
|
||||
if (privacyState.isTierUnavailable) {
|
||||
return (
|
||||
<Box flexDirection="column" marginY={1}>
|
||||
<Text bold color={theme.text.accent}>
|
||||
Gemini Code Assist Privacy Notice
|
||||
</Text>
|
||||
<Newline />
|
||||
<Text color={theme.text.primary}>
|
||||
The data collection opt-in isn't available for this account
|
||||
because it doesn't have a Gemini Code Assist for Individuals
|
||||
(free) tier.
|
||||
</Text>
|
||||
<Newline />
|
||||
<Text color={theme.text.primary}>
|
||||
If you're on a Google Workspace or enterprise account, use the
|
||||
Vertex AI / Google Cloud path instead by setting the
|
||||
GOOGLE_CLOUD_PROJECT environment variable to your Google Cloud
|
||||
project.
|
||||
</Text>
|
||||
<Newline />
|
||||
<Text color={theme.text.primary}>
|
||||
Learn more: https://geminicli.com/docs/get-started/authentication/
|
||||
</Text>
|
||||
<Newline />
|
||||
<Text color={theme.text.secondary}>Press Esc to exit.</Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
if (privacyState.isFreeTier === false) {
|
||||
return (
|
||||
<Box flexDirection="column" marginY={1}>
|
||||
|
||||
@@ -1,10 +1,74 @@
|
||||
(version 1)
|
||||
|
||||
;; allow everything by default
|
||||
(allow default)
|
||||
;; permissive-open: uses (deny default) and explicitly allows the operations the
|
||||
;; CLI needs, matching the restrictive-* / strict-* profiles. Keep the allow-list
|
||||
;; minimal and reviewed; do not switch to (allow default).
|
||||
;;
|
||||
;; Keep the non-network rules in sync with sandbox-macos-permissive-proxied.sb:
|
||||
;; the two profiles are intentionally identical except for their network rules
|
||||
;; ("open" allows broad outbound; "proxied" routes outbound through the proxy).
|
||||
(deny default)
|
||||
|
||||
;; deny all writes EXCEPT under specific paths
|
||||
(deny file-write*)
|
||||
;; allow reading files from anywhere on host
|
||||
(allow file-read*)
|
||||
|
||||
;; allow exec/fork (children inherit this policy, so they stay sandboxed)
|
||||
(allow process-exec)
|
||||
(allow process-fork)
|
||||
|
||||
;; allow signals to self, e.g. SIGPIPE on write to closed pipe
|
||||
(allow signal (target self))
|
||||
|
||||
;; allow read access to specific information about system
|
||||
;; from https://source.chromium.org/chromium/chromium/src/+/main:sandbox/policy/mac/common.sb;l=273-319;drc=7b3962fe2e5fc9e2ee58000dc8fbf3429d84d3bd
|
||||
(allow sysctl-read
|
||||
(sysctl-name "hw.activecpu")
|
||||
(sysctl-name "hw.busfrequency_compat")
|
||||
(sysctl-name "hw.byteorder")
|
||||
(sysctl-name "hw.cacheconfig")
|
||||
(sysctl-name "hw.cachelinesize_compat")
|
||||
(sysctl-name "hw.cpufamily")
|
||||
(sysctl-name "hw.cpufrequency_compat")
|
||||
(sysctl-name "hw.cputype")
|
||||
(sysctl-name "hw.l1dcachesize_compat")
|
||||
(sysctl-name "hw.l1icachesize_compat")
|
||||
(sysctl-name "hw.l2cachesize_compat")
|
||||
(sysctl-name "hw.l3cachesize_compat")
|
||||
(sysctl-name "hw.logicalcpu_max")
|
||||
(sysctl-name "hw.machine")
|
||||
(sysctl-name "hw.ncpu")
|
||||
(sysctl-name "hw.nperflevels")
|
||||
(sysctl-name "hw.optional.arm.FEAT_BF16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_DotProd")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FCMA")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FHM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FP16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_I8MM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_JSCVT")
|
||||
(sysctl-name "hw.optional.arm.FEAT_LSE")
|
||||
(sysctl-name "hw.optional.arm.FEAT_RDM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_SHA512")
|
||||
(sysctl-name "hw.optional.armv8_2_sha512")
|
||||
(sysctl-name "hw.packages")
|
||||
(sysctl-name "hw.pagesize_compat")
|
||||
(sysctl-name "hw.physicalcpu_max")
|
||||
(sysctl-name "hw.tbfrequency_compat")
|
||||
(sysctl-name "hw.vectorunit")
|
||||
(sysctl-name "kern.hostname")
|
||||
(sysctl-name "kern.maxfilesperproc")
|
||||
(sysctl-name "kern.osproductversion")
|
||||
(sysctl-name "kern.osrelease")
|
||||
(sysctl-name "kern.ostype")
|
||||
(sysctl-name "kern.osvariant_status")
|
||||
(sysctl-name "kern.osversion")
|
||||
(sysctl-name "kern.secure_kernel")
|
||||
(sysctl-name "kern.usrstack64")
|
||||
(sysctl-name "kern.version")
|
||||
(sysctl-name "sysctl.proc_cputype")
|
||||
(sysctl-name-prefix "hw.perflevel")
|
||||
)
|
||||
|
||||
;; allow writes only to specific paths (deny default already blocks the rest)
|
||||
(allow file-write*
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
@@ -12,7 +76,6 @@
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gitconfig"))
|
||||
;; Allow writes to included directories from --include-directories
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
@@ -25,3 +88,48 @@
|
||||
(literal "/dev/ptmx")
|
||||
(regex #"^/dev/ttys[0-9]*$")
|
||||
)
|
||||
|
||||
;; allow the mach services normal workflows need under deny-default: sysmond for
|
||||
;; process listing (pgrep), plus DNS resolution (mDNSResponder), directory
|
||||
;; services (opendirectoryd), and certificate validation (trustd/ocspd).
|
||||
;; This set mirrors the deny-default profile in
|
||||
;; packages/core/src/sandbox/macos/baseProfile.ts (its NETWORK_SEATBELT_PROFILE),
|
||||
;; which restrictive-open reaches only implicitly via (allow network-outbound).
|
||||
;; Keep this allow-list minimal and reviewed.
|
||||
(allow mach-lookup
|
||||
(global-name "com.apple.sysmond")
|
||||
(global-name "com.apple.system.opendirectoryd.libinfo")
|
||||
(global-name "com.apple.system.opendirectoryd.membership")
|
||||
(global-name "com.apple.bsd.dirhelper")
|
||||
(global-name "com.apple.SecurityServer")
|
||||
(global-name "com.apple.networkd")
|
||||
(global-name "com.apple.ocspd")
|
||||
(global-name "com.apple.trustd")
|
||||
(global-name "com.apple.trustd.agent")
|
||||
(global-name "com.apple.mDNSResponder")
|
||||
(global-name "com.apple.mDNSResponderHelper")
|
||||
(global-name "com.apple.SystemConfiguration.DNSConfiguration")
|
||||
(global-name "com.apple.SystemConfiguration.configd")
|
||||
)
|
||||
|
||||
;; AF_SYSTEM socket used by the network stack (from baseProfile.ts)
|
||||
(allow system-socket
|
||||
(require-all
|
||||
(socket-domain AF_SYSTEM)
|
||||
(socket-protocol 2)
|
||||
)
|
||||
)
|
||||
|
||||
;; enable terminal access required by ink
|
||||
;; fixes setRawMode EPERM failure (at node:tty:81:24)
|
||||
(allow file-ioctl (regex #"^/dev/tty.*"))
|
||||
|
||||
;; allow inbound network traffic (local dev/test servers, the debugger on :9229,
|
||||
;; OAuth localhost callbacks)
|
||||
(allow network-inbound (local ip "*:*"))
|
||||
|
||||
;; allow binding local ports (dev/test servers, OAuth localhost listeners)
|
||||
(allow network-bind (local ip "*:*"))
|
||||
|
||||
;; allow all outbound network traffic
|
||||
(allow network-outbound)
|
||||
|
||||
@@ -1,10 +1,76 @@
|
||||
(version 1)
|
||||
|
||||
;; allow everything by default
|
||||
(allow default)
|
||||
;; permissive-proxied: uses (deny default) and explicitly allows the operations
|
||||
;; the CLI needs, matching the restrictive-* / strict-* profiles. Keep the
|
||||
;; allow-list minimal and reviewed; do not switch to (allow default).
|
||||
;;
|
||||
;; Keep the non-network rules in sync with sandbox-macos-permissive-open.sb:
|
||||
;; the two profiles are intentionally identical except for their network rules
|
||||
;; ("open" allows broad outbound; "proxied" routes outbound through the proxy).
|
||||
(deny default)
|
||||
|
||||
;; deny all writes EXCEPT under specific paths
|
||||
(deny file-write*)
|
||||
;; allow reading files from anywhere on host
|
||||
(allow file-read*)
|
||||
|
||||
;; allow exec/fork (children inherit this policy, so they stay sandboxed)
|
||||
(allow process-exec)
|
||||
(allow process-fork)
|
||||
|
||||
;; allow signals to self, e.g. SIGPIPE on write to closed pipe
|
||||
(allow signal (target self))
|
||||
|
||||
;; allow read access to specific information about system
|
||||
;; from https://source.chromium.org/chromium/chromium/src/+/main:sandbox/policy/mac/common.sb;l=273-319;drc=7b3962fe2e5fc9e2ee58000dc8fbf3429d84d3bd
|
||||
(allow sysctl-read
|
||||
(sysctl-name "hw.activecpu")
|
||||
(sysctl-name "hw.busfrequency_compat")
|
||||
(sysctl-name "hw.byteorder")
|
||||
(sysctl-name "hw.cacheconfig")
|
||||
(sysctl-name "hw.cachelinesize_compat")
|
||||
(sysctl-name "hw.cpufamily")
|
||||
(sysctl-name "hw.cpufrequency_compat")
|
||||
(sysctl-name "hw.cputype")
|
||||
(sysctl-name "hw.l1dcachesize_compat")
|
||||
(sysctl-name "hw.l1icachesize_compat")
|
||||
(sysctl-name "hw.l2cachesize_compat")
|
||||
(sysctl-name "hw.l3cachesize_compat")
|
||||
(sysctl-name "hw.logicalcpu_max")
|
||||
(sysctl-name "hw.machine")
|
||||
(sysctl-name "hw.ncpu")
|
||||
(sysctl-name "hw.nperflevels")
|
||||
(sysctl-name "hw.optional.arm.FEAT_BF16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_DotProd")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FCMA")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FHM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FP16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_I8MM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_JSCVT")
|
||||
(sysctl-name "hw.optional.arm.FEAT_LSE")
|
||||
(sysctl-name "hw.optional.arm.FEAT_RDM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_SHA512")
|
||||
(sysctl-name "hw.optional.armv8_2_sha512")
|
||||
(sysctl-name "hw.packages")
|
||||
(sysctl-name "hw.pagesize_compat")
|
||||
(sysctl-name "hw.physicalcpu_max")
|
||||
(sysctl-name "hw.tbfrequency_compat")
|
||||
(sysctl-name "hw.vectorunit")
|
||||
(sysctl-name "kern.hostname")
|
||||
(sysctl-name "kern.maxfilesperproc")
|
||||
(sysctl-name "kern.osproductversion")
|
||||
(sysctl-name "kern.osrelease")
|
||||
(sysctl-name "kern.ostype")
|
||||
(sysctl-name "kern.osvariant_status")
|
||||
(sysctl-name "kern.osversion")
|
||||
(sysctl-name "kern.secure_kernel")
|
||||
(sysctl-name "kern.usrstack64")
|
||||
(sysctl-name "kern.version")
|
||||
(sysctl-name "sysctl.proc_cputype")
|
||||
(sysctl-name-prefix "hw.perflevel")
|
||||
)
|
||||
|
||||
;; allow writes only to specific paths (deny default already blocks the rest).
|
||||
;; Mirrors permissive-open, including /dev/ptmx and the /dev/ttys regex needed
|
||||
;; for PTY support under deny default.
|
||||
(allow file-write*
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
@@ -12,7 +78,6 @@
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gitconfig"))
|
||||
;; Allow writes to included directories from --include-directories
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
@@ -22,16 +87,52 @@
|
||||
(literal "/dev/stdout")
|
||||
(literal "/dev/stderr")
|
||||
(literal "/dev/null")
|
||||
(literal "/dev/ptmx")
|
||||
(regex #"^/dev/ttys[0-9]*$")
|
||||
)
|
||||
|
||||
;; deny all inbound network traffic EXCEPT on debugger port
|
||||
(deny network-inbound)
|
||||
;; allow the mach services normal workflows need under deny-default: sysmond for
|
||||
;; process listing (pgrep), plus DNS resolution (mDNSResponder), directory
|
||||
;; services (opendirectoryd), and certificate validation (trustd/ocspd).
|
||||
;; This set mirrors the deny-default profile in
|
||||
;; packages/core/src/sandbox/macos/baseProfile.ts (its NETWORK_SEATBELT_PROFILE),
|
||||
;; which restrictive-proxied reaches only implicitly via (allow network-outbound).
|
||||
;; Keep this allow-list minimal and reviewed.
|
||||
(allow mach-lookup
|
||||
(global-name "com.apple.sysmond")
|
||||
(global-name "com.apple.system.opendirectoryd.libinfo")
|
||||
(global-name "com.apple.system.opendirectoryd.membership")
|
||||
(global-name "com.apple.bsd.dirhelper")
|
||||
(global-name "com.apple.SecurityServer")
|
||||
(global-name "com.apple.networkd")
|
||||
(global-name "com.apple.ocspd")
|
||||
(global-name "com.apple.trustd")
|
||||
(global-name "com.apple.trustd.agent")
|
||||
(global-name "com.apple.mDNSResponder")
|
||||
(global-name "com.apple.mDNSResponderHelper")
|
||||
(global-name "com.apple.SystemConfiguration.DNSConfiguration")
|
||||
(global-name "com.apple.SystemConfiguration.configd")
|
||||
)
|
||||
|
||||
;; AF_SYSTEM socket used by the network stack (from baseProfile.ts)
|
||||
(allow system-socket
|
||||
(require-all
|
||||
(socket-domain AF_SYSTEM)
|
||||
(socket-protocol 2)
|
||||
)
|
||||
)
|
||||
|
||||
;; enable terminal access required by ink
|
||||
;; fixes setRawMode EPERM failure (at node:tty:81:24)
|
||||
(allow file-ioctl (regex #"^/dev/tty.*"))
|
||||
|
||||
;; allow inbound network traffic on debugger port
|
||||
(allow network-inbound (local ip "localhost:9229"))
|
||||
|
||||
;; allow binding local ports (dev/test servers, OAuth localhost listeners)
|
||||
(allow network-bind (local ip "*:*"))
|
||||
|
||||
;; deny all outbound network traffic EXCEPT through proxy on localhost:8877
|
||||
;; set `GEMINI_SANDBOX_PROXY_COMMAND=<command>` to run proxy alongside sandbox
|
||||
;; proxy must listen on :::8877 (see docs/examples/proxy-script.md)
|
||||
(deny network-outbound)
|
||||
(allow network-outbound (remote tcp "localhost:8877"))
|
||||
|
||||
(allow network-bind (local ip "*:*"))
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
/**
|
||||
* @license
|
||||
* Copyright 2026 Google LLC
|
||||
* SPDX-License-Identifier: Apache-2.0
|
||||
*/
|
||||
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import path from 'node:path';
|
||||
|
||||
const utilsDir = path.dirname(fileURLToPath(import.meta.url));
|
||||
|
||||
/**
|
||||
* Strip SBPL comments (`; ...` to end of line) so assertions run against the
|
||||
* actual sandbox rules rather than any keywords that happen to appear in the
|
||||
* explanatory comments.
|
||||
*/
|
||||
function readRules(profile: string): string {
|
||||
return readFileSync(path.join(utilsDir, profile), 'utf8')
|
||||
.split('\n')
|
||||
.map((line) => {
|
||||
const commentStart = line.indexOf(';');
|
||||
return commentStart === -1 ? line : line.slice(0, commentStart);
|
||||
})
|
||||
.join('\n');
|
||||
}
|
||||
|
||||
const PERMISSIVE_PROFILES = [
|
||||
'sandbox-macos-permissive-open.sb',
|
||||
'sandbox-macos-permissive-proxied.sb',
|
||||
];
|
||||
|
||||
// These two profiles are the default macOS Seatbelt profiles, so the invariants
|
||||
// below must never silently regress. Keep them deny-default and confirm the
|
||||
// reviewed allow-list stays in place.
|
||||
describe('macOS permissive Seatbelt profiles', () => {
|
||||
describe.each(PERMISSIVE_PROFILES)('%s', (profile) => {
|
||||
const rules = readRules(profile);
|
||||
|
||||
it('uses a deny-default foundation', () => {
|
||||
expect(rules).toContain('(deny default)');
|
||||
});
|
||||
|
||||
it('does not use an allow-default foundation', () => {
|
||||
expect(rules).not.toContain('(allow default)');
|
||||
});
|
||||
|
||||
it('does not permit filesystem (un)mounts', () => {
|
||||
expect(rules).not.toMatch(/file-mount/);
|
||||
expect(rules).not.toMatch(/file-unmount/);
|
||||
});
|
||||
|
||||
it('does not grant broad service lookups', () => {
|
||||
expect(rules).not.toMatch(/launchd/);
|
||||
expect(rules).not.toMatch(/launchservices/i);
|
||||
});
|
||||
|
||||
it('allows binding local ports for dev/test servers', () => {
|
||||
expect(rules).toContain('(allow network-bind (local ip "*:*"))');
|
||||
});
|
||||
});
|
||||
|
||||
it('permissive-open keeps broad inbound and outbound network', () => {
|
||||
const rules = readRules('sandbox-macos-permissive-open.sb');
|
||||
expect(rules).toContain('(allow network-inbound (local ip "*:*"))');
|
||||
expect(rules).toMatch(/\(allow network-outbound\)/);
|
||||
});
|
||||
|
||||
it('permissive-proxied confines outbound to the proxy', () => {
|
||||
const rules = readRules('sandbox-macos-permissive-proxied.sb');
|
||||
expect(rules).toContain(
|
||||
'(allow network-outbound (remote tcp "localhost:8877"))',
|
||||
);
|
||||
// Proxied mode must never grant unrestricted outbound network.
|
||||
expect(rules).not.toMatch(/\(allow network-outbound\)/);
|
||||
});
|
||||
});
|
||||
@@ -70,7 +70,6 @@
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gitconfig"))
|
||||
;; Allow writes to included directories from --include-directories
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
|
||||
@@ -70,7 +70,6 @@
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gitconfig"))
|
||||
;; Allow writes to included directories from --include-directories
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
|
||||
@@ -105,7 +105,6 @@
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(literal (string-append (param "HOME_DIR") "/.gitconfig"))
|
||||
;; Allow writes to included directories from --include-directories
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
|
||||
@@ -105,7 +105,6 @@
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(literal (string-append (param "HOME_DIR") "/.gitconfig"))
|
||||
;; Allow writes to included directories from --include-directories
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
|
||||
@@ -292,6 +292,140 @@ describe('sandbox', () => {
|
||||
await expect(start_sandbox(config)).rejects.toThrow(FatalSandboxError);
|
||||
});
|
||||
|
||||
it('should fall back to embedded profile if the .sb file is missing on disk', async () => {
|
||||
vi.mocked(os.platform).mockReturnValue('darwin');
|
||||
vi.mocked(fs.existsSync).mockImplementation((p) =>
|
||||
String(p).includes(
|
||||
'gemini-sandbox-macos-permissive-open-a1b2c3d4e5f6.sb',
|
||||
),
|
||||
);
|
||||
|
||||
const config: SandboxConfig = createMockSandboxConfig({
|
||||
command: 'sandbox-exec',
|
||||
image: 'some-image',
|
||||
});
|
||||
|
||||
const onSpy = vi.spyOn(process, 'on');
|
||||
const offSpy = vi.spyOn(process, 'off');
|
||||
|
||||
interface MockProcess extends EventEmitter {
|
||||
stdout: EventEmitter;
|
||||
stderr: EventEmitter;
|
||||
}
|
||||
const mockSpawnProcess = new EventEmitter() as MockProcess;
|
||||
mockSpawnProcess.stdout = new EventEmitter();
|
||||
mockSpawnProcess.stderr = new EventEmitter();
|
||||
vi.mocked(spawn).mockReturnValue(
|
||||
mockSpawnProcess as unknown as ReturnType<typeof spawn>,
|
||||
);
|
||||
|
||||
const promise = start_sandbox(config, [], undefined, ['arg1']);
|
||||
|
||||
setTimeout(() => {
|
||||
mockSpawnProcess.emit('close', 0);
|
||||
}, 10);
|
||||
|
||||
await expect(promise).resolves.toBe(0);
|
||||
|
||||
// Verify fs.writeFileSync was called with the temp profile file, content, and 0o600 permissions
|
||||
expect(fs.writeFileSync).toHaveBeenCalledWith(
|
||||
expect.stringContaining(
|
||||
'gemini-sandbox-macos-permissive-open-a1b2c3d4e5f6.sb',
|
||||
),
|
||||
expect.stringContaining('deny default'),
|
||||
expect.objectContaining({
|
||||
encoding: 'utf8',
|
||||
mode: 0o600,
|
||||
}),
|
||||
);
|
||||
|
||||
// Verify spawn was called with the temp profile file
|
||||
expect(spawn).toHaveBeenCalledWith(
|
||||
'sandbox-exec',
|
||||
expect.arrayContaining([
|
||||
'-f',
|
||||
expect.stringContaining(
|
||||
'gemini-sandbox-macos-permissive-open-a1b2c3d4e5f6.sb',
|
||||
),
|
||||
]),
|
||||
expect.objectContaining({ stdio: 'inherit' }),
|
||||
);
|
||||
|
||||
// Verify process on/off hooks were called for exit, SIGINT, and SIGTERM cleanups
|
||||
expect(onSpy).toHaveBeenCalledWith('exit', expect.any(Function));
|
||||
expect(onSpy).toHaveBeenCalledWith('SIGINT', expect.any(Function));
|
||||
expect(onSpy).toHaveBeenCalledWith('SIGTERM', expect.any(Function));
|
||||
|
||||
expect(offSpy).toHaveBeenCalledWith('exit', expect.any(Function));
|
||||
expect(offSpy).toHaveBeenCalledWith('SIGINT', expect.any(Function));
|
||||
expect(offSpy).toHaveBeenCalledWith('SIGTERM', expect.any(Function));
|
||||
|
||||
// Verify fs.unlinkSync was called to clean up the temp file
|
||||
expect(fs.unlinkSync).toHaveBeenCalledWith(
|
||||
expect.stringContaining(
|
||||
'gemini-sandbox-macos-permissive-open-a1b2c3d4e5f6.sb',
|
||||
),
|
||||
);
|
||||
});
|
||||
|
||||
it.each([
|
||||
'permissive-open',
|
||||
'permissive-closed',
|
||||
'permissive-proxied',
|
||||
'restrictive-open',
|
||||
'restrictive-closed',
|
||||
'restrictive-proxied',
|
||||
'strict-open',
|
||||
'strict-proxied',
|
||||
])(
|
||||
'should fall back to embedded content successfully for profile "%s"',
|
||||
async (profile) => {
|
||||
vi.mocked(os.platform).mockReturnValue('darwin');
|
||||
// Mock existsSync to return false for the profile file but true for temp directories
|
||||
vi.mocked(fs.existsSync).mockImplementation((p) =>
|
||||
String(p).includes('gemini-sandbox-macos-'),
|
||||
);
|
||||
|
||||
vi.stubEnv('SEATBELT_PROFILE', profile);
|
||||
|
||||
const config: SandboxConfig = createMockSandboxConfig({
|
||||
command: 'sandbox-exec',
|
||||
image: 'some-image',
|
||||
});
|
||||
|
||||
interface MockProcess extends EventEmitter {
|
||||
stdout: EventEmitter;
|
||||
stderr: EventEmitter;
|
||||
}
|
||||
const mockSpawnProcess = new EventEmitter() as MockProcess;
|
||||
mockSpawnProcess.stdout = new EventEmitter();
|
||||
mockSpawnProcess.stderr = new EventEmitter();
|
||||
vi.mocked(spawn).mockReturnValue(
|
||||
mockSpawnProcess as unknown as ReturnType<typeof spawn>,
|
||||
);
|
||||
|
||||
const promise = start_sandbox(config, [], undefined, ['arg1']);
|
||||
|
||||
setTimeout(() => {
|
||||
mockSpawnProcess.emit('close', 0);
|
||||
}, 10);
|
||||
|
||||
await expect(promise).resolves.toBe(0);
|
||||
|
||||
// Verify fs.writeFileSync was called with the correct file mode and content for the profile
|
||||
expect(fs.writeFileSync).toHaveBeenCalledWith(
|
||||
expect.stringContaining(`gemini-sandbox-macos-${profile}-`),
|
||||
expect.stringContaining('deny default'),
|
||||
expect.objectContaining({
|
||||
encoding: 'utf8',
|
||||
mode: 0o600,
|
||||
}),
|
||||
);
|
||||
|
||||
vi.unstubAllEnvs();
|
||||
},
|
||||
);
|
||||
|
||||
it('should handle Docker execution', async () => {
|
||||
const config: SandboxConfig = createMockSandboxConfig({
|
||||
command: 'docker',
|
||||
|
||||
+212
-149
@@ -39,6 +39,7 @@ import {
|
||||
SANDBOX_PROXY_NAME,
|
||||
BUILTIN_SEATBELT_PROFILES,
|
||||
} from './sandboxUtils.js';
|
||||
import { BUILTIN_SEATBELT_PROFILE_CONTENTS } from './sandboxBuiltinProfiles.js';
|
||||
|
||||
const execAsync = promisify(exec);
|
||||
const execFileAsync = promisify(execFile);
|
||||
@@ -56,6 +57,41 @@ export async function start_sandbox(
|
||||
patcher.patch();
|
||||
|
||||
let stopProxy: (() => void) | undefined = undefined;
|
||||
let tempProfileFile: string | null = null;
|
||||
|
||||
const cleanup = () => {
|
||||
if (tempProfileFile && fs.existsSync(tempProfileFile)) {
|
||||
try {
|
||||
fs.unlinkSync(tempProfileFile);
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
tempProfileFile = null;
|
||||
}
|
||||
if (stopProxy) {
|
||||
try {
|
||||
stopProxy();
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const sigintHandler = () => {
|
||||
cleanup();
|
||||
process.off('SIGINT', sigintHandler);
|
||||
process.kill(process.pid, 'SIGINT');
|
||||
};
|
||||
|
||||
const sigtermHandler = () => {
|
||||
cleanup();
|
||||
process.off('SIGTERM', sigtermHandler);
|
||||
process.kill(process.pid, 'SIGTERM');
|
||||
};
|
||||
|
||||
process.on('exit', cleanup);
|
||||
process.on('SIGINT', sigintHandler);
|
||||
process.on('SIGTERM', sigtermHandler);
|
||||
|
||||
try {
|
||||
if (config.command === 'sandbox-exec') {
|
||||
@@ -81,161 +117,193 @@ export async function start_sandbox(
|
||||
profileFile = fs.existsSync(userProfileFile)
|
||||
? userProfileFile
|
||||
: projectProfileFile;
|
||||
}
|
||||
if (!fs.existsSync(profileFile)) {
|
||||
throw new FatalSandboxError(
|
||||
`Missing macos seatbelt profile file '${profileFile}'`,
|
||||
);
|
||||
}
|
||||
debugLogger.log(`using macos seatbelt (profile: ${profile}) ...`);
|
||||
// if DEBUG is set, convert to --inspect-brk in NODE_OPTIONS
|
||||
const nodeOptions = [
|
||||
...(process.env['DEBUG'] ? ['--inspect-brk'] : []),
|
||||
...nodeArgs,
|
||||
].join(' ');
|
||||
|
||||
const args = [
|
||||
'-D',
|
||||
`TARGET_DIR=${fs.realpathSync(process.cwd())}`,
|
||||
'-D',
|
||||
`TMP_DIR=${fs.realpathSync(os.tmpdir())}`,
|
||||
'-D',
|
||||
`HOME_DIR=${fs.realpathSync(homedir())}`,
|
||||
'-D',
|
||||
`CACHE_DIR=${fs.realpathSync((await execAsync('getconf DARWIN_USER_CACHE_DIR')).stdout.trim())}`,
|
||||
];
|
||||
|
||||
// Add included directories from the workspace context
|
||||
// Always add 5 INCLUDE_DIR parameters to ensure .sb files can reference them
|
||||
const MAX_INCLUDE_DIRS = 5;
|
||||
const targetDir = fs.realpathSync(cliConfig?.getTargetDir() || '');
|
||||
const includedDirs: string[] = [];
|
||||
|
||||
if (cliConfig) {
|
||||
const workspaceContext = cliConfig.getWorkspaceContext();
|
||||
const directories = workspaceContext.getDirectories();
|
||||
|
||||
// Filter out TARGET_DIR
|
||||
for (const dir of directories) {
|
||||
const realDir = fs.realpathSync(dir);
|
||||
if (realDir !== targetDir) {
|
||||
includedDirs.push(realDir);
|
||||
} else {
|
||||
// For builtin profiles, if the file doesn't exist on disk (e.g. bundled or bazel environments),
|
||||
// write the embedded profile content to a temporary file.
|
||||
if (!fs.existsSync(profileFile)) {
|
||||
const content = BUILTIN_SEATBELT_PROFILE_CONTENTS[profile];
|
||||
if (content) {
|
||||
try {
|
||||
const tempDir = fs.realpathSync(os.tmpdir());
|
||||
const rand = randomBytes(8).toString('hex');
|
||||
tempProfileFile = path.join(
|
||||
tempDir,
|
||||
`gemini-sandbox-macos-${profile}-${rand}.sb`,
|
||||
);
|
||||
fs.writeFileSync(tempProfileFile, content, {
|
||||
encoding: 'utf8',
|
||||
mode: 0o600,
|
||||
});
|
||||
profileFile = tempProfileFile;
|
||||
} catch (err) {
|
||||
debugLogger.warn(
|
||||
`Failed to write temporary seatbelt profile: ${err}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Add custom allowed paths from config
|
||||
if (config.allowedPaths) {
|
||||
for (const hostPath of config.allowedPaths) {
|
||||
if (
|
||||
hostPath &&
|
||||
path.isAbsolute(hostPath) &&
|
||||
fs.existsSync(hostPath)
|
||||
) {
|
||||
const realDir = fs.realpathSync(hostPath);
|
||||
if (!includedDirs.includes(realDir) && realDir !== targetDir) {
|
||||
try {
|
||||
if (!fs.existsSync(profileFile)) {
|
||||
throw new FatalSandboxError(
|
||||
`Missing macos seatbelt profile file '${profileFile}'`,
|
||||
);
|
||||
}
|
||||
debugLogger.log(`using macos seatbelt (profile: ${profile}) ...`);
|
||||
// if DEBUG is set, convert to --inspect-brk in NODE_OPTIONS
|
||||
const nodeOptions = [
|
||||
...(process.env['DEBUG'] ? ['--inspect-brk'] : []),
|
||||
...nodeArgs,
|
||||
].join(' ');
|
||||
|
||||
const args = [
|
||||
'-D',
|
||||
`TARGET_DIR=${fs.realpathSync(process.cwd())}`,
|
||||
'-D',
|
||||
`TMP_DIR=${fs.realpathSync(os.tmpdir())}`,
|
||||
'-D',
|
||||
`HOME_DIR=${fs.realpathSync(homedir())}`,
|
||||
'-D',
|
||||
`CACHE_DIR=${fs.realpathSync((await execAsync('getconf DARWIN_USER_CACHE_DIR')).stdout.trim())}`,
|
||||
];
|
||||
|
||||
// Add included directories from the workspace context
|
||||
// Always add 5 INCLUDE_DIR parameters to ensure .sb files can reference them
|
||||
const MAX_INCLUDE_DIRS = 5;
|
||||
const targetDir = fs.realpathSync(cliConfig?.getTargetDir() || '');
|
||||
const includedDirs: string[] = [];
|
||||
|
||||
if (cliConfig) {
|
||||
const workspaceContext = cliConfig.getWorkspaceContext();
|
||||
const directories = workspaceContext.getDirectories();
|
||||
|
||||
// Filter out TARGET_DIR
|
||||
for (const dir of directories) {
|
||||
const realDir = fs.realpathSync(dir);
|
||||
if (realDir !== targetDir) {
|
||||
includedDirs.push(realDir);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (let i = 0; i < MAX_INCLUDE_DIRS; i++) {
|
||||
let dirPath = '/dev/null'; // Default to a safe path that won't cause issues
|
||||
|
||||
if (i < includedDirs.length) {
|
||||
dirPath = includedDirs[i];
|
||||
}
|
||||
|
||||
args.push('-D', `INCLUDE_DIR_${i}=${dirPath}`);
|
||||
}
|
||||
|
||||
const finalArgv = cliArgs;
|
||||
|
||||
args.push(
|
||||
'-f',
|
||||
profileFile,
|
||||
'sh',
|
||||
'-c',
|
||||
[
|
||||
`SANDBOX=sandbox-exec`,
|
||||
`NODE_OPTIONS="${nodeOptions}"`,
|
||||
...finalArgv.map((arg) => quote([arg])),
|
||||
].join(' '),
|
||||
);
|
||||
// start and set up proxy if GEMINI_SANDBOX_PROXY_COMMAND is set
|
||||
const proxyCommand = process.env['GEMINI_SANDBOX_PROXY_COMMAND'];
|
||||
let proxyProcess: ChildProcess | undefined = undefined;
|
||||
let sandboxProcess: ChildProcess | undefined = undefined;
|
||||
const sandboxEnv = { ...process.env };
|
||||
if (proxyCommand) {
|
||||
const proxy =
|
||||
process.env['HTTPS_PROXY'] ||
|
||||
process.env['https_proxy'] ||
|
||||
process.env['HTTP_PROXY'] ||
|
||||
process.env['http_proxy'] ||
|
||||
'http://localhost:8877';
|
||||
sandboxEnv['HTTPS_PROXY'] = proxy;
|
||||
sandboxEnv['https_proxy'] = proxy; // lower-case can be required, e.g. for curl
|
||||
sandboxEnv['HTTP_PROXY'] = proxy;
|
||||
sandboxEnv['http_proxy'] = proxy;
|
||||
const noProxy = process.env['NO_PROXY'] || process.env['no_proxy'];
|
||||
if (noProxy) {
|
||||
sandboxEnv['NO_PROXY'] = noProxy;
|
||||
sandboxEnv['no_proxy'] = noProxy;
|
||||
}
|
||||
proxyProcess = spawn(proxyCommand, {
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
shell: true,
|
||||
detached: true,
|
||||
});
|
||||
// install handlers to stop proxy on exit/signal
|
||||
stopProxy = () => {
|
||||
debugLogger.log('stopping proxy ...');
|
||||
if (proxyProcess?.pid) {
|
||||
try {
|
||||
process.kill(-proxyProcess.pid, 'SIGTERM');
|
||||
} catch {
|
||||
// ignore
|
||||
// Add custom allowed paths from config
|
||||
if (config.allowedPaths) {
|
||||
for (const hostPath of config.allowedPaths) {
|
||||
if (
|
||||
hostPath &&
|
||||
path.isAbsolute(hostPath) &&
|
||||
fs.existsSync(hostPath)
|
||||
) {
|
||||
const realDir = fs.realpathSync(hostPath);
|
||||
if (!includedDirs.includes(realDir) && realDir !== targetDir) {
|
||||
includedDirs.push(realDir);
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
process.on('exit', stopProxy);
|
||||
process.on('SIGINT', stopProxy);
|
||||
process.on('SIGTERM', stopProxy);
|
||||
}
|
||||
|
||||
// commented out as it disrupts ink rendering
|
||||
// proxyProcess.stdout?.on('data', (data) => {
|
||||
// console.info(data.toString());
|
||||
// });
|
||||
proxyProcess.stderr?.on('data', (data) => {
|
||||
debugLogger.debug(`[PROXY STDERR]: ${data.toString().trim()}`);
|
||||
});
|
||||
proxyProcess.on('close', (code, signal) => {
|
||||
if (sandboxProcess?.pid) {
|
||||
process.kill(-sandboxProcess.pid, 'SIGTERM');
|
||||
for (let i = 0; i < MAX_INCLUDE_DIRS; i++) {
|
||||
let dirPath = '/dev/null'; // Default to a safe path that won't cause issues
|
||||
|
||||
if (i < includedDirs.length) {
|
||||
dirPath = includedDirs[i];
|
||||
}
|
||||
throw new FatalSandboxError(
|
||||
`Proxy command '${proxyCommand}' exited with code ${code}, signal ${signal}`,
|
||||
);
|
||||
});
|
||||
debugLogger.log('waiting for proxy to start ...');
|
||||
await execAsync(
|
||||
`until timeout 0.25 curl -s http://localhost:8877; do sleep 0.25; done`,
|
||||
|
||||
args.push('-D', `INCLUDE_DIR_${i}=${dirPath}`);
|
||||
}
|
||||
|
||||
const finalArgv = cliArgs;
|
||||
|
||||
args.push(
|
||||
'-f',
|
||||
profileFile,
|
||||
'sh',
|
||||
'-c',
|
||||
[
|
||||
`SANDBOX=sandbox-exec`,
|
||||
'NODE_OPTIONS=' + quote([nodeOptions]),
|
||||
...finalArgv.map((arg) => quote([arg])),
|
||||
].join(' '),
|
||||
);
|
||||
}
|
||||
// spawn child and let it inherit stdio
|
||||
process.stdin.pause();
|
||||
sandboxProcess = spawn(config.command, args, {
|
||||
stdio: 'inherit',
|
||||
});
|
||||
return await new Promise((resolve, reject) => {
|
||||
sandboxProcess?.on('error', reject);
|
||||
sandboxProcess?.on('close', (code) => {
|
||||
process.stdin.resume();
|
||||
resolve(code ?? 1);
|
||||
// start and set up proxy if GEMINI_SANDBOX_PROXY_COMMAND is set
|
||||
const proxyCommand = process.env['GEMINI_SANDBOX_PROXY_COMMAND'];
|
||||
let proxyProcess: ChildProcess | undefined = undefined;
|
||||
let sandboxProcess: ChildProcess | undefined = undefined;
|
||||
const sandboxEnv = { ...process.env };
|
||||
if (proxyCommand) {
|
||||
const proxy =
|
||||
process.env['HTTPS_PROXY'] ||
|
||||
process.env['https_proxy'] ||
|
||||
process.env['HTTP_PROXY'] ||
|
||||
process.env['http_proxy'] ||
|
||||
'http://localhost:8877';
|
||||
sandboxEnv['HTTPS_PROXY'] = proxy;
|
||||
sandboxEnv['https_proxy'] = proxy; // lower-case can be required, e.g. for curl
|
||||
sandboxEnv['HTTP_PROXY'] = proxy;
|
||||
sandboxEnv['http_proxy'] = proxy;
|
||||
const noProxy = process.env['NO_PROXY'] || process.env['no_proxy'];
|
||||
if (noProxy) {
|
||||
sandboxEnv['NO_PROXY'] = noProxy;
|
||||
sandboxEnv['no_proxy'] = noProxy;
|
||||
}
|
||||
proxyProcess = spawn(proxyCommand, {
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
shell: true,
|
||||
detached: true,
|
||||
});
|
||||
// install handlers to stop proxy on exit/signal
|
||||
stopProxy = () => {
|
||||
debugLogger.log('stopping proxy ...');
|
||||
if (proxyProcess?.pid) {
|
||||
try {
|
||||
process.kill(-proxyProcess.pid, 'SIGTERM');
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// commented out as it disrupts ink rendering
|
||||
// proxyProcess.stdout?.on('data', (data) => {
|
||||
// console.info(data.toString());
|
||||
// });
|
||||
proxyProcess.stderr?.on('data', (data) => {
|
||||
debugLogger.debug(`[PROXY STDERR]: ${data.toString().trim()}`);
|
||||
});
|
||||
proxyProcess.on('close', (code, signal) => {
|
||||
if (sandboxProcess?.pid) {
|
||||
process.kill(-sandboxProcess.pid, 'SIGTERM');
|
||||
}
|
||||
throw new FatalSandboxError(
|
||||
`Proxy command '${proxyCommand}' exited with code ${code}, signal ${signal}`,
|
||||
);
|
||||
});
|
||||
debugLogger.log('waiting for proxy to start ...');
|
||||
await execAsync(
|
||||
`until timeout 0.25 curl -s http://localhost:8877; do sleep 0.25; done`,
|
||||
);
|
||||
}
|
||||
// spawn child and let it inherit stdio
|
||||
process.stdin.pause();
|
||||
sandboxProcess = spawn(config.command, args, {
|
||||
stdio: 'inherit',
|
||||
});
|
||||
});
|
||||
return await new Promise((resolve, reject) => {
|
||||
sandboxProcess?.on('error', (err) => {
|
||||
cleanup();
|
||||
reject(err);
|
||||
});
|
||||
sandboxProcess?.on('close', (code) => {
|
||||
process.stdin.resume();
|
||||
cleanup();
|
||||
resolve(code ?? 1);
|
||||
});
|
||||
});
|
||||
} catch (err) {
|
||||
cleanup();
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
if (config.command === 'lxc') {
|
||||
@@ -768,9 +836,6 @@ export async function start_sandbox(
|
||||
// ignore
|
||||
}
|
||||
};
|
||||
process.on('exit', stopProxy);
|
||||
process.on('SIGINT', stopProxy);
|
||||
process.on('SIGTERM', stopProxy);
|
||||
|
||||
// commented out as it disrupts ink rendering
|
||||
// proxyProcess.stdout?.on('data', (data) => {
|
||||
@@ -821,12 +886,10 @@ export async function start_sandbox(
|
||||
});
|
||||
});
|
||||
} finally {
|
||||
if (stopProxy) {
|
||||
stopProxy();
|
||||
process.off('exit', stopProxy);
|
||||
process.off('SIGINT', stopProxy);
|
||||
process.off('SIGTERM', stopProxy);
|
||||
}
|
||||
process.off('exit', cleanup);
|
||||
process.off('SIGINT', sigintHandler);
|
||||
process.off('SIGTERM', sigtermHandler);
|
||||
cleanup();
|
||||
patcher.cleanup();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,555 @@
|
||||
/**
|
||||
* @license
|
||||
* Copyright 2026 Google LLC
|
||||
* SPDX-License-Identifier: Apache-2.0
|
||||
*/
|
||||
|
||||
export const BUILTIN_SEATBELT_PROFILE_CONTENTS: Record<string, string> = {
|
||||
'permissive-open': `(version 1)
|
||||
(deny default)
|
||||
(allow file-read*)
|
||||
(allow process-exec)
|
||||
(allow process-fork)
|
||||
(allow signal (target self))
|
||||
(allow sysctl-read
|
||||
(sysctl-name "hw.activecpu")
|
||||
(sysctl-name "hw.busfrequency_compat")
|
||||
(sysctl-name "hw.byteorder")
|
||||
(sysctl-name "hw.cacheconfig")
|
||||
(sysctl-name "hw.cachelinesize_compat")
|
||||
(sysctl-name "hw.cpufamily")
|
||||
(sysctl-name "hw.cpufrequency_compat")
|
||||
(sysctl-name "hw.cputype")
|
||||
(sysctl-name "hw.l1dcachesize_compat")
|
||||
(sysctl-name "hw.l1icachesize_compat")
|
||||
(sysctl-name "hw.l2cachesize_compat")
|
||||
(sysctl-name "hw.l3cachesize_compat")
|
||||
(sysctl-name "hw.logicalcpu_max")
|
||||
(sysctl-name "hw.machine")
|
||||
(sysctl-name "hw.ncpu")
|
||||
(sysctl-name "hw.nperflevels")
|
||||
(sysctl-name "hw.optional.arm.FEAT_BF16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_DotProd")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FCMA")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FHM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FP16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_I8MM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_JSCVT")
|
||||
(sysctl-name "hw.optional.arm.FEAT_LSE")
|
||||
(sysctl-name "hw.optional.arm.FEAT_RDM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_SHA512")
|
||||
(sysctl-name "hw.optional.armv8_2_sha512")
|
||||
(sysctl-name "hw.packages")
|
||||
(sysctl-name "hw.pagesize_compat")
|
||||
(sysctl-name "hw.physicalcpu_max")
|
||||
(sysctl-name "hw.tbfrequency_compat")
|
||||
(sysctl-name "hw.vectorunit")
|
||||
(sysctl-name "kern.hostname")
|
||||
(sysctl-name "kern.maxfilesperproc")
|
||||
(sysctl-name "kern.osproductversion")
|
||||
(sysctl-name "kern.osrelease")
|
||||
(sysctl-name "kern.ostype")
|
||||
(sysctl-name "kern.osvariant_status")
|
||||
(sysctl-name "kern.osversion")
|
||||
(sysctl-name "kern.secure_kernel")
|
||||
(sysctl-name "kern.usrstack64")
|
||||
(sysctl-name "kern.version")
|
||||
(sysctl-name "sysctl.proc_cputype")
|
||||
(sysctl-name-prefix "hw.perflevel")
|
||||
)
|
||||
(allow file-write*
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
(subpath (param "CACHE_DIR"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
(subpath (param "INCLUDE_DIR_2"))
|
||||
(subpath (param "INCLUDE_DIR_3"))
|
||||
(subpath (param "INCLUDE_DIR_4"))
|
||||
(literal "/dev/stdout")
|
||||
(literal "/dev/stderr")
|
||||
(literal "/dev/null")
|
||||
(literal "/dev/ptmx")
|
||||
(regex #"^/dev/ttys[0-9]*$")
|
||||
)
|
||||
(allow mach-lookup
|
||||
(global-name "com.apple.sysmond")
|
||||
(global-name "com.apple.system.opendirectoryd.libinfo")
|
||||
(global-name "com.apple.system.opendirectoryd.membership")
|
||||
(global-name "com.apple.bsd.dirhelper")
|
||||
(global-name "com.apple.SecurityServer")
|
||||
(global-name "com.apple.networkd")
|
||||
(global-name "com.apple.ocspd")
|
||||
(global-name "com.apple.trustd")
|
||||
(global-name "com.apple.trustd.agent")
|
||||
(global-name "com.apple.mDNSResponder")
|
||||
(global-name "com.apple.mDNSResponderHelper")
|
||||
(global-name "com.apple.SystemConfiguration.DNSConfiguration")
|
||||
(global-name "com.apple.SystemConfiguration.configd")
|
||||
)
|
||||
(allow system-socket
|
||||
(require-all
|
||||
(socket-domain AF_SYSTEM)
|
||||
(socket-protocol 2)
|
||||
)
|
||||
)
|
||||
(allow file-ioctl (regex #"^/dev/tty.*"))
|
||||
(allow network-inbound (local ip "*:*"))
|
||||
(allow network-bind (local ip "*:*"))
|
||||
(allow network-outbound)`,
|
||||
|
||||
'permissive-proxied': `(version 1)
|
||||
(deny default)
|
||||
(allow file-read*)
|
||||
(allow process-exec)
|
||||
(allow process-fork)
|
||||
(allow signal (target self))
|
||||
(allow sysctl-read
|
||||
(sysctl-name "hw.activecpu")
|
||||
(sysctl-name "hw.busfrequency_compat")
|
||||
(sysctl-name "hw.byteorder")
|
||||
(sysctl-name "hw.cacheconfig")
|
||||
(sysctl-name "hw.cachelinesize_compat")
|
||||
(sysctl-name "hw.cpufamily")
|
||||
(sysctl-name "hw.cpufrequency_compat")
|
||||
(sysctl-name "hw.cputype")
|
||||
(sysctl-name "hw.l1dcachesize_compat")
|
||||
(sysctl-name "hw.l1icachesize_compat")
|
||||
(sysctl-name "hw.l2cachesize_compat")
|
||||
(sysctl-name "hw.l3cachesize_compat")
|
||||
(sysctl-name "hw.logicalcpu_max")
|
||||
(sysctl-name "hw.machine")
|
||||
(sysctl-name "hw.ncpu")
|
||||
(sysctl-name "hw.nperflevels")
|
||||
(sysctl-name "hw.optional.arm.FEAT_BF16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_DotProd")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FCMA")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FHM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FP16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_I8MM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_JSCVT")
|
||||
(sysctl-name "hw.optional.arm.FEAT_LSE")
|
||||
(sysctl-name "hw.optional.arm.FEAT_RDM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_SHA512")
|
||||
(sysctl-name "hw.optional.armv8_2_sha512")
|
||||
(sysctl-name "hw.packages")
|
||||
(sysctl-name "hw.pagesize_compat")
|
||||
(sysctl-name "hw.physicalcpu_max")
|
||||
(sysctl-name "hw.tbfrequency_compat")
|
||||
(sysctl-name "hw.vectorunit")
|
||||
(sysctl-name "kern.hostname")
|
||||
(sysctl-name "kern.maxfilesperproc")
|
||||
(sysctl-name "kern.osproductversion")
|
||||
(sysctl-name "kern.osrelease")
|
||||
(sysctl-name "kern.ostype")
|
||||
(sysctl-name "kern.osvariant_status")
|
||||
(sysctl-name "kern.osversion")
|
||||
(sysctl-name "kern.secure_kernel")
|
||||
(sysctl-name "kern.usrstack64")
|
||||
(sysctl-name "kern.version")
|
||||
(sysctl-name "sysctl.proc_cputype")
|
||||
(sysctl-name-prefix "hw.perflevel")
|
||||
)
|
||||
(allow file-write*
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
(subpath (param "CACHE_DIR"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
(subpath (param "INCLUDE_DIR_2"))
|
||||
(subpath (param "INCLUDE_DIR_3"))
|
||||
(subpath (param "INCLUDE_DIR_4"))
|
||||
(literal "/dev/stdout")
|
||||
(literal "/dev/stderr")
|
||||
(literal "/dev/null")
|
||||
(literal "/dev/ptmx")
|
||||
(regex #"^/dev/ttys[0-9]*$")
|
||||
)
|
||||
(allow mach-lookup
|
||||
(global-name "com.apple.sysmond")
|
||||
(global-name "com.apple.system.opendirectoryd.libinfo")
|
||||
(global-name "com.apple.system.opendirectoryd.membership")
|
||||
(global-name "com.apple.bsd.dirhelper")
|
||||
(global-name "com.apple.SecurityServer")
|
||||
(global-name "com.apple.networkd")
|
||||
(global-name "com.apple.ocspd")
|
||||
(global-name "com.apple.trustd")
|
||||
(global-name "com.apple.trustd.agent")
|
||||
(global-name "com.apple.mDNSResponder")
|
||||
(global-name "com.apple.mDNSResponderHelper")
|
||||
(global-name "com.apple.SystemConfiguration.DNSConfiguration")
|
||||
(global-name "com.apple.SystemConfiguration.configd")
|
||||
)
|
||||
(allow system-socket
|
||||
(require-all
|
||||
(socket-domain AF_SYSTEM)
|
||||
(socket-protocol 2)
|
||||
)
|
||||
)
|
||||
(allow file-ioctl (regex #"^/dev/tty.*"))
|
||||
(allow network-inbound (local ip "localhost:9229"))
|
||||
(allow network-bind (local ip "*:*"))
|
||||
(allow network-outbound (remote tcp "localhost:8877"))`,
|
||||
|
||||
'restrictive-open': `(version 1)
|
||||
(deny default)
|
||||
(allow file-read*)
|
||||
(allow process-exec)
|
||||
(allow process-fork)
|
||||
(allow signal (target self))
|
||||
(allow sysctl-read
|
||||
(sysctl-name "hw.activecpu")
|
||||
(sysctl-name "hw.busfrequency_compat")
|
||||
(sysctl-name "hw.byteorder")
|
||||
(sysctl-name "hw.cacheconfig")
|
||||
(sysctl-name "hw.cachelinesize_compat")
|
||||
(sysctl-name "hw.cpufamily")
|
||||
(sysctl-name "hw.cpufrequency_compat")
|
||||
(sysctl-name "hw.cputype")
|
||||
(sysctl-name "hw.l1dcachesize_compat")
|
||||
(sysctl-name "hw.l1icachesize_compat")
|
||||
(sysctl-name "hw.l2cachesize_compat")
|
||||
(sysctl-name "hw.l3cachesize_compat")
|
||||
(sysctl-name "hw.logicalcpu_max")
|
||||
(sysctl-name "hw.machine")
|
||||
(sysctl-name "hw.ncpu")
|
||||
(sysctl-name "hw.nperflevels")
|
||||
(sysctl-name "hw.optional.arm.FEAT_BF16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_DotProd")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FCMA")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FHM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FP16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_I8MM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_JSCVT")
|
||||
(sysctl-name "hw.optional.arm.FEAT_LSE")
|
||||
(sysctl-name "hw.optional.arm.FEAT_RDM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_SHA512")
|
||||
(sysctl-name "hw.optional.armv8_2_sha512")
|
||||
(sysctl-name "hw.packages")
|
||||
(sysctl-name "hw.pagesize_compat")
|
||||
(sysctl-name "hw.physicalcpu_max")
|
||||
(sysctl-name "hw.tbfrequency_compat")
|
||||
(sysctl-name "hw.vectorunit")
|
||||
(sysctl-name "kern.hostname")
|
||||
(sysctl-name "kern.maxfilesperproc")
|
||||
(sysctl-name "kern.osproductversion")
|
||||
(sysctl-name "kern.osrelease")
|
||||
(sysctl-name "kern.ostype")
|
||||
(sysctl-name "kern.osvariant_status")
|
||||
(sysctl-name "kern.osversion")
|
||||
(sysctl-name "kern.secure_kernel")
|
||||
(sysctl-name "kern.usrstack64")
|
||||
(sysctl-name "kern.version")
|
||||
(sysctl-name "sysctl.proc_cputype")
|
||||
(sysctl-name-prefix "hw.perflevel")
|
||||
)
|
||||
(allow file-write*
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
(subpath (param "CACHE_DIR"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
(subpath (param "INCLUDE_DIR_2"))
|
||||
(subpath (param "INCLUDE_DIR_3"))
|
||||
(subpath (param "INCLUDE_DIR_4"))
|
||||
(literal "/dev/stdout")
|
||||
(literal "/dev/stderr")
|
||||
(literal "/dev/null")
|
||||
)
|
||||
(allow mach-lookup (global-name "com.apple.sysmond"))
|
||||
(allow file-ioctl (regex #"^/dev/tty.*"))
|
||||
(allow network-inbound (local ip "localhost:9229"))
|
||||
(allow network-outbound)`,
|
||||
|
||||
'restrictive-proxied': `(version 1)
|
||||
(deny default)
|
||||
(allow file-read*)
|
||||
(allow process-exec)
|
||||
(allow process-fork)
|
||||
(allow signal (target self))
|
||||
(allow sysctl-read
|
||||
(sysctl-name "hw.activecpu")
|
||||
(sysctl-name "hw.busfrequency_compat")
|
||||
(sysctl-name "hw.byteorder")
|
||||
(sysctl-name "hw.cacheconfig")
|
||||
(sysctl-name "hw.cachelinesize_compat")
|
||||
(sysctl-name "hw.cpufamily")
|
||||
(sysctl-name "hw.cpufrequency_compat")
|
||||
(sysctl-name "hw.cputype")
|
||||
(sysctl-name "hw.l1dcachesize_compat")
|
||||
(sysctl-name "hw.l1icachesize_compat")
|
||||
(sysctl-name "hw.l2cachesize_compat")
|
||||
(sysctl-name "hw.l3cachesize_compat")
|
||||
(sysctl-name "hw.logicalcpu_max")
|
||||
(sysctl-name "hw.machine")
|
||||
(sysctl-name "hw.ncpu")
|
||||
(sysctl-name "hw.nperflevels")
|
||||
(sysctl-name "hw.optional.arm.FEAT_BF16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_DotProd")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FCMA")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FHM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FP16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_I8MM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_JSCVT")
|
||||
(sysctl-name "hw.optional.arm.FEAT_LSE")
|
||||
(sysctl-name "hw.optional.arm.FEAT_RDM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_SHA512")
|
||||
(sysctl-name "hw.optional.armv8_2_sha512")
|
||||
(sysctl-name "hw.packages")
|
||||
(sysctl-name "hw.pagesize_compat")
|
||||
(sysctl-name "hw.physicalcpu_max")
|
||||
(sysctl-name "hw.tbfrequency_compat")
|
||||
(sysctl-name "hw.vectorunit")
|
||||
(sysctl-name "kern.hostname")
|
||||
(sysctl-name "kern.maxfilesperproc")
|
||||
(sysctl-name "kern.osproductversion")
|
||||
(sysctl-name "kern.osrelease")
|
||||
(sysctl-name "kern.ostype")
|
||||
(sysctl-name "kern.osvariant_status")
|
||||
(sysctl-name "kern.osversion")
|
||||
(sysctl-name "kern.secure_kernel")
|
||||
(sysctl-name "kern.usrstack64")
|
||||
(sysctl-name "kern.version")
|
||||
(sysctl-name "sysctl.proc_cputype")
|
||||
(sysctl-name-prefix "hw.perflevel")
|
||||
)
|
||||
(allow file-write*
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
(subpath (param "CACHE_DIR"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
(subpath (param "INCLUDE_DIR_2"))
|
||||
(subpath (param "INCLUDE_DIR_3"))
|
||||
(subpath (param "INCLUDE_DIR_4"))
|
||||
(literal "/dev/stdout")
|
||||
(literal "/dev/stderr")
|
||||
(literal "/dev/null")
|
||||
)
|
||||
(allow mach-lookup (global-name "com.apple.sysmond"))
|
||||
(allow file-ioctl (regex #"^/dev/tty.*"))
|
||||
(allow network-inbound (local ip "localhost:9229"))
|
||||
(allow network-outbound (remote tcp "localhost:8877"))`,
|
||||
|
||||
'strict-open': `(version 1)
|
||||
(deny default)
|
||||
(allow file-read*
|
||||
(literal "/")
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
(subpath (param "CACHE_DIR"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(literal (string-append (param "HOME_DIR") "/.gitconfig"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.nvm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.fnm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.node"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.config"))
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
(subpath (param "INCLUDE_DIR_2"))
|
||||
(subpath (param "INCLUDE_DIR_3"))
|
||||
(subpath (param "INCLUDE_DIR_4"))
|
||||
(subpath "/usr")
|
||||
(subpath "/bin")
|
||||
(subpath "/sbin")
|
||||
(subpath "/Library")
|
||||
(subpath "/System")
|
||||
(subpath "/private")
|
||||
(subpath "/dev")
|
||||
(subpath "/etc")
|
||||
(subpath "/opt")
|
||||
(subpath "/Applications")
|
||||
)
|
||||
(allow file-read-metadata)
|
||||
(allow process-exec)
|
||||
(allow process-fork)
|
||||
(allow signal (target self))
|
||||
(allow sysctl-read
|
||||
(sysctl-name "hw.activecpu")
|
||||
(sysctl-name "hw.busfrequency_compat")
|
||||
(sysctl-name "hw.byteorder")
|
||||
(sysctl-name "hw.cacheconfig")
|
||||
(sysctl-name "hw.cachelinesize_compat")
|
||||
(sysctl-name "hw.cpufamily")
|
||||
(sysctl-name "hw.cpufrequency_compat")
|
||||
(sysctl-name "hw.cputype")
|
||||
(sysctl-name "hw.l1dcachesize_compat")
|
||||
(sysctl-name "hw.l1icachesize_compat")
|
||||
(sysctl-name "hw.l2cachesize_compat")
|
||||
(sysctl-name "hw.l3cachesize_compat")
|
||||
(sysctl-name "hw.logicalcpu_max")
|
||||
(sysctl-name "hw.machine")
|
||||
(sysctl-name "hw.ncpu")
|
||||
(sysctl-name "hw.nperflevels")
|
||||
(sysctl-name "hw.optional.arm.FEAT_BF16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_DotProd")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FCMA")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FHM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FP16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_I8MM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_JSCVT")
|
||||
(sysctl-name "hw.optional.arm.FEAT_LSE")
|
||||
(sysctl-name "hw.optional.arm.FEAT_RDM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_SHA512")
|
||||
(sysctl-name "hw.optional.armv8_2_sha512")
|
||||
(sysctl-name "hw.packages")
|
||||
(sysctl-name "hw.pagesize_compat")
|
||||
(sysctl-name "hw.physicalcpu_max")
|
||||
(sysctl-name "hw.tbfrequency_compat")
|
||||
(sysctl-name "hw.vectorunit")
|
||||
(sysctl-name "kern.hostname")
|
||||
(sysctl-name "kern.maxfilesperproc")
|
||||
(sysctl-name "kern.osproductversion")
|
||||
(sysctl-name "kern.osrelease")
|
||||
(sysctl-name "kern.ostype")
|
||||
(sysctl-name "kern.osvariant_status")
|
||||
(sysctl-name "kern.osversion")
|
||||
(sysctl-name "kern.secure_kernel")
|
||||
(sysctl-name "kern.usrstack64")
|
||||
(sysctl-name "kern.version")
|
||||
(sysctl-name "sysctl.proc_cputype")
|
||||
(sysctl-name-prefix "hw.perflevel")
|
||||
)
|
||||
(allow file-write*
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
(subpath (param "CACHE_DIR"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
(subpath (param "INCLUDE_DIR_2"))
|
||||
(subpath (param "INCLUDE_DIR_3"))
|
||||
(subpath (param "INCLUDE_DIR_4"))
|
||||
(literal "/dev/stdout")
|
||||
(literal "/dev/stderr")
|
||||
(literal "/dev/null")
|
||||
)
|
||||
(allow mach-lookup (global-name "com.apple.sysmond"))
|
||||
(allow file-ioctl (regex #"^/dev/tty.*"))
|
||||
(allow network-inbound (local ip "localhost:9229"))
|
||||
(allow network-outbound)`,
|
||||
|
||||
'strict-proxied': `(version 1)
|
||||
(deny default)
|
||||
(allow file-read*
|
||||
(literal "/")
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
(subpath (param "CACHE_DIR"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(literal (string-append (param "HOME_DIR") "/.gitconfig"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.nvm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.fnm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.node"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.config"))
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
(subpath (param "INCLUDE_DIR_2"))
|
||||
(subpath (param "INCLUDE_DIR_3"))
|
||||
(subpath (param "INCLUDE_DIR_4"))
|
||||
(subpath "/usr")
|
||||
(subpath "/bin")
|
||||
(subpath "/sbin")
|
||||
(subpath "/Library")
|
||||
(subpath "/System")
|
||||
(subpath "/private")
|
||||
(subpath "/dev")
|
||||
(subpath "/etc")
|
||||
(subpath "/opt")
|
||||
(subpath "/Applications")
|
||||
)
|
||||
(allow file-read-metadata)
|
||||
(allow process-exec)
|
||||
(allow process-fork)
|
||||
(allow signal (target self))
|
||||
(allow sysctl-read
|
||||
(sysctl-name "hw.activecpu")
|
||||
(sysctl-name "hw.busfrequency_compat")
|
||||
(sysctl-name "hw.byteorder")
|
||||
(sysctl-name "hw.cacheconfig")
|
||||
(sysctl-name "hw.cachelinesize_compat")
|
||||
(sysctl-name "hw.cpufamily")
|
||||
(sysctl-name "hw.cpufrequency_compat")
|
||||
(sysctl-name "hw.cputype")
|
||||
(sysctl-name "hw.l1dcachesize_compat")
|
||||
(sysctl-name "hw.l1icachesize_compat")
|
||||
(sysctl-name "hw.l2cachesize_compat")
|
||||
(sysctl-name "hw.l3cachesize_compat")
|
||||
(sysctl-name "hw.logicalcpu_max")
|
||||
(sysctl-name "hw.machine")
|
||||
(sysctl-name "hw.ncpu")
|
||||
(sysctl-name "hw.nperflevels")
|
||||
(sysctl-name "hw.optional.arm.FEAT_BF16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_DotProd")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FCMA")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FHM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_FP16")
|
||||
(sysctl-name "hw.optional.arm.FEAT_I8MM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_JSCVT")
|
||||
(sysctl-name "hw.optional.arm.FEAT_LSE")
|
||||
(sysctl-name "hw.optional.arm.FEAT_RDM")
|
||||
(sysctl-name "hw.optional.arm.FEAT_SHA512")
|
||||
(sysctl-name "hw.optional.armv8_2_sha512")
|
||||
(sysctl-name "hw.packages")
|
||||
(sysctl-name "hw.pagesize_compat")
|
||||
(sysctl-name "hw.physicalcpu_max")
|
||||
(sysctl-name "hw.tbfrequency_compat")
|
||||
(sysctl-name "hw.vectorunit")
|
||||
(sysctl-name "kern.hostname")
|
||||
(sysctl-name "kern.maxfilesperproc")
|
||||
(sysctl-name "kern.osproductversion")
|
||||
(sysctl-name "kern.osrelease")
|
||||
(sysctl-name "kern.ostype")
|
||||
(sysctl-name "kern.osvariant_status")
|
||||
(sysctl-name "kern.osversion")
|
||||
(sysctl-name "kern.secure_kernel")
|
||||
(sysctl-name "kern.usrstack64")
|
||||
(sysctl-name "kern.version")
|
||||
(sysctl-name "sysctl.proc_cputype")
|
||||
(sysctl-name-prefix "hw.perflevel")
|
||||
)
|
||||
(allow file-write*
|
||||
(subpath (param "TARGET_DIR"))
|
||||
(subpath (param "TMP_DIR"))
|
||||
(subpath (param "CACHE_DIR"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.gemini"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.npm"))
|
||||
(subpath (string-append (param "HOME_DIR") "/.cache"))
|
||||
(subpath (param "INCLUDE_DIR_0"))
|
||||
(subpath (param "INCLUDE_DIR_1"))
|
||||
(subpath (param "INCLUDE_DIR_2"))
|
||||
(subpath (param "INCLUDE_DIR_3"))
|
||||
(subpath (param "INCLUDE_DIR_4"))
|
||||
(literal "/dev/stdout")
|
||||
(literal "/dev/stderr")
|
||||
(literal "/dev/null")
|
||||
)
|
||||
(allow mach-lookup (global-name "com.apple.sysmond"))
|
||||
(allow file-ioctl (regex #"^/dev/tty.*"))
|
||||
(allow network-inbound (local ip "localhost:9229"))
|
||||
(allow network-outbound (remote tcp "localhost:8877"))`,
|
||||
};
|
||||
|
||||
// Map standard 'closed' profiles to their strict counterparts for backward compatibility and fallback support
|
||||
BUILTIN_SEATBELT_PROFILE_CONTENTS['permissive-closed'] =
|
||||
BUILTIN_SEATBELT_PROFILE_CONTENTS['strict-open'];
|
||||
BUILTIN_SEATBELT_PROFILE_CONTENTS['restrictive-closed'] =
|
||||
BUILTIN_SEATBELT_PROFILE_CONTENTS['strict-proxied'];
|
||||
@@ -15,8 +15,10 @@ export const SANDBOX_NETWORK_NAME = 'gemini-cli-sandbox';
|
||||
export const SANDBOX_PROXY_NAME = 'gemini-cli-sandbox-proxy';
|
||||
export const BUILTIN_SEATBELT_PROFILES = [
|
||||
'permissive-open',
|
||||
'permissive-closed',
|
||||
'permissive-proxied',
|
||||
'restrictive-open',
|
||||
'restrictive-closed',
|
||||
'restrictive-proxied',
|
||||
'strict-open',
|
||||
'strict-proxied',
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@google/gemini-cli-core",
|
||||
"version": "0.51.0-nightly.20260625.g3fbf93e26",
|
||||
"version": "0.56.0-nightly.20260806.g761f604c1",
|
||||
"description": "Gemini CLI Core",
|
||||
"license": "Apache-2.0",
|
||||
"repository": {
|
||||
@@ -64,7 +64,7 @@
|
||||
"fdir": "6.4.6",
|
||||
"fzf": "0.5.2",
|
||||
"glob": "12.0.0",
|
||||
"google-auth-library": "9.11.0",
|
||||
"google-auth-library": "10.9.0",
|
||||
"html-to-text": "9.0.5",
|
||||
"http-proxy-agent": "7.0.2",
|
||||
"https-proxy-agent": "7.0.6",
|
||||
|
||||
@@ -516,10 +516,34 @@ describe('translateEvent', () => {
|
||||
});
|
||||
|
||||
describe('InvalidStream events', () => {
|
||||
it('emits fatal error', () => {
|
||||
it('emits fatal error with specific message from event', () => {
|
||||
state.streamStartEmitted = true;
|
||||
const event: ServerGeminiStreamEvent = {
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_RESPONSE_TEXT',
|
||||
message: 'Empty response',
|
||||
},
|
||||
};
|
||||
const result = translateEvent(event, state);
|
||||
expect(result).toHaveLength(1);
|
||||
const err = result[0] as AgentEvent<'error'>;
|
||||
expect(err.status).toBe('INTERNAL');
|
||||
expect(err.message).toBe('Empty response');
|
||||
expect(err.fatal).toBe(true);
|
||||
expect(err._meta?.['code']).toBe('INVALID_STREAM');
|
||||
expect(err._meta?.['errorType']).toBe('NO_RESPONSE_TEXT');
|
||||
expect(err._meta?.['rawMessage']).toBe('Empty response');
|
||||
});
|
||||
|
||||
it('falls back to default message when message is missing', () => {
|
||||
state.streamStartEmitted = true;
|
||||
const event: ServerGeminiStreamEvent = {
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_RESPONSE_TEXT',
|
||||
message: '',
|
||||
},
|
||||
};
|
||||
const result = translateEvent(event, state);
|
||||
expect(result).toHaveLength(1);
|
||||
|
||||
@@ -222,8 +222,15 @@ export function translateEvent(
|
||||
out.push(
|
||||
makeEvent('error', state, {
|
||||
status: 'INTERNAL',
|
||||
message: 'Invalid stream received from model',
|
||||
message:
|
||||
event.value?.message?.trim() ||
|
||||
'Invalid stream received from model',
|
||||
fatal: true,
|
||||
_meta: {
|
||||
code: 'INVALID_STREAM',
|
||||
errorType: event.value?.type,
|
||||
rawMessage: event.value?.message,
|
||||
},
|
||||
}),
|
||||
);
|
||||
break;
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
*/
|
||||
|
||||
import { GeminiEventType } from '../core/turn.js';
|
||||
import type { Part } from '@google/genai';
|
||||
import type { Part, FinishReason } from '@google/genai';
|
||||
import type { GeminiClient } from '../core/client.js';
|
||||
import type { Config } from '../config/config.js';
|
||||
import type { ToolCallRequestInfo } from '../scheduler/types.js';
|
||||
@@ -192,6 +192,7 @@ export class LegacyAgentProtocol implements AgentProtocol {
|
||||
}
|
||||
|
||||
const toolCallRequests: ToolCallRequestInfo[] = [];
|
||||
let finishedReason: FinishReason | undefined = undefined;
|
||||
const responseStream = this._client.sendMessageStream(
|
||||
currentParts,
|
||||
this._abortController.signal,
|
||||
@@ -220,10 +221,7 @@ export class LegacyAgentProtocol implements AgentProtocol {
|
||||
this._finishStream('failed');
|
||||
return;
|
||||
case GeminiEventType.Finished:
|
||||
if (toolCallRequests.length === 0) {
|
||||
this._finishStream(mapFinishReason(event.value.reason));
|
||||
return;
|
||||
}
|
||||
finishedReason = event.value.reason;
|
||||
break;
|
||||
case GeminiEventType.AgentExecutionStopped:
|
||||
case GeminiEventType.UserCancelled:
|
||||
@@ -241,7 +239,11 @@ export class LegacyAgentProtocol implements AgentProtocol {
|
||||
}
|
||||
|
||||
if (toolCallRequests.length === 0) {
|
||||
this._finishStream('completed');
|
||||
if (finishedReason !== undefined) {
|
||||
this._finishStream(mapFinishReason(finishedReason));
|
||||
} else {
|
||||
this._finishStream('completed');
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
/**
|
||||
* @license
|
||||
* Copyright 2026 Google LLC
|
||||
* SPDX-License-Identifier: Apache-2.0
|
||||
*/
|
||||
|
||||
import { describe, it, expect, vi, beforeEach, type Mock } from 'vitest';
|
||||
import { GoogleCredentialsAuthProvider } from './google-credentials-provider.js';
|
||||
import type { GoogleCredentialsAuthConfig } from './types.js';
|
||||
import { GoogleAuth } from 'google-auth-library';
|
||||
|
||||
vi.mock('google-auth-library', () => ({
|
||||
GoogleAuth: vi.fn(),
|
||||
}));
|
||||
|
||||
describe('Credential Leak Prevention (RCA / PoC Verification)', () => {
|
||||
const mockConfig: GoogleCredentialsAuthConfig = {
|
||||
type: 'google-credentials',
|
||||
};
|
||||
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
(GoogleAuth as unknown as Mock).mockImplementation(() => ({
|
||||
getClient: vi.fn().mockResolvedValue({
|
||||
getAccessToken: vi.fn().mockResolvedValue({ token: 'leaked-token' }),
|
||||
credentials: { expiry_date: Date.now() + 3600 * 1000 },
|
||||
}),
|
||||
getIdTokenClient: vi.fn().mockResolvedValue({
|
||||
idTokenProvider: {
|
||||
fetchIdToken: vi.fn().mockResolvedValue('leaked-id-token'),
|
||||
},
|
||||
}),
|
||||
}));
|
||||
});
|
||||
|
||||
it('should FAIL (throw error) when trying to initialize with an untrusted arbitrary remote agent URL (reproducing vulnerability prevention)', () => {
|
||||
// This test simulates the reproduction scenario: registering a remote agent with an arbitrary external URL
|
||||
// e.g., http://127.0.0.1:1337 or https://malicious-agent.evil.com
|
||||
const untrustedUrls = [
|
||||
{
|
||||
url: 'http://127.0.0.1:1337/.well-known/agent.json',
|
||||
error: /requires HTTPS/,
|
||||
},
|
||||
{
|
||||
url: 'https://malicious-agent.evil.com/card',
|
||||
error: /is not an allowed host/,
|
||||
},
|
||||
{
|
||||
url: 'https://untrusted-third-party.com/agent',
|
||||
error: /is not an allowed host/,
|
||||
},
|
||||
];
|
||||
|
||||
for (const item of untrustedUrls) {
|
||||
expect(() => {
|
||||
new GoogleCredentialsAuthProvider(mockConfig, item.url);
|
||||
}).toThrow(item.error);
|
||||
}
|
||||
});
|
||||
|
||||
it('should SUCCEED only for allowed Google Services (proving the allowlist constraint)', () => {
|
||||
const trustedUrls = [
|
||||
'https://language.googleapis.com/v1/models',
|
||||
'https://vertex-ai-agent.googleapis.com/agent',
|
||||
'https://my-secure-service-abc.run.app/card',
|
||||
];
|
||||
|
||||
for (const url of trustedUrls) {
|
||||
expect(() => {
|
||||
new GoogleCredentialsAuthProvider(mockConfig, url);
|
||||
}).not.toThrow();
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -82,6 +82,24 @@ describe('GoogleCredentialsAuthProvider', () => {
|
||||
),
|
||||
).not.toThrow();
|
||||
});
|
||||
|
||||
it('throws if the protocol is not HTTPS', () => {
|
||||
expect(
|
||||
() =>
|
||||
new GoogleCredentialsAuthProvider(
|
||||
mockConfig,
|
||||
'http://language.googleapis.com/v1/models',
|
||||
),
|
||||
).toThrow(/requires HTTPS/);
|
||||
|
||||
expect(
|
||||
() =>
|
||||
new GoogleCredentialsAuthProvider(
|
||||
mockConfig,
|
||||
'http://my-cloud-run-service.run.app',
|
||||
),
|
||||
).toThrow(/requires HTTPS/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Token Fetching', () => {
|
||||
|
||||
@@ -40,7 +40,14 @@ export class GoogleCredentialsAuthProvider extends BaseA2AAuthProvider {
|
||||
);
|
||||
}
|
||||
|
||||
const hostname = new URL(targetUrl).hostname;
|
||||
const urlObj = new URL(targetUrl);
|
||||
if (urlObj.protocol !== 'https:') {
|
||||
throw new Error(
|
||||
`Protocol "${urlObj.protocol}" is not secure. Google Credential provider requires HTTPS.`,
|
||||
);
|
||||
}
|
||||
|
||||
const hostname = urlObj.hostname;
|
||||
const isRunAppHost = CLOUD_RUN_HOST_REGEX.test(hostname);
|
||||
|
||||
if (isRunAppHost) {
|
||||
|
||||
@@ -414,4 +414,86 @@ describe('Auto Routing Fallback Integration', () => {
|
||||
'Pro success',
|
||||
);
|
||||
});
|
||||
|
||||
it('should rotate session ID on fallback and retry successfully with the Flash model', async () => {
|
||||
const originalSessionId = 'test-session-rotate-id';
|
||||
config = new Config({
|
||||
sessionId: originalSessionId,
|
||||
targetDir: '/test',
|
||||
debugMode: false,
|
||||
cwd: '/test',
|
||||
model: PREVIEW_GEMINI_MODEL_AUTO,
|
||||
});
|
||||
|
||||
vi.spyOn(config, 'isInteractive').mockReturnValue(true);
|
||||
|
||||
client = new BaseLlmClient(
|
||||
fakeGenerator,
|
||||
config,
|
||||
AuthType.LOGIN_WITH_GOOGLE,
|
||||
);
|
||||
|
||||
let attemptsPro = 0;
|
||||
let attemptsFlash = 0;
|
||||
|
||||
const mockGoogleApiError = {
|
||||
code: 429,
|
||||
message:
|
||||
'Automatically switching from gemini-2.5-pro to gemini-2.5-flash for faster responses for the remainder of this session. Possible reasons for this are...',
|
||||
details: [],
|
||||
};
|
||||
|
||||
vi.spyOn(fakeGenerator, 'generateContent').mockImplementation(
|
||||
async (params) => {
|
||||
if (params.model === PREVIEW_GEMINI_MODEL) {
|
||||
attemptsPro++;
|
||||
throw new RetryableQuotaError(
|
||||
'Quota exceeded for Pro',
|
||||
mockGoogleApiError,
|
||||
0,
|
||||
);
|
||||
} else if (params.model === PREVIEW_GEMINI_FLASH_MODEL) {
|
||||
attemptsFlash++;
|
||||
return {
|
||||
candidates: [
|
||||
{
|
||||
content: {
|
||||
role: 'model',
|
||||
parts: [{ text: 'Flash success after rotation' }],
|
||||
},
|
||||
},
|
||||
],
|
||||
} as unknown as GenerateContentResponse;
|
||||
}
|
||||
throw new Error(`Unexpected model: ${params.model}`);
|
||||
},
|
||||
);
|
||||
|
||||
config.setFallbackModelHandler(
|
||||
async (_failed, _fallback, _error): Promise<FallbackIntent | null> =>
|
||||
'retry_always', // Approve switch to Flash
|
||||
);
|
||||
|
||||
const promise = client.generateContent({
|
||||
modelConfigKey: { model: PREVIEW_GEMINI_MODEL, isChatModel: true },
|
||||
contents: [{ role: 'user', parts: [{ text: 'test query' }] }],
|
||||
abortSignal: new AbortController().signal,
|
||||
promptId: 'test-prompt',
|
||||
role: LlmRole.UTILITY_TOOL,
|
||||
});
|
||||
|
||||
await vi.runAllTimersAsync();
|
||||
const result = await promise;
|
||||
|
||||
// Verify it resolved to Flash success instead of failing with Please submit a new query
|
||||
expect(result.candidates?.[0]?.content?.parts?.[0]?.text).toBe(
|
||||
'Flash success after rotation',
|
||||
);
|
||||
expect(attemptsPro).toBe(3);
|
||||
expect(attemptsFlash).toBe(1);
|
||||
|
||||
// Verify session ID has been rotated
|
||||
expect(config.getSessionId()).not.toBe(originalSessionId);
|
||||
expect(config.getSessionId()).toBeDefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -680,6 +680,64 @@ describe('oauth2', () => {
|
||||
expect(mockFromJSON).toHaveBeenCalledWith(byoidCredentials);
|
||||
expect(client).toBe(mockExternalAccountClient);
|
||||
});
|
||||
|
||||
it('should fall back to GOOGLE_APPLICATION_CREDENTIALS if default cached credentials are invalid or expired', async () => {
|
||||
// Setup default cached credentials that are expired/invalid
|
||||
const defaultCreds = { refresh_token: 'expired-token' };
|
||||
const defaultCredsPath = path.join(
|
||||
tempHomeDir,
|
||||
GEMINI_DIR,
|
||||
'oauth_creds.json',
|
||||
);
|
||||
await fs.promises.mkdir(path.dirname(defaultCredsPath), {
|
||||
recursive: true,
|
||||
});
|
||||
await fs.promises.writeFile(
|
||||
defaultCredsPath,
|
||||
JSON.stringify(defaultCreds),
|
||||
);
|
||||
|
||||
// Setup valid fallback credentials via environment variable
|
||||
const envCreds = { refresh_token: 'valid-env-token' };
|
||||
const envCredsPath = path.join(tempHomeDir, 'env_creds.json');
|
||||
await fs.promises.writeFile(envCredsPath, JSON.stringify(envCreds));
|
||||
vi.stubEnv('GOOGLE_APPLICATION_CREDENTIALS', envCredsPath);
|
||||
|
||||
let currentCredentials: Credentials | null = null;
|
||||
const mockClient = {
|
||||
setCredentials: vi.fn((creds) => {
|
||||
currentCredentials = creds as Credentials;
|
||||
}),
|
||||
getAccessToken: vi.fn(async () => {
|
||||
if (
|
||||
currentCredentials &&
|
||||
currentCredentials.refresh_token === 'expired-token'
|
||||
) {
|
||||
throw new Error('Token is expired or revoked');
|
||||
}
|
||||
return { token: 'valid-token' };
|
||||
}),
|
||||
getTokenInfo: vi.fn(async (_token) => {
|
||||
if (
|
||||
currentCredentials &&
|
||||
currentCredentials.refresh_token === 'expired-token'
|
||||
) {
|
||||
throw new Error('Token is expired or revoked');
|
||||
}
|
||||
return {};
|
||||
}),
|
||||
on: vi.fn(),
|
||||
};
|
||||
|
||||
vi.mocked(OAuth2Client).mockImplementation(
|
||||
() => mockClient as unknown as OAuth2Client,
|
||||
);
|
||||
|
||||
await getOauthClient(AuthType.LOGIN_WITH_GOOGLE, mockConfig);
|
||||
|
||||
// Assert that fallback envCreds were eventually loaded and used
|
||||
expect(mockClient.setCredentials).toHaveBeenCalledWith(envCreds);
|
||||
});
|
||||
});
|
||||
|
||||
describe('with GCP environment variables', () => {
|
||||
|
||||
@@ -113,46 +113,52 @@ function getUseEncryptedStorageFlag() {
|
||||
return process.env[FORCE_ENCRYPTED_FILE_ENV_VAR] === 'true';
|
||||
}
|
||||
|
||||
/**
|
||||
* Determines whether the given credentials object represents ADC credentials.
|
||||
*/
|
||||
function isAdcCredentials(
|
||||
credentials: unknown,
|
||||
): credentials is JWTInput & { type: string } {
|
||||
if (credentials && typeof credentials === 'object' && 'type' in credentials) {
|
||||
const type = credentials.type;
|
||||
return typeof type === 'string' && type !== 'authorized_user';
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
async function initOauthClient(
|
||||
authType: AuthType,
|
||||
config: Config,
|
||||
): Promise<AuthClient> {
|
||||
const credentials = await fetchCachedCredentials();
|
||||
function createBaseOAuth2Client(): OAuth2Client {
|
||||
const client = new OAuth2Client({
|
||||
clientId: OAUTH_CLIENT_ID,
|
||||
clientSecret: OAUTH_CLIENT_SECRET,
|
||||
transporterOptions: {
|
||||
proxy: config.getProxy(),
|
||||
},
|
||||
});
|
||||
const useEncryptedStorage = getUseEncryptedStorageFlag();
|
||||
|
||||
if (
|
||||
credentials &&
|
||||
typeof credentials === 'object' &&
|
||||
'type' in credentials &&
|
||||
(credentials.type === 'external_account_authorized_user' ||
|
||||
credentials.type === 'service_account')
|
||||
) {
|
||||
const auth = new GoogleAuth({
|
||||
scopes: OAUTH_SCOPE,
|
||||
client.on('tokens', async (tokens: Credentials) => {
|
||||
if (useEncryptedStorage) {
|
||||
await OAuthCredentialStorage.saveCredentials(tokens);
|
||||
} else {
|
||||
await cacheCredentials(tokens);
|
||||
}
|
||||
|
||||
await triggerPostAuthCallbacks(tokens);
|
||||
});
|
||||
const byoidClient = auth.fromJSON({
|
||||
...credentials,
|
||||
refresh_token: credentials.refresh_token ?? undefined,
|
||||
});
|
||||
const token = await byoidClient.getAccessToken();
|
||||
if (token) {
|
||||
debugLogger.debug(`Created ${credentials.type} auth client.`);
|
||||
return byoidClient;
|
||||
}
|
||||
|
||||
return client;
|
||||
}
|
||||
|
||||
const client = new OAuth2Client({
|
||||
clientId: OAUTH_CLIENT_ID,
|
||||
clientSecret: OAUTH_CLIENT_SECRET,
|
||||
transporterOptions: {
|
||||
proxy: config.getProxy(),
|
||||
},
|
||||
});
|
||||
const useEncryptedStorage = getUseEncryptedStorageFlag();
|
||||
|
||||
// 1. Try GOOGLE_CLOUD_ACCESS_TOKEN override first if configured
|
||||
if (
|
||||
process.env['GOOGLE_GENAI_USE_GCA'] &&
|
||||
process.env['GOOGLE_CLOUD_ACCESS_TOKEN']
|
||||
) {
|
||||
const client = createBaseOAuth2Client();
|
||||
client.setCredentials({
|
||||
access_token: process.env['GOOGLE_CLOUD_ACCESS_TOKEN'],
|
||||
});
|
||||
@@ -160,49 +166,70 @@ async function initOauthClient(
|
||||
return client;
|
||||
}
|
||||
|
||||
client.on('tokens', async (tokens: Credentials) => {
|
||||
if (useEncryptedStorage) {
|
||||
await OAuthCredentialStorage.saveCredentials(tokens);
|
||||
} else {
|
||||
await cacheCredentials(tokens);
|
||||
}
|
||||
const credentialsList = await fetchCachedCredentialsList();
|
||||
|
||||
await triggerPostAuthCallbacks(tokens);
|
||||
});
|
||||
|
||||
if (credentials) {
|
||||
client.setCredentials(credentials as Credentials);
|
||||
try {
|
||||
// This will verify locally that the credentials look good.
|
||||
const { token } = await client.getAccessToken();
|
||||
if (token) {
|
||||
// This will check with the server to see if it hasn't been revoked.
|
||||
await client.getTokenInfo(token);
|
||||
|
||||
if (!userAccountManager.getCachedGoogleAccount()) {
|
||||
try {
|
||||
await fetchAndCacheUserInfo(client);
|
||||
} catch (error) {
|
||||
// Non-fatal, continue with existing auth.
|
||||
debugLogger.warn(
|
||||
'Failed to fetch user info:',
|
||||
getErrorMessage(error),
|
||||
);
|
||||
}
|
||||
// 2. Iterate sequentially over the credentials list in their natural priority order
|
||||
for (const credentials of credentialsList) {
|
||||
if (isAdcCredentials(credentials)) {
|
||||
try {
|
||||
const auth = new GoogleAuth({
|
||||
scopes: OAUTH_SCOPE,
|
||||
});
|
||||
const adcClient = auth.fromJSON({
|
||||
...credentials,
|
||||
refresh_token: credentials.refresh_token ?? undefined,
|
||||
});
|
||||
const response = await adcClient.getAccessToken();
|
||||
const token = response.token ?? null;
|
||||
if (token) {
|
||||
debugLogger.debug('Created ' + credentials.type + ' auth client.');
|
||||
return adcClient;
|
||||
}
|
||||
debugLogger.log('Loaded cached credentials.');
|
||||
await triggerPostAuthCallbacks(credentials as Credentials);
|
||||
|
||||
return client;
|
||||
} catch (error) {
|
||||
debugLogger.debug(
|
||||
'ADC credentials verification failed:',
|
||||
getErrorMessage(error),
|
||||
);
|
||||
}
|
||||
} else if (credentials) {
|
||||
const client = createBaseOAuth2Client();
|
||||
client.setCredentials(credentials as Credentials);
|
||||
try {
|
||||
// This will verify locally that the credentials look good.
|
||||
const { token } = await client.getAccessToken();
|
||||
if (token) {
|
||||
// This will check with the server to see if it hasn't been revoked.
|
||||
await client.getTokenInfo(token);
|
||||
|
||||
if (!userAccountManager.getCachedGoogleAccount()) {
|
||||
try {
|
||||
await fetchAndCacheUserInfo(client);
|
||||
} catch (error) {
|
||||
// Non-fatal, continue with existing auth.
|
||||
debugLogger.warn(
|
||||
'Failed to fetch user info:',
|
||||
getErrorMessage(error),
|
||||
);
|
||||
}
|
||||
}
|
||||
debugLogger.log('Loaded cached credentials.');
|
||||
await triggerPostAuthCallbacks(
|
||||
client.credentials || (credentials as Credentials),
|
||||
);
|
||||
|
||||
return client;
|
||||
}
|
||||
} catch (error) {
|
||||
debugLogger.debug(
|
||||
'Cached credentials are not valid:',
|
||||
getErrorMessage(error),
|
||||
);
|
||||
}
|
||||
} catch (error) {
|
||||
debugLogger.debug(
|
||||
`Cached credentials are not valid:`,
|
||||
getErrorMessage(error),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const client = createBaseOAuth2Client();
|
||||
|
||||
// In Google Compute Engine based environments (including Cloud Shell), we can
|
||||
// use Application Default Credentials (ADC) provided via its metadata server
|
||||
// to authenticate non-interactively using the identity of the logged-in user.
|
||||
@@ -663,16 +690,27 @@ export function getAvailablePort(): Promise<number> {
|
||||
});
|
||||
}
|
||||
|
||||
async function fetchCachedCredentials(): Promise<
|
||||
Credentials | JWTInput | null
|
||||
async function fetchCachedCredentialsList(): Promise<
|
||||
Array<Credentials | JWTInput>
|
||||
> {
|
||||
const credentialsList: Array<Credentials | JWTInput> = [];
|
||||
const useEncryptedStorage = getUseEncryptedStorageFlag();
|
||||
if (useEncryptedStorage) {
|
||||
return OAuthCredentialStorage.loadCredentials();
|
||||
try {
|
||||
const creds = await OAuthCredentialStorage.loadCredentials();
|
||||
if (creds) {
|
||||
credentialsList.push(creds);
|
||||
}
|
||||
} catch (error) {
|
||||
debugLogger.debug(
|
||||
'Failed to load credentials from encrypted storage:',
|
||||
error,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const pathsToTry = [
|
||||
Storage.getOAuthCredsPath(),
|
||||
...(!useEncryptedStorage ? [Storage.getOAuthCredsPath()] : []),
|
||||
process.env['GOOGLE_APPLICATION_CREDENTIALS'],
|
||||
].filter((p): p is string => !!p);
|
||||
|
||||
@@ -683,9 +721,10 @@ async function fetchCachedCredentials(): Promise<
|
||||
const isOAuthCreds = (val: unknown): val is Credentials | JWTInput =>
|
||||
typeof val === 'object' && val !== null;
|
||||
if (isOAuthCreds(parsed)) {
|
||||
return parsed;
|
||||
credentialsList.push(parsed);
|
||||
} else {
|
||||
throw new Error('Invalid credentials format');
|
||||
}
|
||||
throw new Error('Invalid credentials format');
|
||||
} catch (error) {
|
||||
// Log specific error for debugging, but continue trying other paths
|
||||
debugLogger.debug(
|
||||
@@ -695,7 +734,7 @@ async function fetchCachedCredentials(): Promise<
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
return credentialsList;
|
||||
}
|
||||
|
||||
export function clearOauthClientCache() {
|
||||
|
||||
@@ -86,6 +86,10 @@ export class CodeAssistServer implements ContentGenerator {
|
||||
readonly config?: Config,
|
||||
) {}
|
||||
|
||||
getEffectiveSessionId(): string | undefined {
|
||||
return this.config?.getSessionId() ?? this.sessionId;
|
||||
}
|
||||
|
||||
async generateContentStream(
|
||||
req: GenerateContentParameters,
|
||||
userPromptId: string,
|
||||
@@ -117,7 +121,7 @@ export class CodeAssistServer implements ContentGenerator {
|
||||
req,
|
||||
userPromptId,
|
||||
this.projectId,
|
||||
this.sessionId,
|
||||
this.getEffectiveSessionId(),
|
||||
enabledCreditTypes,
|
||||
),
|
||||
req.config?.abortSignal,
|
||||
@@ -153,7 +157,7 @@ export class CodeAssistServer implements ContentGenerator {
|
||||
translatedResponse,
|
||||
streamingLatency,
|
||||
req.config?.abortSignal,
|
||||
server.sessionId, // Use sessionId as trajectoryId
|
||||
server.getEffectiveSessionId(), // Use sessionId as trajectoryId
|
||||
);
|
||||
|
||||
if (response.consumedCredits) {
|
||||
@@ -204,7 +208,7 @@ export class CodeAssistServer implements ContentGenerator {
|
||||
req,
|
||||
userPromptId,
|
||||
this.projectId,
|
||||
this.sessionId,
|
||||
this.getEffectiveSessionId(),
|
||||
undefined,
|
||||
),
|
||||
req.config?.abortSignal,
|
||||
@@ -224,7 +228,7 @@ export class CodeAssistServer implements ContentGenerator {
|
||||
translatedResponse,
|
||||
streamingLatency,
|
||||
req.config?.abortSignal,
|
||||
this.sessionId, // Use sessionId as trajectoryId
|
||||
this.getEffectiveSessionId(), // Use sessionId as trajectoryId
|
||||
);
|
||||
|
||||
if (response.remainingCredits) {
|
||||
|
||||
@@ -3198,6 +3198,24 @@ describe('Config Quota & Preview Model Access', () => {
|
||||
expect(config.getHasAccessToPreviewModel()).toBe(false);
|
||||
});
|
||||
|
||||
it('should reverse-map gemini-3-flash back to gemini-3.5-flash in modelQuotas', async () => {
|
||||
mockCodeAssistServer.retrieveUserQuota.mockResolvedValue({
|
||||
buckets: [
|
||||
{
|
||||
modelId: 'gemini-3-flash',
|
||||
remainingAmount: '90',
|
||||
remainingFraction: 0.9,
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
config.setModel('gemini-3.5-flash');
|
||||
await config.refreshUserQuota();
|
||||
|
||||
expect(config.getQuotaRemaining()).toBe(90);
|
||||
expect(config.getQuotaLimit()).toBe(100);
|
||||
});
|
||||
|
||||
it('should calculate pooled quota correctly for auto models', async () => {
|
||||
mockCodeAssistServer.retrieveUserQuota.mockResolvedValue({
|
||||
buckets: [
|
||||
@@ -4134,7 +4152,9 @@ describe('Plans Directory Initialization', () => {
|
||||
|
||||
const plansDir = config.storage.getPlansDir();
|
||||
// Should NOT create the directory eagerly
|
||||
expect(fs.promises.mkdir).not.toHaveBeenCalled();
|
||||
expect(fs.promises.mkdir).not.toHaveBeenCalledWith(plansDir, {
|
||||
recursive: true,
|
||||
});
|
||||
// Should check if it exists
|
||||
expect(fs.promises.access).toHaveBeenCalledWith(plansDir);
|
||||
|
||||
@@ -4152,7 +4172,9 @@ describe('Plans Directory Initialization', () => {
|
||||
await config.initialize();
|
||||
|
||||
const plansDir = config.storage.getPlansDir();
|
||||
expect(fs.promises.mkdir).not.toHaveBeenCalled();
|
||||
expect(fs.promises.mkdir).not.toHaveBeenCalledWith(plansDir, {
|
||||
recursive: true,
|
||||
});
|
||||
expect(fs.promises.access).toHaveBeenCalledWith(plansDir);
|
||||
|
||||
const context = config.getWorkspaceContext();
|
||||
|
||||
@@ -87,6 +87,8 @@ import {
|
||||
PREVIEW_GEMINI_FLASH_MODEL,
|
||||
resolveModel,
|
||||
setFlashModels,
|
||||
DEFAULT_GEMINI_3_5_FLASH_MODEL,
|
||||
SECONDARY_GEMINI_3_5_FLASH_MODEL,
|
||||
} from './models.js';
|
||||
import { shouldAttemptBrowserLaunch } from '../utils/browser.js';
|
||||
import type { MCPOAuthConfig } from '../mcp/oauth-provider.js';
|
||||
@@ -678,6 +680,7 @@ export interface ConfigParameters {
|
||||
truncateToolOutputThreshold?: number;
|
||||
eventEmitter?: EventEmitter;
|
||||
useWriteTodos?: boolean;
|
||||
env?: Record<string, string>;
|
||||
workspacePoliciesDir?: string;
|
||||
policyEngineConfig?: PolicyEngineConfig;
|
||||
directWebFetch?: boolean;
|
||||
@@ -896,6 +899,7 @@ export class Config implements McpContext, AgentLoopContext {
|
||||
private readonly useTerminalBuffer: boolean;
|
||||
private readonly useRenderProcess: boolean;
|
||||
private shellExecutionConfig: ShellExecutionConfig;
|
||||
readonly env?: Record<string, string>;
|
||||
private readonly extensionManagement: boolean = true;
|
||||
private readonly extensionRegistryURI: string | undefined;
|
||||
private readonly truncateToolOutputThreshold: number;
|
||||
@@ -1119,6 +1123,7 @@ export class Config implements McpContext, AgentLoopContext {
|
||||
this.checkpointing = params.checkpointing ?? false;
|
||||
this.proxy = params.proxy;
|
||||
this.cwd = params.cwd ?? process.cwd();
|
||||
this.env = params.env;
|
||||
this.fileDiscoveryService = params.fileDiscoveryService ?? null;
|
||||
this.bugCommand = params.bugCommand;
|
||||
this.model = params.model;
|
||||
@@ -1858,6 +1863,10 @@ export class Config implements McpContext, AgentLoopContext {
|
||||
}
|
||||
}
|
||||
|
||||
rotateSessionId(sessionId: string): void {
|
||||
this._sessionId = sessionId;
|
||||
}
|
||||
|
||||
resetNewSessionState(sessionId: string): void {
|
||||
this.setSessionId(sessionId);
|
||||
}
|
||||
@@ -1931,6 +1940,9 @@ export class Config implements McpContext, AgentLoopContext {
|
||||
}
|
||||
|
||||
activateFallbackMode(model: string, failedModel?: string): void {
|
||||
debugLogger.log(
|
||||
`Model fallback activated: switching from ${failedModel ?? 'unknown'} to ${model}`,
|
||||
);
|
||||
if (this.getActiveModel() !== model) {
|
||||
this.setModel(model, true);
|
||||
}
|
||||
@@ -2310,6 +2322,11 @@ export class Config implements McpContext, AgentLoopContext {
|
||||
continue;
|
||||
}
|
||||
|
||||
let modelId = bucket.modelId;
|
||||
if (modelId === SECONDARY_GEMINI_3_5_FLASH_MODEL) {
|
||||
modelId = DEFAULT_GEMINI_3_5_FLASH_MODEL;
|
||||
}
|
||||
|
||||
let remaining: number;
|
||||
let limit: number;
|
||||
|
||||
@@ -2318,7 +2335,7 @@ export class Config implements McpContext, AgentLoopContext {
|
||||
limit =
|
||||
bucket.remainingFraction > 0
|
||||
? Math.round(remaining / bucket.remainingFraction)
|
||||
: (this.modelQuotas.get(bucket.modelId)?.limit ?? 0);
|
||||
: (this.modelQuotas.get(modelId)?.limit ?? 0);
|
||||
} else {
|
||||
// Server only sent remainingFraction — use a normalized scale.
|
||||
limit = 100;
|
||||
@@ -2326,7 +2343,7 @@ export class Config implements McpContext, AgentLoopContext {
|
||||
}
|
||||
|
||||
if (!isNaN(remaining) && Number.isFinite(limit) && limit > 0) {
|
||||
this.modelQuotas.set(bucket.modelId, {
|
||||
this.modelQuotas.set(modelId, {
|
||||
remaining,
|
||||
limit,
|
||||
resetTime: bucket.resetTime,
|
||||
|
||||
@@ -135,6 +135,29 @@ describe('AgentHistoryProvider', () => {
|
||||
);
|
||||
});
|
||||
|
||||
it('should use unambiguous label in fallback summary to avoid LLM confusion', async () => {
|
||||
providerConfig.maxTokens = 60000;
|
||||
providerConfig.retainedTokens = 60000;
|
||||
vi.spyOn(config, 'getContextManagementConfig').mockReturnValue({
|
||||
enabled: true,
|
||||
} as unknown as ContextManagementConfig);
|
||||
vi.mocked(estimateTokenCountSync).mockImplementation(
|
||||
(parts: Part[]) => parts.length * 4000,
|
||||
);
|
||||
generateContentMock.mockRejectedValue(new Error('API Error'));
|
||||
|
||||
const history = createMockHistory(35);
|
||||
const result = await provider.manageHistory(history);
|
||||
|
||||
expect(generateContentMock).toHaveBeenCalled();
|
||||
expect(result.length).toBe(15);
|
||||
// The fallback summary should use clear and unambiguous phrasing
|
||||
expect(result[0].parts![0].text).toContain(
|
||||
'Previous User Intent (Truncated):',
|
||||
);
|
||||
expect(result[0].parts![0].text).not.toContain('Last User Intent:');
|
||||
});
|
||||
|
||||
it('should pass the contextual bridge to the summarizer', async () => {
|
||||
vi.spyOn(config, 'getContextManagementConfig').mockReturnValue({
|
||||
enabled: true,
|
||||
|
||||
@@ -267,7 +267,9 @@ export class AgentHistoryProvider {
|
||||
];
|
||||
|
||||
if (lastUserText) {
|
||||
summaryParts.push(`- **Last User Intent:** "${lastUserText}"`);
|
||||
summaryParts.push(
|
||||
`- **Previous User Intent (Truncated):** "${lastUserText}"`,
|
||||
);
|
||||
}
|
||||
|
||||
if (actionPath) {
|
||||
|
||||
@@ -165,6 +165,16 @@ ONLY use the built-in \`exit_plan_mode\` tool to present the plan for formal app
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -361,6 +371,16 @@ An approved plan is available for this task at \`../plans/feature-x.md\`.
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -671,6 +691,16 @@ ONLY use the built-in \`exit_plan_mode\` tool to present the plan for formal app
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -845,6 +875,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -1005,6 +1045,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -1148,6 +1198,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -1837,6 +1897,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -2011,6 +2081,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -2189,6 +2269,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -2367,6 +2457,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -2541,6 +2641,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -2709,6 +2819,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -2851,6 +2971,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -3025,6 +3155,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -3345,6 +3485,16 @@ You are operating with a persistent file-based task tracking system located at \
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -3774,6 +3924,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -3948,6 +4108,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -4241,6 +4411,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
@@ -4415,6 +4595,16 @@ Operate using a **Research -> Strategy -> Execution** lifecycle. For the Executi
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. \`replace\`, \`write_file\`), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the \`replace\` tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the \`run_shell_command\` tool for running shell commands, remembering the safety rule to explain modifying commands first.
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
/**
|
||||
* @license
|
||||
* Copyright 2026 Google LLC
|
||||
* SPDX-License-Identifier: Apache-2.0
|
||||
*/
|
||||
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { AgentChatHistory, type HistoryTurn } from './agentChatHistory.js';
|
||||
|
||||
describe('AgentChatHistory', () => {
|
||||
const dummyTurns: HistoryTurn[] = [
|
||||
{
|
||||
id: 'turn-1',
|
||||
content: { role: 'user', parts: [{ text: 'Hello' }] },
|
||||
},
|
||||
{
|
||||
id: 'turn-2',
|
||||
content: { role: 'model', parts: [{ text: 'Hi there' }] },
|
||||
},
|
||||
{
|
||||
id: 'turn-3',
|
||||
content: { role: 'user', parts: [{ text: 'How are you?' }] },
|
||||
},
|
||||
];
|
||||
|
||||
it('should initialize with empty history by default', () => {
|
||||
const history = new AgentChatHistory();
|
||||
expect(history.length).toBe(0);
|
||||
expect(history.get()).toEqual([]);
|
||||
});
|
||||
|
||||
it('should initialize with provided turns', () => {
|
||||
const history = new AgentChatHistory(dummyTurns);
|
||||
expect(history.length).toBe(3);
|
||||
expect(history.get()).toEqual(dummyTurns);
|
||||
});
|
||||
|
||||
it('should push new turns', () => {
|
||||
const history = new AgentChatHistory();
|
||||
history.push(dummyTurns[0]);
|
||||
expect(history.length).toBe(1);
|
||||
expect(history.get()[0]).toEqual(dummyTurns[0]);
|
||||
});
|
||||
|
||||
it('should set and overwrite history turns', () => {
|
||||
const history = new AgentChatHistory(dummyTurns.slice(0, 1));
|
||||
expect(history.length).toBe(1);
|
||||
history.set(dummyTurns);
|
||||
expect(history.length).toBe(3);
|
||||
expect(history.get()).toEqual(dummyTurns);
|
||||
});
|
||||
|
||||
it('should clear history', () => {
|
||||
const history = new AgentChatHistory(dummyTurns);
|
||||
expect(history.length).toBe(3);
|
||||
history.clear();
|
||||
expect(history.length).toBe(0);
|
||||
expect(history.get()).toEqual([]);
|
||||
});
|
||||
|
||||
describe('rollback', () => {
|
||||
it('should roll back history to a specified length', () => {
|
||||
const history = new AgentChatHistory(dummyTurns);
|
||||
history.rollback(1);
|
||||
expect(history.length).toBe(1);
|
||||
expect(history.get()).toEqual([dummyTurns[0]]);
|
||||
});
|
||||
|
||||
it('should roll back to 0', () => {
|
||||
const history = new AgentChatHistory(dummyTurns);
|
||||
history.rollback(0);
|
||||
expect(history.length).toBe(0);
|
||||
expect(history.get()).toEqual([]);
|
||||
});
|
||||
|
||||
it('should do nothing if rollback length is out of bounds (negative)', () => {
|
||||
const history = new AgentChatHistory(dummyTurns);
|
||||
history.rollback(-1);
|
||||
expect(history.length).toBe(3);
|
||||
expect(history.get()).toEqual(dummyTurns);
|
||||
});
|
||||
|
||||
it('should do nothing if rollback length is out of bounds (greater than current history length)', () => {
|
||||
const history = new AgentChatHistory(dummyTurns);
|
||||
history.rollback(5);
|
||||
expect(history.length).toBe(3);
|
||||
expect(history.get()).toEqual(dummyTurns);
|
||||
});
|
||||
});
|
||||
|
||||
it('should return raw Content array via getContents()', () => {
|
||||
const history = new AgentChatHistory(dummyTurns);
|
||||
expect(history.getContents()).toEqual(
|
||||
dummyTurns.map((turn) => turn.content),
|
||||
);
|
||||
});
|
||||
|
||||
it('should support mapping and flatMapping operations', () => {
|
||||
const history = new AgentChatHistory(dummyTurns);
|
||||
const mappedIds = history.map((turn) => turn.id);
|
||||
expect(mappedIds).toEqual(['turn-1', 'turn-2', 'turn-3']);
|
||||
|
||||
const flatMappedParts = history.flatMap((turn) => turn.content.parts || []);
|
||||
expect(flatMappedParts).toEqual([
|
||||
{ text: 'Hello' },
|
||||
{ text: 'Hi there' },
|
||||
{ text: 'How are you?' },
|
||||
]);
|
||||
});
|
||||
});
|
||||
@@ -46,6 +46,16 @@ export class AgentChatHistory {
|
||||
this.history = [];
|
||||
}
|
||||
|
||||
/**
|
||||
* Rolls back the history to a specified length.
|
||||
* Useful when a stream fails and we need to remove the un-responded turn(s).
|
||||
*/
|
||||
rollback(length: number) {
|
||||
if (length >= 0 && length <= this.history.length) {
|
||||
this.history = this.history.slice(0, length);
|
||||
}
|
||||
}
|
||||
|
||||
get(): readonly HistoryTurn[] {
|
||||
return this.history;
|
||||
}
|
||||
|
||||
@@ -212,7 +212,7 @@ describe('Gemini Client (client.ts)', () => {
|
||||
.fn()
|
||||
.mockReturnValue(contentGeneratorConfig),
|
||||
getToolRegistry: vi.fn().mockReturnValue(mockToolRegistry),
|
||||
getModel: vi.fn().mockReturnValue('test-model'),
|
||||
getModel: vi.fn().mockReturnValue('gemini-1.5-pro'),
|
||||
getUserTier: vi.fn().mockReturnValue(undefined),
|
||||
getEmbeddingModel: vi.fn().mockReturnValue('test-embedding-model'),
|
||||
getApiKey: vi.fn().mockReturnValue('test-key'),
|
||||
|
||||
@@ -154,6 +154,13 @@ export async function createContentGeneratorConfig(
|
||||
vertexAiRouting,
|
||||
};
|
||||
|
||||
const getEnv = (key: string) => {
|
||||
if (config?.env && config.env[key] !== undefined) {
|
||||
return config.env[key];
|
||||
}
|
||||
return process.env[key];
|
||||
};
|
||||
|
||||
// If we are using Google auth or we are in Cloud Shell, there is nothing else to validate for now.
|
||||
// Return before touching the API-key keychain: on Linux without a Secret Service
|
||||
// (WSL/SSH/Docker/CI) keytar can block indefinitely on its functional probe.
|
||||
@@ -165,16 +172,13 @@ export async function createContentGeneratorConfig(
|
||||
}
|
||||
|
||||
const geminiApiKey =
|
||||
apiKey ||
|
||||
process.env['GEMINI_API_KEY'] ||
|
||||
(await loadApiKey()) ||
|
||||
undefined;
|
||||
const googleApiKey = process.env['GOOGLE_API_KEY'] || undefined;
|
||||
apiKey || getEnv('GEMINI_API_KEY') || (await loadApiKey()) || undefined;
|
||||
const googleApiKey = getEnv('GOOGLE_API_KEY') || undefined;
|
||||
const googleCloudProject =
|
||||
process.env['GOOGLE_CLOUD_PROJECT'] ||
|
||||
process.env['GOOGLE_CLOUD_PROJECT_ID'] ||
|
||||
getEnv('GOOGLE_CLOUD_PROJECT') ||
|
||||
getEnv('GOOGLE_CLOUD_PROJECT_ID') ||
|
||||
undefined;
|
||||
const googleCloudLocation = process.env['GOOGLE_CLOUD_LOCATION'] || undefined;
|
||||
const googleCloudLocation = getEnv('GOOGLE_CLOUD_LOCATION') || undefined;
|
||||
|
||||
if (authType === AuthType.USE_GEMINI && geminiApiKey) {
|
||||
contentGeneratorConfig.apiKey = geminiApiKey;
|
||||
@@ -194,8 +198,7 @@ export async function createContentGeneratorConfig(
|
||||
}
|
||||
|
||||
if (authType === AuthType.GATEWAY) {
|
||||
contentGeneratorConfig.apiKey =
|
||||
apiKey || process.env['GEMINI_API_KEY'] || '';
|
||||
contentGeneratorConfig.apiKey = apiKey || getEnv('GEMINI_API_KEY') || '';
|
||||
contentGeneratorConfig.vertexai = false;
|
||||
|
||||
return contentGeneratorConfig;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -30,7 +30,11 @@ import {
|
||||
getRetryErrorType,
|
||||
} from '../utils/retry.js';
|
||||
import type { ValidationRequiredError } from '../utils/googleQuotaErrors.js';
|
||||
import { resolveModel, supportsModernFeatures } from '../config/models.js';
|
||||
import {
|
||||
resolveModel,
|
||||
supportsModernFeatures,
|
||||
isGemini2Model,
|
||||
} from '../config/models.js';
|
||||
import { hasCycleInSchema } from '../tools/tools.js';
|
||||
import type { StructuredError } from './turn.js';
|
||||
import type { CompletedToolCall } from '../scheduler/types.js';
|
||||
@@ -104,6 +108,13 @@ const MID_STREAM_RETRY_OPTIONS: MidStreamRetryOptions = {
|
||||
|
||||
export const SYNTHETIC_THOUGHT_SIGNATURE = 'skip_thought_signature_validator';
|
||||
|
||||
/**
|
||||
* Stands in for a model turn that never arrived because the stream failed
|
||||
* after a tool response was already committed to history.
|
||||
*/
|
||||
export const INTERRUPTED_RESPONSE_PLACEHOLDER =
|
||||
'[The previous response was interrupted before it completed.]';
|
||||
|
||||
/**
|
||||
* Internal interface for parts that carry the magic 'callIndex' property
|
||||
* used during model response consolidation.
|
||||
@@ -221,7 +232,12 @@ export class InvalidStreamError extends Error {
|
||||
| 'NO_FINISH_REASON'
|
||||
| 'NO_RESPONSE_TEXT'
|
||||
| 'MALFORMED_FUNCTION_CALL'
|
||||
| 'UNEXPECTED_TOOL_CALL';
|
||||
| 'UNEXPECTED_TOOL_CALL'
|
||||
| 'MAX_TOKENS_EXCEEDED'
|
||||
| 'SAFETY_BLOCKED'
|
||||
| 'RECITATION_BLOCKED'
|
||||
| 'OTHER_BLOCKED'
|
||||
| 'THINKING_ONLY_RESPONSE';
|
||||
|
||||
constructor(
|
||||
message: string,
|
||||
@@ -229,7 +245,12 @@ export class InvalidStreamError extends Error {
|
||||
| 'NO_FINISH_REASON'
|
||||
| 'NO_RESPONSE_TEXT'
|
||||
| 'MALFORMED_FUNCTION_CALL'
|
||||
| 'UNEXPECTED_TOOL_CALL',
|
||||
| 'UNEXPECTED_TOOL_CALL'
|
||||
| 'MAX_TOKENS_EXCEEDED'
|
||||
| 'SAFETY_BLOCKED'
|
||||
| 'RECITATION_BLOCKED'
|
||||
| 'OTHER_BLOCKED'
|
||||
| 'THINKING_ONLY_RESPONSE',
|
||||
) {
|
||||
super(message);
|
||||
this.name = 'InvalidStreamError';
|
||||
@@ -383,6 +404,9 @@ export class GeminiChat {
|
||||
): Promise<AsyncGenerator<StreamEvent>> {
|
||||
await this.sendPromise;
|
||||
|
||||
const historyLengthBefore = this.agentHistory.length;
|
||||
const baselinePromptTokenCount = this.lastPromptTokenCount;
|
||||
|
||||
let streamDoneResolver: () => void;
|
||||
const streamDonePromise = new Promise<void>((resolve) => {
|
||||
streamDoneResolver = resolve;
|
||||
@@ -390,6 +414,17 @@ export class GeminiChat {
|
||||
this.sendPromise = streamDonePromise;
|
||||
|
||||
let userContent = createUserContent(message);
|
||||
const isOriginalFunctionResponse = isFunctionResponse(userContent);
|
||||
|
||||
// A turn can end leaving history on an unanswered tool response: a stream
|
||||
// error after the response was committed, or a cancelled tool call. Close
|
||||
// it before recording a genuinely new user message, otherwise the two user
|
||||
// turns are coalesced into one and the model continues the trailing text
|
||||
// instead of answering it.
|
||||
if (!isOriginalFunctionResponse) {
|
||||
this.closeUnansweredToolResponseTurn();
|
||||
}
|
||||
|
||||
const { model } =
|
||||
this.context.config.modelConfigService.getResolvedConfig(modelConfigKey);
|
||||
|
||||
@@ -398,7 +433,7 @@ export class GeminiChat {
|
||||
|
||||
// Record user input - capture complete message with all parts (text, files, images, etc.)
|
||||
// but skip recording function responses (tool call results) as they should be stored in tool call records
|
||||
if (!isFunctionResponse(userContent)) {
|
||||
if (!isOriginalFunctionResponse) {
|
||||
const userMessageParts = userContent.parts || [];
|
||||
const userMessageContent = partListUnionToString(userMessageParts);
|
||||
|
||||
@@ -515,6 +550,7 @@ export class GeminiChat {
|
||||
): AsyncGenerator<StreamEvent, void, void> {
|
||||
try {
|
||||
const maxAttempts = this.context.config.getMaxAttempts();
|
||||
let lastStreamError: unknown = undefined;
|
||||
|
||||
for (let attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
let isConnectionPhase = true;
|
||||
@@ -526,7 +562,7 @@ export class GeminiChat {
|
||||
// If this is a retry, update the key with the new context.
|
||||
const currentConfigKey =
|
||||
attempt > 0
|
||||
? { ...modelConfigKey, isRetry: true }
|
||||
? { ...modelConfigKey, isRetry: true, lastStreamError }
|
||||
: modelConfigKey;
|
||||
|
||||
isConnectionPhase = true;
|
||||
@@ -545,6 +581,10 @@ export class GeminiChat {
|
||||
|
||||
return;
|
||||
} catch (error) {
|
||||
if (error instanceof InvalidStreamError) {
|
||||
lastStreamError = error;
|
||||
}
|
||||
|
||||
if (error instanceof AgentExecutionStoppedError) {
|
||||
yield {
|
||||
type: StreamEventType.AGENT_EXECUTION_STOPPED,
|
||||
@@ -581,8 +621,7 @@ export class GeminiChat {
|
||||
);
|
||||
|
||||
const isContentError = error instanceof InvalidStreamError;
|
||||
const isRetryableContentError =
|
||||
isContentError && error.type !== 'NO_RESPONSE_TEXT';
|
||||
const isRetryableContentError = isContentError;
|
||||
const errorType = isContentError
|
||||
? error.type
|
||||
: getRetryErrorType(error);
|
||||
@@ -644,6 +683,15 @@ export class GeminiChat {
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
if (!isOriginalFunctionResponse) {
|
||||
this.agentHistory.rollback(historyLengthBefore);
|
||||
this.chatRecordingService.updateMessagesFromHistory(
|
||||
this.agentHistory.get(),
|
||||
);
|
||||
this.lastPromptTokenCount = baselinePromptTokenCount;
|
||||
}
|
||||
throw error;
|
||||
} finally {
|
||||
streamDoneResolver!();
|
||||
}
|
||||
@@ -652,6 +700,28 @@ export class GeminiChat {
|
||||
return streamWithRetries.call(this);
|
||||
}
|
||||
|
||||
/**
|
||||
* Appends a closing model turn when history ends with an unanswered tool
|
||||
* response, so the next user message stays a turn of its own.
|
||||
*/
|
||||
private closeUnansweredToolResponseTurn(): void {
|
||||
const turns = this.agentHistory.get();
|
||||
const last = turns[turns.length - 1];
|
||||
if (
|
||||
last?.content.role !== 'user' ||
|
||||
!last.content.parts?.some((part) => !!part.functionResponse)
|
||||
) {
|
||||
return;
|
||||
}
|
||||
this.agentHistory.push({
|
||||
id: randomUUID(),
|
||||
content: {
|
||||
role: 'model',
|
||||
parts: [{ text: INTERRUPTED_RESPONSE_PLACEHOLDER }],
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
private extractBinaryInjections(
|
||||
parts: Part[] | undefined,
|
||||
): Part[] | undefined {
|
||||
@@ -683,10 +753,13 @@ export class GeminiChat {
|
||||
): Promise<AsyncGenerator<GenerateContentResponse>> {
|
||||
// Last mile scrubbing to remove internal tracking properties (e.g. callIndex)
|
||||
// before sending to the Gemini API. This whitelists only standard Gemini fields.
|
||||
const scrubbedHistory = this.context.config.isContextManagementEnabled()
|
||||
let scrubbedHistory = this.context.config.isContextManagementEnabled()
|
||||
? scrubHistory([...requestHistory])
|
||||
: [...requestHistory];
|
||||
|
||||
// Always coalesce consecutive roles to prevent 400 Bad Request errors
|
||||
scrubbedHistory = coalesceConsecutiveRoles(scrubbedHistory);
|
||||
|
||||
const scrubbedContents = scrubbedHistory.map((h) => h.content);
|
||||
|
||||
const requestContents = apiHistoryOverride
|
||||
@@ -763,9 +836,34 @@ export class GeminiChat {
|
||||
abortSignal,
|
||||
};
|
||||
|
||||
let contentsToUse: Content[] = supportsModernFeatures(modelToUse)
|
||||
? [...contentsForPreviewModel]
|
||||
: [...requestContents];
|
||||
// Apply Context-Aware Retries (On-Retry Nudging) to guide the model out of silent loops
|
||||
if (
|
||||
modelConfigKey.isRetry &&
|
||||
modelConfigKey.lastStreamError instanceof InvalidStreamError
|
||||
) {
|
||||
const lastError = modelConfigKey.lastStreamError;
|
||||
let nudgeMessage = '';
|
||||
if (lastError.type === 'THINKING_ONLY_RESPONSE') {
|
||||
nudgeMessage =
|
||||
'\n[System: You previously generated thoughts but failed to provide a final user-facing response. Please ensure you provide your final answer or call a tool now.]';
|
||||
} else if (lastError.type === 'NO_RESPONSE_TEXT') {
|
||||
nudgeMessage =
|
||||
'\n[System: You previously returned an empty response with no text or thoughts. Please ensure you provide your final answer or call a tool now.]';
|
||||
}
|
||||
|
||||
if (nudgeMessage) {
|
||||
if (typeof config.systemInstruction === 'string') {
|
||||
config.systemInstruction += nudgeMessage;
|
||||
} else if (config.systemInstruction === undefined) {
|
||||
config.systemInstruction = nudgeMessage;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let contentsToUse: Content[] =
|
||||
supportsModernFeatures(modelToUse) || isGemini2Model(modelToUse)
|
||||
? [...contentsForPreviewModel]
|
||||
: [...requestContents];
|
||||
|
||||
const hookSystem = this.context.config.getHookSystem();
|
||||
if (hookSystem) {
|
||||
@@ -807,9 +905,10 @@ export class GeminiChat {
|
||||
);
|
||||
lastModelToUse = modelToUse;
|
||||
// Re-evaluate contentsToUse based on the new model's feature support
|
||||
contentsToUse = supportsModernFeatures(modelToUse)
|
||||
? [...contentsForPreviewModel]
|
||||
: [...requestContents];
|
||||
contentsToUse =
|
||||
supportsModernFeatures(modelToUse) || isGemini2Model(modelToUse)
|
||||
? [...contentsForPreviewModel]
|
||||
: [...requestContents];
|
||||
}
|
||||
if (beforeModelResult.modifiedConfig) {
|
||||
Object.assign(config, beforeModelResult.modifiedConfig);
|
||||
@@ -953,9 +1052,16 @@ export class GeminiChat {
|
||||
? extractCuratedHistory(this.agentHistory.get())
|
||||
: [...this.agentHistory.get()];
|
||||
|
||||
return this.context.config.isContextManagementEnabled()
|
||||
? scrubHistory(history)
|
||||
: history;
|
||||
if (this.context.config.isContextManagementEnabled()) {
|
||||
return scrubHistory(history);
|
||||
}
|
||||
|
||||
const model = this.context.config.getModel();
|
||||
if (isGemini2Model(model) || supportsModernFeatures(model)) {
|
||||
return coalesceConsecutiveRoles(stripThoughts(history));
|
||||
}
|
||||
|
||||
return history;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1028,11 +1134,19 @@ export class GeminiChat {
|
||||
requestContents: readonly Content[],
|
||||
): readonly Content[] {
|
||||
// First, find the start of the active loop by finding the last user turn
|
||||
// with a text message, i.e. that is not a function response.
|
||||
// with a text message, i.e. that is not a function response. Testing for
|
||||
// text alone is not enough: `coalesceConsecutiveRoles` can merge a function
|
||||
// response turn with the prompt that follows it, and starting the loop at
|
||||
// such a turn starts it later than the API starts the turn, leaving earlier
|
||||
// function calls unsigned but still validated.
|
||||
let activeLoopStartIndex = -1;
|
||||
for (let i = requestContents.length - 1; i >= 0; i--) {
|
||||
const content = requestContents[i];
|
||||
if (content.role === 'user' && content.parts?.some((part) => part.text)) {
|
||||
if (
|
||||
content.role === 'user' &&
|
||||
content.parts?.some((part) => part.text) &&
|
||||
!content.parts?.some((part) => part.functionResponse)
|
||||
) {
|
||||
activeLoopStartIndex = i;
|
||||
break;
|
||||
}
|
||||
@@ -1119,6 +1233,13 @@ export class GeminiChat {
|
||||
let hasThoughts = false;
|
||||
let finishReason: FinishReason | undefined;
|
||||
|
||||
// Buffers to prevent failed stream attempts from polluting telemetry and logs
|
||||
const bufferedThoughts: Array<{ subject: string; description: string }> =
|
||||
[];
|
||||
let bufferedUsageMetadata:
|
||||
| GenerateContentResponse['usageMetadata']
|
||||
| undefined = undefined;
|
||||
|
||||
// The SDK provides fully assembled FunctionCall objects in chunk.functionCalls
|
||||
// We use a Map to ensure we only keep the latest version of each call (by ID)
|
||||
const finalFunctionCallsMap = new Map<string, FunctionCall>();
|
||||
@@ -1174,7 +1295,10 @@ export class GeminiChat {
|
||||
if (content.parts.some((part) => part.thought)) {
|
||||
// Record thoughts
|
||||
hasThoughts = true;
|
||||
this.recordThoughtFromContent(content);
|
||||
const thought = this.extractThoughtFromContent(content);
|
||||
if (thought) {
|
||||
bufferedThoughts.push(thought);
|
||||
}
|
||||
}
|
||||
if (content.parts.some((part) => part.functionCall)) {
|
||||
hasToolCall = true;
|
||||
@@ -1202,12 +1326,9 @@ export class GeminiChat {
|
||||
}
|
||||
}
|
||||
|
||||
// Record token usage if this chunk has usageMetadata
|
||||
// Buffer token usage if this chunk has usageMetadata
|
||||
if (chunk.usageMetadata) {
|
||||
this.chatRecordingService.recordMessageTokens(chunk.usageMetadata);
|
||||
if (chunk.usageMetadata.promptTokenCount !== undefined) {
|
||||
this.lastPromptTokenCount = chunk.usageMetadata.promptTokenCount;
|
||||
}
|
||||
bufferedUsageMetadata = chunk.usageMetadata;
|
||||
}
|
||||
|
||||
const hookSystem = this.context.config.getHookSystem();
|
||||
@@ -1294,29 +1415,22 @@ export class GeminiChat {
|
||||
}
|
||||
}
|
||||
|
||||
const responseText = consolidatedParts
|
||||
const rawResponseText = consolidatedParts
|
||||
.filter((part) => part.text)
|
||||
.map((part) => part.text)
|
||||
.join('')
|
||||
.trim();
|
||||
.join('');
|
||||
|
||||
let id: string;
|
||||
// Record model response text from the collected parts.
|
||||
// Also flush when there are thoughts or a tool call (even with no text)
|
||||
// so that BeforeTool hooks always see the latest transcript state.
|
||||
if (responseText || hasThoughts || hasToolCall) {
|
||||
id = this.chatRecordingService.recordMessage({
|
||||
model,
|
||||
type: 'gemini',
|
||||
content: responseText,
|
||||
});
|
||||
} else {
|
||||
// Still need a durable ID even if response is empty (e.g. only tool calls)
|
||||
id = this.chatRecordingService.recordSyntheticMessage(
|
||||
'gemini',
|
||||
consolidatedParts,
|
||||
);
|
||||
}
|
||||
// Clean zero-width/invisible characters and HTML comments to determine actual printable/visible content
|
||||
let responseText = rawResponseText.replace(
|
||||
/[\u200B-\u200D\uFEFF\u200E\u200F]/g,
|
||||
'',
|
||||
);
|
||||
let previous: string;
|
||||
do {
|
||||
previous = responseText;
|
||||
responseText = responseText.replace(/<!--[\s\S]*?-->/g, '');
|
||||
} while (responseText !== previous);
|
||||
responseText = responseText.trim();
|
||||
|
||||
// Stream validation logic: A stream is considered successful if:
|
||||
// 1. There's a tool call OR
|
||||
@@ -1346,6 +1460,36 @@ export class GeminiChat {
|
||||
);
|
||||
}
|
||||
if (!responseText) {
|
||||
if (finishReason === FinishReason.MAX_TOKENS) {
|
||||
throw new InvalidStreamError(
|
||||
'Model stream ended due to token limit exhaustion (MAX_TOKENS) with empty response text.',
|
||||
'MAX_TOKENS_EXCEEDED',
|
||||
);
|
||||
}
|
||||
if (finishReason === FinishReason.SAFETY) {
|
||||
throw new InvalidStreamError(
|
||||
'Model stream ended due to safety settings (SAFETY) with empty response text.',
|
||||
'SAFETY_BLOCKED',
|
||||
);
|
||||
}
|
||||
if (finishReason === FinishReason.RECITATION) {
|
||||
throw new InvalidStreamError(
|
||||
'Model stream ended due to recitation settings (RECITATION) with empty response text.',
|
||||
'RECITATION_BLOCKED',
|
||||
);
|
||||
}
|
||||
if (finishReason === FinishReason.OTHER) {
|
||||
throw new InvalidStreamError(
|
||||
'Model stream ended due to other settings (OTHER) with empty response text.',
|
||||
'OTHER_BLOCKED',
|
||||
);
|
||||
}
|
||||
if (hasThoughts) {
|
||||
throw new InvalidStreamError(
|
||||
'Model stream ended with empty response text but contained reasoning thoughts.',
|
||||
'THINKING_ONLY_RESPONSE',
|
||||
);
|
||||
}
|
||||
throw new InvalidStreamError(
|
||||
'Model stream ended with empty response text.',
|
||||
'NO_RESPONSE_TEXT',
|
||||
@@ -1353,6 +1497,37 @@ export class GeminiChat {
|
||||
}
|
||||
}
|
||||
|
||||
// Flush buffered thoughts from the successful attempt
|
||||
for (const thought of bufferedThoughts) {
|
||||
this.chatRecordingService.recordThought(thought);
|
||||
}
|
||||
|
||||
// Flush buffered usage metadata and token counts from the successful attempt
|
||||
if (bufferedUsageMetadata) {
|
||||
this.chatRecordingService.recordMessageTokens(bufferedUsageMetadata);
|
||||
if (bufferedUsageMetadata.promptTokenCount !== undefined) {
|
||||
this.lastPromptTokenCount = bufferedUsageMetadata.promptTokenCount;
|
||||
}
|
||||
}
|
||||
|
||||
let id: string;
|
||||
// Record model response text from the collected parts.
|
||||
// Also flush when there are thoughts or a tool call (even with no text)
|
||||
// so that BeforeTool hooks always see the latest transcript state.
|
||||
if (responseText || hasThoughts || hasToolCall) {
|
||||
id = this.chatRecordingService.recordMessage({
|
||||
model,
|
||||
type: 'gemini',
|
||||
content: responseText,
|
||||
});
|
||||
} else {
|
||||
// Still need a durable ID even if response is empty (e.g. only tool calls)
|
||||
id = this.chatRecordingService.recordSyntheticMessage(
|
||||
'gemini',
|
||||
consolidatedParts,
|
||||
);
|
||||
}
|
||||
|
||||
this.agentHistory.push({
|
||||
id,
|
||||
content: { role: 'model', parts: consolidatedParts },
|
||||
@@ -1407,11 +1582,13 @@ export class GeminiChat {
|
||||
}
|
||||
|
||||
/**
|
||||
* Extracts and records thought from thought content.
|
||||
* Extracts thought from thought content.
|
||||
*/
|
||||
private recordThoughtFromContent(content: Content): void {
|
||||
private extractThoughtFromContent(
|
||||
content: Content,
|
||||
): { subject: string; description: string } | undefined {
|
||||
if (!content.parts || content.parts.length === 0) {
|
||||
return;
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const thoughtPart = content.parts[0];
|
||||
@@ -1424,11 +1601,12 @@ export class GeminiChat {
|
||||
: '';
|
||||
const description = rawText.replace(/\*\*(.*?)\*\*/s, '').trim();
|
||||
|
||||
this.chatRecordingService.recordThought({
|
||||
return {
|
||||
subject,
|
||||
description,
|
||||
});
|
||||
};
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1472,3 +1650,73 @@ export function stripToolCallIdPrefixes(contents: Content[]): Content[] {
|
||||
}),
|
||||
}));
|
||||
}
|
||||
|
||||
export function coalesceConsecutiveRoles(
|
||||
history: HistoryTurn[],
|
||||
): HistoryTurn[] {
|
||||
const result: HistoryTurn[] = [];
|
||||
for (const turn of history) {
|
||||
const lastIdx = result.length - 1;
|
||||
const last = result[lastIdx];
|
||||
if (last && last.content.role && last.content.role === turn.content.role) {
|
||||
const hasParts = last.content.parts || turn.content.parts;
|
||||
result[lastIdx] = {
|
||||
id: last.id,
|
||||
content: {
|
||||
...last.content,
|
||||
parts: hasParts
|
||||
? [...(last.content.parts || []), ...(turn.content.parts || [])]
|
||||
: undefined,
|
||||
},
|
||||
};
|
||||
} else {
|
||||
result.push({
|
||||
id: turn.id,
|
||||
content: { ...turn.content },
|
||||
});
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
export function stripThoughts(history: HistoryTurn[]): HistoryTurn[] {
|
||||
return history
|
||||
.map((turn) => {
|
||||
if (!turn.content.parts) return turn;
|
||||
const hasThought = turn.content.parts.some((p) => p && p.thought);
|
||||
if (!hasThought) return turn;
|
||||
|
||||
const nonThoughtParts = turn.content.parts.filter((p) => p && !p.thought);
|
||||
|
||||
// The thoughtSignature the API requires on the first functionCall of a
|
||||
// model turn is sometimes only carried by the thought part we just
|
||||
// removed, not by the functionCall part itself. Without it, replaying
|
||||
// this turn in a later request gets rejected with a 400 "missing
|
||||
// thought_signature" error, so inject a synthetic one if needed.
|
||||
let patchedFirstCall = false;
|
||||
const finalParts =
|
||||
turn.content.role === 'model'
|
||||
? nonThoughtParts.map((p) => {
|
||||
if (!patchedFirstCall && p.functionCall) {
|
||||
patchedFirstCall = true;
|
||||
if (!p.thoughtSignature) {
|
||||
return {
|
||||
...p,
|
||||
thoughtSignature: SYNTHETIC_THOUGHT_SIGNATURE,
|
||||
};
|
||||
}
|
||||
}
|
||||
return p;
|
||||
})
|
||||
: nonThoughtParts;
|
||||
|
||||
return {
|
||||
...turn,
|
||||
content: {
|
||||
...turn.content,
|
||||
parts: finalParts,
|
||||
},
|
||||
};
|
||||
})
|
||||
.filter((turn) => !turn.content.parts || turn.content.parts.length > 0);
|
||||
}
|
||||
|
||||
@@ -254,7 +254,15 @@ describe('Turn', () => {
|
||||
events.push(event);
|
||||
}
|
||||
|
||||
expect(events).toEqual([{ type: GeminiEventType.InvalidStream }]);
|
||||
expect(events).toEqual([
|
||||
{
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: 'NO_FINISH_REASON',
|
||||
message: 'Test invalid stream',
|
||||
},
|
||||
},
|
||||
]);
|
||||
expect(turn.getDebugResponses().length).toBe(0);
|
||||
expect(reportError).not.toHaveBeenCalled(); // Should not report as error
|
||||
});
|
||||
|
||||
@@ -105,6 +105,19 @@ export type ServerGeminiContextWindowWillOverflowEvent = {
|
||||
|
||||
export type ServerGeminiInvalidStreamEvent = {
|
||||
type: GeminiEventType.InvalidStream;
|
||||
value: {
|
||||
type:
|
||||
| 'NO_FINISH_REASON'
|
||||
| 'NO_RESPONSE_TEXT'
|
||||
| 'MALFORMED_FUNCTION_CALL'
|
||||
| 'UNEXPECTED_TOOL_CALL'
|
||||
| 'MAX_TOKENS_EXCEEDED'
|
||||
| 'SAFETY_BLOCKED'
|
||||
| 'RECITATION_BLOCKED'
|
||||
| 'OTHER_BLOCKED'
|
||||
| 'THINKING_ONLY_RESPONSE';
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
|
||||
export type ServerGeminiModelInfoEvent = {
|
||||
@@ -408,7 +421,13 @@ export class Turn {
|
||||
}
|
||||
|
||||
if (e instanceof InvalidStreamError) {
|
||||
yield { type: GeminiEventType.InvalidStream };
|
||||
yield {
|
||||
type: GeminiEventType.InvalidStream,
|
||||
value: {
|
||||
type: e.type,
|
||||
message: e.message,
|
||||
},
|
||||
};
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
@@ -67,6 +67,7 @@ const createMockConfig = (overrides: Partial<Config> = {}): Config =>
|
||||
setActiveModel: vi.fn(),
|
||||
setModel: vi.fn(),
|
||||
activateFallbackMode: vi.fn(),
|
||||
rotateSessionId: vi.fn(),
|
||||
getModelAvailabilityService: vi.fn(() =>
|
||||
createAvailabilityServiceMock({
|
||||
selectedModel: FALLBACK_MODEL,
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
*/
|
||||
|
||||
import type { Config } from '../config/config.js';
|
||||
import { createSessionId } from '../utils/session.js';
|
||||
import {
|
||||
openBrowserSecurely,
|
||||
shouldLaunchBrowser,
|
||||
@@ -161,8 +162,9 @@ async function processIntent(
|
||||
): Promise<boolean> {
|
||||
switch (intent) {
|
||||
case 'retry_always':
|
||||
// TODO(telemetry): Implement generic fallback event logging. Existing
|
||||
// logFlashFallback is specific to a single Model.
|
||||
// Rotate the session ID to ensure the backend treats the retried request
|
||||
// as a brand-new session, preventing stateful model-switching errors.
|
||||
config.rotateSessionId(createSessionId());
|
||||
config.activateFallbackMode(fallbackModel, failedModel);
|
||||
return true;
|
||||
|
||||
|
||||
@@ -31,6 +31,21 @@ vi.mock('node:fs', async (importOriginal) => {
|
||||
...actual.promises,
|
||||
readFile: vi.fn(),
|
||||
readdir: vi.fn(),
|
||||
realpath: vi.fn((p) => Promise.resolve(p)),
|
||||
stat: vi.fn(() =>
|
||||
Promise.resolve({ uid: process.getuid ? process.getuid() : 1000 }),
|
||||
),
|
||||
open: vi.fn((filePath: string) =>
|
||||
Promise.resolve({
|
||||
stat: () => fs.promises.stat(filePath),
|
||||
readFile: (options?: string | { encoding?: string }) =>
|
||||
fs.promises.readFile(
|
||||
filePath,
|
||||
options as unknown as BufferEncoding | undefined,
|
||||
),
|
||||
close: () => Promise.resolve(),
|
||||
} as unknown as fs.promises.FileHandle),
|
||||
),
|
||||
},
|
||||
realpathSync: (p: string) => p,
|
||||
existsSync: vi.fn(() => false),
|
||||
@@ -430,6 +445,141 @@ describe('ide-connection-utils', () => {
|
||||
|
||||
expect(result).toEqual(config2);
|
||||
});
|
||||
|
||||
it('should NOT filter out config if all found config files are mismatched/invalid workspaces, returning the best sorted match so that the correct Directory Mismatch error is raised downstream', async () => {
|
||||
const invalidConfig1 = {
|
||||
port: '1111',
|
||||
workspacePath: '/invalid/workspace1',
|
||||
};
|
||||
const invalidConfig2 = {
|
||||
port: '2222',
|
||||
workspacePath: '/invalid/workspace2',
|
||||
};
|
||||
vi.mocked(fs.promises.readFile).mockRejectedValueOnce(
|
||||
new Error('not found'),
|
||||
);
|
||||
(
|
||||
vi.mocked(fs.promises.readdir) as Mock<
|
||||
(path: fs.PathLike) => Promise<string[]>
|
||||
>
|
||||
).mockResolvedValue([
|
||||
'gemini-ide-server-12345-111.json',
|
||||
'gemini-ide-server-12345-222.json',
|
||||
]);
|
||||
vi.mocked(fs.promises.readFile)
|
||||
.mockResolvedValueOnce(JSON.stringify(invalidConfig1))
|
||||
.mockResolvedValueOnce(JSON.stringify(invalidConfig2));
|
||||
|
||||
const result = await getConnectionConfigFromFile(12345);
|
||||
|
||||
expect(result).toEqual(invalidConfig1);
|
||||
});
|
||||
|
||||
it('should prioritize the config matching the port from the environment variable when all found config files are mismatched/invalid workspaces', async () => {
|
||||
vi.stubEnv('GEMINI_CLI_IDE_SERVER_PORT', '2222');
|
||||
const invalidConfig1 = {
|
||||
port: '1111',
|
||||
workspacePath: '/invalid/workspace1',
|
||||
};
|
||||
const invalidConfig2 = {
|
||||
port: '2222',
|
||||
workspacePath: '/invalid/workspace2',
|
||||
};
|
||||
vi.mocked(fs.promises.readFile).mockRejectedValueOnce(
|
||||
new Error('not found'),
|
||||
);
|
||||
(
|
||||
vi.mocked(fs.promises.readdir) as Mock<
|
||||
(path: fs.PathLike) => Promise<string[]>
|
||||
>
|
||||
).mockResolvedValue([
|
||||
'gemini-ide-server-12345-111.json',
|
||||
'gemini-ide-server-12345-222.json',
|
||||
]);
|
||||
vi.mocked(fs.promises.readFile)
|
||||
.mockResolvedValueOnce(JSON.stringify(invalidConfig1))
|
||||
.mockResolvedValueOnce(JSON.stringify(invalidConfig2));
|
||||
|
||||
const result = await getConnectionConfigFromFile(12345);
|
||||
|
||||
expect(result).toEqual(invalidConfig2);
|
||||
});
|
||||
|
||||
it.runIf(process.getuid !== undefined)(
|
||||
'should reject and ignore config files owned by a different user UID to prevent hijacking/information disclosure',
|
||||
async () => {
|
||||
const config1 = {
|
||||
port: '1111',
|
||||
workspacePath: '/test/workspace',
|
||||
};
|
||||
vi.mocked(fs.promises.readFile).mockRejectedValueOnce(
|
||||
new Error('not found'),
|
||||
);
|
||||
(
|
||||
vi.mocked(fs.promises.readdir) as Mock<
|
||||
(path: fs.PathLike) => Promise<string[]>
|
||||
>
|
||||
).mockResolvedValue(['gemini-ide-server-12345-111.json']);
|
||||
vi.mocked(fs.promises.readFile).mockResolvedValueOnce(
|
||||
JSON.stringify(config1),
|
||||
);
|
||||
|
||||
const otherUid = (process.getuid ? process.getuid() : 1000) + 1;
|
||||
vi.mocked(fs.promises.stat).mockResolvedValueOnce({
|
||||
uid: otherUid,
|
||||
} as unknown as fs.Stats);
|
||||
|
||||
const result = await getConnectionConfigFromFile(12345);
|
||||
|
||||
expect(result).toBeUndefined();
|
||||
},
|
||||
);
|
||||
|
||||
it('should accept and parse config files owned by the current user UID', async () => {
|
||||
const config1 = {
|
||||
port: '1111',
|
||||
workspacePath: '/test/workspace',
|
||||
};
|
||||
vi.mocked(fs.promises.readFile).mockRejectedValueOnce(
|
||||
new Error('not found'),
|
||||
);
|
||||
(
|
||||
vi.mocked(fs.promises.readdir) as Mock<
|
||||
(path: fs.PathLike) => Promise<string[]>
|
||||
>
|
||||
).mockResolvedValue(['gemini-ide-server-12345-111.json']);
|
||||
vi.mocked(fs.promises.readFile).mockResolvedValueOnce(
|
||||
JSON.stringify(config1),
|
||||
);
|
||||
|
||||
const currentUid = process.getuid ? process.getuid() : 1000;
|
||||
vi.mocked(fs.promises.stat).mockResolvedValueOnce({
|
||||
uid: currentUid,
|
||||
} as unknown as fs.Stats);
|
||||
|
||||
const result = await getConnectionConfigFromFile(12345);
|
||||
|
||||
expect(result).toEqual(config1);
|
||||
});
|
||||
|
||||
it('should reject and ignore config files if fs.promises.open throws an error', async () => {
|
||||
vi.mocked(fs.promises.readFile).mockRejectedValueOnce(
|
||||
new Error('not found'),
|
||||
);
|
||||
(
|
||||
vi.mocked(fs.promises.readdir) as Mock<
|
||||
(path: fs.PathLike) => Promise<string[]>
|
||||
>
|
||||
).mockResolvedValue(['gemini-ide-server-12345-111.json']);
|
||||
|
||||
vi.mocked(fs.promises.open).mockRejectedValueOnce(
|
||||
new Error('symlink loop / permission denied'),
|
||||
);
|
||||
|
||||
const result = await getConnectionConfigFromFile(12345);
|
||||
|
||||
expect(result).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('validateWorkspacePath', () => {
|
||||
|
||||
@@ -109,6 +109,26 @@ export function getStdioConfigFromEnv(): StdioConfig | undefined {
|
||||
|
||||
const IDE_SERVER_FILE_REGEX = /^gemini-ide-server-(\d+)-\d+\.json$/;
|
||||
|
||||
async function verifyAndReadFile(
|
||||
filePath: string,
|
||||
): Promise<string | undefined> {
|
||||
let handle: fs.promises.FileHandle | undefined;
|
||||
try {
|
||||
handle = await fs.promises.open(filePath, 'r');
|
||||
const stat = await handle.stat();
|
||||
if (process.getuid && stat.uid !== process.getuid()) {
|
||||
return undefined;
|
||||
}
|
||||
return await handle.readFile('utf8');
|
||||
} catch {
|
||||
return undefined;
|
||||
} finally {
|
||||
if (handle) {
|
||||
await handle.close();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export async function getConnectionConfigFromFile(
|
||||
pid: number,
|
||||
): Promise<
|
||||
@@ -122,7 +142,10 @@ export async function getConnectionConfigFromFile(
|
||||
'ide',
|
||||
`gemini-ide-server-${pid}.json`,
|
||||
);
|
||||
const portFileContents = await fs.promises.readFile(portFile, 'utf8');
|
||||
const portFileContents = await verifyAndReadFile(portFile);
|
||||
if (!portFileContents) {
|
||||
throw new Error('Verification failed or file not found');
|
||||
}
|
||||
const parsed: unknown = JSON.parse(portFileContents);
|
||||
type ConfigType = ConnectionConfig & {
|
||||
workspacePath?: string;
|
||||
@@ -164,23 +187,21 @@ export async function getConnectionConfigFromFile(
|
||||
|
||||
sortConnectionFiles(matchingFiles, pid);
|
||||
|
||||
let fileContents: string[];
|
||||
try {
|
||||
fileContents = await Promise.all(
|
||||
matchingFiles.map((file) =>
|
||||
fs.promises.readFile(path.join(portFileDir, file), 'utf8'),
|
||||
),
|
||||
);
|
||||
} catch (e) {
|
||||
logger.debug('Failed to read IDE connection config file(s):', e);
|
||||
return undefined;
|
||||
}
|
||||
const fileContents = await Promise.all(
|
||||
matchingFiles.map((file) =>
|
||||
verifyAndReadFile(path.join(portFileDir, file)),
|
||||
),
|
||||
);
|
||||
|
||||
const parsedContents = fileContents.map(
|
||||
(
|
||||
content,
|
||||
):
|
||||
| (ConnectionConfig & { workspacePath?: string; ideInfo?: IdeInfo })
|
||||
| undefined => {
|
||||
if (!content) {
|
||||
return undefined;
|
||||
}
|
||||
try {
|
||||
const parsed: unknown = JSON.parse(content);
|
||||
type ConfigType = ConnectionConfig & {
|
||||
@@ -219,6 +240,31 @@ export async function getConnectionConfigFromFile(
|
||||
);
|
||||
|
||||
if (validWorkspaces.length === 0) {
|
||||
// If no workspace matches the current CWD, but we found and parsed
|
||||
// valid connection config file(s), return the best-sorted config.
|
||||
// This lets downstream connection logic raise a helpful, detailed
|
||||
// "Directory mismatch" warning instead of a generic connection error.
|
||||
let fileIndex = -1;
|
||||
const portFromEnv = getPortFromEnv();
|
||||
if (portFromEnv) {
|
||||
fileIndex = parsedContents.findIndex(
|
||||
(content) =>
|
||||
!!content &&
|
||||
content.port !== undefined &&
|
||||
String(content.port) === portFromEnv,
|
||||
);
|
||||
}
|
||||
if (fileIndex === -1) {
|
||||
fileIndex = parsedContents.findIndex((content) => !!content);
|
||||
}
|
||||
|
||||
if (fileIndex !== -1) {
|
||||
const selected = parsedContents[fileIndex]!;
|
||||
logger.debug(
|
||||
`Selected best mismatched IDE connection file: ${matchingFiles[fileIndex]}`,
|
||||
);
|
||||
return selected;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
@@ -234,7 +280,8 @@ export async function getConnectionConfigFromFile(
|
||||
const portFromEnv = getPortFromEnv();
|
||||
if (portFromEnv) {
|
||||
const matchingPortIndex = validWorkspaces.findIndex(
|
||||
(content) => String(content.port) === portFromEnv,
|
||||
(content) =>
|
||||
content.port !== undefined && String(content.port) === portFromEnv,
|
||||
);
|
||||
if (matchingPortIndex !== -1) {
|
||||
const selected = validWorkspaces[matchingPortIndex];
|
||||
|
||||
@@ -203,6 +203,7 @@ describe('MCPOAuthProvider', () => {
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
vi.unstubAllEnvs();
|
||||
});
|
||||
|
||||
describe('authenticate', () => {
|
||||
@@ -440,6 +441,100 @@ describe('MCPOAuthProvider', () => {
|
||||
);
|
||||
});
|
||||
|
||||
it('should perform dynamic client registration with Cloud Workstations proxy redirect URI when running in Google Cloud Workstations', async () => {
|
||||
vi.stubEnv('GOOGLE_CLOUD_WORKSTATIONS', 'true');
|
||||
vi.stubEnv(
|
||||
'WEB_HOST',
|
||||
'my-workstation.cluster.workstations.cloud.google.com',
|
||||
);
|
||||
|
||||
const configWithoutClient: MCPOAuthConfig = {
|
||||
...mockConfig,
|
||||
registrationUrl: 'https://auth.example.com/register',
|
||||
};
|
||||
delete configWithoutClient.clientId;
|
||||
delete configWithoutClient.redirectUri;
|
||||
|
||||
const mockRegistrationResponse: OAuthClientRegistrationResponse = {
|
||||
client_id: 'dynamic_client_id',
|
||||
client_secret: 'dynamic_client_secret',
|
||||
redirect_uris: [
|
||||
'https://7777-my-workstation.cluster.workstations.cloud.google.com/oauth/callback',
|
||||
],
|
||||
grant_types: ['authorization_code', 'refresh_token'],
|
||||
response_types: ['code'],
|
||||
token_endpoint_auth_method: 'none',
|
||||
};
|
||||
|
||||
mockFetch.mockResolvedValueOnce(
|
||||
createMockResponse({
|
||||
ok: true,
|
||||
contentType: 'application/json',
|
||||
text: JSON.stringify(mockRegistrationResponse),
|
||||
json: mockRegistrationResponse,
|
||||
}),
|
||||
);
|
||||
|
||||
// Setup callback handler
|
||||
let callbackHandler: unknown;
|
||||
vi.mocked(http.createServer).mockImplementation((handler) => {
|
||||
callbackHandler = handler;
|
||||
return mockHttpServer as unknown as http.Server;
|
||||
});
|
||||
|
||||
mockHttpServer.listen.mockImplementation((port, callback) => {
|
||||
callback?.();
|
||||
setTimeout(() => {
|
||||
const mockReq = {
|
||||
url: '/oauth/callback?code=auth_code_123&state=bW9ja19zdGF0ZV8xNl9ieXRlcw',
|
||||
};
|
||||
const mockRes = {
|
||||
writeHead: vi.fn(),
|
||||
end: vi.fn(),
|
||||
};
|
||||
(callbackHandler as (req: unknown, res: unknown) => void)(
|
||||
mockReq,
|
||||
mockRes,
|
||||
);
|
||||
}, 10);
|
||||
});
|
||||
|
||||
// Mock token exchange
|
||||
mockFetch.mockResolvedValueOnce(
|
||||
createMockResponse({
|
||||
ok: true,
|
||||
contentType: 'application/json',
|
||||
text: JSON.stringify(mockTokenResponse),
|
||||
json: mockTokenResponse,
|
||||
}),
|
||||
);
|
||||
|
||||
const authProvider = new MCPOAuthProvider();
|
||||
const result = await authProvider.authenticate(
|
||||
'test-server',
|
||||
configWithoutClient,
|
||||
);
|
||||
|
||||
expect(result).toBeDefined();
|
||||
expect(mockFetch).toHaveBeenCalledWith(
|
||||
'https://auth.example.com/register',
|
||||
expect.objectContaining({
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
client_name: 'Gemini CLI MCP Client',
|
||||
redirect_uris: [
|
||||
'https://7777-my-workstation.cluster.workstations.cloud.google.com/oauth/callback',
|
||||
],
|
||||
grant_types: ['authorization_code', 'refresh_token'],
|
||||
response_types: ['code'],
|
||||
token_endpoint_auth_method: 'none',
|
||||
scope: 'read write',
|
||||
}),
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it('should perform OAuth discovery and dynamic client registration when no client ID or registration URL provided', async () => {
|
||||
const configWithoutClient: MCPOAuthConfig = { ...mockConfig };
|
||||
delete configWithoutClient.clientId;
|
||||
@@ -1458,6 +1553,50 @@ describe('MCPOAuthProvider', () => {
|
||||
);
|
||||
});
|
||||
|
||||
it('should refresh with the stored client ID when config has none', async () => {
|
||||
const expiredCredentials = {
|
||||
serverName: 'test-server',
|
||||
token: { ...mockToken, expiresAt: Date.now() - 3600000 },
|
||||
clientId: 'registered-client-id',
|
||||
tokenUrl: 'https://auth.example.com/token',
|
||||
updatedAt: Date.now(),
|
||||
};
|
||||
|
||||
const tokenStorage = new MCPOAuthTokenStorage();
|
||||
vi.mocked(tokenStorage.getCredentials).mockResolvedValue(
|
||||
expiredCredentials,
|
||||
);
|
||||
vi.mocked(tokenStorage.isTokenExpired).mockReturnValue(true);
|
||||
|
||||
mockFetch.mockResolvedValueOnce(
|
||||
createMockResponse({
|
||||
ok: true,
|
||||
contentType: 'application/json',
|
||||
text: JSON.stringify(mockTokenResponse),
|
||||
json: mockTokenResponse,
|
||||
}),
|
||||
);
|
||||
|
||||
const authProvider = new MCPOAuthProvider();
|
||||
const result = await authProvider.getValidToken('test-server', {
|
||||
...mockConfig,
|
||||
clientId: undefined,
|
||||
});
|
||||
|
||||
expect(result).toBe('access_token_123');
|
||||
expect(mockFetch.mock.calls[0][1].body).toContain(
|
||||
'client_id=registered-client-id',
|
||||
);
|
||||
expect(tokenStorage.saveToken).toHaveBeenCalledWith(
|
||||
'test-server',
|
||||
expect.objectContaining({ accessToken: 'access_token_123' }),
|
||||
'registered-client-id',
|
||||
'https://auth.example.com/token',
|
||||
undefined,
|
||||
);
|
||||
expect(tokenStorage.deleteCredentials).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('should return null when no credentials exist', async () => {
|
||||
const tokenStorage = new MCPOAuthTokenStorage();
|
||||
vi.mocked(tokenStorage.getCredentials).mockResolvedValue(null);
|
||||
@@ -1542,6 +1681,90 @@ describe('MCPOAuthProvider', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('getValidTokenWithMetadata', () => {
|
||||
it('should refresh with the stored client ID when config is empty', async () => {
|
||||
// An empty config is what DynamicStoredOAuthProvider passes for servers
|
||||
// configured via OAuth discovery and dynamic client registration.
|
||||
const expiredCredentials = {
|
||||
serverName: 'test-server',
|
||||
token: { ...mockToken, expiresAt: Date.now() - 3600000 },
|
||||
clientId: 'registered-client-id',
|
||||
tokenUrl: 'https://auth.example.com/token',
|
||||
updatedAt: Date.now(),
|
||||
};
|
||||
|
||||
const tokenStorage = new MCPOAuthTokenStorage();
|
||||
vi.mocked(tokenStorage.getCredentials).mockResolvedValue(
|
||||
expiredCredentials,
|
||||
);
|
||||
vi.mocked(tokenStorage.isTokenExpired).mockReturnValue(true);
|
||||
|
||||
mockFetch.mockResolvedValueOnce(
|
||||
createMockResponse({
|
||||
ok: true,
|
||||
contentType: 'application/json',
|
||||
text: JSON.stringify(mockTokenResponse),
|
||||
json: mockTokenResponse,
|
||||
}),
|
||||
);
|
||||
|
||||
const authProvider = new MCPOAuthProvider();
|
||||
const result = await authProvider.getValidTokenWithMetadata(
|
||||
'test-server',
|
||||
{},
|
||||
);
|
||||
|
||||
expect(result?.accessToken).toBe('access_token_123');
|
||||
expect(mockFetch.mock.calls[0][1].body).toContain(
|
||||
'client_id=registered-client-id',
|
||||
);
|
||||
expect(tokenStorage.saveToken).toHaveBeenCalledWith(
|
||||
'test-server',
|
||||
expect.objectContaining({ accessToken: 'access_token_123' }),
|
||||
'registered-client-id',
|
||||
'https://auth.example.com/token',
|
||||
undefined,
|
||||
);
|
||||
expect(tokenStorage.deleteCredentials).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('should prefer the config client ID over the stored one', async () => {
|
||||
const expiredCredentials = {
|
||||
serverName: 'test-server',
|
||||
token: { ...mockToken, expiresAt: Date.now() - 3600000 },
|
||||
clientId: 'registered-client-id',
|
||||
tokenUrl: 'https://auth.example.com/token',
|
||||
updatedAt: Date.now(),
|
||||
};
|
||||
|
||||
const tokenStorage = new MCPOAuthTokenStorage();
|
||||
vi.mocked(tokenStorage.getCredentials).mockResolvedValue(
|
||||
expiredCredentials,
|
||||
);
|
||||
vi.mocked(tokenStorage.isTokenExpired).mockReturnValue(true);
|
||||
|
||||
mockFetch.mockResolvedValueOnce(
|
||||
createMockResponse({
|
||||
ok: true,
|
||||
contentType: 'application/json',
|
||||
text: JSON.stringify(mockTokenResponse),
|
||||
json: mockTokenResponse,
|
||||
}),
|
||||
);
|
||||
|
||||
const authProvider = new MCPOAuthProvider();
|
||||
const result = await authProvider.getValidTokenWithMetadata(
|
||||
'test-server',
|
||||
{ clientId: 'configured-client-id' },
|
||||
);
|
||||
|
||||
expect(result?.accessToken).toBe('access_token_123');
|
||||
expect(mockFetch.mock.calls[0][1].body).toContain(
|
||||
'client_id=configured-client-id',
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe('PKCE parameter generation', () => {
|
||||
it('should generate valid PKCE parameters', async () => {
|
||||
// Test is implicit in the authenticate flow tests, but we can verify
|
||||
|
||||
@@ -21,7 +21,7 @@ import {
|
||||
buildAuthorizationUrl,
|
||||
exchangeCodeForToken,
|
||||
refreshAccessToken as refreshAccessTokenShared,
|
||||
REDIRECT_PATH,
|
||||
getRedirectUri,
|
||||
type OAuthFlowConfig,
|
||||
type OAuthTokenResponse,
|
||||
} from '../utils/oauth-flow.js';
|
||||
@@ -99,8 +99,7 @@ export class MCPOAuthProvider {
|
||||
config: MCPOAuthConfig,
|
||||
redirectPort: number,
|
||||
): Promise<OAuthClientRegistrationResponse> {
|
||||
const redirectUri =
|
||||
config.redirectUri || `http://localhost:${redirectPort}${REDIRECT_PATH}`;
|
||||
const redirectUri = getRedirectUri(config, redirectPort);
|
||||
|
||||
const registrationRequest: OAuthClientRegistrationRequest = {
|
||||
client_name: 'Gemini CLI MCP Client',
|
||||
@@ -568,15 +567,18 @@ ${authUrl}
|
||||
return token.accessToken;
|
||||
}
|
||||
|
||||
// Try to refresh if we have a refresh token
|
||||
if (token.refreshToken && config.clientId && credentials.tokenUrl) {
|
||||
// Try to refresh if we have a refresh token. Fall back to the client ID
|
||||
// persisted during dynamic client registration when the static config
|
||||
// does not provide one.
|
||||
const clientId = config.clientId ?? credentials.clientId;
|
||||
if (token.refreshToken && clientId && credentials.tokenUrl) {
|
||||
try {
|
||||
debugLogger.log(
|
||||
`Refreshing expired token for MCP server: ${serverName}`,
|
||||
);
|
||||
|
||||
const newTokenResponse = await this.refreshAccessToken(
|
||||
config,
|
||||
{ ...config, clientId },
|
||||
token.refreshToken,
|
||||
credentials.tokenUrl,
|
||||
credentials.mcpServerUrl,
|
||||
@@ -597,7 +599,7 @@ ${authUrl}
|
||||
await this.tokenStorage.saveToken(
|
||||
serverName,
|
||||
newToken,
|
||||
config.clientId,
|
||||
clientId,
|
||||
credentials.tokenUrl,
|
||||
credentials.mcpServerUrl,
|
||||
);
|
||||
@@ -636,7 +638,7 @@ ${authUrl}
|
||||
if (current.refreshToken && clientId && credentials.tokenUrl) {
|
||||
try {
|
||||
const newTokenResponse = await this.refreshAccessToken(
|
||||
config,
|
||||
{ ...config, clientId },
|
||||
current.refreshToken,
|
||||
credentials.tokenUrl,
|
||||
credentials.mcpServerUrl,
|
||||
|
||||
@@ -109,23 +109,90 @@ priority = 50
|
||||
modes = ["plan"]
|
||||
interactive = true
|
||||
|
||||
# Allow write_file and replace for .md files in the plans directory (cross-platform)
|
||||
# We split this into two rules to avoid ReDoS checker issues with nested optional segments.
|
||||
# This rule handles the case where there is a session ID in the plan file path
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"[^\"]+[\\\\/]+\\.gemini[\\\\/]+tmp[\\\\/]+[\\w-]+[\\\\/]+[\\w-]+[\\\\/]+plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
# Allow write_file and replace for .md files in the plans directory (cross-platform).
|
||||
# This rule employs split, traversal-safe, and ReDoS-safe patterns to provide defense-in-depth:
|
||||
# - Absolute paths must strictly be inside the designated plans directory under `.gemini/tmp/`
|
||||
# - Relative paths must be clean (no path traversal `..` and no absolute prefixes)
|
||||
|
||||
# This rule handles the case where there isn't a session ID in the plan file path
|
||||
# 1. Absolute paths with session ID
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"[^\"]+[\\\\/]+\\.gemini[\\\\/]+tmp[\\\\/]+[\\w-]+[\\\\/]+plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
argsPattern = "\\x00\"file_path\":\"[^\\\"]+[\\\\/]+\\.gemini[\\\\/]+tmp[\\\\/]+[\\w-]+[\\\\/]+[\\w-]+[\\\\/]+plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# 2. Absolute paths without session ID
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"[^\\\"]+[\\\\/]+\\.gemini[\\\\/]+tmp[\\\\/]+[\\w-]+[\\\\/]+plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# 3. Relative paths starting with .gemini (with session ID)
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"\\.gemini[\\\\/]+tmp[\\\\/]+[\\w-]+[\\\\/]+[\\w-]+[\\\\/]+plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# 4. Relative paths starting with .gemini (without session ID)
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"\\.gemini[\\\\/]+tmp[\\\\/]+[\\w-]+[\\\\/]+plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# 5. Relative paths starting with ./.gemini (with session ID)
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"\\.[\\\\/]+\\.gemini[\\\\/]+tmp[\\\\/]+[\\w-]+[\\\\/]+[\\w-]+[\\\\/]+plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# 6. Relative paths starting with ./.gemini (without session ID)
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"\\.[\\\\/]+\\.gemini[\\\\/]+tmp[\\\\/]+[\\w-]+[\\\\/]+plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# 7. Clean relative filename (no directories, e.g. plan.md)
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# 8. Clean relative path starting with ./ (no directories, e.g. ./plan.md)
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"\\.[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# 9. Clean relative path under plans directory (e.g. plans/plan.md)
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# 10. Clean relative path under plans directory starting with ./ (e.g. ./plans/plan.md)
|
||||
[[rule]]
|
||||
toolName = ["write_file", "replace"]
|
||||
decision = "allow"
|
||||
priority = 70
|
||||
modes = ["plan"]
|
||||
argsPattern = "\\x00\"file_path\":\"\\.[\\\\/]+plans[\\\\/]+[\\w-]+\\.md\"\\x00"
|
||||
|
||||
# Explicitly Deny other write operations in Plan mode with a clear message.
|
||||
[[rule]]
|
||||
|
||||
@@ -413,6 +413,16 @@ export function renderOperationalGuidelines(
|
||||
- **Security First:** Always apply security best practices. Never introduce code that exposes, logs, or commits secrets, API keys, or other sensitive information.
|
||||
|
||||
## Tool Usage
|
||||
- **Tool Execution Response Rules:**
|
||||
1. After receiving a \`functionResponse\`, you MUST ALWAYS execute one of the following two actions:
|
||||
a) Call another tool to proceed with the task.
|
||||
b) Provide a user-facing text response explaining the tool output, your analysis, and next steps.
|
||||
2. You MUST NEVER return an empty response with no text and no tool calls.
|
||||
- **Post-Edit Response Rules:**
|
||||
1. After an edit tool execution (e.g. ${formatToolName(EDIT_TOOL_NAME)}, ${formatToolName(WRITE_FILE_TOOL_NAME)}), you MUST ALWAYS generate a user-facing text response summarizing:
|
||||
- What changes were made to the file.
|
||||
- Your verification plan or next steps (e.g. running tests).
|
||||
2. You MUST NEVER return an empty response with 0 text tokens after completing an edit.
|
||||
- **Parallelism & Sequencing:** Tools execute in parallel by default. Execute multiple independent tool calls in parallel when feasible (e.g., searching, reading files, independent shell commands, or editing *different* files). If a tool depends on the output or side-effects of a previous tool in the same turn (e.g., running a shell command that depends on the success of a previous command), you MUST set the \`wait_for_previous\` parameter to \`true\` on the dependent tool to ensure sequential execution.
|
||||
- **File Editing Collisions:** Do NOT make multiple calls to the ${formatToolName(EDIT_TOOL_NAME)} tool for the SAME file in a single turn. To make multiple edits to the same file, you MUST perform them sequentially across multiple conversational turns to prevent race conditions and ensure the file state is accurate before each edit.
|
||||
- **Command Execution:** Use the ${formatToolName(SHELL_TOOL_NAME)} tool for running shell commands, remembering the safety rule to explain modifying commands first.${toolUsageInteractive(
|
||||
|
||||
@@ -34,6 +34,10 @@ describe('CheckerRunner', () => {
|
||||
|
||||
beforeEach(() => {
|
||||
mockContextBuilder = new ContextBuilder({} as Config);
|
||||
vi.spyOn(mockContextBuilder, 'config', 'get').mockReturnValue({
|
||||
env: {},
|
||||
getWorkingDir: vi.fn().mockReturnValue('/mock/cwd'),
|
||||
} as unknown as Config);
|
||||
mockRegistry = new CheckerRegistry('/mock/dist');
|
||||
CheckerRegistry.prototype.resolveInProcess = vi.fn();
|
||||
|
||||
|
||||
@@ -168,6 +168,8 @@ export class CheckerRunner {
|
||||
return new Promise((resolve) => {
|
||||
const child = spawn(checkerPath, [], {
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
cwd: this.contextBuilder.config.getWorkingDir(),
|
||||
env: { ...process.env, ...this.contextBuilder.config.env },
|
||||
});
|
||||
|
||||
let stdout = '';
|
||||
|
||||
@@ -15,6 +15,10 @@ import type { AgentLoopContext } from '../config/agent-loop-context.js';
|
||||
export class ContextBuilder {
|
||||
constructor(private readonly context: AgentLoopContext) {}
|
||||
|
||||
get config() {
|
||||
return this.context.config;
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds the full context object with all available data.
|
||||
*/
|
||||
|
||||
@@ -630,8 +630,8 @@ describe('Scheduler (Orchestrator)', () => {
|
||||
CoreToolCallStatus.Cancelled,
|
||||
'Operation cancelled by user',
|
||||
);
|
||||
// finalizeCall is handled by the processing loop, not synchronously by cancelAll
|
||||
// expect(mockStateManager.finalizeCall).toHaveBeenCalledWith('call-1');
|
||||
// finalizeCall is called synchronously by cancelAll to ensure completedBatch is populated and isActive is updated immediately
|
||||
expect(mockStateManager.finalizeCall).toHaveBeenCalledWith('call-1');
|
||||
expect(mockStateManager.cancelAllQueued).toHaveBeenCalledWith(
|
||||
'Operation cancelled by user',
|
||||
);
|
||||
|
||||
@@ -278,6 +278,7 @@ export class Scheduler {
|
||||
CoreToolCallStatus.Cancelled,
|
||||
'Operation cancelled by user',
|
||||
);
|
||||
this.state.finalizeCall(activeCall.request.callId);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -438,6 +439,14 @@ export class Scheduler {
|
||||
*/
|
||||
private async _processNextItem(signal: AbortSignal): Promise<boolean> {
|
||||
if (signal.aborted || this.isCancelling) {
|
||||
// Finalize active calls that are terminal
|
||||
const activeCalls = this.state.allActiveCalls;
|
||||
for (const call of activeCalls) {
|
||||
if (this.isTerminal(call.status)) {
|
||||
this.state.finalizeCall(call.request.callId);
|
||||
}
|
||||
}
|
||||
|
||||
this.state.cancelAllQueued('Operation cancelled');
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -308,6 +308,125 @@ describe('ChatRecordingService', () => {
|
||||
)) as ConversationRecord;
|
||||
expect(conversation.sessionId).toBe('old-session-id');
|
||||
});
|
||||
|
||||
it('should fall back to the in-memory conversation when the file cannot be reloaded', async () => {
|
||||
// Regression test for the `/compress` "Failed to load resumed session
|
||||
// data from file" bug: when resuming with a filePath that cannot be
|
||||
// loaded from disk, initialize must NOT throw. It should adopt the
|
||||
// in-memory conversation it was handed and rewrite a clean file.
|
||||
const chatsDir = path.join(testTempDir, 'chats');
|
||||
fs.mkdirSync(chatsDir, { recursive: true });
|
||||
const missingFile = path.join(chatsDir, 'missing-session.jsonl');
|
||||
expect(fs.existsSync(missingFile)).toBe(false);
|
||||
|
||||
const inMemoryConversation = {
|
||||
sessionId: 'resumed-session-id',
|
||||
projectHash: 'resumed-project-hash',
|
||||
startTime: new Date().toISOString(),
|
||||
lastUpdated: new Date().toISOString(),
|
||||
messages: [
|
||||
{
|
||||
id: 'msg-1',
|
||||
type: 'user',
|
||||
timestamp: new Date().toISOString(),
|
||||
content: 'hello from memory',
|
||||
},
|
||||
],
|
||||
} as unknown as ConversationRecord;
|
||||
|
||||
await expect(
|
||||
chatRecordingService.initialize({
|
||||
filePath: missingFile,
|
||||
conversation: inMemoryConversation,
|
||||
}),
|
||||
).resolves.not.toThrow();
|
||||
|
||||
// The in-memory conversation is adopted.
|
||||
expect(chatRecordingService.getConversation()?.sessionId).toBe(
|
||||
'resumed-session-id',
|
||||
);
|
||||
|
||||
// A clean, loadable file is rewritten from the in-memory copy so future
|
||||
// loads and appends succeed.
|
||||
const reloaded = (await loadConversationRecord(
|
||||
missingFile,
|
||||
)) as ConversationRecord;
|
||||
expect(reloaded).not.toBeNull();
|
||||
expect(reloaded.sessionId).toBe('resumed-session-id');
|
||||
expect(reloaded.projectHash).toBe('resumed-project-hash');
|
||||
expect(reloaded.messages).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('should preserve an unreadable session file instead of destroying it', async () => {
|
||||
// The reload may have failed only transiently, so the original bytes
|
||||
// must survive the recovery rewrite.
|
||||
const chatsDir = path.join(testTempDir, 'chats');
|
||||
fs.mkdirSync(chatsDir, { recursive: true });
|
||||
const sessionFile = path.join(chatsDir, 'unreadable.jsonl');
|
||||
|
||||
// No usable metadata line => loadConversationRecord() returns null.
|
||||
const originalBytes = '{"not":"a valid metadata line"}\n';
|
||||
fs.writeFileSync(sessionFile, originalBytes);
|
||||
|
||||
await chatRecordingService.initialize({
|
||||
filePath: sessionFile,
|
||||
conversation: {
|
||||
sessionId: 'recovered-session-id',
|
||||
projectHash: 'recovered-project-hash',
|
||||
startTime: new Date().toISOString(),
|
||||
lastUpdated: new Date().toISOString(),
|
||||
messages: [],
|
||||
} as unknown as ConversationRecord,
|
||||
});
|
||||
|
||||
// The rewritten file is loadable again...
|
||||
const reloaded = (await loadConversationRecord(
|
||||
sessionFile,
|
||||
)) as ConversationRecord;
|
||||
expect(reloaded.sessionId).toBe('recovered-session-id');
|
||||
|
||||
// ...and the original bytes were kept alongside it.
|
||||
const preserved = fs
|
||||
.readdirSync(chatsDir)
|
||||
.filter((f) => f.startsWith('unreadable.jsonl.unreadable-'));
|
||||
expect(preserved).toHaveLength(1);
|
||||
expect(fs.readFileSync(path.join(chatsDir, preserved[0]), 'utf-8')).toBe(
|
||||
originalBytes,
|
||||
);
|
||||
});
|
||||
|
||||
it('should not leave a temp file behind when the rewrite fails', async () => {
|
||||
const chatsDir = path.join(testTempDir, 'chats');
|
||||
fs.mkdirSync(chatsDir, { recursive: true });
|
||||
const sessionFile = path.join(chatsDir, 'rewrite-fails.jsonl');
|
||||
|
||||
// Fail the rename that publishes the temp file, leaving it orphaned.
|
||||
const realRename = fs.renameSync;
|
||||
vi.spyOn(fs, 'renameSync').mockImplementation((from, to) => {
|
||||
if (String(from).includes('.tmp-')) {
|
||||
throw new Error('simulated rename failure');
|
||||
}
|
||||
return realRename(from, to);
|
||||
});
|
||||
|
||||
await expect(
|
||||
chatRecordingService.initialize({
|
||||
filePath: sessionFile,
|
||||
conversation: {
|
||||
sessionId: 'temp-cleanup-session',
|
||||
projectHash: 'temp-cleanup-hash',
|
||||
startTime: new Date().toISOString(),
|
||||
lastUpdated: new Date().toISOString(),
|
||||
messages: [],
|
||||
} as unknown as ConversationRecord,
|
||||
}),
|
||||
).rejects.toThrow('simulated rename failure');
|
||||
|
||||
const leftovers = fs
|
||||
.readdirSync(chatsDir)
|
||||
.filter((f) => f.includes('.tmp-'));
|
||||
expect(leftovers).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('recordMessage', () => {
|
||||
|
||||
@@ -462,7 +462,16 @@ export class ChatRecordingService {
|
||||
// Update the session ID in the existing file
|
||||
this.updateMetadata({ sessionId: this.sessionId });
|
||||
} else {
|
||||
throw new Error('Failed to load resumed session data from file');
|
||||
// The file could not be reloaded (missing, corrupt metadata, or an
|
||||
// I/O error). Fall back to the in-memory conversation we were handed
|
||||
// rather than failing the caller, and rewrite a clean file from it.
|
||||
debugLogger.warn(
|
||||
'Failed to reload resumed session data from file; falling back ' +
|
||||
'to the in-memory conversation.',
|
||||
);
|
||||
this.cachedConversation = resumedSessionData.conversation;
|
||||
this.projectHash = this.cachedConversation.projectHash;
|
||||
this.rewriteConversationFile(this.cachedConversation);
|
||||
}
|
||||
} else {
|
||||
// Create new session
|
||||
@@ -563,6 +572,73 @@ export class ChatRecordingService {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Rewrites the session file from an in-memory record. Any existing
|
||||
* (unreadable) file is preserved alongside rather than destroyed, and the
|
||||
* new file is written atomically (temp file + rename).
|
||||
*/
|
||||
private rewriteConversationFile(conversation: ConversationRecord): void {
|
||||
if (!this.conversationFile) return;
|
||||
|
||||
// Normalize legacy `.json` paths to the `.jsonl` format we write.
|
||||
if (this.conversationFile.endsWith('.json')) {
|
||||
this.conversationFile = this.conversationFile + 'l';
|
||||
}
|
||||
|
||||
const { messages, memoryScratchpad, ...metadata } = conversation;
|
||||
const lines: string[] = [JSON.stringify(metadata)];
|
||||
for (const msg of messages) {
|
||||
lines.push(JSON.stringify(msg));
|
||||
}
|
||||
if (memoryScratchpad) {
|
||||
lines.push(JSON.stringify({ $set: { memoryScratchpad } }));
|
||||
}
|
||||
const content = lines.join('\n') + '\n';
|
||||
|
||||
try {
|
||||
fs.mkdirSync(path.dirname(this.conversationFile), { recursive: true });
|
||||
|
||||
// The existing file was unreadable, but it may have been only
|
||||
// transiently so (a lock or I/O blip) rather than truly corrupt. Keep
|
||||
// its bytes rather than destroying them.
|
||||
if (fs.existsSync(this.conversationFile)) {
|
||||
const backup = `${this.conversationFile}.unreadable-${Date.now()}`;
|
||||
try {
|
||||
fs.renameSync(this.conversationFile, backup);
|
||||
debugLogger.warn(
|
||||
`Preserved the unreadable session file at ${backup}.`,
|
||||
);
|
||||
} catch (backupError) {
|
||||
debugLogger.error(
|
||||
'Failed to preserve the unreadable session file.',
|
||||
backupError,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const tempFile = `${this.conversationFile}.tmp-${process.pid}`;
|
||||
try {
|
||||
fs.writeFileSync(tempFile, content);
|
||||
fs.renameSync(tempFile, this.conversationFile);
|
||||
} catch (error) {
|
||||
// The rename did not complete, so the temp file would be left behind.
|
||||
try {
|
||||
fs.unlinkSync(tempFile);
|
||||
} catch {
|
||||
// Ignore cleanup errors so the original failure still surfaces.
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
} catch (error) {
|
||||
if (isNodeError(error) && error.code === 'ENOSPC') {
|
||||
this.conversationFile = null;
|
||||
debugLogger.warn(ENOSPC_WARNING_MESSAGE);
|
||||
} else {
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private updateMetadata(updates: Partial<ConversationRecord>): void {
|
||||
if (!this.cachedConversation) return;
|
||||
Object.assign(this.cachedConversation, updates);
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user