diff --git a/.changeset/allow-unknown-fields.md b/.changeset/allow-unknown-fields.md new file mode 100644 index 000000000..cf9db0878 --- /dev/null +++ b/.changeset/allow-unknown-fields.md @@ -0,0 +1,8 @@ +--- +"@googleworkspace/cli": minor +--- + +Add `--allow-unknown-fields` to raw API methods with JSON request bodies. Explicitly +allow fields absent from Discovery recursively, including in dry runs, while +preserving validation of known fields, required fields, JSON, URLs and file paths. +Handwritten helpers retain strict validation. diff --git a/.changeset/current-clippy-baseline.md b/.changeset/current-clippy-baseline.md new file mode 100644 index 000000000..a14e86242 --- /dev/null +++ b/.changeset/current-clippy-baseline.md @@ -0,0 +1,5 @@ +--- +"@googleworkspace/cli": patch +--- + +Keep Apps Script file selection compatible with the current Clippy checks. diff --git a/.changeset/docs-review-bundle.md b/.changeset/docs-review-bundle.md new file mode 100644 index 000000000..286e9e7f0 --- /dev/null +++ b/.changeset/docs-review-bundle.md @@ -0,0 +1,11 @@ +--- +"@googleworkspace/cli": minor +--- + +Add a standalone Python companion for visual Google Docs review bundles with +native exports, safe DOCX raster extraction, local HTML, optional PDF page +previews with explicitly unverified page coverage, revision observations, +and an offline fixture workflow. Preserve nested image occurrences and +legitimate asset reuse, and keep oversized optional comments from failing +the required bundle. Verify each export's exact canonical destination against +the real CLI receipt. diff --git a/.changeset/docs-review-workflow.md b/.changeset/docs-review-workflow.md new file mode 100644 index 000000000..eee647198 --- /dev/null +++ b/.changeset/docs-review-workflow.md @@ -0,0 +1,9 @@ +--- +"@googleworkspace/cli": minor +--- + +Add a standalone Python Docs review example that plans one literal text replacement, +binds it to a source revision and tab, applies it through existing gws commands, +and verifies the result without retrying ambiguous writes. Validate structures +and text ranges across all tabs before normalization, and preserve attempted or +confirmed mutation outcomes through final output failures and interruptions. diff --git a/.changeset/docs-structured-read.md b/.changeset/docs-structured-read.md new file mode 100644 index 000000000..41b070876 --- /dev/null +++ b/.changeset/docs-structured-read.md @@ -0,0 +1,9 @@ +--- +"@googleworkspace/cli": minor +--- + +Add `gws docs +read` to translate documents into compact structured content with +recursive tabs, headings and an outline, styled text, suggestions, nested tables, +figure metadata, and reference markers. Preserve API indices and revisions, +reject partial field masks, and support the existing formatters, sanitization, +and credential-free dry-run. diff --git a/.changeset/maintained-public-fork.md b/.changeset/maintained-public-fork.md new file mode 100644 index 000000000..874117e7b --- /dev/null +++ b/.changeset/maintained-public-fork.md @@ -0,0 +1,7 @@ +--- +"@googleworkspace/cli": patch +--- + +Publish the integrated Docs workflow improvements in the independent ratovarius +fork with upstream attribution, source-install instructions, contribution +tracking, and credential-independent CI. diff --git a/.changeset/offline-dry-run.md b/.changeset/offline-dry-run.md new file mode 100644 index 000000000..ca86d5669 --- /dev/null +++ b/.changeset/offline-dry-run.md @@ -0,0 +1,9 @@ +--- +"@googleworkspace/cli": patch +--- + +Skip authentication for Discovery-generated API and `docs +write` dry-runs. +Validate and preview requests without accessing the keyring or reading, changing, +or deleting stored credentials and token caches. Dry-runs work offline with a +fresh cached Discovery schema; schema fetching on first use or cache expiry is +unchanged. Real requests retain their existing authentication and error handling. diff --git a/.changeset/preserve-credentials.md b/.changeset/preserve-credentials.md new file mode 100644 index 000000000..4b7bafe37 --- /dev/null +++ b/.changeset/preserve-credentials.md @@ -0,0 +1,8 @@ +--- +"@googleworkspace/cli": patch +--- + +Preserve saved encrypted credentials and token caches when credential loading, +decryption, or keyring access fails. Report recovery guidance and stop authentication +instead of silently selecting plaintext credentials or another account through ADC. +Explicit token and credentials-file overrides and intentional logout remain unchanged. diff --git a/.changeset/scoped-file-roots.md b/.changeset/scoped-file-roots.md new file mode 100644 index 000000000..6e17edcc3 --- /dev/null +++ b/.changeset/scoped-file-roots.md @@ -0,0 +1,13 @@ +--- +"@googleworkspace/cli": minor +--- + +Allow operators to set `GOOGLE_WORKSPACE_CLI_FILE_ROOT` to an existing directory +for `--output` and `--upload` paths while keeping CWD confinement by default. +Relative CLI paths remain CWD-relative. Reject invalid roots, parent traversal +with an explicit root, control characters, and symlink escapes, including +dangling symlinks. Directory flags retain their existing boundaries. + +Reject canonical file paths that cannot be represented as UTF-8 at the CLI +string boundary, so explicit output/upload paths cannot silently become omitted +arguments. diff --git a/.changeset/yaml-empty-collections.md b/.changeset/yaml-empty-collections.md new file mode 100644 index 000000000..2ea985ea5 --- /dev/null +++ b/.changeset/yaml-empty-collections.md @@ -0,0 +1,7 @@ +--- +"@googleworkspace/cli": patch +--- + +Fix YAML mapping values containing empty arrays or objects by separating their +inline collection syntax from the mapping colon. This also fixes structured +Docs reader output with empty outlines, child tabs, or style maps. diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index 3c8b319f2..6252656b8 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -1,10 +1,3 @@ -# Codeowners - -# Core engine code strictly requires your review -# Isolates agents to `skills/` or `src/helpers/` unless absolutely necessary -/src/main.rs @jpoehnelt -/src/executor.rs @jpoehnelt -/src/discovery.rs @jpoehnelt -/src/commands.rs @jpoehnelt -/src/auth.rs @jpoehnelt -/src/schema.rs @jpoehnelt +# Review ownership for the independently maintained ratovarius fork. +# Original project authorship is preserved in FORK.md and the source history. +* @ratovarius diff --git a/.github/upstream-workflows/README.md b/.github/upstream-workflows/README.md new file mode 100644 index 000000000..8a0b025a7 --- /dev/null +++ b/.github/upstream-workflows/README.md @@ -0,0 +1,21 @@ +# Upstream automation reference + +These eleven workflow files are preserved unchanged from +`googleworkspace/cli` at `a3768d0e82ad83cca2da97724e46bea4ff0e6dbd`. +GitHub Actions only loads workflows from `.github/workflows`, so these archived +copies do not run in this fork. + +The upstream workflows include Google-specific CLA, bot, package publishing, +live-account smoke tests, and policy integrations. The fork instead runs +`fork-ci.yml`, `docs-review.yml`, and `docs-review-bundle.yml` with read-only +permissions and synthetic test data. They need no Google account credentials. + +This initial fork CI covers Rust tests/builds and the two Python companions on +Linux and macOS, plus Rust formatting and Clippy. It does not reproduce the +upstream release matrix, Nix checks, coverage reporting, dependency audit, or +scheduled skill regeneration. + +When syncing upstream, review changes here and any newly introduced active +workflows. Port useful checks deliberately; restore publishing only after +configuring a distinct fork release identity and destinations. See +[`CONTRIBUTING.md`](../../CONTRIBUTING.md). diff --git a/.github/workflows/audit.yml b/.github/upstream-workflows/audit.yml similarity index 100% rename from .github/workflows/audit.yml rename to .github/upstream-workflows/audit.yml diff --git a/.github/workflows/automation.yml b/.github/upstream-workflows/automation.yml similarity index 100% rename from .github/workflows/automation.yml rename to .github/upstream-workflows/automation.yml diff --git a/.github/workflows/ci.yml b/.github/upstream-workflows/ci.yml similarity index 100% rename from .github/workflows/ci.yml rename to .github/upstream-workflows/ci.yml diff --git a/.github/workflows/cla.yml b/.github/upstream-workflows/cla.yml similarity index 100% rename from .github/workflows/cla.yml rename to .github/upstream-workflows/cla.yml diff --git a/.github/workflows/coverage.yml b/.github/upstream-workflows/coverage.yml similarity index 100% rename from .github/workflows/coverage.yml rename to .github/upstream-workflows/coverage.yml diff --git a/.github/workflows/generate-skills.yml b/.github/upstream-workflows/generate-skills.yml similarity index 100% rename from .github/workflows/generate-skills.yml rename to .github/upstream-workflows/generate-skills.yml diff --git a/.github/workflows/policy.yml b/.github/upstream-workflows/policy.yml similarity index 100% rename from .github/workflows/policy.yml rename to .github/upstream-workflows/policy.yml diff --git a/.github/workflows/publish-skills.yml b/.github/upstream-workflows/publish-skills.yml similarity index 100% rename from .github/workflows/publish-skills.yml rename to .github/upstream-workflows/publish-skills.yml diff --git a/.github/workflows/release-changesets.yml b/.github/upstream-workflows/release-changesets.yml similarity index 100% rename from .github/workflows/release-changesets.yml rename to .github/upstream-workflows/release-changesets.yml diff --git a/.github/workflows/release.yml b/.github/upstream-workflows/release.yml similarity index 100% rename from .github/workflows/release.yml rename to .github/upstream-workflows/release.yml diff --git a/.github/workflows/stale.yml b/.github/upstream-workflows/stale.yml similarity index 100% rename from .github/workflows/stale.yml rename to .github/upstream-workflows/stale.yml diff --git a/.github/workflows/docs-review-bundle.yml b/.github/workflows/docs-review-bundle.yml new file mode 100644 index 000000000..a0f651c61 --- /dev/null +++ b/.github/workflows/docs-review-bundle.yml @@ -0,0 +1,54 @@ +# Copyright 2026 Google LLC +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +name: Docs Review Bundle + +on: + push: + branches: [main] + pull_request: + branches: [main] + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: ${{ github.ref != 'refs/heads/main' }} + +jobs: + docs-review-bundle: + name: Docs Review Bundle (Python) + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [ubuntu-latest, macos-latest] + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + persist-credentials: false + - name: Install Rust + uses: dtolnay/rust-toolchain@d1031067263f94b142dd6c0ce24c5eb9d02d52a0 # master + with: + toolchain: stable + - name: Cache cargo + uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 + with: + key: docs-review-bundle-${{ matrix.os }} + - name: Build CLI for export contract regression + run: cargo build --locked + - name: Test stdlib companion with synthetic fixtures + run: python3 -B -m unittest discover -s examples/docs-review-bundle -p 'test_*.py' -v + env: + GWS_TEST_BINARY: ${{ github.workspace }}/target/debug/gws diff --git a/.github/workflows/docs-review.yml b/.github/workflows/docs-review.yml new file mode 100644 index 000000000..a530b13f8 --- /dev/null +++ b/.github/workflows/docs-review.yml @@ -0,0 +1,46 @@ +# Copyright 2026 Google LLC +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +name: Docs Review Example + +on: + push: + branches: [main] + pull_request: + branches: [main] + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: ${{ github.ref != 'refs/heads/main' }} + +jobs: + docs-review: + name: Docs Review Python Example + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [ubuntu-latest, macos-latest] + env: + PYTHONDONTWRITEBYTECODE: "1" + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + persist-credentials: false + - name: Test with synthetic fixtures and stub gws + run: | + python3 --version + python3 -m unittest discover -s examples/docs-review -p 'test_*.py' -v diff --git a/.github/workflows/fork-ci.yml b/.github/workflows/fork-ci.yml new file mode 100644 index 000000000..a92b56149 --- /dev/null +++ b/.github/workflows/fork-ci.yml @@ -0,0 +1,64 @@ +name: Fork CI + +on: + push: + branches: [main] + pull_request: + branches: [main] + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: ${{ github.ref != 'refs/heads/main' }} + +env: + CARGO_TERM_COLOR: always + +jobs: + test: + name: Rust tests (${{ matrix.os }}) + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-latest] + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + persist-credentials: false + - name: Install Rust + uses: dtolnay/rust-toolchain@d1031067263f94b142dd6c0ce24c5eb9d02d52a0 # master + with: + toolchain: stable + - name: Cache cargo + uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 + with: + key: fork-test-${{ matrix.os }} + - name: Test workspace with synthetic fixtures + run: cargo test --workspace --locked + - name: Build workspace + run: cargo build --workspace --locked + + lint: + name: Rust formatting and Clippy + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + persist-credentials: false + - name: Install Rust + uses: dtolnay/rust-toolchain@d1031067263f94b142dd6c0ce24c5eb9d02d52a0 # master + with: + toolchain: stable + components: rustfmt, clippy + - name: Cache cargo + uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 + with: + key: fork-lint + - name: Check formatting + run: cargo fmt --all -- --check + - name: Clippy + run: cargo clippy --workspace --locked -- -D warnings diff --git a/AGENTS.md b/AGENTS.md index 722112264..8c2347d19 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,5 +1,9 @@ # AGENTS.md +This is the independently maintained `ratovarius/cli` fork. Read +[FORK.md](FORK.md) for upstream attribution and the feature ledger, and +[CONTRIBUTING.md](CONTRIBUTING.md) for fork PRs, upstream submissions, and CI. + ## Project Overview `gws` is a Rust CLI tool for interacting with Google Workspace APIs. It dynamically generates its command surface at runtime by parsing Google Discovery Service JSON documents. @@ -13,7 +17,7 @@ ## Build & Test > [!IMPORTANT] -> **Test Coverage**: The `codecov/patch` check requires that new or modified lines are covered by tests. When adding code, extract testable helper functions rather than embedding logic in `main`/`run` where it's hard to unit-test. Run `cargo test` locally and verify new branches are exercised. +> **Test Coverage**: Cover new or modified behavior with tests. When adding code, extract testable helper functions rather than embedding logic in `main`/`run` where it's hard to unit-test. Run `cargo test` locally and verify new branches are exercised. Upstream submissions may additionally require `codecov/patch`; this fork's active checks are documented in `CONTRIBUTING.md`. ```bash cargo build # Build in dev mode @@ -33,7 +37,7 @@ Every PR must include a changeset file. Create one at `.changeset/gws +

gws — ratovarius fork

+ +**An independently maintained public fork of +[googleworkspace/cli](https://github.com/googleworkspace/cli).** +Built on the original project's work, with additional Google Docs reading, +review, export, and credential-handling improvements. Original authorship, +history, and the Apache-2.0 license are preserved. + +See [what this fork adds and its upstream PRs](FORK.md), or +[contribute here](CONTRIBUTING.md). This fork can develop independently while +focused improvements are offered back to the original project. **One CLI for all of Google Workspace — built for humans and AI agents.**
Drive, Gmail, Calendar, and every Workspace API. Zero boilerplate. Structured JSON output. 40+ agent skills included. > [!NOTE] -> This is **not** an officially supported Google product. +> This fork is maintained by [ratovarius](https://github.com/ratovarius). +> It is **not** an officially supported Google product.

- npm version - license - CI status - install size + license + Fork CI status


-⬇️ **[Download the latest release for your OS](https://github.com/googleworkspace/cli/releases)** +**[Install this fork from source](#installation)** · +[Report an issue](https://github.com/ratovarius/cli/issues) `gws` doesn't ship a static list of commands. It reads Google's own [Discovery Service](https://developers.google.com/discovery) at runtime and builds its entire command surface dynamically. When Google Workspace adds an API endpoint or method, `gws` picks it up automatically. @@ -38,37 +48,38 @@ Drive, Gmail, Calendar, and every Workspace API. Zero boilerplate. Structured JS ## Prerequisites -- **Node.js 18+** — for `npm install` (or download a pre-built binary from [GitHub Releases](https://github.com/googleworkspace/cli/releases)) +- **Stable Rust and Cargo** — to build this fork from source +- **Python 3.10+** — for the optional Docs review and export companions (POSIX systems) - **A Google Cloud project** — required for OAuth credentials. You can create one via the [Google Cloud Console](https://console.cloud.google.com/) or with the [`gcloud` CLI](https://cloud.google.com/sdk/docs/install) or with the `gws auth setup` command. - **A Google account** with access to Google Workspace ## Installation -The recommended way to install `gws` is to download the pre-built binary for your OS and architecture from the **[GitHub Releases](https://github.com/googleworkspace/cli/releases)** page. Extract the archive and place the `gws` binary in your `$PATH`. - -For convenience, you can also use `npm` to automate downloading the appropriate binary from GitHub Releases: +Install the maintained fork's `main` from source: ```bash -npm install -g @googleworkspace/cli +cargo install --git https://github.com/ratovarius/cli --branch main --locked google-workspace-cli ``` -Or build from source: +To use the Docs review companions as well, keep a source checkout: ```bash -cargo install --git https://github.com/googleworkspace/cli --locked +git clone https://github.com/ratovarius/cli.git +cd cli +cargo build --workspace --locked +export PATH="$PWD/target/debug:$PATH" ``` -A Nix flake is also available at `github:googleworkspace/cli` +See [Docs review](examples/docs-review/README.md) and +[visual review bundles](examples/docs-review-bundle/README.md) for their commands. +Check `command -v gws` to confirm which installed binary your shell will use. -```bash -nix run github:googleworkspace/cli -``` - -On macOS and Linux, you can also install via [Homebrew](https://brew.sh/): - -```bash -brew install googleworkspace-cli -``` +This initial fork publication provides source on GitHub. The upstream +`@googleworkspace/cli` npm package, `google-workspace-cli` crates.io package, +Homebrew package, and [upstream releases](https://github.com/googleworkspace/cli/releases) +install the original distribution and do not include fork-only improvements. +For a reproducible source build, replace `--branch main` with `--rev COMMIT_SHA` +using the fork commit you have reviewed. ## Quick Start @@ -104,6 +115,60 @@ gws schema drive.files.list gws drive files list --params '{"pageSize": 100}' --page-all | jq -r '.files[].name' ``` +Discovery-generated API commands and `gws docs +write` support credential-free +`--dry-run`: they validate inputs and display the request without obtaining a +token, accessing the keyring, reading or changing stored credentials, or sending +the API request. + +```bash +# Preview a Docs append without signing in +gws docs +write --document DOC_ID --text 'Hello, world!' --dry-run +``` + +These previews work offline with a fresh cached Discovery schema (24-hour TTL). +First use or an expired cache can still fetch the schema over the network. +Other helpers may need authenticated reads to prepare their plans; this guarantee +applies to raw API commands and `docs +write`. + +### Fields absent from Discovery + +Raw API methods with a request body accept `--allow-unknown-fields` alongside +`--json`. Use it explicitly when an API supports fields that its public Discovery +document does not yet describe. It allows unknown properties recursively, +including nested objects and array elements, and forwards their values unchanged. +JSON is still parsed and serialized normally; whitespace and key order may change. + +Validation remains strict by default. With the flag, known-field types, enums and +required fields are still checked, as are JSON syntax, required URL parameters and +file paths. It does not allow new enum values on a known field. The flag is local +to raw methods and does not apply to handwritten `+` helpers. + +For example, Docs suggestions and comments require a Cloud project enrolled in the +[Google Workspace Developer Preview Program](https://developers.google.com/workspace/preview). +Google still enforces API availability, OAuth scopes, document permissions and +server-side validation. This flag grants no additional access. + +```bash +# Preview a suggested insertion (Docs Developer Preview). +gws docs documents batchUpdate \ + --params '{"documentId":"DOCUMENT_ID"}' \ + --json '{"requests":[{"insertText":{"location":{"index":1},"text":"Suggested text"}}],"writeControl":{"writeMode":"SUGGEST"}}' \ + --allow-unknown-fields --dry-run + +# Preview a comment anchored to existing text; adjust the range for your document. +gws docs documents batchUpdate \ + --params '{"documentId":"DOCUMENT_ID"}' \ + --json '{"requests":[{"insertComment":{"content":"Please review this text.","range":{"startIndex":1,"endIndex":5}}}]}' \ + --allow-unknown-fields --dry-run +``` + +`--dry-run` uses the same validation policy and shows the request without sending +it. It cannot verify preview enrollment or server acceptance. Remove `--dry-run` +to submit a request. See the Docs +[request reference](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents/request#InsertCommentRequest) +for preview field requirements. + + ## Authentication The CLI supports multiple auth workflows so it works on your laptop, in CI, and on a server. @@ -215,17 +280,37 @@ export GOOGLE_WORKSPACE_CLI_TOKEN=$(gcloud auth print-access-token) Environment variables can also live in a `.env` file. +### Troubleshooting saved credentials + +If `gws` cannot read or decrypt `credentials.enc` (including a keyring access +failure), it returns an authentication error and preserves that file, +`token_cache.json`, and `sa_token_cache.json`. It does not silently switch to +plaintext credentials or Application Default Credentials (ADC). This applies to +the default configuration directory and `GOOGLE_WORKSPACE_CLI_CONFIG_DIR`. + +Check that you are using the original configuration directory and can access its +original OS keyring or encryption key. Back up the configuration before changing +key storage or replacing credentials. Preservation does not recover a lost key. +If you intentionally want to discard saved credentials and sign in again, use +`gws auth logout` followed by `gws auth login`; logout still removes saved +credentials and token caches. + +An explicit `GOOGLE_WORKSPACE_CLI_TOKEN` or +`GOOGLE_WORKSPACE_CLI_CREDENTIALS_FILE` still takes precedence. A missing or invalid +explicit credentials file is an error. When no encrypted credentials file exists, +the usual plaintext and ADC fallback remains available. + ## AI Agent Skills The repo ships 100+ Agent Skills (`SKILL.md` files) — one for every supported API, plus higher-level helpers for common workflows and 50 curated recipes for Gmail, Drive, Docs, Calendar, and Sheets. See the full [Skills Index](docs/skills.md) for the complete list. ```bash # Install all skills at once -npx skills add https://github.com/googleworkspace/cli +npx skills add https://github.com/ratovarius/cli # Or pick only what you need -npx skills add https://github.com/googleworkspace/cli/tree/main/skills/gws-drive -npx skills add https://github.com/googleworkspace/cli/tree/main/skills/gws-gmail +npx skills add https://github.com/ratovarius/cli/tree/main/skills/gws-drive +npx skills add https://github.com/ratovarius/cli/tree/main/skills/gws-gmail ```
@@ -239,7 +324,7 @@ ln -s $(pwd)/skills/gws-* ~/.openclaw/skills/ cp -r skills/gws-drive skills/gws-gmail ~/.openclaw/skills/ ``` -The `gws-shared` skill includes an `install` block so OpenClaw auto-installs the CLI via `npm` if `gws` isn't on PATH. +Install this fork using the source instructions above before using agent skills. An agent installer that uses the upstream npm package will install the original distribution.
@@ -253,7 +338,7 @@ The `gws-shared` skill includes an `install` block so OpenClaw auto-installs the 2. Install the extension into the Gemini CLI: ```bash - gemini extensions install https://github.com/googleworkspace/cli + gemini extensions install https://github.com/ratovarius/cli ``` Installing this extension gives your Gemini CLI agent direct access to all `gws` commands and Google Workspace agent skills. Because `gws` handles its own authentication securely, you simply need to authenticate your terminal once prior to using the agent, and the extension will automatically inherit your credentials. @@ -266,6 +351,39 @@ Installing this extension gives your Gemini CLI agent direct access to all `gws` gws drive files create --json '{"name": "report.pdf"}' --upload ./report.pdf ``` +### Output and upload file roots + +`--output` and `--upload` accept paths within the current working directory +(CWD) by default, including absolute paths that resolve inside CWD. To allow +files elsewhere, set a trusted operator environment variable to an existing +directory: + +```bash +mkdir -p /tmp/gws-files +export GOOGLE_WORKSPACE_CLI_FILE_ROOT=/tmp/gws-files +gws drive files get --params '{"fileId":"FILE_ID","alt":"media"}' \ + --output /tmp/gws-files/report.pdf +gws drive files create --json '{"name":"report.pdf"}' \ + --upload /tmp/gws-files/report.pdf +``` + +The root replaces the allowed file boundary; relative CLI paths still resolve +from CWD. For example, `--output report.pdf` is rejected if CWD is outside the +configured root. The root is canonicalized and must exist as a directory; an +empty or invalid value fails validation. Relative root settings resolve from +CWD too. With an explicit root, CLI paths containing `..` components are +rejected. Control characters and symlinks escaping the boundary are rejected; +symlinks resolving inside it are allowed, but dangling symlinks are rejected. +These CLI file flags require a UTF-8 canonical path. If a symlink resolves to a +path with unsupported encoding, the command returns a validation error rather +than dropping the upload or selecting the default output file. + +This setting affects only these file flags, not `--dir` or `--output-dir`. +It does not create parent directories or change the default download filename +when `--output` is omitted. Validation cannot prevent another local process +from replacing a path component between validation and I/O; choose a root +whose directories you control. Unset the variable to restore the CWD boundary. + ### Pagination | Flag | Description | Default | @@ -382,6 +500,7 @@ All variables are optional. See [`.env.example`](.env.example) for a copy-paste | `GOOGLE_WORKSPACE_CLI_CLIENT_ID` | OAuth client ID (alternative to `client_secret.json`) | | `GOOGLE_WORKSPACE_CLI_CLIENT_SECRET` | OAuth client secret (paired with `CLIENT_ID`) | | `GOOGLE_WORKSPACE_CLI_CONFIG_DIR` | Override config directory (default: `~/.config/gws`) | +| `GOOGLE_WORKSPACE_CLI_FILE_ROOT` | Existing directory allowed for `--output` / `--upload` paths (default: CWD); relative CLI paths remain CWD-relative | | `GOOGLE_WORKSPACE_CLI_SANITIZE_TEMPLATE` | Default Model Armor template | | `GOOGLE_WORKSPACE_CLI_SANITIZE_MODE` | `warn` (default) or `block` | | `GOOGLE_WORKSPACE_CLI_LOG` | Log level for stderr (e.g., `gws=debug`). Off by default. | diff --git a/SECURITY.md b/SECURITY.md index 07bc436f3..13c40a975 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -1,6 +1,12 @@ # Report a security issue -To report a security issue, please use [https://g.co/vulnz](https://g.co/vulnz). We use -[https://g.co/vulnz](https://g.co/vulnz) for our intake, and do coordination and disclosure here on -GitHub (including using GitHub Security Advisory). The Google Security Team will -respond within 5 working days of your report on [https://g.co/vulnz](https://g.co/vulnz). +For vulnerabilities in this independently maintained fork, use +[GitHub's private vulnerability reporting](https://github.com/ratovarius/cli/security/advisories/new). +Include the affected commit, reproduction steps, and impact. Do not include +real OAuth tokens, document contents, or other credentials in a public issue. + +The fork maintainer coordinates reports for changes maintained here. This fork +does not promise a Google Security Team response or a fixed response time. + +For vulnerabilities also affecting the original project, follow the +[upstream security policy](https://github.com/googleworkspace/cli/security/policy). diff --git a/crates/google-workspace-cli/Cargo.toml b/crates/google-workspace-cli/Cargo.toml index 058b109e6..78929df0d 100644 --- a/crates/google-workspace-cli/Cargo.toml +++ b/crates/google-workspace-cli/Cargo.toml @@ -18,8 +18,8 @@ version = "0.22.5" edition = "2021" description = "Google Workspace CLI — dynamic command surface from Discovery Service" license = "Apache-2.0" -repository = "https://github.com/googleworkspace/cli" -homepage = "https://github.com/googleworkspace/cli" +repository = "https://github.com/ratovarius/cli" +homepage = "https://github.com/ratovarius/cli" readme = "README.md" authors = ["Justin Poehnelt"] keywords = ["cli", "google-workspace", "google", "drive", "gmail"] diff --git a/crates/google-workspace-cli/README.md b/crates/google-workspace-cli/README.md index 901fae46a..ae06d9d31 100644 --- a/crates/google-workspace-cli/README.md +++ b/crates/google-workspace-cli/README.md @@ -4,18 +4,20 @@ `gws` dynamically generates its command surface at runtime by reading Google's [Discovery Service](https://developers.google.com/discovery). Drive, Gmail, Calendar, and every Workspace API — zero boilerplate, structured JSON output, 40+ agent skills included. -## Install +## Install this fork -Download the pre-built binary for your OS and architecture from the **[GitHub Releases](https://github.com/googleworkspace/cli/releases)** page. - -Alternatively, you can use package managers as a convenience layer: +This is the independently maintained [ratovarius fork](https://github.com/ratovarius/cli) +of [googleworkspace/cli](https://github.com/googleworkspace/cli). +Original authorship and the Apache-2.0 license are retained. ```bash -npm install -g @googleworkspace/cli # npm (downloads GitHub release binary) -cargo install google-workspace-cli # crates.io -nix run github:googleworkspace/cli # nix +cargo install --git https://github.com/ratovarius/cli --branch main --locked google-workspace-cli ``` +Upstream registry packages and binary releases do not include fork-only changes. +See the [fork ledger](https://github.com/ratovarius/cli/blob/main/FORK.md) +for improvements and upstream PRs. + ## Quick Start ```bash @@ -26,8 +28,8 @@ gws gmail users.messages list --params '{"maxResults": 3}' ## Documentation -See the [full README](https://github.com/googleworkspace/cli#readme) for authentication setup, helper commands, agent skills, and more. +See the [full README](https://github.com/ratovarius/cli#readme) for authentication setup, helper commands, agent skills, and more. ## License -Apache-2.0 — see [LICENSE](https://github.com/googleworkspace/cli/blob/main/LICENSE). +Apache-2.0 — see [LICENSE](https://github.com/ratovarius/cli/blob/main/LICENSE). diff --git a/crates/google-workspace-cli/src/auth.rs b/crates/google-workspace-cli/src/auth.rs index 9d8847e4b..6b27e3f92 100644 --- a/crates/google-workspace-cli/src/auth.rs +++ b/crates/google-workspace-cli/src/auth.rs @@ -336,6 +336,22 @@ async fn load_credentials_inner( env_file: Option<&str>, enc_path: &std::path::Path, default_path: &std::path::Path, +) -> anyhow::Result { + load_credentials_with_loader( + env_file, + enc_path, + default_path, + credential_store::load_encrypted_from_path, + ) + .await +} + +// Keep credential selection testable without accessing the OS keyring. +async fn load_credentials_with_loader( + env_file: Option<&str>, + enc_path: &std::path::Path, + default_path: &std::path::Path, + load_encrypted: impl FnOnce(&std::path::Path) -> anyhow::Result, ) -> anyhow::Result { // 1. Explicit env var — plaintext file (User or Service Account) if let Some(path) = env_file { @@ -353,40 +369,21 @@ async fn load_credentials_inner( // 2. Encrypted credentials if enc_path.exists() { - match credential_store::load_encrypted_from_path(enc_path) { - Ok(json_str) => { - return parse_credential_file(enc_path, &json_str).await; - } - Err(e) => { - // Decryption failed — the encryption key likely changed (e.g. after - // an upgrade that migrated keys between keyring and file storage). - // Remove the stale file so the next `gws auth login` starts fresh, - // and fall through to other credential sources (plaintext, ADC). - eprintln!( - "Warning: removing undecryptable credentials file ({}): {e:#}", - enc_path.display() - ); - if let Err(err) = tokio::fs::remove_file(enc_path).await { - eprintln!( - "Warning: failed to remove stale credentials file '{}': {err}", - enc_path.display() - ); - } - // Also remove stale token caches that used the old key. - for cache_file in ["token_cache.json", "sa_token_cache.json"] { - let path = enc_path.with_file_name(cache_file); - if let Err(err) = tokio::fs::remove_file(&path).await { - if err.kind() != std::io::ErrorKind::NotFound { - eprintln!( - "Warning: failed to remove stale token cache '{}': {err}", - path.display() - ); - } - } - } - // Fall through to remaining credential sources below. - } - } + // A read, decryption, or keyring failure does not mean the files are + // disposable. Stop here so a retry cannot silently select another account. + // Do not render backend error details, which may contain sensitive data. + let json_str = load_encrypted(enc_path).map_err(|_| { + anyhow::anyhow!( + "Failed to read or decrypt saved credentials at {}. \ + Check access to the original OS keyring or encryption key and verify \ + GOOGLE_WORKSPACE_CLI_CONFIG_DIR. Credentials and token caches have been \ + preserved; no fallback credentials were used. Back up the configuration \ + before intentionally replacing credentials with `gws auth logout` and \ + `gws auth login`.", + crate::output::sanitize_for_terminal(&enc_path.display().to_string()) + ) + })?; + return parse_credential_file(enc_path, &json_str).await; } // 3. Plaintext credentials at default path (AuthorizedUser) @@ -825,75 +822,214 @@ mod tests { #[tokio::test] #[serial_test::serial] - async fn test_load_credentials_corrupt_encrypted_file_is_removed() { - // When credentials.enc cannot be decrypted, the file should be removed - // automatically and the function should fall through to other sources. - let tmp = tempfile::tempdir().unwrap(); - let _home_guard = EnvVarGuard::set("HOME", tmp.path()); - let _adc_guard = EnvVarGuard::remove("GOOGLE_APPLICATION_CREDENTIALS"); + async fn test_load_credentials_preserves_failed_encrypted_credentials_and_caches() { + let dir = tempfile::tempdir().unwrap(); + let enc_path = dir.path().join("credentials.enc"); + let token_path = dir.path().join("token_cache.json"); + let service_token_path = dir.path().join("sa_token_cache.json"); + let absent_path = dir.path().join("missing.json"); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &absent_path); + + // A short invalid payload fails before accessing any OS keyring. + std::fs::write(&enc_path, b"bad").unwrap(); + std::fs::write(&token_path, b"synthetic-user-cache").unwrap(); + std::fs::write(&service_token_path, b"synthetic-service-cache").unwrap(); + + for _ in 0..2 { + let result = load_credentials_inner(None, &enc_path, &absent_path).await; + + assert!(result.is_err()); + assert!( + enc_path.exists(), + "Authentication failure must preserve saved encrypted credentials" + ); + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); + assert_eq!(std::fs::read(&token_path).unwrap(), b"synthetic-user-cache"); + assert_eq!( + std::fs::read(&service_token_path).unwrap(), + b"synthetic-service-cache" + ); + } + } + #[tokio::test] + #[serial_test::serial] + async fn test_load_credentials_corrupt_encrypted_reports_safe_remediation() { let dir = tempfile::tempdir().unwrap(); let enc_path = dir.path().join("credentials.enc"); + let absent_path = dir.path().join("missing.json"); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &absent_path); + std::fs::write(&enc_path, b"bad").unwrap(); - // Write garbage data that cannot be decrypted. - tokio::fs::write(&enc_path, b"not-valid-encrypted-data-at-all-1234567890") + let err = load_credentials_inner(None, &enc_path, &absent_path) .await - .unwrap(); - assert!(enc_path.exists()); - - let result = - load_credentials_inner(None, &enc_path, &PathBuf::from("/does/not/exist")).await; + .unwrap_err(); + let msg = format!("{err:#}"); + + assert!(msg.contains("decrypt"), "{msg}"); + assert!(msg.contains("keyring"), "{msg}"); + assert!(msg.contains("preserved"), "{msg}"); + assert!(msg.contains("Back up"), "{msg}"); + assert!(!msg.contains("No credentials found"), "{msg}"); + assert!(!msg.contains("bad"), "{msg}"); + } - // Should fall through to "No credentials found" (not a decryption error). - assert!(result.is_err()); - let msg = result.unwrap_err().to_string(); - assert!( - msg.contains("No credentials found"), - "Should fall through to final error, got: {msg}" - ); - assert!( - !enc_path.exists(), - "Stale credentials.enc must be removed after decryption failure" - ); + #[tokio::test] + #[serial_test::serial] + async fn test_load_credentials_corrupt_encrypted_blocks_plaintext_and_adc() { + // Exercise both a default-style layout and a configured directory, + // without changing HOME or touching the actual default directory. + let dir = tempfile::tempdir().unwrap(); + let fallback_json = r#"{ + "client_id": "different-account", + "client_secret": "synthetic-secret", + "refresh_token": "synthetic-refresh", + "type": "authorized_user" + }"#; + let adc_path = dir.path().join("adc.json"); + std::fs::write(&adc_path, fallback_json).unwrap(); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &adc_path); + + for layout in [".config/gws", "custom-config"] { + let config = dir.path().join(layout); + std::fs::create_dir_all(&config).unwrap(); + let enc_path = config.join("credentials.enc"); + let plain_path = config.join("credentials.json"); + std::fs::write(&enc_path, b"bad").unwrap(); + std::fs::write(&plain_path, fallback_json).unwrap(); + + for fallback in [&plain_path, &config.join("missing.json")] { + let err = load_credentials_inner(None, &enc_path, fallback) + .await + .expect_err("Broken encrypted credentials must block another account"); + assert!(err.to_string().contains("decrypt")); + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); + assert_eq!(std::fs::read_to_string(&plain_path).unwrap(), fallback_json); + assert_eq!(std::fs::read_to_string(&adc_path).unwrap(), fallback_json); + } + } } #[tokio::test] #[serial_test::serial] - async fn test_load_credentials_corrupt_encrypted_falls_through_to_plaintext() { - // When credentials.enc is corrupt but a valid plaintext file exists, - // the function should fall through and use the plaintext credentials. + async fn test_load_credentials_keyring_failure_preserves_files_and_blocks_adc() { let dir = tempfile::tempdir().unwrap(); let enc_path = dir.path().join("credentials.enc"); let plain_path = dir.path().join("credentials.json"); + let adc_path = dir.path().join("adc.json"); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &adc_path); + let sentinels: &[(&str, &[u8])] = &[ + ("credentials.enc", b"synthetic-encrypted-credentials"), + ("token_cache.json", b"synthetic-user-cache"), + ("sa_token_cache.json", b"synthetic-service-cache"), + (".encryption_key", b"synthetic-key"), + ]; + for (name, bytes) in sentinels { + std::fs::write(dir.path().join(name), bytes).unwrap(); + } + std::fs::write( + &adc_path, + r#"{"type":"authorized_user","client_id":"other","client_secret":"secret","refresh_token":"refresh"}"#, + ) + .unwrap(); - // Write garbage encrypted data. - tokio::fs::write(&enc_path, b"not-valid-encrypted-data-at-all-1234567890") + for _ in 0..2 { + let err = load_credentials_with_loader(None, &enc_path, &plain_path, |_| { + anyhow::bail!("OS keyring unavailable: synthetic-sensitive-detail") + }) .await - .unwrap(); + .expect_err("Key acquisition failure must stop credential selection"); + for (name, bytes) in sentinels { + assert_eq!(std::fs::read(dir.path().join(name)).unwrap(), *bytes); + } + let msg = format!("{err:#}"); + assert!(msg.contains("keyring"), "{msg}"); + assert!(msg.contains("preserved"), "{msg}"); + assert!(!msg.contains("synthetic-sensitive-detail"), "{msg}"); + assert!(!msg.contains("synthetic-key"), "{msg}"); + } + } - // Write valid plaintext credentials. - let plain_json = r#"{ - "client_id": "fallback_id", - "client_secret": "fallback_secret", - "refresh_token": "fallback_refresh", - "type": "authorized_user" - }"#; - tokio::fs::write(&plain_path, plain_json).await.unwrap(); + #[tokio::test] + #[serial_test::serial] + async fn test_load_credentials_explicit_file_precedes_broken_encrypted_credentials() { + let dir = tempfile::tempdir().unwrap(); + let enc_path = dir.path().join("credentials.enc"); + let explicit_path = dir.path().join("explicit.json"); + let missing_path = dir.path().join("missing.json"); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &missing_path); + std::fs::write(&enc_path, b"bad").unwrap(); + std::fs::write( + &explicit_path, + r#"{"type":"authorized_user","client_id":"explicit","client_secret":"secret","refresh_token":"refresh"}"#, + ) + .unwrap(); - let res = load_credentials_inner(None, &enc_path, &plain_path) + let creds = load_credentials_inner(explicit_path.to_str(), &enc_path, &missing_path) .await .unwrap(); + match creds { + Credential::AuthorizedUser(secret) => assert_eq!(secret.client_id, "explicit"), + _ => panic!("Expected explicitly selected account"), + } + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); - match res { - Credential::AuthorizedUser(secret) => { - assert_eq!( - secret.client_id, "fallback_id", - "Should fall through to plaintext credentials" - ); + // A missing or malformed explicit file must not select another account. + for contents in [None, Some("invalid-json")] { + if let Some(contents) = contents { + std::fs::write(&missing_path, contents).unwrap(); } - _ => panic!("Expected AuthorizedUser from plaintext fallback"), + let err = load_credentials_inner(missing_path.to_str(), &enc_path, &explicit_path) + .await + .unwrap_err(); + assert!(err.to_string().contains(if contents.is_some() { + "Failed to parse" + } else { + "does not exist" + })); + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); + } + } + + #[tokio::test] + #[serial_test::serial] + async fn test_get_token_preserves_configured_credentials_until_explicit_logout() { + let dir = tempfile::tempdir().unwrap(); + let enc_path = dir.path().join("credentials.enc"); + let missing_path = dir.path().join("missing.json"); + let _config_guard = EnvVarGuard::set("GOOGLE_WORKSPACE_CLI_CONFIG_DIR", dir.path()); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &missing_path); + let _file_guard = EnvVarGuard::remove("GOOGLE_WORKSPACE_CLI_CREDENTIALS_FILE"); + let _token_guard = EnvVarGuard::set("GOOGLE_WORKSPACE_CLI_TOKEN", ""); + let names = [ + "credentials.enc", + "credentials.json", + "token_cache.json", + "sa_token_cache.json", + ]; + for name in names { + std::fs::write(dir.path().join(name), b"bad").unwrap(); + } + + let err = get_token(&[]).await.unwrap_err(); + assert!(err.to_string().contains("decrypt"), "{err}"); + for name in names { + assert_eq!(std::fs::read(dir.path().join(name)).unwrap(), b"bad"); + } + + // An explicit token still takes precedence over all credential files. + { + let _token_guard = EnvVarGuard::set("GOOGLE_WORKSPACE_CLI_TOKEN", "synthetic-token"); + assert_eq!(get_token(&[]).await.unwrap(), "synthetic-token"); + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); + } + + crate::auth_commands::handle_auth_command(&["logout".into()]) + .await + .unwrap(); + for name in names { + assert!(!dir.path().join(name).exists(), "Logout must remove {name}"); } - assert!(!enc_path.exists(), "Stale credentials.enc must be removed"); } #[tokio::test] diff --git a/crates/google-workspace-cli/src/commands.rs b/crates/google-workspace-cli/src/commands.rs index 27324e42b..e559ebe10 100644 --- a/crates/google-workspace-cli/src/commands.rs +++ b/crates/google-workspace-cli/src/commands.rs @@ -112,12 +112,20 @@ fn build_resource_command(name: &str, resource: &RestResource) -> Option, body_json: Option<&str>, is_media_upload: bool, + validation_policy: BodyValidationPolicy, ) -> Result { let params: Map = if let Some(p) = params_json { serde_json::from_str(p) @@ -112,7 +120,7 @@ fn parse_and_validate_inputs( if let Some(ref req_ref) = method.request { if let Some(ref schema_name) = req_ref.schema_ref { - validate_body_against_schema(&val, schema_name, doc)?; + validate_body_against_schema(&val, schema_name, doc, validation_policy)?; } } @@ -411,7 +419,54 @@ pub async fn execute_method( output_format: &crate::formatter::OutputFormat, capture_output: bool, ) -> Result, GwsError> { - let input = parse_and_validate_inputs(doc, method, params_json, body_json, upload.is_some())?; + execute_method_with_policy( + doc, + method, + params_json, + body_json, + token, + auth_method, + output_path, + upload, + dry_run, + pagination, + sanitize_template, + sanitize_mode, + output_format, + capture_output, + BodyValidationPolicy::Strict, + ) + .await +} + +/// Executes a raw API method with an explicit request-body validation policy. +/// Handwritten helpers use [`execute_method`] to retain strict validation. +#[allow(clippy::too_many_arguments)] +pub async fn execute_method_with_policy( + doc: &RestDescription, + method: &RestMethod, + params_json: Option<&str>, + body_json: Option<&str>, + token: Option<&str>, + auth_method: AuthMethod, + output_path: Option<&str>, + upload: Option>, + dry_run: bool, + pagination: &PaginationConfig, + sanitize_template: Option<&str>, + sanitize_mode: &crate::helpers::modelarmor::SanitizeMode, + output_format: &crate::formatter::OutputFormat, + capture_output: bool, + validation_policy: BodyValidationPolicy, +) -> Result, GwsError> { + let input = parse_and_validate_inputs( + doc, + method, + params_json, + body_json, + upload.is_some(), + validation_policy, + )?; if dry_run { let dry_run_info = json!({ @@ -996,9 +1051,10 @@ fn validate_body_against_schema( body: &Value, schema_name: &str, doc: &RestDescription, + validation_policy: BodyValidationPolicy, ) -> Result<(), GwsError> { let mut errors = Vec::new(); - validate_value(body, schema_name, doc, "$", &mut errors); + validate_value(body, schema_name, doc, "$", &mut errors, validation_policy); if !errors.is_empty() { return Err(GwsError::Validation(format!( @@ -1016,6 +1072,7 @@ fn validate_value( doc: &RestDescription, path: &str, errors: &mut Vec, + validation_policy: BodyValidationPolicy, ) { let schema = match doc.schemas.get(schema_ref_name) { Some(s) => s, @@ -1028,7 +1085,15 @@ fn validate_value( // If the top-level schema is an object if schema.schema_type.as_deref() == Some("object") || !schema.properties.is_empty() { if let Value::Object(obj) = value { - validate_properties(obj, &schema.properties, &schema.required, doc, path, errors); + validate_properties( + obj, + &schema.properties, + &schema.required, + doc, + path, + errors, + validation_policy, + ); } else { errors.push(format!("{path}: Expected object")); } @@ -1042,6 +1107,7 @@ fn validate_properties( doc: &RestDescription, path: &str, errors: &mut Vec, + validation_policy: BodyValidationPolicy, ) { let valid_keys: std::collections::HashSet<&String> = properties.keys().collect(); @@ -1060,15 +1126,24 @@ fn validate_properties( }; if !valid_keys.contains(key) { - errors.push(format!( - "{current_path}: Unknown property. Valid properties: {:?}", - valid_keys.iter().map(|k| k.as_str()).collect::>() - )); + if validation_policy == BodyValidationPolicy::Strict { + errors.push(format!( + "{current_path}: Unknown property. Valid properties: {:?}", + valid_keys.iter().map(|k| k.as_str()).collect::>() + )); + } continue; } let prop_schema = &properties[key]; - validate_property(val, prop_schema, doc, ¤t_path, errors); + validate_property( + val, + prop_schema, + doc, + ¤t_path, + errors, + validation_policy, + ); } } @@ -1078,10 +1153,11 @@ fn validate_property( doc: &RestDescription, path: &str, errors: &mut Vec, + validation_policy: BodyValidationPolicy, ) { // 1. Resolve $ref if present if let Some(ref_name) = &prop_schema.schema_ref { - validate_value(value, ref_name, doc, path, errors); + validate_value(value, ref_name, doc, path, errors, validation_policy); return; } @@ -1113,7 +1189,14 @@ fn validate_property( if let Value::Array(arr) = value { for (i, item) in arr.iter().enumerate() { let item_path = format!("{path}[{i}]"); - validate_property(item, items_schema, doc, &item_path, errors); + validate_property( + item, + items_schema, + doc, + &item_path, + errors, + validation_policy, + ); } } } @@ -1122,7 +1205,15 @@ fn validate_property( // 4. Object properties validation if prop_schema.prop_type.as_deref() == Some("object") && !prop_schema.properties.is_empty() { if let Value::Object(obj) = value { - validate_properties(obj, &prop_schema.properties, &[], doc, path, errors); + validate_properties( + obj, + &prop_schema.properties, + &[], + doc, + path, + errors, + validation_policy, + ); } } @@ -1186,6 +1277,332 @@ pub fn mime_to_extension(mime: &str) -> &str { } } +#[cfg(test)] +mod preview_fields_tests { + use super::*; + + fn fixture() -> (RestDescription, RestMethod) { + let doc = serde_json::from_value(json!({ + "name": "docs", + "version": "v1", + "rootUrl": "https://example.invalid/", + "servicePath": "v1/", + "schemas": { + "Body": { + "type": "object", + "required": ["name"], + "properties": { + "name": {"type": "string"}, + "mode": {"type": "string", "enum": ["ACTIVE"]}, + "count": {"type": "integer"}, + "tags": {"type": "array", "items": {"type": "string"}}, + "writeControl": { + "type": "object", + "properties": {"requiredRevisionId": {"type": "string"}} + }, + "child": {"$ref": "Child"}, + "requests": {"type": "array", "items": {"$ref": "Request"}}, + "children": { + "type": "array", + "items": {"type": "object", "properties": {"id": {"type": "string"}}} + } + } + }, + "Child": { + "type": "object", "required": ["id"], + "properties": {"id": {"type": "string"}} + }, + "Request": { + "type": "object", + "properties": { + "insertText": { + "type": "object", "properties": {"text": {"type": "string"}} + } + } + } + } + })) + .unwrap(); + let method = serde_json::from_value(json!({ + "httpMethod": "POST", + "path": "documents/{+documentId}:batchUpdate", + "parameterOrder": ["documentId"], + "parameters": { + "documentId": {"type": "string", "location": "path", "required": true}, + "view": {"type": "string", "location": "query", "required": true} + }, + "request": {"$ref": "Body"} + })) + .unwrap(); + (doc, method) + } + + const PARAMS: &str = r#"{"documentId":"test-document","view":"preview"}"#; + // Canonical JSON lets the request test assert exact emitted bytes as well as values. + const PREVIEW_BODY: &str = r#"{"name":"demo","preview":[null,true,1.25,9223372036854775807,{"text":"café\n\"quoted\""}],"requests":[{"insertComment":{"content":"Review"}}],"writeControl":{"writeMode":"SUGGEST"}}"#; + + #[test] + fn unknown_properties_require_opt_in_at_every_depth() { + let (doc, _) = fixture(); + for (body, path) in [ + (json!({"name": "demo", "preview": true}), "preview"), + ( + json!({"name": "demo", "writeControl": {"writeMode": "SUGGEST"}}), + "writeControl.writeMode", + ), + ( + json!({"name": "demo", "child": {"id": "one", "preview": null}}), + "child.preview", + ), + ( + json!({"name": "demo", "requests": [{"insertComment": {"content": "Review"}}]}), + "requests[0].insertComment", + ), + ( + json!({"name": "demo", "children": [{"id": "one", "preview": [1, true]}]}), + "children[0].preview", + ), + ] { + let err = + validate_body_against_schema(&body, "Body", &doc, BodyValidationPolicy::Strict) + .unwrap_err(); + assert!(err + .to_string() + .contains(&format!("{path}: Unknown property"))); + let result = validate_body_against_schema( + &body, + "Body", + &doc, + BodyValidationPolicy::AllowUnknownFields, + ); + assert!(result.is_ok(), "{path}: {result:?}"); + } + } + + #[test] + fn opt_in_preserves_known_field_and_required_validation() { + let (doc, _) = fixture(); + for (body, expected) in [ + ( + json!({"name": 42, "preview": true}), + "name: Expected type 'string'", + ), + ( + json!({"name": "demo", "count": 1.5, "preview": true}), + "count: Expected type 'integer'", + ), + ( + json!({"name": "demo", "mode": "PREVIEW", "preview": true}), + "not a valid enum member", + ), + (json!({"preview": true}), "Missing required property 'name'"), + ( + json!({"name": "demo", "child": {"preview": true}}), + "child: Missing required property 'id'", + ), + ( + json!({"name": "demo", "child": []}), + "child: Expected object", + ), + ( + json!({"name": "demo", "writeControl": {"requiredRevisionId": 42, "preview": true}}), + "writeControl.requiredRevisionId: Expected type 'string'", + ), + ( + json!({"name": "demo", "requests": [{"insertText": {"text": 42}, "preview": true}]}), + "requests[0].insertText.text: Expected type 'string'", + ), + ( + json!({"name": "demo", "children": [{"id": 42, "preview": true}]}), + "children[0].id: Expected type 'string'", + ), + ( + json!({"name": "demo", "tags": [true], "preview": true}), + "tags[0]: Expected type 'string'", + ), + ( + json!({"name": "demo", "requests": {}, "preview": true}), + "requests: Expected type 'array'", + ), + (json!([]), "$: Expected object"), + ] { + for policy in [ + BodyValidationPolicy::Strict, + BodyValidationPolicy::AllowUnknownFields, + ] { + let err = validate_body_against_schema(&body, "Body", &doc, policy).unwrap_err(); + assert!(err.to_string().contains(expected), "{policy:?}: {err}"); + } + } + } + + #[test] + fn opt_in_preserves_missing_schema_errors() { + let (doc, _) = fixture(); + let err = validate_body_against_schema( + &json!({}), + "Missing", + &doc, + BodyValidationPolicy::AllowUnknownFields, + ) + .unwrap_err(); + assert!(err.to_string().contains("Schema 'Missing' not found")); + } + + #[tokio::test] + async fn opt_in_dry_run_preserves_preview_body() { + let (doc, method) = fixture(); + let output = execute_method_with_policy( + &doc, + &method, + Some(PARAMS), + Some(PREVIEW_BODY), + None, + AuthMethod::None, + None, + None, + true, + &PaginationConfig::default(), + None, + &crate::helpers::modelarmor::SanitizeMode::Warn, + &crate::formatter::OutputFormat::Json, + true, + BodyValidationPolicy::AllowUnknownFields, + ) + .await + .unwrap() + .unwrap(); + assert_eq!( + output, + json!({ + "dry_run": true, + "url": "https://example.invalid/v1/documents/test%2Ddocument:batchUpdate", + "method": "POST", + "query_params": [["view", "preview"]], + "body": { + "name": "demo", + "preview": [null, true, 1.25, 9223372036854775807_i64, {"text": "café\n\"quoted\""}], + "requests": [{"insertComment": {"content": "Review"}}], + "writeControl": {"writeMode": "SUGGEST"} + }, + "is_multipart_upload": false + }) + ); + } + + #[tokio::test] + async fn helper_executor_entry_point_remains_strict() { + let (doc, method) = fixture(); + let err = execute_method( + &doc, + &method, + Some(PARAMS), + Some(PREVIEW_BODY), + None, + AuthMethod::None, + None, + None, + true, + &PaginationConfig::default(), + None, + &crate::helpers::modelarmor::SanitizeMode::Warn, + &crate::formatter::OutputFormat::Json, + true, + ) + .await + .unwrap_err(); + assert!(err.to_string().contains("Unknown property")); + } + + #[tokio::test] + #[serial_test::serial] + async fn opt_in_preserves_request_body_bytes() { + let (doc, method) = fixture(); + let input = parse_and_validate_inputs( + &doc, + &method, + Some(PARAMS), + Some(PREVIEW_BODY), + false, + BodyValidationPolicy::AllowUnknownFields, + ) + .unwrap(); + // Avoid native roots and all credential lookup. Build only; never send. + let client = reqwest::Client::builder() + .tls_built_in_root_certs(false) + .build() + .unwrap(); + let previous_project = std::env::var_os("GOOGLE_WORKSPACE_PROJECT_ID"); + std::env::set_var("GOOGLE_WORKSPACE_PROJECT_ID", "test-project"); + let request = build_http_request( + &client, + &method, + &input, + None, + &AuthMethod::None, + None, + 0, + &None, + ) + .await; + match previous_project { + Some(value) => std::env::set_var("GOOGLE_WORKSPACE_PROJECT_ID", value), + None => std::env::remove_var("GOOGLE_WORKSPACE_PROJECT_ID"), + } + let request = request.unwrap().build().unwrap(); + assert_eq!( + request.body().unwrap().as_bytes().unwrap(), + PREVIEW_BODY.as_bytes() + ); + assert_eq!(request.method(), reqwest::Method::POST); + assert_eq!( + request.url().as_str(), + "https://example.invalid/v1/documents/test%2Ddocument:batchUpdate?view=preview" + ); + assert!(!request.headers().contains_key("authorization")); + } + + #[test] + fn opt_in_preserves_json_parameter_and_url_errors() { + let (doc, method) = fixture(); + for (params, body, expected) in [ + (Some(PARAMS), "{", "Invalid --json body"), + (Some("{"), PREVIEW_BODY, "Invalid --params JSON"), + (Some("[]"), PREVIEW_BODY, "Invalid --params JSON"), + (None, PREVIEW_BODY, "Required path parameter documentId"), + ( + Some(r#"{"documentId":"test-document"}"#), + PREVIEW_BODY, + "Required parameter 'view'", + ), + ( + Some(r#"{"documentId":"../secret","view":"preview"}"#), + PREVIEW_BODY, + "path traversal", + ), + ( + Some(r#"{"documentId":"document?injected=true","view":"preview"}"#), + PREVIEW_BODY, + "must not contain '?'", + ), + ] { + let result = parse_and_validate_inputs( + &doc, + &method, + params, + Some(body), + false, + BodyValidationPolicy::AllowUnknownFields, + ); + let err = result.err().expect("unsafe input must fail"); + assert!( + err.to_string().contains(expected), + "expected {expected}: {err}" + ); + } + } +} + #[cfg(test)] mod tests { use super::*; @@ -1253,7 +1670,9 @@ mod tests { }; let body = json!({ "name": "My File" }); - assert!(validate_body_against_schema(&body, "File", &doc).is_ok()); + assert!( + validate_body_against_schema(&body, "File", &doc, BodyValidationPolicy::Strict).is_ok() + ); } #[test] @@ -1283,7 +1702,8 @@ mod tests { }; let body = json!({ "name": "My File", "invalidField": 123 }); - let result = validate_body_against_schema(&body, "File", &doc); + let result = + validate_body_against_schema(&body, "File", &doc, BodyValidationPolicy::Strict); assert!(result.is_err()); assert!(result.unwrap_err().to_string().contains("Unknown property")); } @@ -1373,23 +1793,39 @@ mod tests { "tags": ["one", "two"], "parent": { "id": "123" } }); - assert!(validate_body_against_schema(&body, "File", &doc).is_ok()); + assert!( + validate_body_against_schema(&body, "File", &doc, BodyValidationPolicy::Strict).is_ok() + ); // Missing Required Field let body_missing = json!({ "name": "My File" }); - let err = validate_body_against_schema(&body_missing, "File", &doc).unwrap_err(); + let err = + validate_body_against_schema(&body_missing, "File", &doc, BodyValidationPolicy::Strict) + .unwrap_err(); assert!(err .to_string() .contains("Missing required property 'status'")); // Invalid Enum Value let body_bad_enum = json!({ "name": "My File", "status": "UNKNOWN" }); - let err = validate_body_against_schema(&body_bad_enum, "File", &doc).unwrap_err(); + let err = validate_body_against_schema( + &body_bad_enum, + "File", + &doc, + BodyValidationPolicy::Strict, + ) + .unwrap_err(); assert!(err.to_string().contains("not a valid enum member")); // Invalid Type let body_bad_type = json!({ "name": "My File", "status": "ACTIVE", "count": "10" }); - let err = validate_body_against_schema(&body_bad_type, "File", &doc).unwrap_err(); + let err = validate_body_against_schema( + &body_bad_type, + "File", + &doc, + BodyValidationPolicy::Strict, + ) + .unwrap_err(); assert!(err .to_string() .contains("Expected type 'integer', found string")); @@ -1400,12 +1836,20 @@ mod tests { "status": "ACTIVE", "parent": { "invalidField": "123" } }); - let err = validate_body_against_schema(&body_bad_ref, "File", &doc).unwrap_err(); + let err = + validate_body_against_schema(&body_bad_ref, "File", &doc, BodyValidationPolicy::Strict) + .unwrap_err(); assert!(err.to_string().contains("Unknown property")); // Expected Object Type Failure let body_not_object = json!([]); - let err = validate_body_against_schema(&body_not_object, "File", &doc).unwrap_err(); + let err = validate_body_against_schema( + &body_not_object, + "File", + &doc, + BodyValidationPolicy::Strict, + ) + .unwrap_err(); assert!(err.to_string().contains("Expected object")); } #[tokio::test] diff --git a/crates/google-workspace-cli/src/formatter.rs b/crates/google-workspace-cli/src/formatter.rs index 08d4d287a..57ae406af 100644 --- a/crates/google-workspace-cli/src/formatter.rs +++ b/crates/google-workspace-cli/src/formatter.rs @@ -319,7 +319,10 @@ fn json_to_yaml(value: &Value, indent: usize) -> String { match val { Value::Object(_) | Value::Array(_) => { let val_str = json_to_yaml(val, indent + 1); - let _ = write!(out, "\n{prefix}{key}:{val_str}"); + // Empty collections use inline flow syntax and need a + // space after the colon; block collections start a line. + let separator = if val_str.starts_with('\n') { "" } else { " " }; + let _ = write!(out, "\n{prefix}{key}:{separator}{val_str}"); } _ => { let val_str = json_to_yaml(val, indent); @@ -637,6 +640,24 @@ mod tests { assert!(output.contains("count: 42")); } + #[test] + fn test_format_yaml_empty_collections_as_mapping_values() { + let value = json!({"array": [], "object": {}, "tail": true}); + assert_eq!( + format_value(&value, &OutputFormat::Yaml), + "\narray: []\nobject: {}\ntail: true" + ); + } + + #[test] + fn test_format_yaml_empty_collections_nested_in_sequences() { + let value = json!({"items": [[], {}, {"array": [], "object": {}}, "tail"]}); + assert_eq!( + format_value(&value, &OutputFormat::Yaml), + "\nitems:\n - []\n - {}\n - \n array: []\n object: {}\n - \"tail\"" + ); + } + #[test] fn test_format_table_empty_array() { let val = json!({"files": []}); diff --git a/crates/google-workspace-cli/src/generate_skills.rs b/crates/google-workspace-cli/src/generate_skills.rs index ce2654077..e99591b4a 100644 --- a/crates/google-workspace-cli/src/generate_skills.rs +++ b/crates/google-workspace-cli/src/generate_skills.rs @@ -747,8 +747,8 @@ gws [sub-resource] [flags] ## Community & Feedback Etiquette -- Encourage agents/users to star the repository when they find the project useful: `https://github.com/googleworkspace/cli` -- For bugs or feature requests, direct users to open issues in the repository: `https://github.com/googleworkspace/cli/issues` +- This is the independently maintained `https://github.com/ratovarius/cli` fork of `https://github.com/googleworkspace/cli`; see `FORK.md` for attribution and upstream contributions. +- For bugs or feature requests, direct users to open issues in the fork: `https://github.com/ratovarius/cli/issues` - Before creating a new issue, **always** search existing issues and feature requests first - If a matching issue already exists, add context by commenting on the existing thread instead of creating a duplicate "# diff --git a/crates/google-workspace-cli/src/helpers/docs.rs b/crates/google-workspace-cli/src/helpers/docs.rs index d3ef7fa21..c42c02feb 100644 --- a/crates/google-workspace-cli/src/helpers/docs.rs +++ b/crates/google-workspace-cli/src/helpers/docs.rs @@ -21,14 +21,21 @@ use serde_json::json; use std::future::Future; use std::pin::Pin; +mod read; + pub struct DocsHelper; +#[cfg(test)] +#[path = "docs/read_tests.rs"] +mod read_tests; + impl Helper for DocsHelper { fn inject_commands( &self, mut cmd: Command, _doc: &crate::discovery::RestDescription, ) -> Command { + cmd = cmd.subcommand(read::command()); cmd = cmd.subcommand( Command::new("+write") .about("[Helper] Append text to a document") @@ -63,17 +70,26 @@ TIPS: &'a self, doc: &'a crate::discovery::RestDescription, matches: &'a ArgMatches, - _sanitize_config: &'a crate::helpers::modelarmor::SanitizeConfig, + sanitize_config: &'a crate::helpers::modelarmor::SanitizeConfig, ) -> Pin> + Send + 'a>> { Box::pin(async move { + if let Some(matches) = matches.subcommand_matches("+read") { + read::handle(doc, matches, sanitize_config).await?; + return Ok(true); + } if let Some(matches) = matches.subcommand_matches("+write") { let (params_str, body_str, scopes) = build_write_request(matches, doc)?; let scope_strs: Vec<&str> = scopes.iter().map(|s| s.as_str()).collect(); - let (token, auth_method) = match auth::get_token(&scope_strs).await { - Ok(t) => (Some(t), executor::AuthMethod::OAuth), - Err(_) if matches.get_flag("dry-run") => (None, executor::AuthMethod::None), - Err(e) => return Err(GwsError::Auth(format!("Docs auth failed: {e}"))), + let dry_run = matches.get_flag("dry-run"); + // Skip auth entirely: even failed auth can mutate stored credentials. + let (token, auth_method) = if dry_run { + (None, executor::AuthMethod::None) + } else { + match auth::get_token(&scope_strs).await { + Ok(t) => (Some(t), executor::AuthMethod::OAuth), + Err(e) => return Err(GwsError::Auth(format!("Docs auth failed: {e}"))), + } }; // Method: documents.batchUpdate @@ -100,7 +116,7 @@ TIPS: auth_method, None, None, - matches.get_flag("dry-run"), + dry_run, &pagination, None, &crate::helpers::modelarmor::SanitizeMode::Warn, diff --git a/crates/google-workspace-cli/src/helpers/docs/read.rs b/crates/google-workspace-cli/src/helpers/docs/read.rs new file mode 100644 index 000000000..4e2e8625c --- /dev/null +++ b/crates/google-workspace-cli/src/helpers/docs/read.rs @@ -0,0 +1,541 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Translate Docs structure, rather than render its visual layout. Keep API +//! indices and metadata; never calculate edit offsets from extracted text. + +use crate::discovery::RestDescription; +use crate::error::GwsError; +use crate::executor::{self, AuthMethod, PaginationConfig}; +use crate::formatter::{format_value, OutputFormat}; +use crate::helpers::modelarmor::SanitizeConfig; +use clap::{Arg, ArgMatches, Command}; +use serde_json::{json, Map, Value}; +use std::future::Future; + +pub(super) fn command() -> Command { + Command::new("+read") + .about("[Helper] Read a document as compact structured content") + .arg(Arg::new("document").long("document").help("Document ID").required(true).value_name("ID")) + .arg(Arg::new("params").long("params").help("Additional documents.get API parameters as JSON").value_name("JSON")) + .after_help( + "\ +EXAMPLES: + gws docs +read --document DOC_ID + gws docs +read --document DOC_ID --format yaml + gws docs +read --document DOC_ID --params '{\"fields\":\"*\"}' --dry-run + gws docs +read --document DOC_ID | jq '.outline' + gws docs +read --document DOC_ID | jq '.. | objects | select(.paragraphStyle?.headingId? == \"HEADING_ID\")' + +TIPS: + Requests all tabs with includeTabsContent=true and suggestionsViewMode=SUGGESTIONS_INLINE. + Only those tab/suggestion options are supported; fields must be absent or exactly \"*\". + Other documents.get options pass through --params; alt must be json. $fields is rejected. + JSON/YAML preserve the structured view; table/CSV use the global formatter's array summary. + tabs[].blocks and childTabs keep API order; outline lists headings with tab IDs and JSON Pointer paths. + Paragraph text concatenates text runs only. elements retain styles, links, suggestion IDs and reference markers. + Tables contain rows[].cells[].blocks recursively. Headers, footers and footnotes have separate blocks. + figures contain image/drawing metadata, including alt text and URIs when returned; images are never downloaded. + Unknown blocks, inline elements and tab types retain type=unknown markers and raw data. + startIndex/endIndex are the API's UTF-16 offsets, scoped to each tab/segment; never offsets into extracted text. + revisionId and suggestionsViewMode are retained when returned. Missing revisionId is not synthesized. + source=legacyBody indicates a fallback response without populated tabs; all-tab coverage cannot be confirmed. + This is a content view, not a layout renderer or lossless API round trip. Inherited styles are not resolved. + Suggestions remain inline, including proposed deletions; this helper does not accept or reject suggestions. + Use raw documents get for unsupported views or field masks. Missing body content produces an error. + --dry-run validates and prints a request plan without acquiring credentials or fetching document content. + --sanitize uses the existing Model Armor policy before normalization and retains _sanitization metadata.", + ) +} + +pub(super) async fn handle( + doc: &RestDescription, + matches: &ArgMatches, + sanitize: &SanitizeConfig, +) -> Result<(), GwsError> { + let output = run(doc, matches, sanitize, async { + crate::auth::get_token(&["https://www.googleapis.com/auth/documents.readonly"]) + .await + .map_err(|e| GwsError::Auth(format!("Docs auth failed: {e}"))) + }) + .await?; + println!("{output}"); + Ok(()) +} + +/// Token acquisition is lazy so local validation and dry-run never read +/// credentials. The executor remains responsible for HTTP and sanitization. +pub(super) async fn run( + doc: &RestDescription, + matches: &ArgMatches, + sanitize: &SanitizeConfig, + token: impl Future>, +) -> Result { + let params = build_params( + matches.get_one::("document").unwrap(), + matches.get_one::("params").map(String::as_str), + )?; + let method = doc + .resources + .get("documents") + .and_then(|r| r.methods.get("get")) + .ok_or_else(|| GwsError::Discovery("Method 'documents.get' not found".into()))?; + let dry_run = matches.get_flag("dry-run"); + let token = if dry_run { None } else { Some(token.await?) }; + let format = matches + .get_one::("format") + .map(|f| OutputFormat::from_str(f)) + .unwrap_or_default(); + let result = executor::execute_method( + doc, + method, + Some(¶ms.to_string()), + None, + token.as_deref(), + if token.is_some() { + AuthMethod::OAuth + } else { + AuthMethod::None + }, + None, + None, + dry_run, + &PaginationConfig::default(), + sanitize.template.as_deref(), + &sanitize.mode, + &format, + true, + ) + .await? + .ok_or_else(|| invalid_content("expected a JSON document response"))?; + let output = if dry_run { result } else { normalize(&result)? }; + Ok(format_value(&output, &format)) +} + +pub(super) fn build_params(document: &str, params: Option<&str>) -> Result { + crate::validate::validate_resource_name(document)?; + let mut params: Map = match params { + Some(raw) => serde_json::from_str(raw) + .map_err(|e| GwsError::Validation(format!("Invalid --params JSON object: {e}")))?, + None => Map::new(), + }; + // A complete response is required for normalization. A narrow allowlist is + // deliberate: parsing arbitrary nested masks cannot prove completeness as + // the Docs API grows. Reject the system-parameter alias as well. + if params.contains_key("$fields") || params.get("fields").is_some_and(|v| v != "*") { + return Err(GwsError::Validation( + "docs +read requires all content: omit fields or use \"*\"; $fields is unsupported" + .into(), + )); + } + for (key, required) in [ + ("documentId", json!(document)), + ("includeTabsContent", json!(true)), + ("suggestionsViewMode", json!("SUGGESTIONS_INLINE")), + ] { + if params.get(key).is_some_and(|v| v != &required) { + return Err(GwsError::Validation(format!( + "docs +read requires {key}={required}" + ))); + } + params.insert(key.into(), required); + } + if params.get("alt").is_some_and(|v| v != "json") { + return Err(GwsError::Validation("docs +read requires alt=json".into())); + } + Ok(Value::Object(params)) +} + +fn invalid_content(detail: &str) -> GwsError { + GwsError::Validation(format!( + "Incomplete or invalid Docs response: {detail}; use raw documents get to inspect it" + )) +} + +fn object(value: &Value) -> Result, GwsError> { + value + .as_object() + .cloned() + .ok_or_else(|| invalid_content("expected an object")) +} + +fn array<'a>(value: &'a Value, field: &str) -> Result<&'a [Value], GwsError> { + value + .get(field) + .and_then(Value::as_array) + .map(Vec::as_slice) + .ok_or_else(|| invalid_content(&format!("missing or invalid {field} array"))) +} + +const TAB_CONTENT: &[&str] = &[ + "body", + "headers", + "footers", + "footnotes", + "inlineObjects", + "positionedObjects", + "lists", + "namedStyles", + "namedRanges", + "suggestedNamedStylesChanges", +]; + +pub(super) fn normalize(document: &Value) -> Result { + let mut result = object(document)?; + let mut outline = Vec::new(); + let tabs = match document.get("tabs") { + Some(_) => array(document, "tabs")?, + None => &[], + }; + let (source, tabs) = if tabs.is_empty() { + let mut tab = Map::from_iter([ + ("tabId".into(), Value::Null), + ("parentTabId".into(), Value::Null), + ("childTabs".into(), json!([])), + ]); + if let Some(title) = document.get("title") { + tab.insert("title".into(), title.clone()); + } + let content: Map = TAB_CONTENT + .iter() + .filter_map(|key| document.get(key).map(|v| ((*key).into(), v.clone()))) + .collect(); + tab.extend(contents( + &Value::Object(content), + &Value::Null, + "/tabs/0", + &mut outline, + )?); + ("legacyBody", vec![Value::Object(tab)]) + } else { + let tabs = tabs + .iter() + .enumerate() + .map(|(i, tab)| normalize_tab(tab, &Value::Null, &format!("/tabs/{i}"), &mut outline)) + .collect::, _>>()?; + ("tabs", tabs) + }; + for field in TAB_CONTENT { + result.remove(*field); + } + result.insert("source".into(), json!(source)); + result.insert("tabs".into(), json!(tabs)); + result.insert("outline".into(), json!(outline)); + Ok(Value::Object(result)) +} + +fn normalize_tab( + tab: &Value, + parent: &Value, + path: &str, + outline: &mut Vec, +) -> Result { + let mut result = tab + .get("tabProperties") + .map(object) + .transpose()? + .unwrap_or_default(); + result + .entry("parentTabId") + .or_insert_with(|| parent.clone()); + let id = result.get("tabId").cloned().unwrap_or(Value::Null); + if let Some(content) = tab.get("documentTab") { + result.extend(contents(content, &id, path, outline)?); + let mut extra = object(tab)?; + for key in ["tabProperties", "documentTab", "childTabs"] { + extra.remove(key); + } + if !extra.is_empty() { + result.insert("metadata".into(), Value::Object(extra)); + } + } else { + result.insert("type".into(), json!("unknown")); + result.insert("data".into(), tab.clone()); + } + let children = if tab.get("childTabs").is_some() { + array(tab, "childTabs")? + } else { + &[] + }; + let children = children + .iter() + .enumerate() + .map(|(i, tab)| normalize_tab(tab, &id, &format!("{path}/childTabs/{i}"), outline)) + .collect::, _>>()?; + result.insert("childTabs".into(), json!(children)); + Ok(Value::Object(result)) +} + +fn contents( + content: &Value, + tab_id: &Value, + path: &str, + outline: &mut Vec, +) -> Result, GwsError> { + let mut result = object(content)?; + let body = result + .remove("body") + .ok_or_else(|| invalid_content("missing body"))?; + result.insert( + "blocks".into(), + blocks( + array(&body, "content")?, + tab_id, + &format!("{path}/blocks"), + outline, + )?, + ); + let mut body_metadata = object(&body)?; + body_metadata.remove("content"); + if !body_metadata.is_empty() { + result.insert("bodyMetadata".into(), Value::Object(body_metadata)); + } + for kind in ["headers", "footers", "footnotes"] { + if let Some(segments) = result.get_mut(kind) { + let mut normalized = Map::new(); + for (id, segment) in object(segments)? { + let mut segment_result = object(&segment)?; + // JSON Pointer escaping, not URL escaping. + let pointer_id = id.replace('~', "~0").replace('/', "~1"); + segment_result.insert( + "blocks".into(), + blocks( + array(&segment, "content")?, + tab_id, + &format!("{path}/{kind}/{pointer_id}/blocks"), + outline, + )?, + ); + segment_result.remove("content"); + normalized.insert(id, Value::Object(segment_result)); + } + *segments = Value::Object(normalized); + } + } + let mut figures = Map::new(); + for (field, properties, placement) in [ + ("inlineObjects", "inlineObjectProperties", "inline"), + ( + "positionedObjects", + "positionedObjectProperties", + "positioned", + ), + ] { + if let Some(objects) = result.remove(field) { + for (id, value) in object(&objects)? { + let mut figure = object(&value)?; + if let Some(properties) = figure.remove(properties) { + figure.extend(object(&properties)?); + } + let kind = if figure + .get("embeddedObject") + .and_then(|v| v.get("imageProperties")) + .is_some() + { + "image" + } else if figure + .get("embeddedObject") + .and_then(|v| v.get("embeddedDrawingProperties")) + .is_some() + { + "drawing" + } else { + "unknown" + }; + figure.insert("type".into(), json!(kind)); + figure.insert("placement".into(), json!(placement)); + figure.entry("objectId").or_insert_with(|| json!(id)); + figures.insert(id, Value::Object(figure)); + } + } + } + if !figures.is_empty() { + result.insert("figures".into(), Value::Object(figures)); + } + Ok(result) +} + +/// Flatten a known union arm, retaining styles, suggestions, source indices and +/// future metadata fields. Unrecognized union arms are retained as raw markers. +fn payload(value: &Value, key: &str, kind: &str) -> Result, GwsError> { + let mut outer = object(value)?; + let mut inner = object( + &outer + .remove(key) + .ok_or_else(|| invalid_content("missing element"))?, + )?; + inner.extend(outer); + inner.insert("type".into(), json!(kind)); + Ok(inner) +} + +fn unknown(value: &Value) -> Value { + let mut marker = json!({"type": "unknown", "data": value}); + for key in ["startIndex", "endIndex"] { + if let Some(index) = value.get(key) { + marker[key] = index.clone(); + } + } + marker +} + +fn element(value: &Value) -> Result { + for (key, kind) in [ + ("textRun", "text"), + ("inlineObjectElement", "figure"), + ("footnoteReference", "footnoteReference"), + ("horizontalRule", "horizontalRule"), + ("pageBreak", "pageBreak"), + ("columnBreak", "columnBreak"), + ("equation", "equation"), + ("autoText", "autoText"), + ] { + if value.get(key).is_some() { + let mut result = payload(value, key, kind)?; + if key == "textRun" { + let text = result + .remove("content") + .filter(Value::is_string) + .ok_or_else(|| invalid_content("textRun missing content"))?; + result.insert("text".into(), text); + } else if key == "inlineObjectElement" { + if let Some(id) = result.remove("inlineObjectId") { + result.insert("objectId".into(), id); + } + } else if key == "autoText" { + // The source's `type` is content, distinct from our union tag. + if let Some(subtype) = value[key].get("type") { + result.insert("autoTextType".into(), subtype.clone()); + } + } + return Ok(Value::Object(result)); + } + } + Ok(unknown(value)) +} + +fn blocks( + content: &[Value], + tab_id: &Value, + path: &str, + outline: &mut Vec, +) -> Result { + content + .iter() + .enumerate() + .map(|(i, block)| { + let path = format!("{path}/{i}"); + if let Some(paragraph) = block.get("paragraph") { + let mut result = payload(block, "paragraph", "paragraph")?; + let elements = array(paragraph, "elements")? + .iter() + .map(element) + .collect::, _>>()?; + let text: String = elements + .iter() + .filter_map(|e| e.get("text").and_then(Value::as_str)) + .collect(); + result.insert("text".into(), json!(text)); + result.insert("elements".into(), json!(elements)); + if let Some(style) = paragraph.get("paragraphStyle") { + if let Some(level) = + style + .get("namedStyleType") + .and_then(Value::as_str) + .filter(|s| { + matches!( + *s, + "TITLE" + | "SUBTITLE" + | "HEADING_1" + | "HEADING_2" + | "HEADING_3" + | "HEADING_4" + | "HEADING_5" + | "HEADING_6" + ) + }) + { + let mut heading = + json!({"tabId": tab_id, "level": level, "text": text, "path": path}); + for key in ["startIndex", "endIndex"] { + if let Some(value) = block.get(key) { + heading[key] = value.clone(); + } + } + if let Some(id) = style.get("headingId") { + heading["headingId"] = id.clone(); + } + outline.push(heading); + } + } + Ok(Value::Object(result)) + } else if let Some(table) = block.get("table") { + let mut result = payload(block, "table", "table")?; + let rows = array(table, "tableRows")? + .iter() + .enumerate() + .map(|(r, row)| { + let mut normalized = object(row)?; + let cells = array(row, "tableCells")? + .iter() + .enumerate() + .map(|(c, cell)| { + let mut normalized = object(cell)?; + normalized.insert( + "blocks".into(), + blocks( + array(cell, "content")?, + tab_id, + &format!("{path}/rows/{r}/cells/{c}/blocks"), + outline, + )?, + ); + normalized.remove("content"); + Ok(Value::Object(normalized)) + }) + .collect::, GwsError>>()?; + normalized.remove("tableCells"); + normalized.insert("cells".into(), json!(cells)); + Ok(Value::Object(normalized)) + }) + .collect::, GwsError>>()?; + if let Some(count) = result.remove("rows") { + result.insert("rowCount".into(), count); + } + result.remove("tableRows"); + result.insert("rows".into(), json!(rows)); + Ok(Value::Object(result)) + } else if let Some(toc) = block.get("tableOfContents") { + let mut result = payload(block, "tableOfContents", "tableOfContents")?; + result.insert( + "blocks".into(), + blocks( + array(toc, "content")?, + tab_id, + &format!("{path}/blocks"), + outline, + )?, + ); + result.remove("content"); + Ok(Value::Object(result)) + } else if block.get("sectionBreak").is_some() { + payload(block, "sectionBreak", "sectionBreak").map(Value::Object) + } else { + Ok(unknown(block)) + } + }) + .collect::, _>>() + .map(Value::Array) +} diff --git a/crates/google-workspace-cli/src/helpers/docs/read_tests.rs b/crates/google-workspace-cli/src/helpers/docs/read_tests.rs new file mode 100644 index 000000000..d0152bf9f --- /dev/null +++ b/crates/google-workspace-cli/src/helpers/docs/read_tests.rs @@ -0,0 +1,651 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use crate::commands::build_cli; +use crate::discovery::RestDescription; +use crate::error::GwsError; +use crate::helpers::modelarmor::{SanitizeConfig, SanitizeMode}; +use serde_json::{json, Value}; + +use super::read; + +fn discovery() -> RestDescription { + serde_json::from_value(serde_json::json!({ + "name": "docs", "version": "v1", "rootUrl": "https://docs.example.invalid/", + "servicePath": "v1/", + "resources": {"documents": {"methods": {"get": { + "id": "docs.documents.get", "httpMethod": "GET", + "path": "documents/{documentId}", "parameterOrder": ["documentId"], + "parameters": {"documentId": {"type": "string", "location": "path", "required": true}}, + "scopes": ["https://www.googleapis.com/auth/documents.readonly"] + }}}} + })) + .unwrap() +} + +#[test] +fn registers_read_alongside_write_with_required_document() { + let cli = build_cli(&discovery()); + assert!(cli.find_subcommand("+write").is_some()); + assert!( + cli.find_subcommand("+read").is_some(), + "missing structured reader" + ); + let error = cli + .clone() + .try_get_matches_from(["gws", "+read"]) + .unwrap_err(); + assert_eq!( + error.kind(), + clap::error::ErrorKind::MissingRequiredArgument + ); + assert!(cli + .try_get_matches_from([ + "gws", + "+read", + "--document", + "synthetic", + "--params", + "{}", + "--format", + "yaml", + "--dry-run" + ]) + .is_ok()); +} + +fn matches(args: &[&str]) -> clap::ArgMatches { + build_cli(&discovery()).try_get_matches_from(args).unwrap() +} + +fn paragraph(text: &str) -> Value { + json!({"startIndex": 1, "endIndex": 5, "paragraph": { + "elements": [{"startIndex": 1, "endIndex": 5, "textRun": {"content": text}}] + }}) +} + +fn legacy() -> Value { + json!({ + "documentId": "synthetic", "title": "Example", "revisionId": "rev-1", + "suggestionsViewMode": "SUGGESTIONS_INLINE", + "body": {"content": [paragraph("Hi😀")] } + }) +} + +#[test] +fn auto_text_preserves_page_number_and_count_with_indices_and_styles() { + let mut outputs = Vec::new(); + for subtype in ["PAGE_NUMBER", "PAGE_COUNT"] { + let mut input = legacy(); + input["body"]["content"][0]["paragraph"]["elements"] = json!([{ + "startIndex": 0, "endIndex": 1, + "autoText": {"type": subtype, "textStyle": {"bold": true}, + "suggestedInsertionIds": ["s1"]} + }]); + let output = read::normalize(&input).unwrap(); + let element = &output["tabs"][0]["blocks"][0]["elements"][0]; + assert_eq!( + element, + &json!({ + "type": "autoText", "autoTextType": subtype, + "startIndex": 0, "endIndex": 1, + "textStyle": {"bold": true}, "suggestedInsertionIds": ["s1"] + }) + ); + outputs.push(output); + } + assert_ne!(outputs[0], outputs[1]); +} + +#[test] +fn request_requires_full_inline_tabs_and_preserves_other_params() { + let params = + read::build_params("synthetic", Some(r#"{"fields":"*","prettyPrint":false}"#)).unwrap(); + assert_eq!( + params, + json!({ + "documentId": "synthetic", "fields": "*", "prettyPrint": false, + "includeTabsContent": true, "suggestionsViewMode": "SUGGESTIONS_INLINE" + }) + ); + assert_eq!( + read::build_params("id", None).unwrap()["includeTabsContent"], + true + ); + assert!(read::build_params("id", Some( + r#"{"includeTabsContent":true,"suggestionsViewMode":"SUGGESTIONS_INLINE","documentId":"id"}"# + )).is_ok()); +} + +#[test] +fn request_rejects_partial_masks_lossy_views_and_parameter_bypasses() { + for params in [ + r#"{"fields":"title"}"#, + r#"{"fields":"tabs(documentTab/body/content)"}"#, + r#"{"fields":""}"#, + r#"{"fields":null}"#, + r#"{"fields": ["*"]}"#, + r#"{"includeTabsContent":false}"#, + r#"{"includeTabsContent":"true"}"#, + r#"{"suggestionsViewMode":"PREVIEW_WITHOUT_SUGGESTIONS"}"#, + r#"{"suggestionsViewMode":"DEFAULT_FOR_CURRENT_ACCESS"}"#, + r#"{"documentId":"other"}"#, + r#"{"alt":"media"}"#, + r#"{"$fields":"title"}"#, + "[]", + "null", + "{", + ] { + assert!(read::build_params("id", Some(params)).is_err(), "{params}"); + } + for id in [ + "", + "../../secret", + "id?fields=title", + "id#fragment", + "id\n", + "%2e%2e", + ] { + assert!(read::build_params(id, None).is_err(), "{id:?}"); + } +} + +#[test] +fn legacy_body_keeps_source_revision_text_and_utf16_indices() { + let output = read::normalize(&legacy()).unwrap(); + assert_eq!(output["documentId"], "synthetic"); + assert_eq!(output["revisionId"], "rev-1"); + assert_eq!(output["suggestionsViewMode"], "SUGGESTIONS_INLINE"); + assert_eq!(output["source"], "legacyBody"); + assert_eq!(output["tabs"][0]["tabId"], Value::Null); + let block = &output["tabs"][0]["blocks"][0]; + assert_eq!(block["type"], "paragraph"); + assert_eq!(block["text"], "Hi😀"); + assert_eq!(block["endIndex"], 5); + assert_eq!(block["elements"][0]["endIndex"], 5); + assert_eq!(block["elements"][0]["text"], "Hi😀"); +} + +#[test] +fn recursively_reads_tabs_and_child_tabs_in_api_order_without_duplicate_legacy_body() { + let mut input = legacy(); + input["tabs"] = json!([ + {"tabProperties": {"tabId": "a", "title": "First", "index": 0}, + "documentTab": {"body": {"content": [paragraph("first")]}}, + "childTabs": [{"tabProperties": {"tabId": "b", "title": "Child", "parentTabId": "a"}, + "documentTab": {"body": {"content": [paragraph("child")]}}, + "childTabs": [{"tabProperties": {"tabId": "c", "title": "Grandchild"}, + "documentTab": {"body": {"content": [paragraph("grandchild")]}}}]}]}, + {"tabProperties": {"tabId": "d", "title": "Last", "index": 1}, + "documentTab": {"body": {"content": [paragraph("last")]}}} + ]); + let output = read::normalize(&input).unwrap(); + assert_eq!(output["source"], "tabs"); + assert_eq!(output["tabs"].as_array().unwrap().len(), 2); + assert_eq!(output["tabs"][0]["blocks"][0]["text"], "first"); + assert_eq!(output["tabs"][0]["childTabs"][0]["parentTabId"], "a"); + assert_eq!( + output["tabs"][0]["childTabs"][0]["childTabs"][0]["parentTabId"], + "b" + ); + assert_eq!(output["tabs"][1]["tabId"], "d"); + input["tabs"] = json!([]); + assert_eq!(read::normalize(&input).unwrap()["source"], "legacyBody"); +} + +#[test] +fn preserves_styled_link_runs_and_inline_suggestion_metadata_in_outline_order() { + let input = json!({"documentId": "id", "title": "Styled", "body": {"content": [{ + "startIndex": 1, "endIndex": 9, "paragraph": { + "paragraphStyle": {"namedStyleType": "HEADING_2", "headingId": "h1"}, + "suggestedParagraphStyleChanges": {"s1": {"paragraphStyle": {"namedStyleType": "HEADING_1"}}}, + "elements": [ + {"startIndex": 1, "endIndex": 5, "textRun": { + "content": "Look", "textStyle": {"bold": true, "link": {"url": "https://example.invalid"}}, + "suggestedInsertionIds": ["s1"], + "suggestedTextStyleChanges": {"s2": {"textStyle": {"italic": true}}}}}, + {"startIndex": 5, "endIndex": 9, "textRun": { + "content": "here", "textStyle": {"italic": true, "link": {"heading": {"id": "h2", "tabId": "t2"}}}, + "suggestedDeletionIds": ["s3"]}} + ] + } + }, {"startIndex": 9, "endIndex": 14, "paragraph": { + "paragraphStyle": {"namedStyleType": "TITLE"}, "elements": [] + }}]}}); + let output = read::normalize(&input).unwrap(); + let block = &output["tabs"][0]["blocks"][0]; + assert_eq!(block["text"], "Lookhere"); + assert_eq!(block["paragraphStyle"]["headingId"], "h1"); + assert!(block["suggestedParagraphStyleChanges"]["s1"].is_object()); + assert_eq!(block["elements"][0]["textStyle"]["bold"], true); + assert_eq!( + block["elements"][0]["textStyle"]["link"]["url"], + "https://example.invalid" + ); + assert_eq!(block["elements"][0]["suggestedInsertionIds"], json!(["s1"])); + assert_eq!(block["elements"][1]["suggestedDeletionIds"], json!(["s3"])); + assert_eq!( + block["elements"][1]["textStyle"]["link"]["heading"]["tabId"], + "t2" + ); + assert!(block["elements"][0]["suggestedTextStyleChanges"]["s2"].is_object()); + assert_eq!( + output["outline"][0], + json!({ + "tabId": null, "level": "HEADING_2", "headingId": "h1", "text": "Lookhere", + "startIndex": 1, "endIndex": 9, "path": "/tabs/0/blocks/0" + }) + ); + assert_eq!(output["outline"][1]["level"], "TITLE"); +} + +#[test] +fn nested_tables_keep_row_cell_order_indices_styles_and_suggestions() { + let input = json!({"documentId": "id", "body": {"content": [{ + "startIndex": 10, "endIndex": 30, "table": {"rows": 1, "columns": 2, + "tableRows": [{"startIndex": 11, "endIndex": 29, "tableCells": [ + {"startIndex": 12, "endIndex": 25, "tableCellStyle": {"rowSpan": 1, "columnSpan": 1}, + "suggestedInsertionIds": ["cell-s"], "content": [ + paragraph("cell"), + {"startIndex": 17, "endIndex": 24, "table": {"rows": 1, "columns": 1, + "tableRows": [{"tableCells": [{"content": [paragraph("nested")]}]}]}} + ]}, + {"content": [paragraph("second")]} + ]}] + } + }, paragraph("after")]}}); + let output = read::normalize(&input).unwrap(); + let table = &output["tabs"][0]["blocks"][0]; + assert_eq!(table["type"], "table"); + assert_eq!(table["startIndex"], 10); + assert_eq!(table["rowCount"], 1); + assert_eq!(table["columns"], 2); + assert_eq!(table["rows"][0]["endIndex"], 29); + let cell = &table["rows"][0]["cells"][0]; + assert_eq!(cell["startIndex"], 12); + assert_eq!(cell["suggestedInsertionIds"], json!(["cell-s"])); + assert_eq!(cell["tableCellStyle"]["columnSpan"], 1); + assert_eq!(cell["blocks"][0]["text"], "cell"); + assert_eq!( + cell["blocks"][1]["rows"][0]["cells"][0]["blocks"][0]["text"], + "nested" + ); + assert_eq!(table["rows"][0]["cells"][1]["blocks"][0]["text"], "second"); + assert_eq!(output["tabs"][0]["blocks"][1]["text"], "after"); +} + +#[test] +fn figures_keep_references_and_metadata_even_without_content_uri() { + let mut input = legacy(); + input["body"]["content"][0]["paragraph"]["elements"] = json!([ + {"startIndex": 1, "endIndex": 2, "inlineObjectElement": { + "inlineObjectId": "image", "suggestedInsertionIds": ["s1"], "textStyle": {"baselineOffset": "SUPERSCRIPT"}}}, + {"startIndex": 2, "endIndex": 3, "inlineObjectElement": {"inlineObjectId": "missing"}} + ]); + input["body"]["content"][0]["paragraph"]["positionedObjectIds"] = json!(["drawing"]); + input["inlineObjects"] = json!({"image": { + "objectId": "image", "inlineObjectProperties": {"embeddedObject": { + "title": "Alt title", "description": "Alt text", "size": {"width": {"magnitude": 42, "unit": "PT"}}, + "imageProperties": {"sourceUri": "https://example.invalid/image.png"} + }}, "suggestedDeletionIds": ["s2"] + }}); + input["positionedObjects"] = json!({"drawing": { + "objectId": "drawing", "positionedObjectProperties": {"embeddedObject": {"embeddedDrawingProperties": {}}} + }}); + let output = read::normalize(&input).unwrap(); + let tab = &output["tabs"][0]; + assert_eq!(tab["blocks"][0]["elements"][0]["type"], "figure"); + assert_eq!(tab["blocks"][0]["elements"][0]["objectId"], "image"); + assert_eq!( + tab["blocks"][0]["elements"][0]["suggestedInsertionIds"], + json!(["s1"]) + ); + assert_eq!(tab["blocks"][0]["elements"][1]["objectId"], "missing"); + assert_eq!(tab["blocks"][0]["positionedObjectIds"], json!(["drawing"])); + assert_eq!(tab["figures"]["image"]["type"], "image"); + assert_eq!( + tab["figures"]["image"]["embeddedObject"]["description"], + "Alt text" + ); + assert!(tab["figures"]["image"]["embeddedObject"]["imageProperties"] + .get("contentUri") + .is_none()); + assert_eq!( + tab["figures"]["image"]["suggestedDeletionIds"], + json!(["s2"]) + ); + assert_eq!(tab["figures"]["drawing"]["placement"], "positioned"); +} + +#[test] +fn preserves_reference_markers_segments_unknown_blocks_and_unknown_inline_elements() { + let mut input = legacy(); + input["body"]["content"] = json!([ + {"endIndex": 1, "sectionBreak": {"sectionStyle": {"columnSeparatorStyle": "NONE"}}}, + {"startIndex": 1, "endIndex": 4, "paragraph": {"elements": [ + {"startIndex": 1, "endIndex": 2, "footnoteReference": {"footnoteId": "f1", "footnoteNumber": "1"}}, + {"startIndex": 2, "endIndex": 3, "person": {"personId": "p1"}}, + {"startIndex": 3, "endIndex": 4, "futureInline": {"label": "unrecognized"}} + ]}}, + {"startIndex": 4, "endIndex": 8, "futureBlock": {"content": "keep me"}}, + {"tableOfContents": {"content": [paragraph("toc")]}} + ]); + input["headers"] = json!({"h1": {"headerId": "h1", "content": [paragraph("header")]}}); + input["footers"] = json!({"f2": {"footerId": "f2", "content": [paragraph("footer")]}}); + input["footnotes"] = json!({"f1": {"footnoteId": "f1", "content": [paragraph("note")]}}); + input["namedStyles"] = json!({"styles": [{"namedStyleType": "NORMAL_TEXT"}]}); + let output = read::normalize(&input).unwrap(); + let tab = &output["tabs"][0]; + assert_eq!(tab["blocks"][0]["type"], "sectionBreak"); + assert!(tab["blocks"][0].get("startIndex").is_none()); + assert_eq!(tab["blocks"][1]["elements"][0]["type"], "footnoteReference"); + assert_eq!(tab["blocks"][1]["elements"][0]["footnoteId"], "f1"); + assert_eq!(tab["blocks"][1]["elements"][1]["type"], "unknown"); + assert_eq!( + tab["blocks"][1]["elements"][1]["data"]["person"]["personId"], + "p1" + ); + assert_eq!( + tab["blocks"][1]["elements"][2]["data"]["futureInline"]["label"], + "unrecognized" + ); + assert_eq!(tab["blocks"][2]["type"], "unknown"); + assert_eq!(tab["blocks"][2]["startIndex"], 4); + assert_eq!( + tab["blocks"][2]["data"]["futureBlock"]["content"], + "keep me" + ); + assert_eq!(tab["blocks"][3]["blocks"][0]["text"], "toc"); + assert_eq!(tab["headers"]["h1"]["blocks"][0]["text"], "header"); + assert_eq!(tab["footers"]["f2"]["blocks"][0]["text"], "footer"); + assert_eq!(tab["footnotes"]["f1"]["blocks"][0]["text"], "note"); + assert_eq!( + tab["namedStyles"]["styles"][0]["namedStyleType"], + "NORMAL_TEXT" + ); +} + +#[test] +fn rejects_missing_content_instead_of_claiming_empty_document() { + for input in [ + json!({"documentId": "id", "title": "metadata only"}), + json!({"body": {}}), + json!({"tabs": [{"tabProperties": {"tabId": "t"}, "documentTab": {}}]}), + json!({"tabs": [{"documentTab": {"body": {"content": "not an array"}}}]}), + json!({"tabs": "not an array", "body": {"content": []}}), + json!({"body": {"content": [{"table": {"rows": 1}}]}}), + ] { + assert!(read::normalize(&input).is_err(), "{input}"); + } + assert!(read::normalize(&json!({"body": {"content": []}})).is_ok()); +} + +#[test] +fn preserves_unknown_tab_and_sanitization_annotation() { + let output = read::normalize(&json!({ + "documentId": "id", "_sanitization": {"filterMatchState": "NO_MATCH_FOUND"}, + "tabs": [{"tabProperties": {"tabId": "future", "title": "Future"}, + "futureTab": {"content": "opaque"}}] + })) + .unwrap(); + assert_eq!( + output["_sanitization"]["filterMatchState"], + "NO_MATCH_FOUND" + ); + assert_eq!(output["tabs"][0]["type"], "unknown"); + assert_eq!(output["tabs"][0]["data"]["futureTab"]["content"], "opaque"); +} + +#[tokio::test] +async fn dry_run_uses_executor_plan_without_polling_auth_or_sanitize() { + let args = matches(&[ + "gws", + "+read", + "--document", + "a/b c", + "--dry-run", + "--params", + r#"{"prettyPrint":false}"#, + ]); + let result = read::run( + &discovery(), + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig { + template: Some("never-call".into()), + mode: SanitizeMode::Block, + }, + async { panic!("dry-run must not poll authentication") }, + ) + .await + .unwrap(); + let output: Value = serde_json::from_str(&result).unwrap(); + assert_eq!(output["dry_run"], true); + assert_eq!(output["method"], "GET"); + assert_eq!( + output["url"], + "https://docs.example.invalid/v1/documents/a%2Fb%20c" + ); + let query = output["query_params"].as_array().unwrap(); + assert!(query.contains(&json!(["includeTabsContent", "true"]))); + assert!(query.contains(&json!(["suggestionsViewMode", "SUGGESTIONS_INLINE"]))); + assert!(query.contains(&json!(["prettyPrint", "false"]))); + assert!(output.get("tabs").is_none()); +} + +#[tokio::test] +async fn rejects_partial_mask_before_polling_authentication() { + let args = matches(&[ + "gws", + "+read", + "--document", + "id", + "--params", + r#"{"fields":"title"}"#, + ]); + let result = read::run( + &discovery(), + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { panic!("invalid request must fail before authentication") }, + ) + .await; + assert!(matches!(result, Err(GwsError::Validation(_)))); +} + +#[tokio::test] +async fn propagates_auth_and_discovery_failures() { + let args = matches(&["gws", "+read", "--document", "id"]); + let result = read::run( + &discovery(), + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { Err(GwsError::Auth("synthetic failure".into())) }, + ) + .await; + assert!(matches!(result, Err(GwsError::Auth(_)))); + let result = read::run( + &RestDescription::default(), + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { panic!("missing method must fail before authentication") }, + ) + .await; + assert!(matches!(result, Err(GwsError::Discovery(_)))); +} + +// The transport is the only fake: real executor, request building, capture, +// normalization and formatting run against a loopback server with synthetic auth. +async fn serve(status: &str, body: String) -> (RestDescription, tokio::task::JoinHandle) { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let mut doc = discovery(); + doc.root_url = format!("http://{}/", listener.local_addr().unwrap()); + let response = format!( + "HTTP/1.1 {status}\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", + body.len() + ); + let task = tokio::spawn(async move { + let (mut stream, _) = listener.accept().await.unwrap(); + let mut request = Vec::new(); + loop { + let mut buffer = [0; 1024]; + let len = stream.read(&mut buffer).await.unwrap(); + assert_ne!(len, 0); + request.extend_from_slice(&buffer[..len]); + if request.windows(4).any(|w| w == b"\r\n\r\n") { + break; + } + } + stream.write_all(response.as_bytes()).await.unwrap(); + String::from_utf8(request).unwrap() + }); + (doc, task) +} + +// Stop the existing executor's quota lookup before it can read host config/ADC. +struct SyntheticQuota(Option); +impl SyntheticQuota { + fn new() -> Self { + let old = std::env::var_os("GOOGLE_WORKSPACE_PROJECT_ID"); + std::env::set_var("GOOGLE_WORKSPACE_PROJECT_ID", "synthetic-project"); + Self(old) + } +} +impl Drop for SyntheticQuota { + fn drop(&mut self) { + if let Some(old) = &self.0 { + std::env::set_var("GOOGLE_WORKSPACE_PROJECT_ID", old); + } else { + std::env::remove_var("GOOGLE_WORKSPACE_PROJECT_ID"); + } + } +} + +#[tokio::test] +#[serial_test::serial] +async fn executor_fetches_full_content_with_auth_and_honors_all_global_formats() { + let _quota = SyntheticQuota::new(); + for format in ["json", "yaml", "table", "csv"] { + let mut input = legacy(); + input["body"]["content"][0]["paragraph"]["elements"][0]["textRun"]["textStyle"] = json!({}); + let (doc, request) = serve("200 OK", input.to_string()).await; + let args = matches(&[ + "gws", + "+read", + "--document", + "synthetic", + "--format", + format, + ]); + let rendered = read::run( + &doc, + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { Ok("synthetic-token".into()) }, + ) + .await + .unwrap(); + let request = request.await.unwrap(); + assert!(request.starts_with("GET /v1/documents/synthetic?")); + assert!(request.contains("includeTabsContent=true")); + assert!(request.contains("suggestionsViewMode=SUGGESTIONS_INLINE")); + assert!(request.contains("authorization: Bearer synthetic-token\r\n")); + match format { + "json" => assert_eq!( + serde_json::from_str::(&rendered).unwrap()["tabs"][0]["blocks"][0]["text"], + "Hi😀" + ), + // Complete consumer-visible YAML, including the empty map and + // arrays that previously lacked a mapping-value separator. + "yaml" => assert_eq!( + rendered, + concat!( + "\ndocumentId: \"synthetic\"", + "\noutline: []", + "\nrevisionId: \"rev-1\"", + "\nsource: \"legacyBody\"", + "\nsuggestionsViewMode: \"SUGGESTIONS_INLINE\"", + "\ntabs:", + "\n - ", + "\n blocks:", + "\n - ", + "\n elements:", + "\n - ", + "\n endIndex: 5", + "\n startIndex: 1", + "\n text: \"Hi😀\"", + "\n textStyle: {}", + "\n type: \"text\"", + "\n endIndex: 5", + "\n startIndex: 1", + "\n text: \"Hi😀\"", + "\n type: \"paragraph\"", + "\n childTabs: []", + "\n parentTabId: null", + "\n tabId: null", + "\n title: \"Example\"", + "\ntitle: \"Example\"" + ) + ), + "table" => assert!(rendered.contains("─") && rendered.contains("blocks")), + "csv" => assert!( + rendered.lines().next().unwrap().contains("blocks,") + && rendered.contains("\"\"text\"\"") + ), + _ => unreachable!(), + } + } +} + +#[tokio::test] +#[serial_test::serial] +async fn executor_propagates_server_errors_and_rejects_non_document_responses() { + let _quota = SyntheticQuota::new(); + for status in ["403 Forbidden", "500 Internal Server Error"] { + let (doc, request) = serve( + status, + json!({"error": {"message": "synthetic denied"}}).to_string(), + ) + .await; + let args = matches(&["gws", "+read", "--document", "synthetic"]); + let result = read::run( + &doc, + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { Ok("synthetic-token".into()) }, + ) + .await; + match result.unwrap_err() { + GwsError::Api { code, message, .. } => { + assert_eq!(code, if status.starts_with("403") { 403 } else { 500 }); + assert_eq!(message, "synthetic denied"); + } + other => panic!("unexpected error: {other:?}"), + } + request.await.unwrap(); + } + for body in [r#"{"documentId":"id","title":"partial"}"#, "invalid JSON"] { + let (doc, request) = serve("200 OK", body.into()).await; + let args = matches(&["gws", "+read", "--document", "synthetic"]); + assert!(read::run( + &doc, + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { Ok("synthetic-token".into()) } + ) + .await + .is_err()); + request.await.unwrap(); + } +} diff --git a/crates/google-workspace-cli/src/helpers/gmail/mod.rs b/crates/google-workspace-cli/src/helpers/gmail/mod.rs index caeb8b6b0..27de3eb9c 100644 --- a/crates/google-workspace-cli/src/helpers/gmail/mod.rs +++ b/crates/google-workspace-cli/src/helpers/gmail/mod.rs @@ -3008,6 +3008,28 @@ mod tests { // --- Attachment tests --- + // Attachment parsing calls the public file validator. Default-policy tests + // must ignore and restore an inherited operator root, including on panic. + // All users of this guard are serialized with the other environment tests. + struct DefaultFileRoot(Option); + + impl DefaultFileRoot { + fn unset() -> Self { + let saved = Self(std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT")); + std::env::remove_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT"); + saved + } + } + + impl Drop for DefaultFileRoot { + fn drop(&mut self) { + match &self.0 { + Some(root) => std::env::set_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT", root), + None => std::env::remove_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), + } + } + } + fn make_attach_matches(args: &[&str]) -> ArgMatches { let cmd = Command::new("test").arg( Arg::new("attach") @@ -3095,14 +3117,18 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_rejects_control_chars() { + let _root = DefaultFileRoot::unset(); let matches = make_attach_matches(&["test", "-a", "file\0name.pdf"]); let err = parse_attachments(&matches).unwrap_err(); assert!(err.to_string().contains("control characters")); } #[test] + #[serial_test::serial] fn test_parse_attachments_rejects_directory() { + let _root = DefaultFileRoot::unset(); // Use a relative directory that exists in CWD let matches = make_attach_matches(&["test", "-a", "src"]); let err = parse_attachments(&matches).unwrap_err(); @@ -3110,14 +3136,18 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_empty_returns_empty_vec() { + let _root = DefaultFileRoot::unset(); let matches = make_attach_matches(&["test"]); let attachments = parse_attachments(&matches).unwrap(); assert!(attachments.is_empty()); } #[test] + #[serial_test::serial] fn test_parse_attachments_reads_real_file() { + let _root = DefaultFileRoot::unset(); use std::io::Write; let cwd = std::env::current_dir().unwrap().canonicalize().unwrap(); let dir = tempfile::tempdir_in(&cwd).unwrap(); @@ -3137,7 +3167,9 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_nonexistent_file() { + let _root = DefaultFileRoot::unset(); let matches = make_attach_matches(&["test", "-a", "nonexistent_file.pdf"]); let err = parse_attachments(&matches).unwrap_err(); assert!( @@ -3148,7 +3180,9 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_unknown_extension_falls_back_to_octet_stream() { + let _root = DefaultFileRoot::unset(); use std::io::Write; let cwd = std::env::current_dir().unwrap().canonicalize().unwrap(); let dir = tempfile::tempdir_in(&cwd).unwrap(); @@ -3165,7 +3199,9 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_size_limit_accumulates() { + let _root = DefaultFileRoot::unset(); let cwd = std::env::current_dir().unwrap().canonicalize().unwrap(); let dir = tempfile::tempdir_in(&cwd).unwrap(); @@ -3193,7 +3229,9 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_rejects_empty_file() { + let _root = DefaultFileRoot::unset(); let cwd = std::env::current_dir().unwrap().canonicalize().unwrap(); let dir = tempfile::tempdir_in(&cwd).unwrap(); let file_path = dir.path().join("empty.txt"); diff --git a/crates/google-workspace-cli/src/helpers/script.rs b/crates/google-workspace-cli/src/helpers/script.rs index 11bcdebec..4b31db62d 100644 --- a/crates/google-workspace-cli/src/helpers/script.rs +++ b/crates/google-workspace-cli/src/helpers/script.rs @@ -169,13 +169,7 @@ fn process_file(path: &Path) -> Result, GwsError> { filename.trim_end_matches(".js").trim_end_matches(".gs"), ), "html" => ("HTML", filename.trim_end_matches(".html")), - "json" => { - if filename == "appsscript.json" { - ("JSON", "appsscript") - } else { - return Ok(None); - } - } + "json" if filename == "appsscript.json" => ("JSON", "appsscript"), _ => return Ok(None), }; diff --git a/crates/google-workspace-cli/src/main.rs b/crates/google-workspace-cli/src/main.rs index 41dcc1e1f..63de1d1c0 100644 --- a/crates/google-workspace-cli/src/main.rs +++ b/crates/google-workspace-cli/src/main.rs @@ -225,7 +225,7 @@ async fn run() -> Result<(), GwsError> { // Validate file paths against traversal before any I/O. // Use the returned canonical paths so the validated path is the one - // actually used for I/O (closes TOCTOU gap). + // actually used for I/O. Local path-replacement races still apply. let upload_path_buf = if let Some(p) = upload_path { Some(crate::validate::validate_safe_file_path(p, "--upload")?) } else { @@ -236,8 +236,8 @@ async fn run() -> Result<(), GwsError> { } else { None }; - let upload_path = upload_path_buf.as_deref().and_then(|p| p.to_str()); - let output_path = output_path_buf.as_deref().and_then(|p| p.to_str()); + let upload_path = optional_file_path_as_str(upload_path_buf.as_deref(), "--upload")?; + let output_path = optional_file_path_as_str(output_path_buf.as_deref(), "--output")?; let upload = { let upload_content_type = matched_args @@ -261,25 +261,30 @@ async fn run() -> Result<(), GwsError> { // to avoid restrictive scopes like gmail.metadata that block query parameters. let scopes: Vec<&str> = select_scope(&method.scopes).into_iter().collect(); - // Authenticate: try OAuth, fail with error if credentials exist but are broken - let (token, auth_method) = match auth::get_token(&scopes).await { - Ok(t) => (Some(t), executor::AuthMethod::OAuth), - Err(e) => { - // If credentials were found but failed (e.g. decryption error, invalid token), - // propagate the error instead of silently falling back to unauthenticated. - // Only fall back to None if no credentials exist at all. - let err_msg = format!("{e:#}"); - // NB: matches the bail!() message in auth::load_credentials_inner - if err_msg.starts_with("No credentials found") { - (None, executor::AuthMethod::None) - } else { - return Err(GwsError::Auth(format!("Authentication failed: {err_msg}"))); + // Dry-runs only need the schema and inputs. Do not load credentials: + // authentication may access the keyring or remove corrupt credential files. + let (token, auth_method) = if dry_run { + (None, executor::AuthMethod::None) + } else { + match auth::get_token(&scopes).await { + Ok(t) => (Some(t), executor::AuthMethod::OAuth), + Err(e) => { + // If credentials were found but failed (e.g. decryption error, invalid token), + // propagate the error instead of silently falling back to unauthenticated. + // Only fall back to None if no credentials exist at all. + let err_msg = format!("{e:#}"); + // NB: matches the bail!() message in auth::load_credentials_inner + if err_msg.starts_with("No credentials found") { + (None, executor::AuthMethod::None) + } else { + return Err(GwsError::Auth(format!("Authentication failed: {err_msg}"))); + } } } }; // Execute - executor::execute_method( + executor::execute_method_with_policy( &doc, method, params_json, @@ -294,11 +299,42 @@ async fn run() -> Result<(), GwsError> { &sanitize_config.mode, &output_format, false, + parse_body_validation_policy(matched_args), ) .await .map(|_| ()) } +fn parse_body_validation_policy(matches: &clap::ArgMatches) -> executor::BodyValidationPolicy { + if matches + .try_get_one::("allow-unknown-fields") + .ok() + .flatten() + .copied() + .unwrap_or(false) + { + executor::BodyValidationPolicy::AllowUnknownFields + } else { + executor::BodyValidationPolicy::Strict + } +} + +// The executor takes strings. An explicit path must never become an omitted +// argument just because canonicalization found a non-UTF-8 component. +fn optional_file_path_as_str<'a>( + path: Option<&'a std::path::Path>, + flag_name: &str, +) -> Result, GwsError> { + path.map(|path| { + path.to_str().ok_or_else(|| { + GwsError::Validation(format!( + "{flag_name} resolves to a path that is not valid UTF-8; choose a path whose canonical components are valid UTF-8" + )) + }) + }) + .transpose() +} + /// Select the best scope from a method's scope list. /// /// Discovery Documents list method scopes as alternatives — any single scope @@ -505,8 +541,9 @@ fn print_usage() { } println!(); println!("COMMUNITY:"); - println!(" Star the repo: https://github.com/googleworkspace/cli"); - println!(" Report bugs / request features: https://github.com/googleworkspace/cli/issues"); + println!(" Fork: https://github.com/ratovarius/cli"); + println!(" Upstream: https://github.com/googleworkspace/cli"); + println!(" Report bugs / request features: https://github.com/ratovarius/cli/issues"); println!(" Please search existing issues first; if one already exists, comment there."); println!(); println!("DISCLAIMER:"); @@ -525,6 +562,99 @@ fn is_version_flag(arg: &str) -> bool { mod tests { use super::*; + #[test] + fn test_parse_body_validation_policy_from_raw_method_flags() { + let doc: discovery::RestDescription = serde_json::from_value(serde_json::json!({ + "name": "test", + "version": "v1", + "rootUrl": "https://example.invalid/", + "servicePath": "", + "resources": {"files": {"methods": { + "create": {"path": "files", "httpMethod": "POST", "request": {"$ref": "File"}}, + "list": {"path": "files", "httpMethod": "GET"} + }}} + })) + .unwrap(); + for (args, expected) in [ + ( + vec!["gws", "files", "list"], + executor::BodyValidationPolicy::Strict, + ), + ( + vec!["gws", "files", "create", "--json", "{}"], + executor::BodyValidationPolicy::Strict, + ), + ( + vec![ + "gws", + "files", + "create", + "--json", + "{}", + "--allow-unknown-fields", + ], + executor::BodyValidationPolicy::AllowUnknownFields, + ), + ] { + let matches = commands::build_cli(&doc) + .try_get_matches_from(args) + .unwrap(); + let (_, method_args) = resolve_method_from_matches(&doc, &matches).unwrap(); + assert_eq!(parse_body_validation_policy(method_args), expected); + } + } + + #[test] + fn file_root_path_encoding_preserves_present_and_absent_paths() { + for flag in ["--output", "--upload"] { + assert_eq!(optional_file_path_as_str(None, flag).unwrap(), None); + assert_eq!( + optional_file_path_as_str(Some(std::path::Path::new("résumé.pdf")), flag).unwrap(), + Some("résumé.pdf") + ); + } + } + + #[cfg(any(unix, windows))] + fn non_utf8_canonical_path() -> std::path::PathBuf { + // Construct an OS path in memory: no filesystem support is required. + #[cfg(unix)] + { + use std::os::unix::ffi::OsStringExt; + std::ffi::OsString::from_vec(b"/files/bytes-\xff/report.pdf".to_vec()).into() + } + #[cfg(windows)] + { + use std::os::windows::ffi::OsStringExt; + let mut units: Vec = r"C:\files\bytes-".encode_utf16().collect(); + units.push(0xD800); // unpaired surrogate + units.extend(r"\report.pdf".encode_utf16()); + std::ffi::OsString::from_wide(&units).into() + } + } + + #[cfg(any(unix, windows))] + #[test] + fn file_root_path_encoding_rejects_explicit_output_instead_of_fallback() { + let path = non_utf8_canonical_path(); + let error = optional_file_path_as_str(Some(&path), "--output").unwrap_err(); + assert!(matches!(error, GwsError::Validation(_))); + let message = error.to_string(); + assert!(message.contains("--output"), "{message}"); + assert!(message.contains("UTF-8"), "{message}"); + } + + #[cfg(any(unix, windows))] + #[test] + fn file_root_path_encoding_rejects_explicit_upload_instead_of_omitting_it() { + let path = non_utf8_canonical_path(); + let error = optional_file_path_as_str(Some(&path), "--upload").unwrap_err(); + assert!(matches!(error, GwsError::Validation(_))); + let message = error.to_string(); + assert!(message.contains("--upload"), "{message}"); + assert!(message.contains("UTF-8"), "{message}"); + } + #[test] fn test_parse_pagination_config_defaults() { let matches = clap::Command::new("test") diff --git a/crates/google-workspace-cli/tests/dry_run.rs b/crates/google-workspace-cli/tests/dry_run.rs new file mode 100644 index 000000000..c760dccd4 --- /dev/null +++ b/crates/google-workspace-cli/tests/dry_run.rs @@ -0,0 +1,651 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use serde_json::{json, Value}; +use std::collections::BTreeMap; +use std::fs; +use std::io::{Read, Write}; +use std::net::TcpListener; +use std::path::{Path, PathBuf}; +use std::process::{Command, Output, Stdio}; +use std::sync::{mpsc, Arc, Mutex}; +use std::thread; +use std::time::{Duration, Instant, SystemTime}; +use tempfile::TempDir; + +const BODY: &str = + r#"{"requests":[{"insertText":{"text":"hello","endOfSegmentLocation":{"segmentId":""}}}]}"#; +const RAW: &[&str] = &[ + "docs", + "documents", + "batchUpdate", + "--params", + r#"{"documentId":"doc /?#","fields":"documentId"}"#, + "--json", + BODY, +]; +const WRITE: &[&str] = &["docs", "+write", "--document", "doc /?#", "--text", "hello"]; + +// Observe every API/proxy connection without contacting Google. Returning an +// error also lets real-request tests verify that API failures stay failures. +struct NetworkTrap { + url: String, + requests: Arc>>, + stop: mpsc::Sender<()>, + worker: Option>, +} + +impl NetworkTrap { + fn new() -> Self { + let listener = TcpListener::bind("127.0.0.1:0").unwrap(); + listener.set_nonblocking(true).unwrap(); + let url = format!("http://{}/", listener.local_addr().unwrap()); + let requests = Arc::new(Mutex::new(Vec::new())); + let captured = Arc::clone(&requests); + let (stop, stopping) = mpsc::channel(); + let worker = thread::spawn(move || loop { + match listener.accept() { + Ok((mut stream, _)) => { + // Accepted sockets may inherit the listener's nonblocking + // mode. Request reads need the timeout below on every OS. + stream.set_nonblocking(false).unwrap(); + // Record the connection even if the client fails before + // sending HTTP (for example, during TLS/proxy setup). + let mut requests = captured.lock().unwrap(); + requests.push(String::new()); + stream + .set_read_timeout(Some(Duration::from_secs(1))) + .unwrap(); + let mut header = Vec::new(); + let mut buffer = [0; 1024]; + while header.len() < 8192 && !header.windows(4).any(|w| w == b"\r\n\r\n") { + match stream.read(&mut buffer) { + Ok(0) | Err(_) => break, + Ok(size) => header.extend_from_slice(&buffer[..size]), + } + } + // TCP may deliver headers and body in separate reads. + // Consume the body before closing the connection, or unread + // request bytes can reset it and hide our synthetic 403. + if let Some(end) = header.windows(4).position(|w| w == b"\r\n\r\n") { + let body_len = String::from_utf8_lossy(&header[..end]) + .lines() + .filter_map(|line| line.split_once(':')) + .find(|(name, _)| name.eq_ignore_ascii_case("content-length")) + .map(|(_, value)| value.trim().parse::().unwrap()) + .unwrap_or(0); + let expected_len = end + 4 + body_len; + while header.len() < expected_len { + let remaining = (expected_len - header.len()).min(buffer.len()); + match stream.read(&mut buffer[..remaining]) { + Ok(0) | Err(_) => break, + Ok(size) => header.extend_from_slice(&buffer[..size]), + } + } + } + *requests.last_mut().unwrap() = String::from_utf8_lossy(&header).into_owned(); + let body = r#"{"error":{"code":403,"message":"Synthetic denial","errors":[{"reason":"forbidden"}]}}"#; + let _ = write!( + stream, + "HTTP/1.1 403 Forbidden\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", + body.len() + ); + } + Err(error) if error.kind() == std::io::ErrorKind::WouldBlock => { + if stopping.recv_timeout(Duration::from_millis(5)) + != Err(mpsc::RecvTimeoutError::Timeout) + { + break; + } + } + Err(error) => panic!("Network trap failed: {error}"), + } + }); + Self { + url, + requests, + stop, + worker: Some(worker), + } + } +} + +impl Drop for NetworkTrap { + fn drop(&mut self) { + let _ = self.stop.send(()); + self.worker.take().unwrap().join().unwrap(); + } +} + +type Snapshot = BTreeMap, SystemTime)>; + +fn snapshot(root: &Path) -> Snapshot { + let mut files = BTreeMap::new(); + for entry in fs::read_dir(root).unwrap() { + let path = entry.unwrap().path(); + if path.is_dir() { + files.extend(snapshot(&path)); + } else { + files.insert( + path.clone(), + ( + fs::read(&path).unwrap(), + fs::metadata(path).unwrap().modified().unwrap(), + ), + ); + } + } + files +} + +struct Fixture { + dir: TempDir, + config: PathBuf, + network: NetworkTrap, +} + +impl Fixture { + fn new(corrupt_credentials: bool) -> Self { + let dir = tempfile::tempdir().unwrap(); + let config = dir.path().join("config"); + fs::create_dir_all(config.join("cache")).unwrap(); + // Stop dotenvy's ancestor search and isolate all credential sources. + fs::write(dir.path().join(".env"), "").unwrap(); + fs::create_dir(dir.path().join("home")).unwrap(); + let fixture = Self { + dir, + config, + network: NetworkTrap::new(), + }; + fixture.write_discovery(&fixture.discovery()); + if corrupt_credentials { + for name in [ + "credentials.enc", + "credentials.json", + ".encryption_key", + "token_cache.json", + "sa_token_cache.json", + ] { + fs::write( + fixture.config.join(name), + format!("corrupt synthetic sentinel: {name}"), + ) + .unwrap(); + } + } + fixture + } + + fn discovery(&self) -> Value { + json!({ + "name": "docs", + "version": "v1", + "rootUrl": self.network.url, + "servicePath": "v1/", + "resources": { + "documents": { + "methods": { + "batchUpdate": { + "id": "docs.documents.batchUpdate", + "httpMethod": "POST", + "path": "documents/{documentId}:batchUpdate", + "parameterOrder": ["documentId"], + "parameters": { + "documentId": {"type": "string", "location": "path", "required": true}, + "fields": {"type": "string", "location": "query"} + }, + "request": {"$ref": "BatchUpdateDocumentRequest"}, + "scopes": ["https://www.googleapis.com/auth/documents"] + } + } + } + }, + "schemas": { + "BatchUpdateDocumentRequest": { + "type": "object", + "required": ["requests"], + "properties": { + "requests": {"type": "array", "items": {"$ref": "Request"}} + } + }, + "Request": { + "type": "object", + "properties": { + "insertText": { + "type": "object", + "properties": { + "text": {"type": "string"}, + "endOfSegmentLocation": { + "type": "object", + "properties": {"segmentId": {"type": "string"}} + } + } + } + } + } + } + }) + } + + fn write_discovery(&self, discovery: &Value) { + // Exercise the real config override, cache filename and freshness check. + fs::write( + self.config.join("cache/docs_v1.json"), + serde_json::to_vec(discovery).unwrap(), + ) + .unwrap(); + } + + fn command(&self, args: &[&str], token: Option<&str>) -> Command { + let mut command = Command::new(env!("CARGO_BIN_EXE_gws")); + command + .args(args) + .current_dir(self.dir.path()) + .env_clear() + .env("HOME", self.dir.path().join("home")) + .env("USERPROFILE", self.dir.path().join("home")) + .env("APPDATA", self.dir.path().join("home")) + .env("XDG_CONFIG_HOME", self.dir.path().join("home")) + .env("USER", "gws-dry-run-test") + .env("USERNAME", "gws-dry-run-test") + .env("GOOGLE_WORKSPACE_CLI_CONFIG_DIR", &self.config) + // Windows known-folder lookup ignores the home overrides above. + // Pin both token loading and quota-project lookup to the fixture. + .env( + "GOOGLE_APPLICATION_CREDENTIALS", + self.dir.path().join("missing-adc.json"), + ) + // Never query an actual OS account, even when testing broken auth. + .env("GOOGLE_WORKSPACE_CLI_KEYRING_BACKEND", "file") + .env("HTTP_PROXY", &self.network.url) + .env("HTTPS_PROXY", &self.network.url) + .env("ALL_PROXY", &self.network.url) + .stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + if let Some(system_root) = std::env::var_os("SystemRoot") { + command.env("SystemRoot", system_root); + } + if let Some(token) = token { + command.env("GOOGLE_WORKSPACE_CLI_TOKEN", token); + } + command + } + + fn run(&self, args: &[&str], token: Option<&str>) -> Output { + Self::run_command(self.command(args, token)) + } + + fn run_command(mut command: Command) -> Output { + let mut child = command.spawn().unwrap(); + let deadline = Instant::now() + Duration::from_secs(15); + while child.try_wait().unwrap().is_none() { + if Instant::now() >= deadline { + child.kill().unwrap(); + let output = child.wait_with_output().unwrap(); + panic!("CLI timed out: {output:?}"); + } + thread::sleep(Duration::from_millis(10)); + } + child.wait_with_output().unwrap() + } + + fn dry_run(&self, args: &[&str], exit_code: i32) -> Value { + let before = snapshot(self.dir.path()); + let mut args = args.to_vec(); + args.push("--dry-run"); + let output = self.run(&args, None); + assert_eq!( + output.status.code(), + Some(exit_code), + "stdout: {}\nstderr: {}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + let after = snapshot(self.dir.path()); + assert_eq!( + after.keys().collect::>(), + before.keys().collect::>(), + "Dry-run created or deleted fixture files" + ); + for (path, expected) in &before { + assert!( + after.get(path) == Some(expected), + "Dry-run changed contents or mtime: {path:?}" + ); + } + assert!( + self.network.requests.lock().unwrap().is_empty(), + "Dry-run attempted an API, auth or Discovery connection" + ); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!( + !stderr.contains("keyring") && !stderr.contains("credentials"), + "Dry-run unexpectedly used auth: {stderr}" + ); + serde_json::from_slice(&output.stdout).unwrap() + } + + fn assert_preview(&self, preview: &Value, query: Value) { + assert_eq!( + preview, + &json!({ + "dry_run": true, + "url": format!("{}v1/documents/doc%20%2F%3F%23:batchUpdate", self.network.url), + "method": "POST", + "query_params": query, + "body": { + "requests": [{ + "insertText": { + "text": "hello", + "endOfSegmentLocation": {"segmentId": ""} + } + }] + }, + "is_multipart_upload": false + }) + ); + } +} + +#[test] +fn raw_dry_run_preserves_corrupt_credentials() { + let fixture = Fixture::new(true); + fixture.assert_preview(&fixture.dry_run(RAW, 0), json!([["fields", "documentId"]])); +} + +#[test] +fn raw_dry_run_needs_no_credentials() { + let fixture = Fixture::new(false); + fixture.assert_preview(&fixture.dry_run(RAW, 0), json!([["fields", "documentId"]])); +} + +#[test] +fn docs_write_dry_run_preserves_corrupt_credentials() { + let fixture = Fixture::new(true); + fixture.assert_preview(&fixture.dry_run(WRITE, 0), json!([])); +} + +#[test] +fn docs_write_dry_run_needs_no_credentials() { + let fixture = Fixture::new(false); + fixture.assert_preview(&fixture.dry_run(WRITE, 0), json!([])); +} + +fn assert_validation(error: &Value, message: &str) { + assert_eq!(error["error"]["reason"], "validationError"); + assert!( + error["error"]["message"] + .as_str() + .unwrap() + .contains(message), + "{error}" + ); +} + +#[test] +fn raw_dry_run_rejects_malformed_body_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + *args.last_mut().unwrap() = "{"; + assert_validation(&fixture.dry_run(&args, 3), "Invalid --json body"); +} + +#[test] +fn raw_dry_run_validates_nested_body_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + *args.last_mut().unwrap() = r#"{"requests":[{"insertText":{"text":42}}]}"#; + assert_validation(&fixture.dry_run(&args, 3), "Expected type 'string'"); +} + +#[test] +fn raw_dry_run_rejects_malformed_params_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + args[4] = "{"; + assert_validation(&fixture.dry_run(&args, 3), "Invalid --params JSON"); +} + +#[test] +fn raw_dry_run_requires_path_parameter_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + args[4] = "{}"; + assert_validation(&fixture.dry_run(&args, 3), "documentId is missing"); +} + +#[test] +fn raw_dry_run_requires_query_parameter_without_auth() { + let fixture = Fixture::new(true); + let mut doc = fixture.discovery(); + doc["resources"]["documents"]["methods"]["batchUpdate"]["parameters"]["revision"] = + json!({"type": "string", "location": "query", "required": true}); + fixture.write_discovery(&doc); + assert_validation(&fixture.dry_run(RAW, 3), "'revision' is missing"); +} + +#[test] +fn raw_dry_run_rejects_resource_traversal_without_auth() { + let fixture = Fixture::new(true); + let mut doc = fixture.discovery(); + doc["resources"]["documents"]["methods"]["batchUpdate"]["path"] = + json!("documents/{+documentId}:batchUpdate"); + fixture.write_discovery(&doc); + let mut args = RAW.to_vec(); + args[4] = r#"{"documentId":"../outside"}"#; + assert_validation(&fixture.dry_run(&args, 3), "traversal"); +} + +#[test] +fn raw_dry_run_rejects_output_traversal_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + args.extend(["--output", "../outside"]); + assert_validation(&fixture.dry_run(&args, 3), "outside the current directory"); +} + +#[test] +fn docs_write_dry_run_requires_document() { + let fixture = Fixture::new(true); + assert_validation( + &fixture.dry_run(&["docs", "+write", "--text", "hello"], 3), + "--document", + ); +} + +#[test] +fn docs_write_dry_run_requires_text() { + let fixture = Fixture::new(true); + assert_validation( + &fixture.dry_run(&["docs", "+write", "--document", "doc"], 3), + "--text", + ); +} + +#[test] +fn docs_write_dry_run_validates_generated_body_without_auth() { + let fixture = Fixture::new(true); + let mut doc = fixture.discovery(); + doc["schemas"]["BatchUpdateDocumentRequest"]["required"] = json!(["requests", "title"]); + fixture.write_discovery(&doc); + assert_validation( + &fixture.dry_run(WRITE, 3), + "Missing required property 'title'", + ); +} + +#[test] +fn docs_write_dry_run_preserves_discovery_errors() { + let fixture = Fixture::new(true); + let mut doc = fixture.discovery(); + doc["resources"]["documents"]["methods"] = json!({}); + fixture.write_discovery(&doc); + let error = fixture.dry_run(WRITE, 4); + assert_eq!(error["error"]["reason"], "discoveryError"); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("batchUpdate")); +} + +fn assert_auth_failure(args: &[&str]) { + let fixture = Fixture::new(true); + let output = fixture.run(args, None); + assert_eq!(output.status.code(), Some(2), "{output:?}"); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert_eq!(error["error"]["reason"], "authError"); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("credentials")); + assert!(fixture.network.requests.lock().unwrap().is_empty()); +} + +#[test] +fn raw_real_request_still_fails_on_broken_credentials() { + assert_auth_failure(RAW); +} + +#[test] +fn docs_write_real_request_still_fails_on_broken_credentials() { + assert_auth_failure(WRITE); +} + +#[test] +fn raw_real_request_rejects_fixture_adc_without_profile_fallback() { + let fixture = Fixture::new(false); + let output = fixture.run(RAW, None); + assert_eq!(output.status.code(), Some(2), "{output:?}"); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert_eq!(error["error"]["reason"], "authError"); + let message = error["error"]["message"].as_str().unwrap(); + assert!( + message.contains("GOOGLE_APPLICATION_CREDENTIALS points to"), + "{error}" + ); + assert!( + message.contains( + fixture + .dir + .path() + .join("missing-adc.json") + .to_str() + .unwrap() + ), + "{error}" + ); + assert!(message.contains("file does not exist"), "{error}"); + assert!(fixture.network.requests.lock().unwrap().is_empty()); +} + +// This case must leave ADC unset to exercise the real no-credentials fallback. +// Only Unix dirs::home_dir() respects our HOME isolation; Windows uses the +// actual profile's known folder, so this case must not run there. +#[cfg(unix)] +#[test] +fn raw_real_request_without_credentials_preserves_access_denied() { + let fixture = Fixture::new(false); + let mut command = fixture.command(RAW, None); + command.env_remove("GOOGLE_APPLICATION_CREDENTIALS"); + let output = Fixture::run_command(command); + assert_eq!(output.status.code(), Some(2), "{output:?}"); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert_eq!(error["error"]["reason"], "authError"); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("No credentials provided")); + let requests = fixture.network.requests.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert!(!requests[0].to_lowercase().contains("authorization:")); +} + +fn assert_authenticated_api_failure(args: &[&str]) { + let fixture = Fixture::new(false); + // Make a mistaken profile fallback observable on Unix without using a real + // profile. Windows known-folder lookup ignores these home overrides, so the + // fixture must explicitly redirect ADC there as well. + let adc_dir = fixture.dir.path().join("home/.config/gcloud"); + fs::create_dir_all(&adc_dir).unwrap(); + fs::write( + adc_dir.join("application_default_credentials.json"), + r#"{"quota_project_id":"synthetic-profile-must-not-be-read"}"#, + ) + .unwrap(); + let output = fixture.run(args, Some("synthetic-test-token")); + assert_eq!(output.status.code(), Some(1), "{output:?}"); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert_eq!(error["error"]["code"], 403); + assert_eq!(error["error"]["message"], "Synthetic denial"); + let requests = fixture.network.requests.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert!(requests[0] + .to_lowercase() + .contains("authorization: bearer synthetic-test-token")); + assert!( + !requests[0].to_lowercase().contains("x-goog-user-project:"), + "Token-authenticated requests must not read quota attribution from profile ADC" + ); +} + +#[test] +fn raw_real_request_uses_token_and_preserves_api_failure() { + assert_authenticated_api_failure(RAW); +} + +#[test] +fn docs_write_real_request_uses_token_and_preserves_api_failure() { + assert_authenticated_api_failure(WRITE); +} + +#[test] +fn network_trap_waits_for_split_request_body_before_responding() { + use std::net::TcpStream; + + let trap = NetworkTrap::new(); + let address = trap + .url + .strip_prefix("http://") + .unwrap() + .trim_end_matches('/'); + let mut stream = TcpStream::connect(address).unwrap(); + stream + .set_read_timeout(Some(Duration::from_millis(100))) + .unwrap(); + stream + .write_all(b"POST / HTTP/1.1\r\nHost: localhost\r\nContent-Length: 5\r\n\r\n") + .unwrap(); + + let mut byte = [0; 1]; + let error = stream + .read(&mut byte) + .expect_err("must consume the body before responding"); + assert!(matches!( + error.kind(), + std::io::ErrorKind::WouldBlock | std::io::ErrorKind::TimedOut + )); + + stream.write_all(b"hello").unwrap(); + stream + .set_read_timeout(Some(Duration::from_secs(2))) + .unwrap(); + let mut response = String::new(); + stream.read_to_string(&mut response).unwrap(); + assert!( + response.starts_with("HTTP/1.1 403 Forbidden\r\n"), + "{response}" + ); + assert!(trap.requests.lock().unwrap()[0].ends_with("hello")); +} diff --git a/crates/google-workspace-cli/tests/file_roots.rs b/crates/google-workspace-cli/tests/file_roots.rs new file mode 100644 index 000000000..9dc516821 --- /dev/null +++ b/crates/google-workspace-cli/tests/file_roots.rs @@ -0,0 +1,274 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Exercise the real CLI with synthetic cached Discovery and child-only env. + +use std::fs; +use std::io::{Read, Write}; +use std::net::TcpListener; +use std::path::{Path, PathBuf}; +use std::process::{Command, Output}; +use std::time::{Duration, Instant}; + +use serde_json::{json, Value}; +use tempfile::{tempdir, TempDir}; + +struct Fixture { + _temp: TempDir, + cwd: PathBuf, + root: PathBuf, + config: PathBuf, +} + +impl Fixture { + fn new(root_url: &str) -> Self { + let temp = tempdir().unwrap(); + let base = temp.path().canonicalize().unwrap(); + let cwd = base.join("working"); + let root = base.join("files"); + let config = base.join("config"); + fs::create_dir(&cwd).unwrap(); + fs::create_dir(&root).unwrap(); + fs::create_dir_all(config.join("cache")).unwrap(); + // Stop dotenv from searching parent directories for real configuration. + fs::write(cwd.join(".env"), "").unwrap(); + fs::write(config.join("cache/drive_v3.json"), json!({ + "name": "drive", "version": "v3", "rootUrl": root_url, + "resources": {"files": {"methods": { + "get": {"httpMethod": "GET", "path": "files/synthetic"}, + "create": {"httpMethod": "POST", "path": "files", "supportsMediaUpload": true, + "mediaUpload": {"protocols": {"simple": {"path": "/upload/files", "multipart": true}}}} + }}} + }).to_string()).unwrap(); + Self { + _temp: temp, + cwd, + root, + config, + } + } + + fn command(&self, configured: bool) -> Command { + let mut command = Command::new(env!("CARGO_BIN_EXE_gws")); + command + .env_clear() + .current_dir(&self.cwd) + .env("HOME", &self.cwd) + .env("USERPROFILE", &self.cwd) + .env("GOOGLE_WORKSPACE_CLI_CONFIG_DIR", &self.config) + .env("GOOGLE_WORKSPACE_CLI_TOKEN", "synthetic-test-token") + .env("GOOGLE_WORKSPACE_CLI_KEYRING_BACKEND", "file") + .env("NO_COLOR", "1") + // A cache regression must fail locally, never fetch real Discovery. + .env("HTTPS_PROXY", "http://127.0.0.1:1") + .env("HTTP_PROXY", "http://127.0.0.1:1") + .env("NO_PROXY", "127.0.0.1,localhost"); + if configured { + command.env("GOOGLE_WORKSPACE_CLI_FILE_ROOT", &self.root); + } + command + } + + fn dry_run(&self, flag: &str, path: &Path, configured: bool) -> Output { + let method = if flag == "--upload" { "create" } else { "get" }; + self.command(configured) + .args(["drive", "files", method, "--dry-run", flag]) + .arg(path) + .output() + .unwrap() + } +} + +fn successful_json(output: &Output) -> Value { + assert!( + output.status.success(), + "status: {:?}\nstdout: {}\nstderr: {}", + output.status.code(), + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + serde_json::from_slice(&output.stdout).unwrap() +} + +#[test] +fn file_root_cli_dry_run_accepts_external_output_and_upload() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + fs::write(fixture.root.join("upload.txt"), "synthetic upload").unwrap(); + for (flag, name) in [("--output", "new.bin"), ("--upload", "upload.txt")] { + let output = fixture.dry_run(flag, &fixture.root.join(name), true); + let result = successful_json(&output); + assert_eq!(result["dry_run"], true); + assert_eq!(result["is_multipart_upload"], flag == "--upload"); + } + assert!(!fixture.root.join("new.bin").exists()); +} + +#[test] +fn file_root_cli_default_rejects_external_paths_but_keeps_local_output() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + let output = fixture.dry_run("--output", &fixture.root.join("new.bin"), false); + assert_eq!(output.status.code(), Some(3)); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("current directory")); + assert_eq!( + successful_json(&fixture.dry_run("--output", Path::new("new.bin"), false))["dry_run"], + true + ); +} + +#[test] +fn file_root_cli_rejects_cwd_relative_path_outside_configured_root() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + let output = fixture.dry_run("--output", Path::new("new.bin"), true); + assert_eq!( + output.status.code(), + Some(3), + "{}", + String::from_utf8_lossy(&output.stdout) + ); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("GOOGLE_WORKSPACE_CLI_FILE_ROOT")); +} + +#[test] +fn file_root_cli_download_propagates_canonical_output() { + let listener = TcpListener::bind("127.0.0.1:0").unwrap(); + listener.set_nonblocking(true).unwrap(); + let fixture = Fixture::new(&format!("http://{}/", listener.local_addr().unwrap())); + let output_path = fixture.root.join("./output.bin"); + let mut child = fixture + .command(true) + .args(["drive", "files", "get", "--output"]) + .arg(&output_path) + .stdout(std::process::Stdio::piped()) + .stderr(std::process::Stdio::piped()) + .spawn() + .unwrap(); + let deadline = Instant::now() + Duration::from_secs(15); + loop { + match listener.accept() { + Ok((mut stream, _)) => { + stream + .set_read_timeout(Some(Duration::from_secs(5))) + .unwrap(); + let mut request = Vec::new(); + let mut buffer = [0; 1024]; + while !request.windows(4).any(|part| part == b"\r\n\r\n") { + let read = stream.read(&mut buffer).unwrap(); + assert!(read > 0, "request ended before headers"); + request.extend_from_slice(&buffer[..read]); + } + assert!(request.starts_with(b"GET /files/synthetic HTTP/1.1\r\n")); + stream.write_all(b"HTTP/1.1 200 OK\r\nContent-Type: application/octet-stream\r\nContent-Length: 9\r\nConnection: close\r\n\r\nsynthetic").unwrap(); + break; + } + Err(err) if err.kind() == std::io::ErrorKind::WouldBlock => { + if child.try_wait().unwrap().is_some() { + break; + } + if Instant::now() >= deadline { + child.kill().unwrap(); + let output = child.wait_with_output().unwrap(); + panic!( + "CLI did not contact local fixture: {}", + String::from_utf8_lossy(&output.stderr) + ); + } + std::thread::sleep(Duration::from_millis(10)); + } + Err(err) => panic!("local fixture failed: {err}"), + } + } + let result = successful_json(&child.wait_with_output().unwrap()); + assert_eq!( + result["saved_file"], + fixture.root.join("output.bin").to_str().unwrap() + ); + assert_eq!(result["bytes"], 9); + assert_eq!( + fs::read(fixture.root.join("output.bin")).unwrap(), + b"synthetic" + ); + assert!(!fixture.cwd.join("output.bin").exists()); +} + +// macOS filesystems commonly reject invalid UTF-8 directory names. The CLI +// conversion itself is covered without filesystem access by main.rs unit tests; +// these Linux regressions exercise the complete canonical symlink handoff. +#[cfg(target_os = "linux")] +fn non_utf8_alias(fixture: &Fixture, filename: &str) -> PathBuf { + use std::os::unix::{ffi::OsStringExt, fs::symlink}; + let target = fixture + .root + .join(std::ffi::OsString::from_vec(b"bytes-\xff".to_vec())); + fs::create_dir(&target).unwrap(); + fs::write(target.join("upload.txt"), b"synthetic upload").unwrap(); + let alias = fixture.root.join("alias"); + symlink(&target, &alias).unwrap(); + alias.join(filename) +} + +#[cfg(target_os = "linux")] +#[test] +fn file_root_cli_rejects_non_utf8_output_without_fallback_write() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + let output_path = non_utf8_alias(&fixture, "new.bin"); + let fallback = fixture.cwd.join("download.bin"); + fs::write(&fallback, b"keep existing download").unwrap(); + // A non-dry run must fail at validation, before HTTP or the default output + // can be selected. No live service or credentials are involved. + let output = fixture + .command(true) + .args(["drive", "files", "get", "--output"]) + .arg(&output_path) + .output() + .unwrap(); + assert_eq!(fs::read(&fallback).unwrap(), b"keep existing download"); + assert!(!output_path.exists()); + assert_eq!( + output.status.code(), + Some(3), + "{}", + String::from_utf8_lossy(&output.stdout) + ); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + let message = error["error"]["message"].as_str().unwrap(); + assert!(message.contains("--output"), "{message}"); + assert!(message.contains("UTF-8"), "{message}"); +} + +#[cfg(target_os = "linux")] +#[test] +fn file_root_cli_rejects_non_utf8_upload_instead_of_omitting_it() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + let upload_path = non_utf8_alias(&fixture, "upload.txt"); + let output = fixture.dry_run("--upload", &upload_path, true); + assert_eq!( + output.status.code(), + Some(3), + "{}", + String::from_utf8_lossy(&output.stdout) + ); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + let message = error["error"]["message"].as_str().unwrap(); + assert!(message.contains("--upload"), "{message}"); + assert!(message.contains("UTF-8"), "{message}"); + assert_eq!(fs::read(&upload_path).unwrap(), b"synthetic upload"); +} diff --git a/crates/google-workspace/Cargo.toml b/crates/google-workspace/Cargo.toml index 3dc9f0830..5351565ed 100644 --- a/crates/google-workspace/Cargo.toml +++ b/crates/google-workspace/Cargo.toml @@ -18,8 +18,8 @@ version = "0.22.5" edition = "2021" description = "Google Workspace API client — Discovery Document types, service registry, and HTTP utilities" license = "Apache-2.0" -repository = "https://github.com/googleworkspace/cli" -homepage = "https://github.com/googleworkspace/cli" +repository = "https://github.com/ratovarius/cli" +homepage = "https://github.com/ratovarius/cli" readme = "README.md" authors = ["Justin Poehnelt"] keywords = ["google-workspace", "google", "discovery", "api-client"] diff --git a/crates/google-workspace/src/validate.rs b/crates/google-workspace/src/validate.rs index 32ef200f9..a951cae56 100644 --- a/crates/google-workspace/src/validate.rs +++ b/crates/google-workspace/src/validate.rs @@ -161,11 +161,13 @@ pub fn validate_safe_dir_path(dir: &str) -> Result { /// Validates that a file path (e.g. `--upload` or `--output`) is safe. /// -/// Rejects paths that escape above CWD via `..` traversal, contain -/// control characters, or follow symlinks to locations outside CWD. -/// Absolute paths are allowed (reading an existing file from a known -/// location is legitimate) but the resolved target must still live -/// under CWD. +/// By default, the resolved target must live under CWD. The trusted operator +/// environment variable `GOOGLE_WORKSPACE_CLI_FILE_ROOT` can select a different +/// boundary: an existing directory, canonicalized before validation. With an +/// explicit root, CLI paths must not contain `..` components. Relative CLI paths +/// always resolve from CWD, not from the configured root. Absolute paths within +/// the boundary are allowed. Control characters and symlink escapes are rejected. +/// Directory validators do not use this setting. /// /// # TOCTOU caveat /// @@ -175,48 +177,127 @@ pub fn validate_safe_dir_path(dir: &str) -> Result { /// TOCTOU would require `openat(O_NOFOLLOW)` on each path component, /// which is tracked as a follow-up for Unix platforms. pub fn validate_safe_file_path(path_str: &str, flag_name: &str) -> Result { - reject_dangerous_chars(path_str, flag_name)?; - - let path = Path::new(path_str); let cwd = std::env::current_dir() .map_err(|e| GwsError::Validation(format!("Failed to determine current directory: {e}")))?; + let file_root = std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT"); + validate_file_path_with_root( + path_str, + flag_name, + &cwd, + file_root.as_deref().map(Path::new), + ) +} - let resolved = if path.is_absolute() { - path.to_path_buf() - } else { - cwd.join(path) - }; +/// Explicit policy keeps filesystem validation independent of process-global env. +fn validate_file_path_with_root( + path_str: &str, + flag_name: &str, + cwd: &Path, + file_root: Option<&Path>, +) -> Result { + reject_dangerous_chars(path_str, flag_name)?; - // For existing files, canonicalize to resolve symlinks. - // For non-existing files, get the prefix canonicalized then normalize - // the remaining components to resolve any `..` or `.` segments. - let canonical = if resolved.exists() { - resolved.canonicalize().map_err(|e| { - GwsError::Validation(format!("Failed to resolve {flag_name} '{}': {e}", path_str)) + let canonical_root = if let Some(root) = file_root { + if root.as_os_str().is_empty() { + return Err(GwsError::Validation( + "GOOGLE_WORKSPACE_CLI_FILE_ROOT must name an existing directory; got an empty value" + .to_string(), + )); + } + // Environment is trusted: relative roots (including `..`) are valid. + let canonical = cwd.join(root).canonicalize().map_err(|e| { + GwsError::Validation(format!( + "GOOGLE_WORKSPACE_CLI_FILE_ROOT {root:?} must name an existing directory: {e}" + )) + })?; + if !canonical.is_dir() { + return Err(GwsError::Validation(format!( + "GOOGLE_WORKSPACE_CLI_FILE_ROOT {root:?} must name an existing directory" + ))); + } + canonical + } else { + cwd.canonicalize().map_err(|e| { + GwsError::Validation(format!("Failed to canonicalize current directory: {e}")) })? + }; + let boundary = if file_root.is_some() { + format!("GOOGLE_WORKSPACE_CLI_FILE_ROOT directory {canonical_root:?}") } else { - let raw = normalize_non_existing(&resolved)?; - // normalize_non_existing does NOT resolve `..` in the non-existent - // suffix. We must resolve them here to prevent bypass via paths like - // `non_existent/../../etc/passwd`. - normalize_dotdot(&raw) + format!("current directory {canonical_root:?}") }; - let canonical_cwd = cwd.canonicalize().map_err(|e| { - GwsError::Validation(format!("Failed to canonicalize current directory: {e}")) + let path = Path::new(path_str); + if file_root.is_some() + && path + .components() + .any(|component| component == std::path::Component::ParentDir) + { + return Err(GwsError::Validation(format!( + "{flag_name} must not contain parent traversal ('..') components within the {boundary}; use a path without '..'" + ))); + } + + // Path::join preserves absolute arguments; relative arguments stay CWD-relative. + let resolved = cwd.join(path); + let canonical = canonicalize_file_path(&resolved).map_err(|e| { + GwsError::Validation(format!( + "Failed to resolve {flag_name} {path_str:?} within the {boundary}: {e}" + )) })?; + // Preserve default handling of paths that normalize safely within CWD. + let canonical = normalize_dotdot(&canonical); - if !canonical.starts_with(&canonical_cwd) { + if !canonical.starts_with(&canonical_root) { return Err(GwsError::Validation(format!( - "{flag_name} '{}' resolves to '{}' which is outside the current directory", - path_str, - canonical.display() + "{flag_name} {path_str:?} resolves to {canonical:?} which is outside the {boundary}; set GOOGLE_WORKSPACE_CLI_FILE_ROOT to an existing directory containing the intended file" ))); } Ok(canonical) } +/// Canonicalize the existing file or nearest existing parent, then append the +/// missing suffix. Unlike `exists()`, symlink_metadata does not mistake dangling +/// symlinks for missing files. Keep this stricter resolver local to file flags. +fn canonicalize_file_path(path: &Path) -> std::io::Result { + let mut current = path; + let mut remaining = Vec::new(); + loop { + match std::fs::symlink_metadata(current) { + Ok(_) => { + let mut canonical = current.canonicalize()?; + if !remaining.is_empty() && !canonical.is_dir() { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "existing file parent must be a directory", + )); + } + for component in remaining.into_iter().rev() { + canonical.push(component); + } + return Ok(canonical); + } + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + let name = current.file_name().ok_or_else(|| { + std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "cannot resolve an existing directory prefix", + ) + })?; + remaining.push(name); + current = current.parent().ok_or_else(|| { + std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "cannot resolve a file parent", + ) + })?; + } + Err(error) => return Err(error), + } + } +} + /// Resolve `.` and `..` components in a path without touching the filesystem. fn normalize_dotdot(path: &Path) -> PathBuf { let mut out = PathBuf::new(); @@ -763,11 +844,9 @@ mod tests { let canonical_dir = dir.path().canonicalize().unwrap(); fs::write(canonical_dir.join("test.txt"), "data").unwrap(); - let saved_cwd = std::env::current_dir().unwrap(); - std::env::set_current_dir(&canonical_dir).unwrap(); + let _environment = FilePathEnvironment::set(&canonical_dir, None); let result = validate_safe_file_path("test.txt", "--upload"); - std::env::set_current_dir(&saved_cwd).unwrap(); assert!(result.is_ok(), "expected Ok, got: {result:?}"); } @@ -778,11 +857,9 @@ mod tests { let dir = tempdir().unwrap(); let canonical_dir = dir.path().canonicalize().unwrap(); - let saved_cwd = std::env::current_dir().unwrap(); - std::env::set_current_dir(&canonical_dir).unwrap(); + let _environment = FilePathEnvironment::set(&canonical_dir, None); let result = validate_safe_file_path("../../etc/passwd", "--upload"); - std::env::set_current_dir(&saved_cwd).unwrap(); assert!(result.is_err(), "path traversal should be rejected"); assert!( @@ -792,7 +869,10 @@ mod tests { } #[test] + #[serial] fn test_file_path_rejects_control_chars() { + let dir = tempdir().unwrap(); + let _environment = FilePathEnvironment::set(dir.path(), None); let result = validate_safe_file_path("file\x00.txt", "--output"); assert!(result.is_err(), "null bytes should be rejected"); } @@ -808,11 +888,9 @@ mod tests { let link_path = canonical_dir.join("escape"); std::os::unix::fs::symlink("/tmp", &link_path).unwrap(); - let saved_cwd = std::env::current_dir().unwrap(); - std::env::set_current_dir(&canonical_dir).unwrap(); + let _environment = FilePathEnvironment::set(&canonical_dir, None); let result = validate_safe_file_path("escape/secret.txt", "--output"); - std::env::set_current_dir(&saved_cwd).unwrap(); assert!(result.is_err(), "symlink escape should be rejected"); } @@ -824,15 +902,337 @@ mod tests { let dir = tempdir().unwrap(); let canonical_dir = dir.path().canonicalize().unwrap(); - let saved_cwd = std::env::current_dir().unwrap(); - std::env::set_current_dir(&canonical_dir).unwrap(); + let _environment = FilePathEnvironment::set(&canonical_dir, None); let result = validate_safe_file_path("doesnt_exist/../../etc/passwd", "--output"); - std::env::set_current_dir(&saved_cwd).unwrap(); assert!( result.is_err(), "traversal via non-existent prefix should be rejected" ); } + + // Default public-validator and directory-scope tests isolate global state; + // scoped file-policy tests pass CWD and the trusted root explicitly. + struct FilePathEnvironment { + cwd: PathBuf, + root: Option, + } + + impl FilePathEnvironment { + fn set(cwd: &Path, root: Option<&Path>) -> Self { + let saved = Self { + cwd: std::env::current_dir().unwrap(), + root: std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), + }; + std::env::set_current_dir(cwd).unwrap(); + match root { + Some(root) => std::env::set_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT", root), + None => std::env::remove_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), + } + saved + } + } + + impl Drop for FilePathEnvironment { + fn drop(&mut self) { + std::env::set_current_dir(&self.cwd).unwrap(); + match &self.root { + Some(root) => std::env::set_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT", root), + None => std::env::remove_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), + } + } + } + + fn file_path_under_root( + path: &Path, + flag: &str, + cwd: &Path, + root: Option<&Path>, + ) -> Result { + validate_file_path_with_root(path.to_str().unwrap(), flag, cwd, root) + } + + #[test] + fn file_root_default_preserves_cwd_boundary_and_resolution() { + let dir = tempdir().unwrap(); + let cwd = dir.path().canonicalize().unwrap(); + fs::create_dir(cwd.join("nested")).unwrap(); + fs::write(cwd.join("upload.txt"), "synthetic upload").unwrap(); + for path in [cwd.join("upload.txt"), PathBuf::from("upload.txt")] { + assert_eq!( + file_path_under_root(&path, "--upload", &cwd, None).unwrap(), + cwd.join("upload.txt") + ); + } + assert_eq!( + file_path_under_root(Path::new("nested/../new.txt"), "--output", &cwd, None).unwrap(), + cwd.join("new.txt") + ); + for path in [ + cwd.parent().unwrap().join("outside.txt"), + PathBuf::from("../outside.txt"), + ] { + let err = file_path_under_root(&path, "--output", &cwd, None) + .unwrap_err() + .to_string(); + assert!(err.contains("outside the current directory"), "{err}"); + assert!(err.contains("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), "{err}"); + } + assert!(file_path_under_root( + Path::new("missing/../../outside.txt"), + "--output", + &cwd, + None + ) + .is_err()); + } + + #[test] + fn file_root_accepts_absolute_output_and_existing_upload() { + let cwd = tempdir().unwrap(); + let root = tempdir().unwrap(); + let canonical_root = root.path().canonicalize().unwrap(); + fs::write(root.path().join("upload.txt"), "synthetic upload").unwrap(); + for (name, flag) in [("new.txt", "--output"), ("upload.txt", "--upload")] { + assert_eq!( + file_path_under_root(&root.path().join(name), flag, cwd.path(), Some(root.path())) + .unwrap(), + canonical_root.join(name) + ); + } + assert!(!root.path().join("new.txt").exists()); + } + + #[test] + fn file_root_rejects_sibling_even_with_shared_name_prefix() { + let dir = tempdir().unwrap(); + // A literal backslash on Unix, a separator on Windows; both must be + // compared using the escaped canonical representation in diagnostics. + let parent = dir.path().join(r"back\slash"); + fs::create_dir_all(&parent).unwrap(); + let root = parent.join("allowed"); + let sibling = parent.join("allowed-sibling"); + fs::create_dir(&root).unwrap(); + fs::create_dir(&sibling).unwrap(); + let err = file_path_under_root( + &sibling.join("new.txt"), + "--output", + dir.path(), + Some(&root), + ) + .unwrap_err() + .to_string(); + assert!(err.contains("outside"), "{err}"); + assert!(err.contains("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), "{err}"); + assert!( + err.contains(&format!("{:?}", root.canonicalize().unwrap())), + "{err}" + ); + } + + #[test] + fn file_root_rejects_parent_components_even_inside_boundary() { + let root = tempdir().unwrap(); + fs::create_dir(root.path().join("nested")).unwrap(); + for path in ["nested/../new.txt", "missing/../new.txt", "../outside.txt"] { + assert!( + file_path_under_root(Path::new(path), "--output", root.path(), Some(root.path())) + .is_err(), + "accepted {path}" + ); + } + } + + #[test] + fn file_root_rejects_control_and_dangerous_unicode_arguments() { + let root = tempdir().unwrap(); + for path in [ + "bad\0.txt", + "bad\n.txt", + "bad\u{202e}.txt", + "bad\u{200b}.txt", + ] { + assert!( + file_path_under_root(Path::new(path), "--output", root.path(), Some(root.path())) + .is_err(), + "accepted {path:?}" + ); + } + } + + #[test] + fn file_root_rejects_invalid_roots_without_falling_back_to_cwd() { + let cwd = tempdir().unwrap(); + let file = cwd.path().join("file.txt"); + fs::write(&file, "synthetic file").unwrap(); + for root in [PathBuf::new(), cwd.path().join("missing"), file] { + let err = + file_path_under_root(Path::new("new.txt"), "--output", cwd.path(), Some(&root)) + .unwrap_err() + .to_string(); + assert!(err.contains("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), "{err}"); + assert!(err.contains("directory"), "{err}"); + } + } + + #[test] + fn file_root_keeps_relative_arguments_cwd_relative() { + let root = tempdir().unwrap(); + let cwd = root.path().join("working"); + fs::create_dir(&cwd).unwrap(); + assert_eq!( + file_path_under_root(Path::new("new.txt"), "--output", &cwd, Some(root.path())) + .unwrap(), + cwd.canonicalize().unwrap().join("new.txt") + ); + let unrelated = tempdir().unwrap(); + assert!(file_path_under_root( + Path::new("new.txt"), + "--output", + unrelated.path(), + Some(root.path()) + ) + .is_err()); + } + + #[test] + fn file_root_canonicalizes_trusted_relative_root_with_parent_components() { + let root = tempdir().unwrap(); + let cwd = root.path().join("working"); + fs::create_dir(&cwd).unwrap(); + assert_eq!( + file_path_under_root( + Path::new("new.txt"), + "--output", + &cwd, + Some(Path::new("..")) + ) + .unwrap(), + cwd.canonicalize().unwrap().join("new.txt") + ); + } + + #[test] + fn file_root_validates_nested_new_file_parents_without_creating_them() { + let cwd = tempdir().unwrap(); + let root = tempdir().unwrap(); + let output = root.path().join("new/nested/output.bin"); + assert_eq!( + file_path_under_root(&output, "--output", cwd.path(), Some(root.path())).unwrap(), + root.path() + .canonicalize() + .unwrap() + .join("new/nested/output.bin") + ); + assert!(!root.path().join("new").exists()); + let file = root.path().join("file.txt"); + fs::write(&file, "synthetic file").unwrap(); + assert!(file_path_under_root( + &file.join("output.bin"), + "--output", + cwd.path(), + Some(root.path()) + ) + .is_err()); + } + + #[test] + #[serial] + fn file_root_does_not_expand_directory_validators() { + let cwd = tempdir().unwrap(); + let root = tempdir().unwrap(); + let _environment = FilePathEnvironment::set(cwd.path(), Some(root.path())); + assert!(validate_safe_output_dir(root.path().to_str().unwrap()).is_err()); + assert!(validate_safe_dir_path(root.path().to_str().unwrap()).is_err()); + assert_eq!( + validate_safe_output_dir("new").unwrap(), + cwd.path().canonicalize().unwrap().join("new") + ); + assert!(validate_safe_dir_path(".").is_ok()); + } + + #[test] + #[serial] + fn file_root_environment_is_restored_on_unwind() { + let cwd = std::env::current_dir().unwrap(); + let root = std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT"); + let dir = tempdir().unwrap(); + let result = std::panic::catch_unwind(|| { + let _environment = FilePathEnvironment::set(dir.path(), Some(dir.path())); + panic!("exercise restoration"); + }); + assert!(result.is_err()); + assert_eq!(std::env::current_dir().unwrap(), cwd); + assert_eq!(std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), root); + } + + #[cfg(unix)] + #[test] + fn file_root_resolves_inside_symlinks_and_rejects_escapes() { + use std::os::unix::fs::symlink; + let root = tempdir().unwrap(); + let outside = tempdir().unwrap(); + fs::create_dir(root.path().join("inside")).unwrap(); + fs::write(root.path().join("inside/upload.txt"), "inside").unwrap(); + fs::write(outside.path().join("upload.txt"), "outside").unwrap(); + symlink(root.path().join("inside"), root.path().join("safe")).unwrap(); + symlink(outside.path(), root.path().join("escape")).unwrap(); + for (suffix, flag) in [("upload.txt", "--upload"), ("new/output.bin", "--output")] { + assert_eq!( + file_path_under_root( + &root.path().join("safe").join(suffix), + flag, + root.path(), + Some(root.path()) + ) + .unwrap(), + root.path() + .canonicalize() + .unwrap() + .join("inside") + .join(suffix) + ); + assert!(file_path_under_root( + &root.path().join("escape").join(suffix), + flag, + root.path(), + Some(root.path()) + ) + .is_err()); + } + // A trusted root may itself be a symlink to an existing directory. + assert_eq!( + file_path_under_root( + &root.path().join("safe/upload.txt"), + "--upload", + outside.path(), + Some(&root.path().join("safe")) + ) + .unwrap(), + root.path() + .canonicalize() + .unwrap() + .join("inside/upload.txt") + ); + } + + #[cfg(unix)] + #[test] + fn file_root_rejects_dangling_symlinks_and_loops() { + use std::os::unix::fs::symlink; + let root = tempdir().unwrap(); + let outside = tempdir().unwrap(); + symlink(outside.path().join("new.txt"), root.path().join("dangling")).unwrap(); + symlink("loop", root.path().join("loop")).unwrap(); + for root_policy in [None, Some(root.path())] { + for name in ["dangling", "dangling/new.txt", "loop", "loop/new.txt"] { + assert!( + file_path_under_root(Path::new(name), "--output", root.path(), root_policy) + .is_err(), + "accepted {name}" + ); + } + } + } } diff --git a/docs/docs-read.md b/docs/docs-read.md new file mode 100644 index 000000000..b95b8a66c --- /dev/null +++ b/docs/docs-read.md @@ -0,0 +1,85 @@ +# Read structured Google Docs content + +```bash +gws docs +read --document DOC_ID +gws docs +read --document DOC_ID --format yaml +gws docs +read --document DOC_ID --dry-run +gws docs +read --document DOC_ID | jq '.outline' +``` + +`+read` translates the Docs API's nested structural elements into ordered +blocks. It uses the existing authentication, request executor, Model Armor +sanitization, and output formatters. `+write` is unchanged. + +The request uses `includeTabsContent=true` and +`suggestionsViewMode=SUGGESTIONS_INLINE`. Pass other API options through +`--params`. To prevent omitted content from looking like an empty document, +`fields` must be absent or exactly `"*"`. `$fields`, other tab/suggestion views, +conflicting document IDs, and non-JSON `alt` responses are rejected before +authentication. Dry-run prints the executor's request plan without acquiring +credentials, fetching document content, or invoking Model Armor. + +## Output + +- The root retains `documentId`, `title`, `revisionId`, `suggestionsViewMode`, + and `_sanitization` when returned. No revision is invented when absent. +- `tabs` and recursive `childTabs` preserve API order, IDs, titles, and parents. + Each tab has ordered `blocks`. `source: "legacyBody"` means the response had + no populated tabs; that fallback cannot confirm coverage of other tabs. +- Paragraph blocks have `text`, ordered `elements`, and `paragraphStyle`. + Text elements retain separate runs, text styles, links (including tab-aware + internal links), and suggested insertion/deletion/style changes. Other + returned paragraph metadata, such as bullets and positioned object IDs, + stays on the block. Automatic-text markers use `type: "autoText"` and retain + their source subtype (`PAGE_NUMBER` or `PAGE_COUNT`) as `autoTextType`. +- Table blocks have `rowCount`, `columns`, and + `rows[].cells[].blocks`, including nested tables. Row/cell styles and + suggestion metadata are retained. +- Each tab's `figures` map contains inline and positioned object metadata. + `embeddedObject` retains alt text, dimensions, and image properties when + available. Figure elements reference `objectId`. Missing metadata or + `contentUri` does not remove the reference. +- Headers, footers, and footnotes remain separate maps with their own `blocks`. + Footnote references and structural markers remain in content order. + Unsupported elements use `type: "unknown"` with their original `data`. +- `outline` contains titles, subtitles, and headings with text, style level, + tab ID, heading ID when present, source indices, and a JSON Pointer `path` + to the normalized paragraph, including headings in tables and segments. + +For example, select a heading by its returned ID: + +```bash +gws docs +read --document DOC_ID | + jq '.. | objects | select(.paragraphStyle?.headingId? == "HEADING_ID")' +``` + +To select the blocks between two top-level headings in a particular tab: + +```bash +gws docs +read --document DOC_ID | + jq --arg tab TAB_ID --argjson start 10 --argjson end 50 \ + '.. | objects | select(.tabId? == $tab and has("blocks")) | + .blocks[] | select(.startIndex >= $start and .startIndex < $end)' +``` + +Use indices actually returned for that tab. `startIndex` and `endIndex` are +UTF-16 offsets in the API's tab/segment, **not** byte or character offsets into +the extracted `text`. The text convenience field concatenates text runs only; +figures and other markers remain in `elements`. + +## Limits + +This is a structured content view, not a visual layout renderer or a lossless +API round trip. It does not resolve inherited styles, render drawings or +equations, download images, or accept/reject suggestions. Proposed deletions +remain inline. Image URIs are included only when returned and may expire. +Use raw `gws docs documents get` for other views or partial field masks. + +JSON is the default and retains the entire normalized tree. YAML uses the +existing serializer. Table and CSV use the existing formatter's first +nonempty-array summary (typically the outline, otherwise tabs); table cells +can be truncated. Use JSON for complete downstream processing. + +Usage and limitations also live in the command's help, the source consumed by +`gws generate-skills`. Generated skill files are maintained by the repository's +Generate Skills workflow. diff --git a/examples/docs-review-bundle/README.md b/examples/docs-review-bundle/README.md new file mode 100644 index 000000000..7975497f3 --- /dev/null +++ b/examples/docs-review-bundle/README.md @@ -0,0 +1,212 @@ +# Visual Docs review bundle + +This standalone companion combines native Docs JSON, Drive PDF/DOCX/Markdown +exports, DOCX raster assets, and a local HTML review index. It uses only the +Python standard library and existing `gws` commands. It does not change `gws`, +authenticate separately, or download document hyperlinks or image `contentUri`s. + +## Run + +Requires Python 3.10+ on Linux or macOS and an authenticated `gws` executable +with read access to the document through both Docs and Drive. + +From the repository root: + +```sh +python3 examples/docs-review-bundle/docs_review_bundle.py \ + --document-id DOCUMENT_ID review-bundle +``` + +`review-bundle` must be a **new relative directory inside the current working +directory**. Existing directories, absolute paths, `..`, symlink components, +control characters, and names outside the portable ASCII subset are refused. +Nested paths work when their parents already exist. The final directory is +created with mode `0700`; keep its parents under your control. + +Optional orchestration flags: + +```sh +python3 examples/docs-review-bundle/docs_review_bundle.py \ + --document-id DOCUMENT_ID --include-comments --render-pages \ + --timeout 120 --gws /trusted/path/to/gws another-review-bundle +``` + +- `--include-comments`: collect every returned Drive comments page. The entire + optional artifact is omitted and marked unavailable if retrieval is incomplete + or fails, or if the final serialized artifact exceeds 20 MiB. Size is checked + before publication, so an oversized optional result does not fail the required + bundle. Comments are a separate observation, not revision-bound or mapped to + PDF coordinates. Deleted comments are not requested. +- `--render-pages`: use a trusted `pdftoppm` on `PATH` to generate 96 DPI PNGs. + Its path is resolved before running inside the bundle, including relative + `PATH` entries. Rendering is opt-in. Missing tools, failures, timeouts, invalid + output and noncontiguous page numbers retain the PDF and report no available + raster previews. Outputs from these detected failures are discarded. After a + successful process exit and validation, previews are labeled `available` + with `coverage: "unverified"`; a missing page suffix cannot be detected. +- `--timeout`: positive finite seconds per subprocess; default 60. This is not + a total workflow deadline. +- `--gws`: trusted executable, resolved before changing subprocess working + directories. Existing `gws` authentication and Model Armor environment + settings are inherited. No credential values or raw process diagnostics + are inserted into HTML or error messages. + +The companion requests `docs documents get` with `includeTabsContent: true` +and uses `drive files export` with `--format json` and fixed relative +`--output` filenames. Every `gws` subprocess runs inside the new bundle. +Export success, MIME type, destination and byte count are checked against the +written artifact. The receipt must name the exact canonical absolute destination, +as returned by `gws`; a matching basename alone is insufficient. This requires +the existing `gws` binary-export receipt format. + +Open `index.html` locally. The index has no JavaScript or remote dependencies. +It contains a sandboxed PDF frame, optional page images, a paragraph/table +outline grouped by native tabs, DOCX figures and nearby text, native object +metadata, and escaped Markdown source. Some browsers block local PDF frames; +use the PDF artifact link in that case. Markdown is readable escaped source, +not rendered Markdown. + +## Artifacts and completion + +| File | Meaning | +| --- | --- | +| `source.json` | Original Docs JSON response, including all returned tabs | +| `revision-after.json` | Final revision observation, when available | +| `document.pdf`, `document.docx`, `document.md` | Required native Drive exports | +| `comments.json` | Optional fully retrieved comments result | +| `assets/` | Recognized DOCX PNG, JPEG, GIF and WebP media | +| `pages/` | Optional successful local page-rendering output | +| `index.html` | Local review index | +| `manifest.json` | State, versions, capabilities, limitations, mappings and hashes | + +The manifest begins as `in-progress` and becomes `complete` only after all +required exports, validation, index generation and artifact hashing succeed. +`complete` means the required bundle files were produced; it does **not** mean +an atomic snapshot, successful optional rendering, or verified export tab +coverage. Check `revisions`, `comments`, and `rendering` independently. + +Page previews are never labeled `complete`. `rendering.status: "available"` +means that local PNG files passed validation, while +`rendering.coverage: "unverified"` means the original PDF page count was not +independently checked. Even a contiguous list beginning with page 1 may omit +later pages. The HTML displays this same coverage limitation. + +Revision status is `mixed` for differing observed Docs revision IDs, `unchanged` +for equal nonempty IDs, and `unknown` if either is missing. Even `unchanged` +does not prove atomicity or that every export represents the same revision. +The manifest always records `atomic_snapshot: false`. + +Required failures return exit code 1 and leave a `failed` manifest with a safe +error code and stage. If writing the failure state also fails (for example, +a full disk), the earlier `in-progress` state can remain. Process termination +can also leave that state. No such bundle should be treated as complete. +Usage errors return 2. Optional failures and mixed/unknown revisions return 0 +when the required bundle completes. Existing output directories are never +overwritten; retries need a new name. + +SHA-256 and byte counts cover each produced artifact, including the index and +assets. The manifest does not hash itself and is not an authenticity signature. + +## Scope and safety limits + +- Native JSON traversal includes nested tabs, body paragraphs, tables and + tables of contents. The outline identifies paragraph styles but does not + recreate full layout. Headers, footers, notes, lists, equations, charts, + suggestions and other document features are not fully represented. +- Google determines the PDF/DOCX/Markdown export layout and tab coverage. + This companion cannot verify that every tab appears in those formats or + associate a PDF page/DOCX figure with an exact native tab. +- DOCX figure order follows individual image occurrences in the main document, + including drawings in tables and nested text boxes. Each occurrence uses its + nearest drawing's alt text and nearest paragraph's text; legitimate repeated + uses of an asset remain separate figures. Alt text and nearby paragraphs are + context, **not exact captions**. Native object IDs are recorded separately; + the companion never fabricates a matching native ID or source tab ID. +- Media basenames are replaced with distinct generated local filenames. + External, missing, traversing and unsupported relationships remain visibly + unavailable. Unreferenced recognized rasters are retained as artifacts. +- ZIP validation rejects traversal, absolute/Windows/control-character paths, + duplicate names (case-insensitive), symlinks and other special files, + encryption and unsupported compression. It never calls `extractall`. + Limits are 2,000 members, 20 MiB per member and 100 MiB total uncompressed. + ZIP paths are restricted to printable ASCII. Only stored/deflate compression + is accepted. +- Every XML/relationship member is parsed after rejecting DTDs, entities, + UTF-16/32 and non-UTF-8 encodings. XML trees are limited to 100 levels and + 100,000 nodes per part. The supported DOCX vocabulary is transitional OOXML + main-document drawings; unsupported content can remain unmapped. +- Each source/export/JSON artifact is capped at 20 MiB. Page output is limited + to 500 files and 100 MiB total. PDF header/EOF and raster signatures are + checked; these are **not full format validation**. Renderer exit success and + contiguous numbering do not independently prove the original PDF page count, + so preview coverage is always labeled unverified. +- Subprocess capture is file-backed, with size checks after exit. Timeouts and + post-render limits do not enforce disk or memory quotas on external tools. + `pdftoppm`, `gws`, local viewers and the operator-controlled parent directory + are trusted. This is not a sandbox against a concurrent local attacker. +- Display text is HTML/attribute escaped; URI references use generated local + allowlisted names. The index has a restrictive content-security policy and + no scripts. Source URLs and recognizable bearer/token strings are redacted + from display text, but this is not a general secret scanner or Model Armor + replacement. Existing `gws` sanitization behavior is preserved; binary + exports are not made safe by JSON sanitization. +- Raw artifacts deliberately retain original content, including any temporary + URLs or sensitive text. Treat the whole directory as sensitive. Review + external links and active content in native viewers separately; the + companion never follows them automatically. + +## Offline fixtures and visual QA + +Offline mode performs no `gws` calls. Supply a relative directory containing +`source.json`, `document.pdf`, `document.docx`, and UTF-8 `document.md`. +`revision-after.json` and `comments.json` are optional; a missing revision +observation remains unknown. Files and path components must not be symlinks. + +Generate entirely synthetic fixtures using the test utility, then create a +review bundle suitable for inspecting the HTML: + +```sh +python3 -B - <<'PY' +import sys +from pathlib import Path +sys.path.insert(0, "examples/docs-review-bundle") +from test_docs_review_bundle import fixtures +fixtures(Path("synthetic-docs-fixture")) +PY + +python3 -B examples/docs-review-bundle/docs_review_bundle.py \ + --from-fixture synthetic-docs-fixture --render-pages synthetic-docs-review +``` + +These generated fixtures intentionally contain HTML injection strings, an +untrusted synthetic URL, duplicate image basenames, missing image URIs, +nested tabs and table content. They are hand-built test data, not an actual +Google export or proof of cross-format fidelity. No real documents, +credentials or network access are needed. Omit `--render-pages` to avoid +running external software. + +## Tests + +```sh +python3 -B -m unittest discover \ + -s examples/docs-review-bundle -p 'test_*.py' -v +``` + +Tests use generated JSON/PDF/DOCX files and explicit `gws`/renderer executables +as stubs. Their subprocess environment omits real authentication settings; +no installed `gws`, real document, or network request is needed for those tests. + +To include the real CLI export-contract regression: + +```sh +cargo build --locked +GWS_TEST_BINARY="$PWD/target/debug/gws" python3 -B -m unittest discover \ + -s examples/docs-review-bundle -p 'test_*.py' -v +``` + +This additional test uses cached synthetic Discovery, a dummy token, isolated +configuration and ADC paths, and a loopback HTTP server. It downloads all three +generated exports through the real CLI and checks that the bundle completes. +It never contacts Google or uses real credentials. The dedicated Linux/macOS CI +job builds `gws` and always enables this test; local runs without +`GWS_TEST_BINARY` explicitly skip it. diff --git a/examples/docs-review-bundle/docs_review_bundle.py b/examples/docs-review-bundle/docs_review_bundle.py new file mode 100755 index 000000000..2ba9ee7c0 --- /dev/null +++ b/examples/docs-review-bundle/docs_review_bundle.py @@ -0,0 +1,783 @@ +#!/usr/bin/env python3 +# Copyright 2026 Google LLC +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy at https://www.apache.org/licenses/LICENSE-2.0 +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Build a local visual review bundle using Python's standard library and gws.""" + +import argparse +import hashlib +import html +import io +import json +import math +import os +from pathlib import Path, PurePosixPath +import re +import shutil +import stat +import subprocess +import sys +import tempfile +import xml.etree.ElementTree as ET +import zipfile + + +VERSION = "1.0" +MIB = 1024 * 1024 +FILE_LIMIT = 20 * MIB +TOTAL_LIMIT = 100 * MIB +MEMBER_COUNT = 2000 +MAX_PAGES = 500 +EXPORTS = { + "document.pdf": "application/pdf", + "document.docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document", + "document.md": "text/markdown", +} +W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}" +A = "{http://schemas.openxmlformats.org/drawingml/2006/main}" +WP = "{http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing}" +R = "{http://schemas.openxmlformats.org/officeDocument/2006/relationships}" +REL = "{http://schemas.openxmlformats.org/package/2006/relationships}" +LIMITATIONS = [ + "Exports are sequential, not an atomic snapshot; unchanged revisions are observations only.", + "Native outline includes body paragraphs, tables and nested tabs, not full layout or styling.", + "Drive export tab coverage is not verified; pages and DOCX figures have no reliable tab mapping.", + "Page preview coverage is unverified; available previews may omit pages.", + "DOCX relationships provide document order and nearby text, " + "not exact captions or native Docs IDs.", + "Only recognized PNG, JPEG, GIF and WebP media are previewed; " + "signatures are not full validation.", + "Headers, footers, notes, charts, vectors and unsupported drawings " + "may be absent from the outline or figures.", + "Content URIs and document hyperlinks are never fetched; URLs are redacted from display text.", + "Raw exports and JSON are sensitive, unsanitized source artifacts " + "and may contain temporary URLs.", + "The companion is not a document sanitizer or a sandbox for PDF/image viewers or pdftoppm.", +] + + +class BundleError(Exception): + """A fixed, non-sensitive failure code safe for manifests and terminals.""" + + +def relative_parts(value): + """Conservative portable path subset; do not normalize away traversal.""" + parts = value.split("/") + if not parts or any( + part in ("", ".", "..") or not re.fullmatch(r"[A-Za-z0-9_. -]+", part) + for part in parts + ): + raise BundleError("unsafe-relative-path") + return parts + + +def directory(value, *, create=False): + """Walk existing parents with no-follow descriptors (Linux/macOS).""" + parts = relative_parts(value) + flags = os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW + fd = os.open(".", flags) + try: + for index, part in enumerate(parts): + if create and index == len(parts) - 1: + os.mkdir(part, mode=0o700, dir_fd=fd) + next_fd = os.open(part, flags, dir_fd=fd) + os.close(fd) + fd = next_fd + except OSError: + raise BundleError("directory-exists-or-unsafe") from None + finally: + os.close(fd) + return Path.cwd().joinpath(*parts) + + +def read_bytes(path, limit=FILE_LIMIT): + """Bound reads and reject special files, including symlinks.""" + try: + fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + with os.fdopen(fd, "rb") as stream: + info = os.fstat(stream.fileno()) + if not stat.S_ISREG(info.st_mode) or info.st_size > limit: + raise BundleError("file-type-or-size-limit") + data = stream.read(limit + 1) + except OSError: + raise BundleError("missing-or-unsafe-file") from None + if len(data) > limit: + raise BundleError("file-size-limit") + return data + + +def write_bytes(path, data): + # Files are new, inside the newly created private bundle directory. + with path.open("xb") as stream: + stream.write(data) + + +def json_bytes(value): + return (json.dumps(value, ensure_ascii=True, indent=2) + "\n").encode("utf-8") + + +def parse_json(data): + try: + value = json.loads(data) + except (ValueError, UnicodeError): + raise BundleError("invalid-json") from None + if not isinstance(value, dict) or "error" in value: + raise BundleError("invalid-json-response") + return value + + +def publish_manifest(bundle, manifest): + temporary = bundle / ".manifest.tmp" + write_bytes(temporary, json_bytes(manifest)) + os.replace(temporary, bundle / "manifest.json") + + +def local_uri(value): + """HTML references are generated file names, never document-supplied URIs.""" + if not re.fullmatch(r"[A-Za-z0-9_-]+(?:[./][A-Za-z0-9_-]+)*", value): + raise BundleError("unsafe-local-uri") + return value + + +def display(value): + text = str(value or "") + text = re.sub(r"(?i)\b(?:https?|ftp|file|data|javascript):[^\s<>\"']+", "[URL omitted]", text) + text = re.sub(r"(?i)\bBearer\s+\S+", "[credential omitted]", text) + text = re.sub( + r"(?i)\b(?:access_token|authorization|token)\s*[:=]\s*[^\s<>\"']+", + "[credential omitted]", + text, + ) + text = "".join(char for char in text if char in "\n\t" or ord(char) >= 32) + return html.escape(text, quote=True) + + +def safe_xml(data): + # Reject UTF-16/32 (including declaration smuggling via NUL bytes), DTDs and + # entities before giving XML to the stdlib parser. DOCX normally uses UTF-8. + if b"\x00" in data or re.search(br"]*\bencoding\s*=\s*['\"]([^'\"]+)", text, re.I) + if declaration and declaration[1].lower() not in ("utf-8", "utf8", "us-ascii"): + raise BundleError("unsupported-xml-encoding") + root = ET.fromstring(text) + except (ET.ParseError, UnicodeError): + raise BundleError("invalid-xml") from None + pending = [(root, 0)] + count = 0 + while pending: + node, depth = pending.pop() + count += 1 + if depth > 100 or count > 100_000: + raise BundleError("xml-complexity-limit") + pending.extend((child, depth + 1) for child in node) + return root + + +def raster_extension(data): + if data.startswith(b"\x89PNG\r\n\x1a\n"): + return "png" + if data.startswith(b"\xff\xd8\xff"): + return "jpg" + if data.startswith((b"GIF87a", b"GIF89a")): + return "gif" + if data.startswith(b"RIFF") and data[8:12] == b"WEBP": + return "webp" + return None + + +def zip_members(source, member_limit, total_limit, member_count): + """Read a bounded archive without ever extracting its member paths.""" + try: + with zipfile.ZipFile(io.BytesIO(read_bytes(source))) as archive: + infos = archive.infolist() + if len(infos) > member_count: + raise BundleError("zip-member-count-limit") + seen = set() + declared_total = 0 + for info in infos: + name = info.orig_filename + parts = name.rstrip("/").split("/") + kind = stat.S_IFMT(info.external_attr >> 16) + if ( + name != info.filename + or name.startswith("/") + or "\\" in name + or ":" in name + or any(part in ("", ".", "..") for part in parts) + or any(ord(char) < 32 or ord(char) > 126 for char in name) + or kind not in (0, stat.S_IFREG, stat.S_IFDIR) + or info.flag_bits & 1 + or info.compress_type not in (zipfile.ZIP_STORED, zipfile.ZIP_DEFLATED) + or name.casefold() in seen + ): + raise BundleError("unsafe-zip-member") + seen.add(name.casefold()) + declared_total += info.file_size + if info.file_size > member_limit or declared_total > total_limit: + raise BundleError("zip-size-limit") + members = {} + actual_total = 0 + for info in infos: + if info.is_dir(): + continue + with archive.open(info) as stream: + data = stream.read(member_limit + 1) + actual_total += len(data) + if ( + len(data) > member_limit + or actual_total > total_limit + or len(data) != info.file_size + ): + raise BundleError("zip-size-limit") + members[info.filename] = data + return members + except (zipfile.BadZipFile, RuntimeError, NotImplementedError, OSError, EOFError): + raise BundleError("invalid-zip") from None + + +def docx_image_occurrences(document): + """Visit each blip once, retaining its nearest paragraph and drawing.""" + paragraph_text = {} + paragraph_index = {} + drawing_alt = {} + occurrences = [] + pending = [(document, None, None)] + while pending: + node, paragraph, drawing = pending.pop() + if node.tag == W + "p": + paragraph = node + paragraph_index[node] = len(paragraph_text) + paragraph_text[node] = [] + elif node.tag == W + "drawing": + drawing = node + elif node.tag == W + "t" and paragraph is not None: + paragraph_text[paragraph].append(node.text or "") + elif node.tag == WP + "docPr" and drawing is not None: + drawing_alt.setdefault( + drawing, " ".join(node.get(field, "") for field in ("title", "descr")).strip() + ) + elif node.tag == A + "blip" and paragraph is not None and drawing is not None: + occurrences.append((node, paragraph, drawing)) + pending.extend((child, paragraph, drawing) for child in reversed(node)) + + texts = ["".join(parts) for parts in paragraph_text.values()] + for blip, paragraph, drawing in occurrences: + index = paragraph_index[paragraph] + nearby = texts[index] or " ".join( + texts[max(0, index - 1):index] + texts[index + 1:index + 2] + ) + yield blip, drawing_alt.get(drawing, ""), nearby + + +def extract_docx( + source, output, *, + member_limit=FILE_LIMIT, total_limit=TOTAL_LIMIT, member_count=MEMBER_COUNT, +): + members = zip_members(source, member_limit, total_limit, member_count) + # Validate even unused XML before any asset writes. + trees = { + name: safe_xml(data) + for name, data in members.items() + if name.lower().endswith((".xml", ".rels")) + } + document = trees.get("word/document.xml") + if document is None or document.tag != W + "document": + raise BundleError("missing-docx-document") + relationships = {} + rels = trees.get("word/_rels/document.xml.rels") + if rels is not None: + for rel in rels.findall(REL + "Relationship"): + identity = rel.get("Id") + if not identity or identity in relationships: + raise BundleError("ambiguous-docx-relationship") + relationships[identity] = rel.attrib + output.mkdir(mode=0o700) + assets = {} + for name, data in members.items(): + extension = raster_extension(data) + if name.startswith("word/media/") and extension: + filename = f"image-{len(assets) + 1}.{extension}" + write_bytes(output / filename, data) + assets[name] = local_uri(f"{output.name}/{filename}") + figures = [] + for blip, alt, nearby in docx_image_occurrences(document): + identity = blip.get(R + "embed") or blip.get(R + "link") + relationship = relationships.get(identity, {}) + asset = None + availability = "missing-or-unsupported" + if relationship.get("TargetMode", "").lower() == "external": + availability = "external-not-fetched" + elif relationship.get("Type") == R[1:-1] + "/image": + target = relationship.get("Target", "") + # Allow only relative media targets in the document's part. + if ( + target.startswith("media/") + and not any(p in ("", ".", "..") for p in target.split("/")) + and not any(char in target for char in "\\:%?#") + ): + asset = assets.get(str(PurePosixPath("word") / target)) + if asset: + availability = "available" + figures.append({ + "order": len(figures) + 1, + "relationship_id": identity, + "asset": asset, + "alt": alt, + "nearby_text": nearby[:1000], + "availability": availability, + "mapping_confidence": "docx-relationship-only", + "native_object_id": None, + "source_tab_id": None, + }) + return {"figures": figures, "assets": list(assets.values())} + + +def native_view(source): + if not isinstance(source.get("body"), dict) and not isinstance(source.get("tabs"), list): + raise BundleError("missing-document-content") + tabs = [] + images = [] + + def add_tab(tab, depth): + if depth > 50 or len(tabs) >= 1000: + raise BundleError("native-tab-limit") + properties = tab.get("tabProperties", {}) + document = tab.get("documentTab", {}) + tab_id = properties.get("tabId") + tabs.append({ + "id": tab_id, + "title": properties.get("title", "Document"), + "depth": depth, + "blocks": document.get("body", {}).get("content", []), + }) + for collection, property_name in ( + ("inlineObjects", "inlineObjectProperties"), + ("positionedObjects", "positionedObjectProperties"), + ): + for object_id, obj in document.get(collection, {}).items(): + embedded = obj.get(property_name, {}).get("embeddedObject", {}) + images.append({ + "object_id": object_id, + "tab_id": tab_id, + "title": embedded.get("title", ""), + "description": embedded.get("description", ""), + "content_uri_available": bool( + embedded.get("imageProperties", {}).get("contentUri") + ), + "asset": None, + "mapping_confidence": "unmapped", + }) + for child in tab.get("childTabs", []): + add_tab(child, depth + 1) + + if source.get("tabs"): + for tab in source["tabs"]: + add_tab(tab, 0) + else: + add_tab({"documentTab": source}, 0) + return tabs, images + + +def outline_html(blocks, depth=0): + if depth > 50: + raise BundleError("native-outline-depth-limit") + parts = [] + for block in blocks: + if "paragraph" in block: + paragraph = block["paragraph"] + text = "".join( + element.get("textRun", {}).get("content", "") + for element in paragraph.get("elements", []) + ) + style = paragraph.get("paragraphStyle", {}).get("namedStyleType", "NORMAL_TEXT") + parts.append(f"

{display(style)} {display(text)}

") + elif "table" in block: + parts.append("") + for row in block["table"].get("tableRows", []): + parts.append("") + for cell in row.get("tableCells", []): + parts.append( + "" + ) + parts.append("") + parts.append("
" + outline_html(cell.get("content", []), depth + 1) + "
") + elif "tableOfContents" in block: + parts.append(outline_html(block["tableOfContents"].get("content", []), depth + 1)) + return "".join(parts) + + +def index_html(source, markdown, tabs, manifest, artifacts): + parts = [ + '', + '', + '", + "Docs review bundle", + "", + f"

{display(source.get('title', 'Docs review bundle'))}

", + f"

Revision observation: {display(manifest['revisions']['status'])}. " + "Sequential exports; not an atomic snapshot.

", + "

Artifacts

    ", + ] + for name in [*artifacts, "manifest.json"]: + parts.append(f'
  • {display(name)}
  • ') + parts.extend( + [ + "

Raw files may contain sensitive content and temporary URLs.

", + "

Native PDF

", + '', + "

If your browser blocks the embedded viewer, open the local PDF artifact.

", + f"

Raster previews: {display(manifest['rendering']['status'])}; " + f"{display(manifest['rendering'].get('reason', ''))}

", + ] + ) + if manifest["rendering"]["pages"]: + parts.append( + f"

Page coverage: {display(manifest['rendering']['coverage'])}. " + "The PDF page count has not been verified; previews may omit pages.

" + ) + for page in manifest["rendering"]["pages"]: + parts.append(f'{display(page)}') + parts.append("

Native outline and tabs

") + for tab in tabs: + parts.append( + f"

{display(tab['title'])}

" + f"

Tab: {display(tab['id'] or 'legacy body')}; " + f"depth: {tab['depth']}

{outline_html(tab['blocks'])}" + ) + parts.append("

DOCX figures

") + for figure in manifest["figures"]: + parts.append(f"

Figure {figure['order']}

") + if figure["asset"]: + parts.append( + f'' + ) + parts.append( + f"
{display(figure['alt'])}
" + f"

Availability: {display(figure['availability'])}

" + f"

Nearby text (not an exact caption): {display(figure['nearby_text'])}

" + "

DOCX relationship only; native object and source tab mapping unknown.

" + ) + parts.append("

Native image metadata

") + for image in manifest["native_images"]: + parts.append( + f"

Object {display(image['object_id'])}, tab {display(image['tab_id'])}: " + f"{display(image['title'])} {display(image['description'])}. " + f"contentUri available: {image['content_uri_available']}; asset mapping: unmapped.

" + ) + parts.append( + "

Readable Markdown source

" + f"
{display(markdown)}
" + ) + parts.append( + f"

Comments

{display(manifest['comments']['status'])}

" + ) + parts.append("

Capabilities and limitations

    ") + parts.extend(f"
  • {display(item)}
  • " for item in LIMITATIONS) + parts.append("
") + return "".join(parts).encode("utf-8") + + +def run_process(argv, bundle, timeout): + # File-backed capture bounds memory. No shell and no untrusted diagnostics + # echoed to the terminal or copied into the manifest/display HTML. + with tempfile.TemporaryFile(dir=bundle) as output: + try: + process = subprocess.run( + argv, + cwd=bundle, + stdout=output, + stderr=subprocess.DEVNULL, + stdin=subprocess.DEVNULL, + timeout=timeout, + check=False, + ) + except subprocess.TimeoutExpired: + raise BundleError("timeout") from None + except OSError: + raise BundleError("process-unavailable") from None + if process.returncode: + raise BundleError("process-failed") + if output.tell() > FILE_LIMIT: + raise BundleError("process-output-limit") + output.seek(0) + return output.read(FILE_LIMIT + 1) + + +def gws_json(executable, bundle, timeout, command, params, output=None): + argv = [executable, *command, "--params", json.dumps(params), "--format", "json"] + if output: + argv.extend(["--output", local_uri(output)]) + data = run_process(argv, bundle, timeout) + return data, parse_json(data) + + +def collect_comments(executable, bundle, timeout, document_id): + comments = [] + seen = set() + token = None + total = 0 + for _ in range(100): + params = {"fileId": document_id, "fields": "nextPageToken,comments", "pageSize": 100} + if token: + params["pageToken"] = token + raw, page = gws_json(executable, bundle, timeout, ["drive", "comments", "list"], params) + total += len(raw) + if total > FILE_LIMIT or not isinstance(page.get("comments", []), list): + raise BundleError("invalid-or-oversized-comments") + comments.extend(page.get("comments", [])) + token = page.get("nextPageToken") + if not token: + return json_bytes({"comments": comments}) + if not isinstance(token, str) or token in seen: + raise BundleError("incomplete-comments") + seen.add(token) + raise BundleError("incomplete-comments") + + +def render_pages(bundle, timeout, requested): + result = {"status": "not-requested", "pages": []} + if not requested: + return result + renderer = shutil.which("pdftoppm") + if not renderer: + return {"status": "unavailable", "reason": "pdftoppm-not-found; PDF retained", "pages": []} + try: + renderer = str(Path(renderer).resolve()) + with tempfile.TemporaryDirectory(prefix=".render-", dir=bundle) as temp: + staging = Path(temp) + prefix = str(staging.relative_to(bundle) / "page") + run_process([renderer, "-png", "-r", "96", "document.pdf", prefix], bundle, timeout) + pages = {} + total = 0 + for path in staging.iterdir(): + match = re.fullmatch(r"page-([0-9]+)\.png", path.name) + if not match: + raise BundleError("invalid-render") + number = int(match[1]) + data = read_bytes(path) + total += len(data) + if number in pages or raster_extension(data) != "png" or total > TOTAL_LIMIT: + raise BundleError("invalid-render") + pages[number] = path.name + if ( + not pages + or len(pages) > MAX_PAGES + or sorted(pages) != list(range(1, len(pages) + 1)) + ): + raise BundleError("invalid-render") + staging.rename(bundle / "pages") + return { + "status": "available", + "coverage": "unverified", + "pages": [local_uri("pages/" + pages[number]) for number in sorted(pages)], + } + except (BundleError, OSError) as error: + return { + "status": "failed", "pages": [], + "reason": "timeout" if str(error) == "timeout" else "invalid-or-failed-render", + } + + +def revision_observation(before, after): + before_id = before.get("revisionId") + after_id = after.get("revisionId") + before_id = before_id if isinstance(before_id, str) and before_id else None + after_id = after_id if isinstance(after_id, str) and after_id else None + status = "unknown" + if before_id and after_id: + status = "unchanged" if before_id == after_id else "mixed" + return {"before": before_id, "after": after_id, "status": status, "atomic_snapshot": False} + + +def build_bundle(args, bundle, manifest): + artifacts = ["source.json", *EXPORTS] + manifest["stage"] = "exports" + if args.from_fixture: + fixture = directory(args.from_fixture) + for name in artifacts: + write_bytes(bundle / name, read_bytes(fixture / name)) + source = parse_json(read_bytes(bundle / "source.json")) + after_path = fixture / "revision-after.json" + after = parse_json(read_bytes(after_path)) if after_path.exists() else {} + if after_path.exists(): + write_bytes(bundle / "revision-after.json", read_bytes(after_path)) + artifacts.append("revision-after.json") + executable = None + else: + executable = shutil.which(args.gws) + if not executable: + raise BundleError("gws-not-found") + executable = str(Path(executable).resolve()) + try: + version = run_process([executable, "--version"], bundle, args.timeout).decode("utf-8") + match = re.fullmatch(r"gws ([0-9][A-Za-z0-9.+-]*)\s*", version) + manifest["versions"]["gws"] = match[1] if match else "unknown" + except BundleError: + manifest["versions"]["gws"] = "unknown" + params = {"documentId": args.document_id, "includeTabsContent": True} + raw, source = gws_json( + executable, bundle, args.timeout, ["docs", "documents", "get"], params + ) + if source.get("documentId") != args.document_id: + raise BundleError("document-id-mismatch") + write_bytes(bundle / "source.json", raw) + for name, mime in EXPORTS.items(): + _, receipt = gws_json( + executable, bundle, args.timeout, ["drive", "files", "export"], + {"fileId": args.document_id, "mimeType": mime}, output=name, + ) + data = read_bytes(bundle / name) + if ( + receipt.get("status") != "success" + or receipt.get("saved_file") != str((bundle / name).resolve()) + or str(receipt.get("mimeType", "")).split(";")[0].strip().lower() != mime + or type(receipt.get("bytes")) is not int + or receipt["bytes"] != len(data) + ): + raise BundleError("invalid-export-receipt") + raw, after = gws_json( + executable, bundle, args.timeout, ["docs", "documents", "get"], + {**params, "fields": "revisionId"}, + ) + write_bytes(bundle / "revision-after.json", raw) + artifacts.append("revision-after.json") + + manifest["revisions"] = revision_observation(source, after) + manifest["comments"] = {"status": "not-requested"} + if args.include_comments: + try: + if executable: + comments = collect_comments(executable, bundle, args.timeout, args.document_id) + else: + comments = read_bytes(fixture / "comments.json") + value = parse_json(comments) + if not isinstance(value.get("comments"), list) or value.get("nextPageToken"): + raise BundleError("incomplete-comments") + if len(comments) > FILE_LIMIT: + raise BundleError("comments-size-limit") + write_bytes(bundle / "comments.json", comments) + artifacts.append("comments.json") + manifest["comments"] = {"status": "available"} + except (BundleError, OSError) as error: + (bundle / "comments.json").unlink(missing_ok=True) + manifest["comments"] = { + "status": "unavailable", + "reason": ( + "comments-size-limit" + if isinstance(error, BundleError) and str(error) == "comments-size-limit" + else "retrieval-incomplete-or-failed" + ), + } + + manifest["stage"] = "validate-and-extract" + pdf = read_bytes(bundle / "document.pdf") + if not pdf.startswith(b"%PDF-") or b"%%EOF" not in pdf[-1024:]: + raise BundleError("invalid-or-truncated-pdf") + markdown = read_bytes(bundle / "document.md").decode("utf-8") + tabs, native_images = native_view(source) + extracted = extract_docx(bundle / "document.docx", bundle / "assets") + manifest["figures"] = extracted["figures"] + manifest["native_images"] = native_images + manifest["tabs"] = [{k: v for k, v in tab.items() if k != "blocks"} for tab in tabs] + artifacts.extend(extracted["assets"]) + manifest["stage"] = "render" + manifest["rendering"] = render_pages(bundle, args.timeout, args.render_pages) + artifacts.extend(manifest["rendering"]["pages"]) + manifest["stage"] = "index" + write_bytes(bundle / "index.html", index_html(source, markdown, tabs, manifest, artifacts)) + artifacts.append("index.html") + manifest["artifacts"] = {} + for name in artifacts: + data = read_bytes(bundle / local_uri(name)) + manifest["artifacts"][name] = { + "bytes": len(data), + "sha256": hashlib.sha256(data).hexdigest(), + } + manifest["status"] = "complete" + manifest["stage"] = "complete" + publish_manifest(bundle, manifest) + + +def positive_timeout(value): + number = float(value) + if not math.isfinite(number) or number <= 0: + raise argparse.ArgumentTypeError("timeout must be a positive finite number") + return number + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + source = parser.add_mutually_exclusive_group(required=True) + source.add_argument("--document-id", help="Google Docs ID (not a URL)") + source.add_argument( + "--from-fixture", help="Relative directory containing captured export files" + ) + parser.add_argument("output_dir", help="New relative directory within CWD; parents must exist") + parser.add_argument("--include-comments", action="store_true") + parser.add_argument( + "--render-pages", action="store_true", help="Opt in to local pdftoppm rendering" + ) + parser.add_argument( + "--timeout", type=positive_timeout, default=60.0, + help="Seconds per process (default: 60)", + ) + parser.add_argument( + "--gws", default="gws", help="Trusted gws executable (default: PATH lookup)" + ) + args = parser.parse_args(argv) + if args.document_id and not re.fullmatch(r"[A-Za-z0-9_-]+", args.document_id): + parser.error("document-id must be a Google Docs ID, not a URL or path") + bundle = None + manifest = { + "schema_version": 1, "status": "in-progress", "stage": "initialize", + "mode": "offline" if args.from_fixture else "gws", + "versions": {"companion": VERSION, "python": sys.version.split()[0], "gws": None}, + "capabilities": { + "all_native_tabs": True, "native_pdf": True, "docx_raster_assets": True, + "native_markdown": True, "network_image_fetch": False, + "exact_native_asset_mapping": False, "atomic_snapshot": False, + }, + "limitations": LIMITATIONS, + } + try: + bundle = directory(args.output_dir, create=True) + publish_manifest(bundle, manifest) + build_bundle(args, bundle, manifest) + except (Exception, KeyboardInterrupt) as error: + code = str(error) if isinstance(error, BundleError) else "invalid-or-incomplete-bundle" + if bundle is not None: + manifest["status"] = "failed" + manifest["error"] = code + try: + publish_manifest(bundle, manifest) + except OSError: + # An earlier in-progress manifest remains non-complete if storage fails. + pass + print(f"Review bundle failed: {code}.", file=sys.stderr) + return 1 + print("Review bundle complete. Open index.html in the new output directory.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/examples/docs-review-bundle/test_docs_review_bundle.py b/examples/docs-review-bundle/test_docs_review_bundle.py new file mode 100644 index 000000000..299ab0614 --- /dev/null +++ b/examples/docs-review-bundle/test_docs_review_bundle.py @@ -0,0 +1,1058 @@ +#!/usr/bin/env python3 +# Copyright 2026 Google LLC +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy at https://www.apache.org/licenses/LICENSE-2.0 +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Hermetic behavior tests: generated documents and executable process stubs only.""" + +import base64 +import contextlib +import hashlib +import http.server +from html.parser import HTMLParser +import importlib.util +import io +import json +import os +from pathlib import Path +import stat +import subprocess +import sys +import tempfile +import threading +import unittest +from unittest import mock +import urllib.parse +import zipfile + + +SCRIPT = Path(__file__).with_name("docs_review_bundle.py") +PNG = base64.b64decode( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8" + "/x8AAwMCAO+jRZkAAAAASUVORK5CYII=" +) +W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" +R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +A = "http://schemas.openxmlformats.org/drawingml/2006/main" +WP = "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing" +REL = "http://schemas.openxmlformats.org/package/2006/relationships" +REMOTE = "https://untrusted.invalid/image?token=SIGNED_SECRET" + + +def paragraph(text, style="NORMAL_TEXT"): + return { + "paragraph": { + "paragraphStyle": {"namedStyleType": style}, + "elements": [ + {"textRun": {"content": text, "textStyle": {"link": {"url": REMOTE}}}} + ], + } + } + + +def native_document(): + return { + "documentId": "synthetic-doc", + "title": 'Review ', + "revisionId": "revision-one", + "tabs": [ + { + "tabProperties": {"tabId": "tab-main", "title": "Main"}, + "documentTab": { + "body": { + "content": [ + paragraph("Heading ", "HEADING_1"), + { + "table": { + "tableRows": [ + { + "tableCells": [ + {"content": [paragraph("Table cell")]} + ] + } + ] + } + }, + { + "paragraph": { + "elements": [ + { + "inlineObjectElement": { + "inlineObjectId": "native-image" + } + } + ] + } + }, + ] + }, + "inlineObjects": { + "native-image": { + "inlineObjectProperties": { + "embeddedObject": { + "title": "Native figure", + "description": "No exact DOCX mapping", + "imageProperties": {"contentUri": REMOTE}, + } + } + }, + "missing-uri": { + "inlineObjectProperties": { + "embeddedObject": { + "description": "No downloadable URI", + "imageProperties": {}, + } + } + }, + }, + }, + "childTabs": [ + { + "tabProperties": {"tabId": "tab-child", "title": "Child"}, + "documentTab": { + "body": {"content": [paragraph("Nested tab paragraph")]} + }, + } + ], + } + ], + } + + +def pdf_bytes(page_count=1): + """A complete synthetic PDF, also usable by real pdftoppm.""" + content = ( + b"BT /F1 22 Tf 48 720 Td (Synthetic Docs review) Tj ET\n" + b"BT /F1 12 Tf 48 687 Td (Local fixture - no Google document) Tj ET\n" + b"0.85 0.92 1 rg 48 435 516 210 re f\n" + b"0.08 0.25 0.5 rg 72 459 120 162 re f\n" + b"0.12 0.45 0.6 rg 216 459 120 105 re f\n" + b"0.1 0.6 0.45 rg 360 459 120 140 re f\n" + b"0 0 0 rg BT /F1 12 Tf 48 402 Td (Synthetic figure and table context) Tj ET\n" + b"0.5 G 48 270 516 90 re S 48 315 m 564 315 l S\n" + b"306 270 m 306 360 l S\n" + b"BT /F1 12 Tf 60 333 Td (Column A) Tj 258 0 Td (Column B) Tj ET\n" + b"BT /F1 12 Tf 60 288 Td (Cell one) Tj 258 0 Td (Cell two) Tj ET\n" + ) + content_id = page_count + 3 + font_id = page_count + 4 + kids = " ".join(f"{number} 0 R" for number in range(3, page_count + 3)) + objects = [ + b"<< /Type /Catalog /Pages 2 0 R >>", + f"<< /Type /Pages /Kids [{kids}] /Count {page_count} >>".encode(), + ] + objects.extend( + ( + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] " + f"/Resources << /Font << /F1 {font_id} 0 R >> >> /Contents {content_id} 0 R >>" + ).encode() + for _ in range(page_count) + ) + objects.extend([ + f"<< /Length {len(content)} >>\nstream\n".encode() + content + b"endstream", + b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", + ]) + data = b"%PDF-1.4\n" + offsets = [0] + for index, obj in enumerate(objects, 1): + offsets.append(len(data)) + data += f"{index} 0 obj\n".encode() + obj + b"\nendobj\n" + startxref = len(data) + object_count = len(objects) + 1 + data += f"xref\n0 {object_count}\n0000000000 65535 f \n".encode() + for offset in offsets[1:]: + data += f"{offset:010d} 00000 n \n".encode() + data += ( + f"trailer\n<< /Size {object_count} /Root 1 0 R >>\nstartxref\n{startxref}\n%%EOF\n" + ).encode() + return data + + +def docx_members(): + xml = f""" + + + +Nearby <script>bad()</script> + + +Table image context + + + + + +""" + rels = f""" + + +""" + return { + "word/document.xml": xml.encode(), + "word/_rels/document.xml.rels": rels.encode(), + "word/styles.xml": f''.encode(), + "word/media/image1.png": PNG, + "word/media/nested/image1.png": PNG + b"different", + } + + +def nested_docx_members(*, outer_image=False): + members = docx_members() + outer_blip = '' if outer_image else "" + members["word/document.xml"] = f""" + +Outer paragraph + +Inner paragraph + + + +{outer_blip} +""".encode() + return members + + +def write_docx(path, members=None): + with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as archive: + for name, data in (members if members is not None else docx_members()).items(): + archive.writestr(name, data) + + +def fixtures(path): + path.mkdir() + (path / "source.json").write_text(json.dumps(native_document()), encoding="utf-8") + (path / "revision-after.json").write_text( + '{"revisionId":"revision-one"}', encoding="utf-8" + ) + (path / "document.pdf").write_bytes(pdf_bytes()) + (path / "document.md").write_text( + "# Markdown\n\n![remote](" + REMOTE + ")\n", + encoding="utf-8", + ) + write_docx(path / "document.docx") + return path + + +GWS_STUB = r''' +import json +import os +from pathlib import Path +import shutil +import sys +import time + +args = sys.argv[1:] +mode = os.environ.get("STUB_MODE", "") +fixture = Path(os.environ["STUB_FIXTURES"]) +with open(os.environ["STUB_LOG"], "a", encoding="utf-8") as log: + log.write(json.dumps({"args": args, "cwd": os.getcwd(), + "sanitize": os.environ.get("GOOGLE_WORKSPACE_CLI_SANITIZE_MODE")}) + "\n") +if args == ["--version"]: + print("gws 0.0.0-synthetic") + sys.exit(0) +params = json.loads(args[args.index("--params") + 1]) +assert args[args.index("--format") + 1] == "json" +if mode == "timeout": + time.sleep(10) +if args[:3] == ["docs", "documents", "get"]: + assert params["documentId"] == "synthetic-doc" + assert params["includeTabsContent"] is True + if params.get("fields") == "revisionId": + source = {"revisionId": "revision-two" if mode == "mixed" else "revision-one"} + if mode == "missing-revision": + source = {} + else: + source = json.loads((fixture / "source.json").read_text()) + print(json.dumps(source)) +elif args[:3] == ["drive", "files", "export"]: + assert params["fileId"] == "synthetic-doc" + expected = { + "application/pdf": "document.pdf", + "application/vnd.openxmlformats-officedocument.wordprocessingml.document": "document.docx", + "text/markdown": "document.md", + } + name = args[args.index("--output") + 1] + assert expected[params["mimeType"]] == name + assert "/" not in name and "\\" not in name + if mode == "failed-export" and name == "document.docx": + Path(name).write_bytes(b"partial") + print("Bearer PRIVATE_TOKEN " + "\x1b[31m", file=sys.stderr) + sys.exit(7) + if mode != "no-export-file": + shutil.copyfile(fixture / name, name) + print(json.dumps({ + "status": "error" if mode == "bad-export-status" else "success", + "saved_file": ( + str(Path.cwd().parent / "other" / name) if mode == "wrong-export-path" + else name if mode == "relative-export-path" + else str(Path(name).resolve()) + ), + "mimeType": params["mimeType"], + "bytes": 1 if mode == "wrong-export-size" else (fixture / name).stat().st_size, + })) +elif args[:3] == ["drive", "comments", "list"]: + assert params["fileId"] == "synthetic-doc" + assert "nextPageToken" in params["fields"] + if mode == "expanded-comments": + # ~8 MiB on the wire; >24 MiB after ensure_ascii=True serialization. + print(json.dumps( + {"comments": [{"id": "large", "content": "é" * (4 * 1024 * 1024)}]}, + ensure_ascii=False, separators=(",", ":"), + )) + sys.exit(0) + if mode == "failed-comments" and params.get("pageToken"): + sys.exit(9) + if params.get("pageToken"): + print(json.dumps({"comments": [{"id": "two", "content": "Second comment"}]})) + else: + print(json.dumps({"nextPageToken": "page-two", "comments": [{"id": "one"}]})) +else: + raise AssertionError(args) +''' + +RENDER_STUB = r''' +import base64 +import os +from pathlib import Path +import sys +import time +assert sys.argv[1:4] == ["-png", "-r", "96"] +assert sys.argv[-2] == "document.pdf" +prefix = Path(sys.argv[-1]) +png = base64.b64decode(os.environ["STUB_PNG"]) +mode = os.environ.get("RENDER_MODE", "") +Path(str(prefix) + "-1.png").write_bytes(png) +if mode == "failed": + sys.exit(2) +if mode == "timeout": + time.sleep(10) +if mode == "gap": + Path(str(prefix) + "-3.png").write_bytes(png) +''' + + +class HTMLInspection(HTMLParser): + def __init__(self, text): + super().__init__() + self.tags = [] + self.references = [] + self.feed(text) + + def handle_starttag(self, tag, attrs): + self.tags.append((tag, dict(attrs))) + for name, value in attrs: + if name in ("src", "href", "data"): + self.references.append(value) + + +class BundleTestCase(unittest.TestCase): + def setUp(self): + # All generated files stay inside this assigned worktree and are removed. + self.temp = tempfile.TemporaryDirectory(dir=SCRIPT.parent) + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name).resolve() + self.fixture = fixtures(self.root / "fixtures") + self.bin = self.root / "bin" + self.bin.mkdir() + self.env = { + "PATH": str(self.bin), + "PYTHONDONTWRITEBYTECODE": "1", + "STUB_FIXTURES": str(self.fixture), + "STUB_LOG": str(self.root / "gws.log"), + "STUB_PNG": base64.b64encode(PNG).decode(), + "GOOGLE_WORKSPACE_CLI_CONFIG_DIR": str(self.root / "unused-config"), + "GOOGLE_WORKSPACE_CLI_SANITIZE_MODE": "block", + } + + def api(self): + self.assertTrue(SCRIPT.is_file(), "Missing executable review bundle companion") + if not hasattr(self, "_api"): + spec = importlib.util.spec_from_file_location("docs_review_bundle", SCRIPT) + self._api = importlib.util.module_from_spec(spec) + spec.loader.exec_module(self._api) + return self._api + + def executable(self, name, source): + path = self.bin / name + path.write_text(f"#!{sys.executable}\n" + source, encoding="utf-8") + path.chmod(0o700) + return path + + def run_bundle(self, *args, live=False, **env): + self.assertTrue(SCRIPT.is_file(), "Missing executable review bundle companion") + if live: + self.executable("gws", GWS_STUB) + source = ["--document-id", "synthetic-doc"] + else: + source = ["--from-fixture", "fixtures"] + return subprocess.run( + [sys.executable, "-B", str(SCRIPT), *source, *args], + cwd=self.root, + env={**self.env, **env}, + text=True, + capture_output=True, + timeout=15, + ) + + def manifest(self, directory="review"): + return json.loads((self.root / directory / "manifest.json").read_text()) + + +class ExportReceiptTests(BundleTestCase): + def test_canonical_receipts_complete_all_exports(self): + result = self.run_bundle("review", live=True) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertEqual(self.manifest()["status"], "complete") + + def test_wrong_destination_or_relative_receipt_is_refused(self): + for mode in ["wrong-export-path", "relative-export-path"]: + with self.subTest(mode=mode): + result = self.run_bundle(mode, live=True, STUB_MODE=mode) + self.assertEqual(result.returncode, 1, result.stderr) + self.assertEqual(self.manifest(mode)["error"], "invalid-export-receipt") + + +@unittest.skipUnless(os.environ.get("GWS_TEST_BINARY"), "Set GWS_TEST_BINARY for real CLI coverage") +class RealCliExportTests(BundleTestCase): + def test_real_cli_exports_complete_bundle_with_canonical_receipts(self): + binary = Path(os.environ["GWS_TEST_BINARY"]).resolve(strict=True) + fixture = self.fixture + requests = [] + exports = { + "application/pdf": "document.pdf", + "application/vnd.openxmlformats-officedocument.wordprocessingml.document": "document.docx", + "text/markdown": "document.md", + } + + class Handler(http.server.BaseHTTPRequestHandler): + def log_message(self, *_args): + pass + + def do_GET(self): + parsed = urllib.parse.urlsplit(self.path) + path = urllib.parse.unquote(parsed.path) + params = urllib.parse.parse_qs(parsed.query) + requests.append((path, params, self.headers.get("Authorization"))) + if path == "/documents/synthetic-doc": + data = (fixture / "source.json").read_bytes() + mime = "application/json" + elif path == "/files/synthetic-doc/export": + mime = params.get("mimeType", [""])[0] + if mime not in exports: + self.send_error(400) + return + data = (fixture / exports[mime]).read_bytes() + else: + self.send_error(404) + return + self.send_response(200) + self.send_header("Content-Type", mime) + self.send_header("Content-Length", str(len(data))) + self.end_headers() + self.wfile.write(data) + + server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + config = self.root / "real-cli-config" + cache = config / "cache" + cache.mkdir(parents=True) + (self.root / ".env").write_text("") + for service, version, resource, method, id_field, path in [ + ("docs", "v1", "documents", "get", "documentId", "documents/{documentId}"), + ("drive", "v3", "files", "export", "fileId", "files/{fileId}/export"), + ]: + discovery = { + "name": service, "version": version, + "rootUrl": f"http://127.0.0.1:{server.server_port}/", + "resources": {resource: {"methods": {method: { + "httpMethod": "GET", "path": path, + "parameters": {id_field: { + "type": "string", "location": "path", "required": True, + }}, + }}}}, + } + (cache / f"{service}_{version}.json").write_text(json.dumps(discovery)) + env = { + **self.env, + "GOOGLE_WORKSPACE_CLI_CONFIG_DIR": str(config), + "GOOGLE_WORKSPACE_CLI_TOKEN": "synthetic-loopback-token", + "GOOGLE_APPLICATION_CREDENTIALS": str(self.root / "absent-adc.json"), + "GOOGLE_WORKSPACE_CLI_KEYRING_BACKEND": "file", + "GOOGLE_WORKSPACE_PROJECT_ID": "synthetic-loopback-project", + "HTTP_PROXY": "http://127.0.0.1:1", + "HTTPS_PROXY": "http://127.0.0.1:1", + "ALL_PROXY": "http://127.0.0.1:1", + "NO_PROXY": "127.0.0.1,localhost", + } + result = subprocess.run( + [sys.executable, "-B", str(SCRIPT), "--document-id", "synthetic-doc", + "--gws", str(binary), "--timeout", "10", "review"], + cwd=self.root, env=env, text=True, capture_output=True, timeout=25, + ) + self.assertEqual(result.returncode, 0, result.stderr) + manifest = self.manifest() + self.assertEqual(manifest["status"], "complete") + self.assertEqual(manifest["revisions"]["status"], "unchanged") + for name in exports.values(): + data = (self.root / "review" / name).read_bytes() + self.assertEqual(data, (fixture / name).read_bytes()) + self.assertEqual(manifest["artifacts"][name]["bytes"], len(data)) + self.assertEqual(len(requests), 5) + self.assertTrue(all(auth == "Bearer synthetic-loopback-token" + for _, _, auth in requests)) + finally: + server.shutdown() + server.server_close() + thread.join(timeout=5) + + +class ExtractionTests(BundleTestCase): + def extract(self, **limits): + return self.api().extract_docx( + self.fixture / "document.docx", self.root / "assets", **limits + ) + + def test_relationships_preserve_order_alt_text_and_table_context(self): + result = self.extract() + figures = result["figures"] + self.assertEqual([f["order"] for f in figures], [1, 2, 3]) + self.assertIn('Figure " onerror="bad()', figures[0]["alt"]) + self.assertIn("Table image context", figures[1]["nearby_text"]) + self.assertIsNone(figures[2]["asset"]) + self.assertEqual(figures[2]["availability"], "external-not-fetched") + self.assertIsNone(figures[0]["native_object_id"]) + self.assertEqual(figures[0]["mapping_confidence"], "docx-relationship-only") + + def test_duplicate_basenames_get_distinct_local_assets(self): + result = self.extract() + paths = [f["asset"] for f in result["figures"][:2]] + self.assertNotEqual(*paths) + self.assertEqual((self.root / paths[0]).read_bytes(), PNG) + self.assertEqual((self.root / paths[1]).read_bytes(), PNG + b"different") + + def test_nested_text_box_image_has_one_occurrence_with_inner_ownership(self): + write_docx(self.fixture / "document.docx", nested_docx_members()) + figures = self.extract()["figures"] + self.assertEqual(len(figures), 1) + self.assertEqual(figures[0]["order"], 1) + self.assertEqual(figures[0]["alt"], "Inner image") + self.assertEqual(figures[0]["nearby_text"], "Inner paragraph") + self.assertEqual((self.root / figures[0]["asset"]).read_bytes(), PNG) + + def test_nested_image_order_and_context_follow_nearest_owners(self): + write_docx(self.fixture / "document.docx", nested_docx_members(outer_image=True)) + figures = self.extract()["figures"] + self.assertEqual([f["relationship_id"] for f in figures], ["rId1", "rId2"]) + self.assertEqual([f["alt"] for f in figures], ["Inner image", "Outer text box"]) + self.assertEqual( + [f["nearby_text"] for f in figures], ["Inner paragraph", "Outer paragraph"] + ) + self.assertEqual([f["order"] for f in figures], [1, 2]) + + def test_repeated_asset_uses_remain_distinct_figure_occurrences(self): + members = docx_members() + members["word/document.xml"] = members["word/document.xml"].replace( + b'r:embed="rId2"', b'r:embed="rId1"' + ) + write_docx(self.fixture / "document.docx", members) + figures = self.extract()["figures"] + self.assertEqual(len(figures), 3) + self.assertEqual([f["order"] for f in figures], [1, 2, 3]) + self.assertEqual(figures[0]["asset"], figures[1]["asset"]) + self.assertEqual([f["relationship_id"] for f in figures[:2]], ["rId1", "rId1"]) + self.assertNotEqual(figures[0]["alt"], figures[1]["alt"]) + self.assertEqual(figures[1]["nearby_text"], "Table image context") + + def test_zip_traversal_absolute_windows_and_control_paths_are_rejected(self): + api = self.api() + for index, name in enumerate( + ["../escape", "/escape", "C:/escape", r"..\escape", "word/../escape", "bad\x01"] + ): + with self.subTest(name=name): + members = {**docx_members(), name: b"bad"} + write_docx(self.fixture / "document.docx", members) + with self.assertRaises(api.BundleError): + api.extract_docx( + self.fixture / "document.docx", self.root / f"assets-{index}" + ) + self.assertFalse((self.root / "escape").exists()) + + def test_zip_symlink_rejected(self): + api = self.api() + with zipfile.ZipFile(self.fixture / "document.docx", "a") as archive: + link = zipfile.ZipInfo("word/media/link.png") + link.create_system = 3 + link.external_attr = (stat.S_IFLNK | 0o777) << 16 + archive.writestr(link, "../../../escape") + with self.assertRaises(api.BundleError): + self.extract() + + def test_member_total_and_count_limits_reject_before_writing_assets(self): + api = self.api() + for limits in ( + {"member_limit": 32}, + {"total_limit": 32}, + {"member_count": 2}, + ): + with self.subTest(limits=limits), self.assertRaises(api.BundleError): + self.extract(**limits) + self.assertFalse((self.root / "assets").exists()) + + def test_duplicate_archive_member_rejected(self): + api = self.api() + with zipfile.ZipFile(self.fixture / "document.docx", "a") as archive: + archive.writestr("WORD/MEDIA/IMAGE1.PNG", PNG) + with self.assertRaises(api.BundleError): + self.extract() + + def test_doctype_entities_and_utf16_are_rejected_even_in_unused_xml(self): + api = self.api() + for data in ( + b']>&x;', + b'', + ']>&x;'.encode("utf-16"), + ): + with self.subTest(data=data[:30]): + members = {**docx_members(), "word/unused.xml": data} + write_docx(self.fixture / "document.docx", members) + with self.assertRaises(api.BundleError): + self.extract() + self.assertFalse((self.root / "assets").exists()) + + def test_xml_encoding_allowlist_handles_declaration_whitespace(self): + api = self.api() + for encoding in ("UTF-7", "UTF-16", "UTF-32", "ISO-8859-1"): + for assignment in (f' = "{encoding}"', f"= '{encoding}'", f'\t=\n"{encoding}"'): + with self.subTest(encoding=encoding, assignment=assignment): + data = f''.encode("utf-8") + with self.assertRaises(api.BundleError) as raised: + api.safe_xml(data) + self.assertEqual(str(raised.exception), "unsupported-xml-encoding") + + def test_xml_utf8_bom_is_accepted_and_utf16_utf32_are_rejected(self): + api = self.api() + data = 'café'.encode("utf-8-sig") + root = api.safe_xml(data) + self.assertEqual(root.tag, "x") + self.assertEqual(root.text, "café") + for encoding in ("utf-16", "utf-32"): + with self.subTest(encoding=encoding), self.assertRaises(api.BundleError): + api.safe_xml("".encode(encoding)) + + def test_relationship_traversal_and_svg_are_not_local_html_assets(self): + members = docx_members() + members["word/_rels/document.xml.rels"] = members[ + "word/_rels/document.xml.rels" + ].replace(b"media/image1.png", b"../outside.png") + members["word/media/nested/image1.png"] = b"" + write_docx(self.fixture / "document.docx", members) + figures = self.extract()["figures"] + self.assertIsNone(figures[0]["asset"]) + self.assertIsNone(figures[1]["asset"]) + # The safe, now unreferenced raster is still preserved; the SVG is not. + files = list((self.root / "assets").iterdir()) + self.assertEqual(len(files), 1) + self.assertEqual(files[0].read_bytes(), PNG) + + def test_malformed_xml_and_corrupt_zip_fail_closed(self): + api = self.api() + for data in (b"not a zip", None): + if data is None: + write_docx( + self.fixture / "document.docx", + {**docx_members(), "word/document.xml": b"", html) + self.assertIn("<script>", html) + self.assertNotIn("SIGNED_SECRET", html) + self.assertNotIn("untrusted.invalid", html) + self.assertTrue(any(tag == "iframe" for tag, _ in inspection.tags)) + for tag, attrs in inspection.tags: + self.assertNotEqual(tag, "script") + self.assertFalse(any(name.startswith("on") for name in attrs)) + if tag == "iframe": + self.assertIn("sandbox", attrs) + for ref in inspection.references: + self.assertNotIn(":", ref) + self.assertNotIn("..", ref) + self.assertNotIn("%", ref) + self.assertTrue((self.root / "review" / ref).is_file(), ref) + + def test_native_outline_includes_tables_child_tabs_styles_and_uncertain_images(self): + result = self.run_bundle("review") + self.assertEqual(result.returncode, 0, result.stderr) + html = (self.root / "review/index.html").read_text() + for text in ("Table cell", "Nested tab paragraph", "HEADING_1", "tab-child"): + self.assertIn(text, html) + self.assertIn(" find.txt +printf '%s' 'reader' > replacement.txt + +python3 examples/docs-review/docs_review.py plan \ + --document SYNTHETIC_DOCUMENT_ID --tab t.synthetic \ + --find find.txt --replacement replacement.txt --out reviewed-plan.json + +# Inspect the full JSON plan: IDs, revision, literal text, diff, target, digest. +cat reviewed-plan.json + +# Local request/diff preview: validates the plan, calls no gws, writes no files. +python3 examples/docs-review/docs_review.py apply \ + --plan reviewed-plan.json --dry-run + +# After review: reread, compare, submit once, reread and verify. +python3 examples/docs-review/docs_review.py apply --plan reviewed-plan.json +``` + +Omit `--tab` only for a document containing exactly one tab. Child tabs count; +titles are not selectors. Use a new `--out` path when regenerating a plan. +An empty replacement file deletes the matched text. Empty finds and no-op +replacements are rejected. Do not use `echo` to prepare text files: it commonly +adds a newline, which this workflow deliberately rejects. + +The versioned plan contains the document/tab IDs, exact source revision, source +fingerprint, find/replacement text, expected count `1`, UTF-16 target offsets, +diff, and deterministic SHA-256 digest. It does not contain document titles, +the complete source, surrounding text, credentials, or command strings. +The diff is the exact removed/inserted text, not a full-document preview. +Plans still contain sensitive text if your inputs do; treat them accordingly. + +The digest detects accidental edits and binds the review fields together; it +is **not a signature or an authorization mechanism**. Anyone who can replace +the entire plan can recompute it. Protect the reviewed file and compare its +digest with your separately retained review record before applying. + +## Supported text and verification + +- Exactly one case-sensitive literal occurrence in one tab. Regex is disabled. + Overlapping occurrences also count as ambiguous. +- The match must fit inside one top-level body paragraph. It may cross text + style runs. Unicode offsets use UTF-16 code units, including emoji. +- Tables, images, headers, footers, footnotes, other paragraphs and other tabs + remain in the document. Text in non-body regions still counts toward + uniqueness: a duplicate in a header or table causes refusal. +- Apply reconstructs the plan from the fresh source and compares every field. + A changed revision, source, target or digest stops before submission. +- Exactly one `replaceAllText` request is sent with + `writeControl.requiredRevisionId` and `tabsCriteria.tabIds`. No block deletion, + full-document reupload, or arbitrary request from a plan is allowed. +- Success requires `occurrencesChanged == 1`, a new returned revision, and a + reread at that revision. The resulting text, paragraph/structure metadata, + shifted body indices, image metadata, other tabs, and untouched character + formatting must match the expected result. + +Google controls formatting inheritance **inside the replaced span**. This +example does not set or promise the replacement's character styles. It checks +the replacement text and all formatting outside that span, accepting changes +in text-run splitting that leave character formatting unchanged. Temporary +image `contentUri` values and gws `_sanitization` annotations are excluded from +fingerprints; other image/style metadata remains checked. The verification is +an API snapshot, not a visual rendering or a guarantee against later edits. + +## Deliberate limitations + +This is a text patch workflow, not a structural document editor. Paragraph +breaks, tabs/control characters, private-use characters and object markers +are rejected in find/replacement files. Moving blocks, inserting tables/images, +editing across images, and targets inside tables, tables of contents, headers, +footers or footnotes are unsupported. Structural requests cannot be supplied +through the plan. + +The selected tab must have no unresolved suggestions, named ranges or +bookmarks. Unknown tab regions, unknown structural blocks, equations, +automatic text, rich links and smart chips in the selected tab are refused. +These strict limits avoid interpreting inaccessible text, unstable anchors, +or unsupported element boundaries as a safe replacement. Other tabs are +fingerprinted and verified, not edited. Before normalization, every tab's +paragraph elements must have valid UTF-16 lengths and contiguous ranges. +Recognized regions and structural blocks must have valid object/list shapes, +including table rows and cell content. Malformed snapshots in any tab are +refused before submission or reported as ambiguous after submission. +Unknown metadata is retained for comparison; unknown structural blocks and +extra text-element fields that would be discarded are refused. New API +structures may require an explicit compatibility update. + +Each text input is limited to 16 KiB, the plan to 1 MiB, and each gws response +to 16 MiB. Source verification additionally limits total text to 250,000 +characters and its estimated expanded style representation to 16 MiB. +Oversized or malformed inputs fail closed. `--timeout SECONDS` bounds each +gws subprocess (default 60, maximum 600); there are no mutation retries. + +## Outcomes and recovery + +Commands emit JSON. `plan`, offline preview, and a verified apply exit `0` +with status `planned`, `preview`, or `applied`. Preview only validates local +data: it cannot establish that a live revision is still current. + +Exit `2` / `refused` means the companion did not submit a document write. +Correct the input or reread the source and review a **new** plan. + +Exit `3` / `ambiguous` means a write **may have applied**, or its result could +not be verified or reported successfully. This includes subprocess timeouts, +nonzero write exits (including a concurrent API 400), malformed responses, missing revisions, +unexpected reply counts, failed/mismatching rereads, and interruptions or +output failures after submission. The diagnostic JSON on stderr includes +`mutation_state: "attempted"` or `"confirmed"`. A confirmed mutation followed +by a closed stdout consumer or failed final reporting still exits `3`; it +does not claim that the write was refused or never submitted. Stdout is +flushed before success is returned, so buffered output failures are handled +inside the same outcome check. + +The original plan stays unchanged. Inspect the document and revision through +your normal tools; do not blindly rerun apply. Keep the failed plan as your review record and +create a new plan if further work is required. The companion does not persist +a separate attempt journal or prevent an operator from manually rerunning it. + +All subprocesses use argument arrays, checked exit codes and timeouts. +Raw gws output is not echoed into errors. Diagnostics produce a fixed, +redacted notice; inspect your gws/Model Armor configuration when it appears. +Model Armor block outcomes stop the workflow; settings are never disabled. +This example uses raw `gws docs documents` methods, whose executor at the +accompanying source revision does not retry requests. + +## Tests + +```sh +python3 -m unittest discover -s examples/docs-review -p 'test_*.py' -v +``` + +The suite executes the real companion CLI with a temporary stub `gws` and +synthetic documents. It passes an isolated environment, never reads real +credentials, and never contacts Google. CI runs it independently of Rust +changes on Linux and macOS. Tests cover the request contract, no-write +planning/preview, revision/source checks, Unicode/tabs, paths and plan +tampering, structures/styles, reply counts, ambiguity and plan retention. + +## API references + +- [Documents.get](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents/get): + `includeTabsContent` and `suggestionsViewMode`. +- [Document structure](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents): + indices are measured in UTF-16 code units; revisions are opaque. +- [ReplaceAllTextRequest](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents/request#ReplaceAllTextRequest): + literal matching and tab criteria. +- [BatchUpdate / WriteControl](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents/batchUpdate): + stale `requiredRevisionId` requests are rejected with HTTP 400. A required + revision in the response identifies the revision after application. diff --git a/examples/docs-review/docs_review.py b/examples/docs-review/docs_review.py new file mode 100755 index 000000000..1f122daaf --- /dev/null +++ b/examples/docs-review/docs_review.py @@ -0,0 +1,710 @@ +#!/usr/bin/env python3 +# Copyright 2026 Google LLC +# SPDX-License-Identifier: Apache-2.0 +"""Review and apply one revision-bound Google Docs text replacement (stdlib). + +Only the gws executable communicates with Google. Plans contain data, never +commands. See README.md for the intentionally limited structural support. +""" + +import argparse +from contextlib import contextmanager +import difflib +import hashlib +import hmac +import json +import math +import os +import re +import stat +import subprocess +import sys +import tempfile + + +MAX_PLAN = 1024 * 1024 +MAX_TEXT = 16 * 1024 +MAX_DOCUMENT = 16 * 1024 * 1024 +PLAN_KEYS = { + "version", "document_id", "tab_id", "revision_id", "source_sha256", + "find", "replacement", "expected_occurrences", "target", "diff", "digest", +} + + +class Refusal(Exception): + """A fixed, safe message; never construct one from subprocess output.""" + + +class Ambiguous(Refusal): + """A submission may have applied; never retry automatically.""" + + +def require(condition, message): + if not condition: + raise Refusal(message) + + +def canonical(value): + return json.dumps(value, sort_keys=True, separators=(",", ":"), + ensure_ascii=True, allow_nan=False).encode("utf-8") + + +def sha256(value): + return hashlib.sha256(canonical(value)).hexdigest() + + +def utf16(text): + return len(text.encode("utf-16-le")) // 2 + + +def unique_object(pairs): + result = {} + for key, value in pairs: + require(key not in result, "Duplicate JSON keys are not supported.") + result[key] = value + return result + + +def parse_json(data): + try: + return json.loads(data, object_pairs_hook=unique_object, + parse_constant=lambda _: invalid_json()) + except (ValueError, UnicodeError, RecursionError): + raise Refusal("Invalid UTF-8 JSON; regenerate the plan or check gws.") from None + + +def invalid_json(): + raise Refusal("Non-finite JSON numbers are not supported.") + + +def identifier(value, tab=False): + pattern = r"[A-Za-z0-9_.-]{1,200}" if tab else r"[A-Za-z0-9_-]{1,200}" + require(isinstance(value, str) and re.fullmatch(pattern, value) is not None + and ".." not in value, "Invalid document or tab ID; supply an ID, not a URL.") + return value + + +def revision(value): + require(isinstance(value, str) and 0 < len(value) <= 4096 + and not any(ord(c) < 32 or 127 <= ord(c) <= 159 for c in value), + "Missing or invalid revision; read an editable document again.") + return value + + +def text_input(value, allow_empty=False): + require(isinstance(value, str), "Find and replacement must be UTF-8 text.") + require(allow_empty or bool(value), "Find text must not be empty.") + # Docs can strip these or treat them as structural edits. Reject rather + # than silently submit a replacement different from the reviewed text. + require(not any( + ord(c) < 32 or 127 <= ord(c) <= 159 + or 0xD800 <= ord(c) <= 0xF8FF or 0xFFF9 <= ord(c) <= 0xFFFF + or ord(c) in (0x2028, 0x2029) + or 0xF0000 <= ord(c) <= 0xFFFFD + or 0x100000 <= ord(c) <= 0x10FFFD + for c in value + ), "Unsupported structural/control character; use single-paragraph plain text.") + require(len(value.encode("utf-8")) <= MAX_TEXT, "Text input exceeds 16 KiB.") + return value + + +@contextmanager +def confined_parent(path): + """Hold directory descriptors so symlink swaps cannot redirect file I/O. + + All symlinks (even inward ones) are rejected. POSIX openat/O_NOFOLLOW is + required; fail closed on platforms without it. + """ + require(isinstance(path, str) and path and not path.startswith("/") + and "\\" not in path and ":" not in path + and not any(ord(c) < 32 or 127 <= ord(c) <= 159 for c in path), + "Use a relative path within the current directory.") + parts = path.split("/") + require(all(p not in ("", ".", "..") for p in parts), + "Path traversal and empty path components are not allowed.") + require(hasattr(os, "O_NOFOLLOW") and os.open in os.supports_dir_fd, + "Safe file access requires a POSIX platform with O_NOFOLLOW.") + descriptors = [] + try: + flags = os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW + descriptors.append(os.open(".", flags)) + for part in parts[:-1]: + descriptors.append(os.open(part, flags, dir_fd=descriptors[-1])) + yield descriptors[-1], parts[-1] + except OSError: + raise Refusal( + "Cannot access path safely; check parents, symlinks and permissions." + ) from None + finally: + for descriptor in reversed(descriptors): + os.close(descriptor) + + +def read_file(path, limit): + with confined_parent(path) as (parent, name): + descriptor = os.open(name, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK, + dir_fd=parent) + with os.fdopen(descriptor, "rb") as stream: + require(stat.S_ISREG(os.fstat(stream.fileno()).st_mode), + "Input must be a regular file.") + data = stream.read(limit + 1) + require(len(data) <= limit, "Input file exceeds the supported size limit.") + try: + return data.decode("utf-8") + except UnicodeError: + raise Refusal("Input file must be valid UTF-8.") from None + + +def unused_output(parent, name): + try: + os.stat(name, dir_fd=parent, follow_symlinks=False) + except FileNotFoundError: + return + raise Refusal("Output already exists; choose a new plan path.") + + +def write_plan(parent, name, plan): + data = json.dumps(plan, ensure_ascii=True, sort_keys=True, indent=2) + "\n" + require(len(data.encode()) <= MAX_PLAN, "Plan exceeds 1 MiB; use a smaller patch.") + # Exclusive creation prevents overwrites/hardlink attacks; plans are private. + descriptor = os.open(name, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, dir_fd=parent) + with os.fdopen(descriptor, "w", encoding="utf-8") as stream: + stream.write(data) + stream.flush() + os.fsync(stream.fileno()) + + +class Gws: + def __init__(self, timeout): + self.timeout = timeout + self.diagnostics = False + self.mutation_state = "not_attempted" + + def call(self, method, document_id, request=None): + params = {"documentId": document_id} + if method == "get": + params.update(includeTabsContent=True, suggestionsViewMode="SUGGESTIONS_INLINE") + args = ["gws", "docs", "documents", method, + "--params", canonical(params).decode(), "--format", "json"] + if request is not None: + args += ["--json", canonical(request).decode()] + try: + # Inherit gws auth/Model Armor settings. Raw diagnostics never reach + # the terminal (they may contain document text or credentials). + with tempfile.TemporaryFile() as output, tempfile.TemporaryFile() as diagnostics: + result = subprocess.run( + args, stdin=subprocess.DEVNULL, stdout=output, + stderr=diagnostics, timeout=self.timeout, check=False, + ) + self.diagnostics |= diagnostics.tell() > 0 + require(result.returncode == 0, + "gws failed; check access, revision and Model Armor settings.") + output.seek(0) + data = output.read(MAX_DOCUMENT + 1) + require(len(data) <= MAX_DOCUMENT, "gws response exceeds 16 MiB.") + value = parse_json(data) + require(isinstance(value, dict) and "error" not in value, + "gws did not return a successful JSON object.") + return value + except subprocess.TimeoutExpired: + raise Refusal("gws timed out; check connectivity and timeout settings.") from None + except OSError: + raise Refusal("Cannot execute gws; check installation and PATH.") from None + + def get(self, document_id): + return self.call("get", document_id) + + +def walk(value, path=()): + if isinstance(value, dict): + yield path, value + for key, item in value.items(): + yield from walk(item, path + (key,)) + elif isinstance(value, list): + for index, item in enumerate(value): + yield from walk(item, path + (index,)) + + +def at(value, path): + for key in path: + value = value[key] + return value + + +def select_tab(document, document_id, tab_id): + require(isinstance(document, dict) and document.get("documentId") == document_id, + "Document identity mismatch.") + revision(document.get("revisionId")) + require(document.get("suggestionsViewMode") == "SUGGESTIONS_INLINE", + "Expected suggestions-inline source; refusing an incomplete view.") + require(isinstance(document.get("tabs"), list) and document["tabs"], + "Missing all-tabs content; check gws Docs discovery support.") + tabs = [] + + def visit(items, path): + require(isinstance(items, list), "Malformed document tabs.") + for i, item in enumerate(items): + require(isinstance(item, dict) + and isinstance(item.get("tabProperties"), dict) + and isinstance(item.get("documentTab"), dict), "Unsupported tab shape.") + identity = identifier(item["tabProperties"].get("tabId"), tab=True) + tabs.append((identity, path + (i, "documentTab"))) + if "childTabs" in item: + visit(item["childTabs"], path + (i, "childTabs")) + + visit(document["tabs"], ("tabs",)) + require(len({identity for identity, _ in tabs}) == len(tabs), "Duplicate tab IDs.") + if tab_id is None: + require(len(tabs) == 1, "Multiple tabs; select exactly one with --tab ID.") + tab_id = tabs[0][0] + matches = [path for identity, path in tabs if identity == tab_id] + require(len(matches) == 1, "Selected tab does not exist.") + return tab_id, matches[0] + + +def check_supported(tab): + require(isinstance(tab.get("body"), dict) + and isinstance(tab["body"].get("content"), list), "Tab body is missing.") + require(set(tab) <= { + "body", "headers", "footers", "footnotes", "documentStyle", "namedStyles", + "lists", "namedRanges", "inlineObjects", "positionedObjects", "bookmarks", + "suggestedDocumentStyleChanges", "suggestedNamedStylesChanges", + }, "Unsupported tab region; this example requires a known document structure.") + for _, obj in walk(tab): + require(not any(k.startswith("suggested") and v for k, v in obj.items()), + "Suggested content in selected tab is unsupported; resolve suggestions first.") + require(not obj.get("namedRanges") and not obj.get("bookmarks"), + "Named ranges and bookmarks in the selected tab are unsupported.") + if "paragraph" not in obj: + continue + for element in obj["paragraph"]["elements"]: + kinds = set(element) - {"startIndex", "endIndex"} + require(len(kinds) == 1 and kinds <= { + "textRun", "inlineObjectElement", "footnoteReference", + "horizontalRule", "pageBreak", "columnBreak"}, + "Unsupported paragraph element; rich links, equations and chips are excluded.") + + +def index_range(value): + start, end = value.get("startIndex", 0), value.get("endIndex") + require(type(start) is int and type(end) is int and 0 <= start < end, + "Invalid structural or paragraph element indices.") + return start, end + + +def check_region(region): + require(isinstance(region, dict) and isinstance(region.get("content"), list), + "Malformed document region; expected structural content.") + + +def check_content(content): + for block in content: + require(isinstance(block, dict), "Malformed structural block.") + kinds = set(block) - {"startIndex", "endIndex"} + require(len(kinds) == 1 and kinds <= { + "paragraph", "sectionBreak", "table", "tableOfContents"}, + "Unsupported document structure.") + require(isinstance(block[next(iter(kinds))], dict), "Malformed structural block value.") + index_range(block) + + +def check_table(table): + require(isinstance(table, dict), "Malformed table.") + require(type(table.get("rows")) is int and table["rows"] > 0 + and type(table.get("columns")) is int and table["columns"] > 0, + "Malformed table dimensions.") + rows = table.get("tableRows") + require(isinstance(rows, list) and len(rows) == table["rows"], "Malformed table rows.") + for row in rows: + require(isinstance(row, dict) and isinstance(row.get("tableCells"), list) + and 0 < len(row["tableCells"]) <= table["columns"], "Malformed table row.") + for cell in row["tableCells"]: + check_region(cell) + + +def check_paragraph(block): + paragraph = block["paragraph"] + require(isinstance(paragraph, dict) + and isinstance(paragraph.get("elements"), list) + and paragraph["elements"], "Malformed paragraph.") + require(isinstance(paragraph.get("paragraphStyle", {}), dict), + "Malformed paragraph style.") + cursor, paragraph_end = index_range(block) + for element in paragraph["elements"]: + require(isinstance(element, dict), "Malformed paragraph element.") + start, end = index_range(element) + require(start == cursor, "Non-contiguous paragraph indices.") + kinds = set(element) - {"startIndex", "endIndex"} + # Extra siblings of textRun would otherwise disappear in normalization. + # Unrecognized non-text unions are retained whole in unselected tabs. + require(len(kinds) == 1 and isinstance(element[next(iter(kinds))], dict), + "Malformed or unsupported paragraph element fields.") + if "textRun" in element: + run = element["textRun"] + require(isinstance(run.get("content"), str) + and isinstance(run.get("textStyle", {}), dict), "Malformed text run.") + require(utf16(run["content"]) == end - start, + "Text run does not match its UTF-16 indices.") + cursor = end + require(paragraph_end == cursor, "Paragraph end index mismatch.") + + +def check_document_structure(document): + """Validate every region/range the normalizer interprets, in every tab. + + Unknown metadata is retained unchanged. Unknown structural blocks or + text-element siblings that cannot be retained safely are refused. + """ + for _, obj in walk(document): + if "documentTab" in obj: + require(isinstance(obj["documentTab"], dict) and "body" in obj["documentTab"], + "Missing document tab body.") + if "body" in obj: + check_region(obj["body"]) + for group in ("headers", "footers", "footnotes"): + if group in obj: + require(isinstance(obj[group], dict), "Malformed document region map.") + for region in obj[group].values(): + check_region(region) + if isinstance(obj.get("content"), list): + check_content(obj["content"]) + if "paragraph" in obj: + check_paragraph(obj) + if "table" in obj: + check_table(obj["table"]) + if "tableOfContents" in obj: + check_region(obj["tableOfContents"]) + if "sectionBreak" in obj: + require(isinstance(obj["sectionBreak"], dict) + and isinstance(obj["sectionBreak"].get("sectionStyle", {}), dict), + "Malformed section break.") + + +def check_document_budget(document): + """Bound character/style expansion before constructing canonical atoms.""" + expanded_bytes = 0 + characters = 0 + for _, obj in walk(document): + if "textRun" not in obj: + continue + run = obj["textRun"] + require(isinstance(run, dict) and isinstance(run.get("content"), str), + "Malformed text run.") + length = len(run["content"]) + characters += length + metadata = {k: v for k, v in run.items() if k != "content"} + expanded_bytes += length * (len(canonical(metadata)) + 64) + require(characters <= 250000 and expanded_bytes <= MAX_DOCUMENT, + "Document is too large for safe text/style verification.") + + +def normalized(document): + # This gate is part of normalization itself, so no caller can accidentally + # compare discarded text-run indices before validating the complete source. + check_document_structure(document) + return _normalized(document) + + +def _normalized(value, path=()): + """Canonical semantic shape; text-run splitting is not a style change.""" + if isinstance(value, list): + return [_normalized(v, path + (i,)) for i, v in enumerate(value)] + if not isinstance(value, dict): + return value + result = {} + for key, item in value.items(): + if not path and key in ("revisionId", "_sanitization"): + continue + # Google refreshes this temporary URL on reads; keep all other image + # metadata, including sourceUri, object IDs, dimensions and crop data. + if key == "contentUri" and path and path[-1] == "imageProperties": + continue + if key == "elements" and path and path[-1] == "paragraph": + elements = [] + for element in item: + if "textRun" in element: + run = element["textRun"] + metadata = {k: v for k, v in run.items() if k != "content"} + metadata.setdefault("textStyle", {}) + for char in run["content"]: + elements.append({"text": char, "format": metadata}) + else: + elements.append(_normalized(element, path + (key,))) + result[key] = elements + else: + result[key] = _normalized(item, path + (key,)) + return result + + +def locate(document, tab_path, find): + """Count overlapping occurrences in every text segment of the chosen tab.""" + tab = at(document, tab_path) + matches = [] + for path, block in walk(tab): + if "paragraph" not in block: + continue + # Only top-level body paragraphs are editable. Tables, TOCs, headers, + # footers and footnotes still participate in ambiguity detection. + supported = (len(path) == 3 and path[:2] == ("body", "content") + and isinstance(path[2], int)) + cursor = block.get("startIndex", 0) + tokens = [] + indices = [] + for element in block["paragraph"]["elements"]: + if "textRun" in element: + for char in element["textRun"]["content"]: + tokens.append(char) + indices.append(cursor) + cursor += utf16(char) + else: + tokens.append("\ufffc") + indices.append(cursor) + cursor = element["endIndex"] + text = "".join(tokens) + offset = text.find(find) + while offset != -1: + matches.append((supported, tab_path + path + ("paragraph", "elements"), + offset, indices[offset])) + offset = text.find(find, offset + 1) + require(len(matches) == 1, + "Expected exactly one match in the selected tab; found zero or multiple.") + supported, path, offset, start = matches[0] + require(supported, "Target is outside a supported body paragraph.") + return path, offset, start + + +def review_diff(find, replacement): + return "".join(difflib.unified_diff( + [find + "\n"], [replacement + "\n"], fromfile="before", tofile="after", + )) + + +def build_plan(document, document_id, tab_id, find, replacement): + identifier(document_id) + text_input(find) + text_input(replacement, allow_empty=True) + require(find != replacement, "No-op replacement; choose different text.") + tab_id, tab_path = select_tab(document, document_id, tab_id) + check_document_budget(document) + source = normalized(document) + check_supported(at(document, tab_path)) + _, _, start = locate(document, tab_path, find) + result = { + "version": 1, "document_id": document_id, "tab_id": tab_id, + "revision_id": document["revisionId"], "source_sha256": sha256(source), + "find": find, "replacement": replacement, "expected_occurrences": 1, + "target": {"start_index": start, "end_index": start + utf16(find)}, + "diff": review_diff(find, replacement), + } + result["digest"] = sha256(result) + return result + + +def validate_plan(plan): + require(isinstance(plan, dict) and set(plan) == PLAN_KEYS, "Unsupported plan schema.") + require(type(plan["version"]) is int and plan["version"] == 1, + "Unsupported plan version.") + require(type(plan["expected_occurrences"]) is int and plan["expected_occurrences"] == 1, + "Plan must specify exactly one replacement.") + identifier(plan["document_id"]) + identifier(plan["tab_id"], tab=True) + revision(plan["revision_id"]) + text_input(plan["find"]) + text_input(plan["replacement"], allow_empty=True) + require(plan["find"] != plan["replacement"], "No-op replacement.") + target = plan["target"] + require(isinstance(target, dict) and set(target) == {"start_index", "end_index"} + and all(type(v) is int and v >= 1 for v in target.values()) + and target["end_index"] - target["start_index"] == utf16(plan["find"]), + "Invalid UTF-16 target range.") + require(plan["diff"] == review_diff(plan["find"], plan["replacement"]), + "Plan diff does not match its replacement.") + for field in ("digest", "source_sha256"): + require(isinstance(plan[field], str) and re.fullmatch(r"[0-9a-f]{64}", plan[field]), + "Invalid SHA-256 field.") + payload = {k: v for k, v in plan.items() if k != "digest"} + require(hmac.compare_digest(plan["digest"], sha256(payload)), + "Plan digest mismatch; regenerate and review a fresh plan.") + return plan + + +def request_body(plan): + return { + "writeControl": {"requiredRevisionId": plan["revision_id"]}, + "requests": [{"replaceAllText": { + "containsText": {"text": plan["find"], "matchCase": True, "searchByRegex": False}, + "replaceText": plan["replacement"], + "tabsCriteria": {"tabIds": [plan["tab_id"]]}, + }}], + } + + +def verify(before, after, plan, write_revision): + _, tab_path = select_tab(before, plan["document_id"], plan["tab_id"]) + select_tab(after, plan["document_id"], plan["tab_id"]) + require(after["revisionId"] == write_revision, "Post-write revision mismatch.") + check_document_budget(after) + expected, actual = normalized(before), normalized(after) + check_supported(at(after, tab_path)) + path, offset, _ = locate(before, tab_path, plan["find"]) + end = plan["target"]["end_index"] + delta = utf16(plan["replacement"]) - utf16(plan["find"]) + # Indices in the body shift; headers/footers/footnotes have separate indices. + for _, obj in walk(at(expected, tab_path)["body"]): + for key in ("startIndex", "endIndex"): + if key in obj and obj[key] >= end: + obj[key] += delta + replacement = [{"text": c} for c in plan["replacement"]] + expected_elements = at(expected, path) + expected_elements[offset:offset + len(plan["find"])] = replacement + actual_elements = at(actual, path) + # Formatting within newly inserted text is Google-controlled. Verify its + # exact characters but compare formatting of every untouched character. + for i in range(offset, offset + len(replacement)): + require(i < len(actual_elements) and "text" in actual_elements[i], + "Replacement missing from expected location.") + actual_elements[i] = {"text": actual_elements[i]["text"]} + require(actual == expected, "Post-write text, structure or untouched style mismatch.") + + +def apply_plan(plan, gws): + before = gws.get(plan["document_id"]) + rebuilt = build_plan(before, plan["document_id"], plan["tab_id"], + plan["find"], plan["replacement"]) + require(rebuilt == plan, + "Source revision or content changed; regenerate and review a new plan.") + try: + gws.mutation_state = "attempted" + reply = gws.call("batchUpdate", plan["document_id"], request_body(plan)) + require(reply.get("documentId") == plan["document_id"], "Write identity mismatch.") + replies = reply.get("replies") + require(isinstance(replies, list) and len(replies) == 1 + and isinstance(replies[0], dict), "Missing replacement reply.") + change = replies[0].get("replaceAllText") + require(isinstance(change, dict) and type(change.get("occurrencesChanged")) is int + and change["occurrencesChanged"] == 1, "Replacement count was not exactly one.") + control = reply.get("writeControl") + require(isinstance(control, dict), "Missing write revision.") + write_revision = revision(control.get("requiredRevisionId")) + require(write_revision != plan["revision_id"], "Write revision did not advance.") + after = gws.get(plan["document_id"]) + verify(before, after, plan, write_revision) + gws.mutation_state = "confirmed" + except (Refusal, OSError, ValueError, KeyError, TypeError, IndexError, RecursionError, + KeyboardInterrupt): + raise Ambiguous( + "Write may have applied; verification did not establish success. " + "Keep the original plan, inspect the document and revision, and do not " + "blindly retry. A concurrent revision rejection requires a newly reviewed plan." + ) from None + return {"status": "applied", "digest": plan["digest"], "revision_id": write_revision} + + +class Parser(argparse.ArgumentParser): + def error(self, message): + raise Refusal("Invalid arguments; use --help for supported options.") + + +def silence_failed_stdout(): + """Prevent a failed buffered write from being retried at interpreter exit.""" + try: + with open(os.devnull, "w") as sink: + os.dup2(sink.fileno(), sys.stdout.fileno()) + except (OSError, ValueError, AttributeError): + pass + + +def report_failure(error, mutation_state): + if mutation_state != "not_attempted": + message = ( + "Write was confirmed, but final reporting failed. " + if mutation_state == "confirmed" else + "Write may have applied; verification did not establish success. " + ) + outcome = { + "status": "ambiguous", "mutation_state": mutation_state, + "message": message + "Keep the original plan, inspect the document and revision, " + "and do not blindly retry. Further changes require a newly reviewed plan.", + } + code = 3 + else: + if isinstance(error, KeyboardInterrupt): + message = "Interrupted before submission." + elif isinstance(error, Refusal): + message = str(error) + else: + message = "Malformed source or inaccessible file; check inputs and regenerate." + outcome = {"status": "refused", "message": message} + code = 2 + try: + print(json.dumps(outcome), file=sys.stderr, flush=True) + except (OSError, KeyboardInterrupt): + # Neither an unavailable diagnostic channel nor another interruption + # changes whether a mutation was attempted. + pass + return code + + +def main(argv=None): + parser = Parser(description=__doc__) + commands = parser.add_subparsers(dest="command", required=True) + plan_parser = commands.add_parser("plan", help="Read a document and create a reviewable plan") + plan_parser.add_argument("--document", required=True) + plan_parser.add_argument( + "--tab", help="Exact tab ID (required when there is more than one tab)") + plan_parser.add_argument("--find", required=True, help="Relative UTF-8 input file") + plan_parser.add_argument("--replacement", required=True, help="Relative UTF-8 input file") + plan_parser.add_argument("--out", required=True, help="New relative plan file") + apply_parser = commands.add_parser("apply", help="Validate, apply once, and verify") + apply_parser.add_argument("--plan", required=True) + apply_parser.add_argument("--dry-run", action="store_true", + help="Offline plan validation and request preview; no gws calls") + for command in (plan_parser, apply_parser): + command.add_argument("--timeout", type=float, default=60, + help="Per-gws-call timeout in seconds (default: 60)") + gws = None + reporting = False + try: + args = parser.parse_args(argv) + require(math.isfinite(args.timeout) and 0 < args.timeout <= 600, + "Timeout must be greater than zero and at most 600 seconds.") + gws = Gws(args.timeout) + if args.command == "plan": + identifier(args.document) + if args.tab is not None: + identifier(args.tab, tab=True) + find = text_input(read_file(args.find, MAX_TEXT)) + replacement = text_input(read_file(args.replacement, MAX_TEXT), allow_empty=True) + with confined_parent(args.out) as (parent, name): + unused_output(parent, name) + plan = build_plan( + gws.get(args.document), args.document, args.tab, find, replacement) + write_plan(parent, name, plan) + result = {"status": "planned", "digest": plan["digest"], + "message": "Review the plan and digest before applying."} + else: + plan = validate_plan(parse_json(read_file(args.plan, MAX_PLAN))) + if args.dry_run: + result = {"status": "preview", "digest": plan["digest"], + "diff": plan["diff"], "request": request_body(plan)} + else: + result = apply_plan(plan, gws) + if gws.diagnostics: + result["gws_diagnostics"] = ( + "gws reported diagnostics; check gws and Model Armor settings. " + "Raw output was suppressed to protect document content." + ) + reporting = True + print(json.dumps(result, ensure_ascii=True, sort_keys=True), flush=True) + return 0 + except (Refusal, OSError, ValueError, KeyError, TypeError, IndexError, + RecursionError, KeyboardInterrupt) as error: + if reporting: + silence_failed_stdout() + mutation_state = gws.mutation_state if gws is not None else "not_attempted" + return report_failure(error, mutation_state) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/examples/docs-review/test_docs_review.py b/examples/docs-review/test_docs_review.py new file mode 100644 index 000000000..03f5af589 --- /dev/null +++ b/examples/docs-review/test_docs_review.py @@ -0,0 +1,771 @@ +#!/usr/bin/env python3 +# Copyright 2026 Google LLC +# SPDX-License-Identifier: Apache-2.0 +"""Behavior tests: all gws calls go to an isolated executable, never Google.""" + +import copy +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + + +SCRIPT = Path(__file__).with_name("docs_review.py") + + +def paragraph(parts, start=1): + elements = [] + cursor = start + for text, style in parts: + end = cursor + len(text.encode("utf-16-le")) // 2 + elements.append({ + "startIndex": cursor, "endIndex": end, + "textRun": {"content": text, "textStyle": style}, + }) + cursor = end + return { + "startIndex": start, "endIndex": cursor, + "paragraph": { + "elements": elements, + "paragraphStyle": {"namedStyleType": "NORMAL_TEXT"}, + }, + } + + +def document(text="Hello world.\n", revision="rev-1"): + return { + "documentId": "synthetic-doc", "revisionId": revision, + "title": "Synthetic review fixture", + "suggestionsViewMode": "SUGGESTIONS_INLINE", + "tabs": [{ + "tabProperties": {"tabId": "t.main", "title": "Main", "index": 0}, + "documentTab": { + "body": {"content": [ + {"endIndex": 1, "sectionBreak": { + "sectionStyle": {"sectionType": "CONTINUOUS"}}}, + paragraph([(text, {})]), + ]}, + }, + }], + } + + +def body(doc): + return doc["tabs"][0]["documentTab"]["body"]["content"] + + +def resign(plan): + payload = {k: v for k, v in plan.items() if k != "digest"} + plan["digest"] = hashlib.sha256(json.dumps( + payload, sort_keys=True, separators=(",", ":"), ensure_ascii=True, + allow_nan=False, + ).encode()).hexdigest() + return plan + + +# The executable checks the actual argv contract and simulates the remote +# boundary only. Its output fixtures are independent of production helpers. +STUB = r''' +import json, os, pathlib, sys, time +root = pathlib.Path(os.environ["STUB_ROOT"]) +args = sys.argv[1:] +with (root / "calls.jsonl").open("a") as f: + f.write(json.dumps(args) + "\n") +assert args[:2] == ["docs", "documents"], args +assert args[args.index("--format") + 1] == "json", args +params = json.loads(args[args.index("--params") + 1]) +assert params["documentId"] == "synthetic-doc", params +method = args[2] +mode = (root / "mode").read_text() +if method == "get": + assert params["includeTabsContent"] is True, params + assert params["suggestionsViewMode"] == "SUGGESTIONS_INLINE", params + name = "after.json" if (root / "submitted").exists() else "before.json" + if mode == "read-error" or (mode == "verify-error" and name == "after.json"): + print("secret-token \x1b[31m remote private content", file=sys.stderr) + sys.exit(1) + if mode == "armor-block": + print(json.dumps({"error": "Content blocked by Model Armor"})) + sys.exit(1) + if mode == "armor-warn": + assert os.environ["GOOGLE_WORKSPACE_CLI_SANITIZE_TEMPLATE"] == "synthetic-template" + print("secret-token \x1b[31m Model Armor warning", file=sys.stderr) + print((root / name).read_text()) +elif method == "batchUpdate": + request = json.loads(args[args.index("--json") + 1]) + (root / "request.json").write_text(json.dumps(request)) + (root / "submitted").touch() + if mode == "timeout": + time.sleep(5) + if mode in ("conflict", "write-error"): + print(json.dumps({"error": {"code": 400 if mode == "conflict" else 503, + "message": "secret-token \x1b[31m private response"}})) + sys.exit(1) + if mode == "bad-response": + print("secret-token malformed") + else: + result = {"documentId": "synthetic-doc", + "replies": [{"replaceAllText": {"occurrencesChanged": 1}}], + "writeControl": {"requiredRevisionId": "rev-2"}} + if mode == "zero": + result["replies"][0]["replaceAllText"]["occurrencesChanged"] = 0 + if mode == "two": + result["replies"][0]["replaceAllText"]["occurrencesChanged"] = 2 + if mode == "missing-reply": + result["replies"] = [] + if mode == "missing-write-revision": + result.pop("writeControl") + print(json.dumps(result)) +else: + raise AssertionError(args) +''' + + +class ReviewCliTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.bin = self.root / "bin" + self.bin.mkdir() + stub = self.bin / "gws" + stub.write_text("#!" + sys.executable + "\n" + STUB) + stub.chmod(0o700) + # Do not pass real credentials/configuration into any child process. + self.env = { + "PATH": str(self.bin) + os.pathsep + os.defpath, + "STUB_ROOT": str(self.root), + "PYTHONDONTWRITEBYTECODE": "1", + "GOOGLE_WORKSPACE_CLI_CONFIG_DIR": str(self.root / "config"), + } + self.put("mode", "ok") + self.put("find.txt", "world") + self.put("replacement.txt", "reader") + self.fixture(document(), document("Hello reader.\n", "rev-2")) + + def put(self, name, value): + (self.root / name).write_text(value, encoding="utf-8") + + def fixture(self, before, after=None): + self.put("before.json", json.dumps(before)) + self.put("after.json", json.dumps(after or before)) + + def cli(self, *args): + return subprocess.run( + [sys.executable, str(SCRIPT), *args], cwd=self.root, + env=self.env, text=True, capture_output=True, timeout=10, + ) + + def plan(self, *extra): + return self.cli("plan", "--document", "synthetic-doc", + "--find", "find.txt", "--replacement", "replacement.txt", + "--out", "plan.json", *extra) + + def apply(self, *extra): + return self.cli("apply", "--plan", "plan.json", *extra) + + def load_plan(self): + return json.loads((self.root / "plan.json").read_text()) + + def calls(self): + path = self.root / "calls.jsonl" + return [json.loads(x) for x in path.read_text().splitlines()] if path.exists() else [] + + def success(self, result): + self.assertEqual(result.returncode, 0, result.stderr) + return json.loads(result.stdout) + + def refused(self, result, status="refused"): + self.assertNotEqual(result.returncode, 0) + self.assertNotIn("secret-token", result.stderr + result.stdout) + self.assertNotIn("\x1b", result.stderr + result.stdout) + try: + payload = json.loads(result.stderr) + except ValueError: + self.fail("Expected a structured refusal, got: " + result.stderr[:200]) + self.assertEqual(payload["status"], status) + + def test_plan_is_deterministic_reviewable_and_reads_only_once(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + plan = self.load_plan() + self.assertEqual(plan["version"], 1) + self.assertEqual(plan["revision_id"], "rev-1") + self.assertEqual(plan["tab_id"], "t.main") + self.assertEqual(plan["expected_occurrences"], 1) + self.assertEqual(plan["target"], {"start_index": 7, "end_index": 12}) + self.assertIn("-world", plan["diff"]) + self.assertIn("+reader", plan["diff"]) + self.assertNotIn("Synthetic review fixture", original.decode()) + self.assertNotIn("Hello", original.decode()) + self.assertEqual(resign(copy.deepcopy(plan)), plan) + self.assertEqual([c[2] for c in self.calls()], ["get"]) + (self.root / "plan.json").unlink() + self.success(self.plan()) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_apply_submits_only_reviewed_revision_and_tab_then_verifies(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + result = self.success(self.apply()) + self.assertEqual(result["status"], "applied") + self.assertEqual(json.loads((self.root / "request.json").read_text()), { + "writeControl": {"requiredRevisionId": "rev-1"}, + "requests": [{"replaceAllText": { + "containsText": {"text": "world", "matchCase": True, + "searchByRegex": False}, + "replaceText": "reader", "tabsCriteria": {"tabIds": ["t.main"]}, + }}], + }) + self.assertEqual([c[2] for c in self.calls()], + ["get", "get", "batchUpdate", "get"]) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_closed_stdout_after_apply_keeps_confirmed_mutation_state(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + for python_flags in [[], ["-u"]]: + with self.subTest(python_flags=python_flags): + (self.root / "submitted").unlink(missing_ok=True) + previous_calls = len(self.calls()) + read_fd, write_fd = os.pipe() + os.close(read_fd) + try: + result = subprocess.run( + [sys.executable, *python_flags, str(SCRIPT), + "apply", "--plan", "plan.json"], + cwd=self.root, env=self.env, stdout=write_fd, + stderr=subprocess.PIPE, text=True, timeout=10, + ) + finally: + os.close(write_fd) + self.assertEqual(result.returncode, 3, result.stderr) + outcome = json.loads(result.stderr) + self.assertEqual(outcome["status"], "ambiguous") + self.assertEqual(outcome["mutation_state"], "confirmed") + self.assertIn("inspect", outcome["message"]) + self.assertIn("do not blindly retry", outcome["message"]) + self.assertEqual([c[2] for c in self.calls()[previous_calls:]], + ["get", "batchUpdate", "get"]) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_interruption_after_apply_returns_keeps_confirmed_mutation_state(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + # Inject only the interruption at the return boundary. The actual + # apply implementation still performs every read/write/verification. + wrapper = """ +import runpy, sys +namespace = runpy.run_path(sys.argv[1]) +real_apply = namespace["apply_plan"] +def interrupted_return(*args, **kwargs): + real_apply(*args, **kwargs) + raise KeyboardInterrupt +namespace["main"].__globals__["apply_plan"] = interrupted_return +sys.exit(namespace["main"](["apply", "--plan", "plan.json"])) +""" + result = subprocess.run( + [sys.executable, "-c", wrapper, str(SCRIPT)], + cwd=self.root, env=self.env, capture_output=True, text=True, timeout=10, + ) + self.assertEqual(result.returncode, 3, result.stderr) + self.refused(result, "ambiguous") + outcome = json.loads(result.stderr) + self.assertEqual(outcome["mutation_state"], "confirmed") + self.assertIn("do not blindly retry", outcome["message"]) + self.assertNotIn("before submission", outcome["message"]) + self.assertEqual([c[2] for c in self.calls()], + ["get", "get", "batchUpdate", "get"]) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_final_serialization_failure_keeps_confirmed_mutation_state(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + wrapper = """ +import json, runpy, sys +namespace = runpy.run_path(sys.argv[1]) +real_dumps = json.dumps +def failed_result(value, *args, **kwargs): + if isinstance(value, dict) and value.get("status") == "applied": + raise ValueError("secret-token final output failure") + return real_dumps(value, *args, **kwargs) +json.dumps = failed_result +sys.exit(namespace["main"](["apply", "--plan", "plan.json"])) +""" + result = subprocess.run( + [sys.executable, "-c", wrapper, str(SCRIPT)], + cwd=self.root, env=self.env, capture_output=True, text=True, timeout=10, + ) + self.assertEqual(result.returncode, 3, result.stderr) + self.refused(result, "ambiguous") + self.assertEqual(json.loads(result.stderr)["mutation_state"], "confirmed") + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_zero_multiple_and_overlapping_matches_refuse_without_write(self): + for text, find in [("Nothing.\n", "world"), + ("world world\n", "world"), ("aaa\n", "aa")]: + with self.subTest(text=text): + self.fixture(document(text)) + self.put("find.txt", find) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + self.assertTrue(all(c[2] == "get" for c in self.calls())) + + def test_unicode_utf16_and_nested_tab_scope(self): + before = document("😀 café world.\n") + before["tabs"][0]["childTabs"] = [{ + "tabProperties": {"tabId": "t.child", "title": "Main", "index": 0}, + "documentTab": {"body": {"content": [paragraph([("world\n", {})])]}} + }] + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + body(after)[1] = paragraph([("😀 café reader.\n", {})]) + self.fixture(before, after) + self.refused(self.plan()) # No silent first-tab selection. + self.success(self.plan("--tab", "t.main")) + self.assertEqual(self.load_plan()["target"], + {"start_index": 9, "end_index": 14}) + self.success(self.apply()) + request = json.loads((self.root / "request.json").read_text()) + self.assertEqual(request["requests"][0]["replaceAllText"]["tabsCriteria"], + {"tabIds": ["t.main"]}) + + def test_replaces_unicode_in_selected_child_without_touching_parent(self): + before = document("😀targetZ\n") + child = copy.deepcopy(before["tabs"][0]) + child["tabProperties"]["tabId"] = "t.child" + before["tabs"][0]["childTabs"] = [child] + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + # Preserve Z/newline styles while allowing Google to style inserted text. + after["tabs"][0]["childTabs"][0]["documentTab"]["body"]["content"][1] = ( + paragraph([("🛰️", {"italic": True}), ("Z\n", {})])) + self.fixture(before, after) + self.put("find.txt", "😀target") + self.put("replacement.txt", "🛰️") + self.success(self.plan("--tab", "t.child")) + self.assertEqual(self.load_plan()["target"], + {"start_index": 1, "end_index": 9}) + self.success(self.apply()) + + def test_literal_matching_does_not_enable_regex_or_shell(self): + literal = "$(touch injected);.*" + self.fixture(document(literal + "\n"), document("done\n", "rev-2")) + self.put("find.txt", literal) + self.put("replacement.txt", "done") + self.success(self.plan()) + self.success(self.apply()) + self.assertFalse((self.root / "injected").exists()) + + def test_preview_is_offline_and_does_not_submit(self): + self.success(self.plan()) + calls = self.calls() + before = (self.root / "plan.json").read_bytes() + self.success(self.apply("--dry-run")) + self.assertEqual(self.calls(), calls) + self.assertEqual((self.root / "plan.json").read_bytes(), before) + + def test_source_revision_missing_or_changed_refuses(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + for doc in [document(revision="rev-new"), document("Different world.\n"), + {k: v for k, v in document().items() if k != "revisionId"}]: + with self.subTest(doc=doc): + self.fixture(doc) + self.refused(self.apply()) + self.assertFalse((self.root / "submitted").exists()) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_plan_without_revision_refuses(self): + doc = document() + doc.pop("revisionId") + self.fixture(doc) + self.refused(self.plan()) + + def test_malformed_tampered_and_oversized_plan_refuse_offline(self): + self.success(self.plan()) + original = self.load_plan() + tampered = copy.deepcopy(original) + tampered["replacement"] = "unreviewed" + unknown = resign(dict(original, command="touch injected")) + wrong_target = copy.deepcopy(original) + wrong_target["target"]["start_index"] = -1 + wrong_version = resign(dict(original, version=True)) + for value in ["{", "[]", '{"version":1,"version":2}', + json.dumps(tampered), json.dumps(unknown), + json.dumps(resign(wrong_target)), json.dumps(wrong_version), + " " * (1024 * 1024 + 1)]: + with self.subTest(value=value[:90]): + self.put("plan.json", value) + before = (self.root / "plan.json").read_bytes() + calls = self.calls() + self.refused(self.apply()) + self.assertEqual(self.calls(), calls) + self.assertEqual((self.root / "plan.json").read_bytes(), before) + + def test_resigned_semantic_tampering_is_reconstructed_before_write(self): + self.success(self.plan()) + plan = self.load_plan() + altered_target = copy.deepcopy(plan) + altered_target["target"] = {"start_index": 8, "end_index": 13} + altered_fingerprint = dict(plan, source_sha256="0" * 64) + for altered in [altered_target, altered_fingerprint]: + with self.subTest(altered=altered): + self.put("plan.json", json.dumps(resign(altered))) + original = (self.root / "plan.json").read_bytes() + previous_calls = len(self.calls()) + self.refused(self.apply()) + self.assertEqual([c[2] for c in self.calls()[previous_calls:]], ["get"]) + self.assertFalse((self.root / "submitted").exists()) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_unsupported_structural_edits_and_noops_refuse(self): + for find, replacement in [("", "new"), ("world", "world"), + ("world", "one\ntwo"), ("world", "\ufffc"), + ("world.\n", "new"), ("world", "\x00")]: + with self.subTest(find=find, replacement=replacement): + self.put("find.txt", find) + self.put("replacement.txt", replacement) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + + def test_table_header_suggestion_and_image_crossing_targets_refuse(self): + table = document() + body(table)[1] = {"startIndex": 1, "endIndex": 14, "table": { + "rows": 1, "columns": 1, "tableRows": [{"tableCells": [{ + "content": [paragraph([("world\n", {})], 3)]}]}]}} + header = document("Other.\n") + header["tabs"][0]["documentTab"]["headers"] = { + "h.1": {"content": [paragraph([("world\n", {})])]}} + suggested = document() + body(suggested)[1]["paragraph"]["elements"][0]["textRun"][ + "suggestedInsertionIds"] = ["suggestion-1"] + image = document() + body(image)[1] = paragraph([("wor", {}), ("ld\n", {})]) + elements = body(image)[1]["paragraph"]["elements"] + elements.insert(1, {"startIndex": 4, "endIndex": 5, + "inlineObjectElement": {"inlineObjectId": "img-1"}}) + elements[2]["startIndex"] += 1 + elements[2]["endIndex"] += 1 + body(image)[1]["endIndex"] += 1 + for doc in [table, header, suggested, image]: + with self.subTest(doc=doc): + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "submitted").exists()) + + def test_duplicate_in_table_or_header_also_refuses(self): + doc = document() + doc["tabs"][0]["documentTab"]["footers"] = { + "f.1": {"content": [paragraph([("world\n", {})])]}} + self.fixture(doc) + self.refused(self.plan()) + + def test_style_run_splitting_and_deletion_are_supported(self): + before = document() + body(before)[1] = paragraph([ + ("Hello wo", {"bold": True}), ("rld", {"italic": True}), (".\n", {})]) + after = document(revision="rev-2") + body(after)[1] = paragraph([("Hello ", {"bold": True}), (".\n", {})]) + self.fixture(before, after) + self.put("replacement.txt", "") + self.success(self.plan()) + self.success(self.apply()) + + def test_preserves_tables_images_styles_and_checks_untouched_content(self): + before = document() + body(before).append({"startIndex": 14, "endIndex": 15, "paragraph": { + "elements": [{"startIndex": 14, "endIndex": 15, + "inlineObjectElement": {"inlineObjectId": "img-1"}}]}}) + before["tabs"][0]["documentTab"]["inlineObjects"] = { + "img-1": {"inlineObjectProperties": {"embeddedObject": { + "imageProperties": {"contentUri": "https://example.invalid/temporary"}, + "size": {"width": {"magnitude": 50, "unit": "PT"}}}}}} + body(before).append({"startIndex": 15, "endIndex": 25, "table": { + "rows": 1, "columns": 1, "tableRows": [{"tableCells": [{ + "content": [paragraph([("cell\n", {"bold": True})], 18)]}]}]}}) + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + body(after)[1] = paragraph([("Hello reader.\n", {})]) + # One extra UTF-16 unit shifts following structures, not their content. + def shift(value): + if isinstance(value, dict): + for k, v in value.items(): + if k in ("startIndex", "endIndex"): + value[k] = v + 1 + else: + shift(v) + elif isinstance(value, list): + for v in value: + shift(v) + shift(body(after)[2:]) + after["tabs"][0]["documentTab"]["inlineObjects"]["img-1"][ + "inlineObjectProperties"]["embeddedObject"]["imageProperties"][ + "contentUri"] = "https://example.invalid/refreshed" + self.fixture(before, after) + self.success(self.plan()) + self.success(self.apply()) + for kind in ["table", "image-id", "image-size"]: + with self.subTest(kind=kind): + changed = copy.deepcopy(after) + if kind == "table": + table_paragraph = body(changed)[3]["table"]["tableRows"][0][ + "tableCells"][0]["content"][0] + table_paragraph["paragraph"]["elements"][0]["textRun"]["content"] = "sell\n" + elif kind == "image-id": + body(changed)[2]["paragraph"]["elements"][0][ + "inlineObjectElement"]["inlineObjectId"] = "different-image" + else: + changed["tabs"][0]["documentTab"]["inlineObjects"]["img-1"][ + "inlineObjectProperties"]["embeddedObject"]["size"]["width"][ + "magnitude"] = 51 + (self.root / "submitted").unlink() + self.fixture(before, changed) + self.refused(self.apply(), "ambiguous") + + def test_post_write_mismatch_and_missing_revision_are_not_success(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + missing = document("Hello reader.\n") + missing.pop("revisionId") + changed_style = document("Hello reader.\n", "rev-2") + body(changed_style)[1]["paragraph"]["elements"][0]["textRun"][ + "textStyle"] = {"bold": True} + for doc in [document("Hello incorrect.\n", "rev-2"), missing, + changed_style, document("Hello reader.\n", "rev-unexpected")]: + with self.subTest(doc=doc): + (self.root / "submitted").unlink(missing_ok=True) + self.fixture(document(), doc) + self.refused(self.apply(), "ambiguous") + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_failed_write_or_verification_never_retries_and_keeps_plan(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + for mode in ["conflict", "write-error", "zero", "two", "missing-reply", + "bad-response", "missing-write-revision", "verify-error"]: + with self.subTest(mode=mode): + (self.root / "submitted").unlink(missing_ok=True) + self.put("mode", mode) + calls = len(self.calls()) + self.refused(self.apply(), "ambiguous") + writes = [c for c in self.calls()[calls:] if c[2] == "batchUpdate"] + self.assertEqual(len(writes), 1) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_timeout_after_submission_reports_possible_application(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + self.put("mode", "timeout") + result = self.apply("--timeout", "0.3") + self.refused(result, "ambiguous") + self.assertIn("may have", json.loads(result.stderr)["message"]) + self.assertEqual(len([c for c in self.calls() if c[2] == "batchUpdate"]), 1) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_paths_are_relative_confined_and_never_overwrite(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + self.refused(self.plan()) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + for path in ["../escape", "/tmp/escape", "sub/../../escape", "bad\nname"]: + with self.subTest(path=path): + self.refused(self.cli("apply", "--plan", path)) + self.refused(self.cli("plan", "--document", "synthetic-doc", + "--find", path, "--replacement", "replacement.txt", + "--out", "new-plan.json")) + self.refused(self.cli("plan", "--document", "synthetic-doc", + "--find", "find.txt", "--replacement", "replacement.txt", + "--out", path)) + with tempfile.TemporaryDirectory() as outside: + (self.root / "escape").symlink_to(outside, target_is_directory=True) + Path(outside, "plan.json").write_bytes(original) + self.refused(self.cli("apply", "--plan", "escape/plan.json")) + self.refused(self.cli("plan", "--document", "synthetic-doc", + "--find", "find.txt", "--replacement", "replacement.txt", + "--out", "escape/new.json")) + self.assertFalse(Path(outside, "new.json").exists()) + + def test_invalid_document_id_and_gws_failure_are_redacted(self): + result = self.cli("plan", "--document", "../secret?token", + "--find", "find.txt", "--replacement", "replacement.txt", + "--out", "plan.json") + self.refused(result) + self.assertEqual(self.calls(), []) + self.put("mode", "read-error") + self.refused(self.plan()) + + def test_model_armor_diagnostics_remain_visible_without_leaking_content(self): + self.env["GOOGLE_WORKSPACE_CLI_SANITIZE_TEMPLATE"] = "synthetic-template" + self.put("mode", "armor-block") + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + self.put("mode", "armor-warn") + result = self.success(self.plan()) + self.assertTrue(result.get("gws_diagnostics")) + self.assertNotIn("secret-token", json.dumps(result)) + self.assertNotIn("\x1b", json.dumps(result)) + + def test_unknown_regions_and_malformed_structures_fail_before_plan(self): + unknown = document() + unknown["tabs"][0]["documentTab"]["futureRegion"] = {"text": "world"} + malformed = document() + body(malformed)[0]["futureBlock"] = {} + for doc in [unknown, malformed]: + with self.subTest(doc=doc): + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + + def test_null_scalar_and_incomplete_structures_refuse_before_plan(self): + cases = [] + for value in [None, 3, {}, {"rows": 1, "columns": 1, "tableRows": None}, + {"rows": 1, "columns": 1, "tableRows": [None]}, + {"rows": 1, "columns": 1, "tableRows": [{"tableCells": [None]}]}, + {"rows": 1, "columns": 1, "tableRows": [ + {"tableCells": [{"content": "not structural content"}]}]}]: + doc = document() + body(doc).append({"startIndex": 14, "endIndex": 15, "table": value}) + cases.append(doc) + for region in ["headers", "footers", "footnotes"]: + for value in [None, [], {"segment-1": None}, + {"segment-1": {"content": "not structural content"}}]: + doc = document() + doc["tabs"][0]["documentTab"][region] = value + cases.append(doc) + for kind in ["sectionBreak", "tableOfContents"]: + doc = document() + body(doc).append({"startIndex": 14, "endIndex": 15, kind: None}) + cases.append(doc) + for doc in cases: + with self.subTest(document=doc): + (self.root / "plan.json").unlink(missing_ok=True) + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + self.assertFalse((self.root / "submitted").exists()) + + def test_unselected_tab_malformed_ranges_refuse_preflight_and_postwrite(self): + before = document() + child = copy.deepcopy(document("😀 untouched\n")["tabs"][0]) + child["tabProperties"]["tabId"] = "t.other" + before["tabs"].append(child) + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + body(after)[1] = paragraph([("Hello reader.\n", {})]) + self.fixture(before, after) + self.success(self.plan("--tab", "t.main")) + original = (self.root / "plan.json").read_bytes() + for phase in ["preflight", "postwrite"]: + for field, value in [("endIndex", 999), ("startIndex", 2), + ("endIndex", True)]: + with self.subTest(phase=phase, field=field, value=value): + (self.root / "submitted").unlink(missing_ok=True) + bad = copy.deepcopy(before if phase == "preflight" else after) + bad["tabs"][1]["documentTab"]["body"]["content"][1][ + "paragraph"]["elements"][0][field] = value + self.fixture(bad if phase == "preflight" else before, + bad if phase == "postwrite" else after) + previous_calls = len(self.calls()) + self.refused(self.apply(), + "refused" if phase == "preflight" else "ambiguous") + expected_calls = (["get"] if phase == "preflight" else + ["get", "batchUpdate", "get"]) + self.assertEqual([c[2] for c in self.calls()[previous_calls:]], + expected_calls) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_unselected_text_run_shape_is_checked_before_discarding_fields(self): + for extra in [{"textRun": None}, {"textRun": {"content": "extra\n", + "textStyle": "invalid"}}, + {"futureElementMetadata": {"value": "must not disappear"}}]: + with self.subTest(extra=extra): + (self.root / "plan.json").unlink(missing_ok=True) + doc = document() + other = copy.deepcopy(document("extra\n")["tabs"][0]) + other["tabProperties"]["tabId"] = "t.other" + other["documentTab"]["body"]["content"][1]["paragraph"]["elements"][0].update( + extra) + doc["tabs"].append(other) + self.fixture(doc) + self.refused(self.plan("--tab", "t.main")) + self.assertFalse((self.root / "plan.json").exists()) + self.assertFalse((self.root / "submitted").exists()) + + def test_omitted_empty_text_style_does_not_fail_verification(self): + before = document() + after = document("Hello reader.\n", "rev-2") + body(after)[1]["paragraph"]["elements"][0]["textRun"].pop("textStyle") + self.fixture(before, after) + self.success(self.plan()) + self.success(self.apply()) + + def test_c1_control_characters_in_paths_are_rejected_before_read(self): + self.put("unsafe\u0085.txt", "world") + self.refused(self.cli( + "plan", "--document", "synthetic-doc", "--find", "unsafe\u0085.txt", + "--replacement", "replacement.txt", "--out", "plan.json", + )) + self.assertEqual(self.calls(), []) + + def test_nonregular_files_and_leaf_symlinks_refuse(self): + os.mkfifo(self.root / "pipe") + (self.root / "link.txt").symlink_to(self.root / "find.txt") + for path in ["pipe", "link.txt"]: + with self.subTest(path=path): + self.refused(self.cli( + "plan", "--document", "synthetic-doc", "--find", path, + "--replacement", "replacement.txt", "--out", "plan.json", + )) + self.assertEqual(self.calls(), []) + + def test_invalid_utf8_oversized_text_and_timeout_values_refuse(self): + for data in [b"\xff", b"x" * (16 * 1024 + 1)]: + with self.subTest(size=len(data)): + (self.root / "find.txt").write_bytes(data) + self.refused(self.plan()) + for timeout in ["nan", "inf", "0", "-1", "601"]: + with self.subTest(timeout=timeout): + self.refused(self.plan("--timeout", timeout)) + self.assertEqual(self.calls(), []) + + def test_malformed_source_and_unsupported_anchors_refuse(self): + for value in ["[]", "{", '{"documentId":NaN}']: + with self.subTest(value=value): + self.put("before.json", value) + self.refused(self.plan()) + for key in ["namedRanges", "bookmarks"]: + doc = document() + doc["tabs"][0]["documentTab"][key] = {"synthetic-anchor": {}} + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "submitted").exists()) + + def test_large_expanded_style_payload_refuses_before_plan(self): + doc = document("world" + "a" * 20000 + "\n") + body(doc)[1]["paragraph"]["elements"][0]["textRun"]["textStyle"] = { + "link": {"url": "https://example.invalid/" + "x" * 2000}} + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + + def test_default_zero_indices_in_untouched_header_are_supported(self): + before = document() + header = paragraph([("Header\n", {})], start=0) + del header["startIndex"] + del header["paragraph"]["elements"][0]["startIndex"] + before["tabs"][0]["documentTab"]["headers"] = {"h.1": {"content": [header]}} + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + body(after)[1] = paragraph([("Hello reader.\n", {})]) + self.fixture(before, after) + self.success(self.plan()) + self.success(self.apply()) + + +if __name__ == "__main__": + unittest.main() diff --git a/package.json b/package.json index 2a556aade..1084846b6 100644 --- a/package.json +++ b/package.json @@ -2,19 +2,19 @@ "name": "@googleworkspace/cli", "version": "0.22.5", "private": true, - "description": "Google Workspace CLI — dynamic command surface from Discovery Service", + "description": "Google Workspace CLI \u2014 dynamic command surface from Discovery Service", "license": "Apache-2.0", "repository": { "type": "git", - "url": "https://github.com/googleworkspace/cli.git" + "url": "https://github.com/ratovarius/cli.git" }, "author": { "name": "Justin Poehnelt", "email": "justin.poehnelt@gmail.com" }, - "homepage": "https://github.com/googleworkspace/cli", + "homepage": "https://github.com/ratovarius/cli", "bugs": { - "url": "https://github.com/googleworkspace/cli/issues" + "url": "https://github.com/ratovarius/cli/issues" }, "scripts": { "test": "cargo test", diff --git a/skills/gws-shared/SKILL.md b/skills/gws-shared/SKILL.md index dc6a32237..b658e05c7 100644 --- a/skills/gws-shared/SKILL.md +++ b/skills/gws-shared/SKILL.md @@ -76,7 +76,7 @@ gws [sub-resource] [flags] ## Community & Feedback Etiquette -- Encourage agents/users to star the repository when they find the project useful: `https://github.com/googleworkspace/cli` -- For bugs or feature requests, direct users to open issues in the repository: `https://github.com/googleworkspace/cli/issues` +- This is the independently maintained `https://github.com/ratovarius/cli` fork of `https://github.com/googleworkspace/cli`; see `FORK.md` for attribution and upstream contributions. +- For bugs or feature requests, direct users to open issues in the fork: `https://github.com/ratovarius/cli/issues` - Before creating a new issue, **always** search existing issues and feature requests first - If a matching issue already exists, add context by commenting on the existing thread instead of creating a duplicate