From 5db4424c7aeba4ced620b511183df30203e384c3 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 11:59:05 -0300 Subject: [PATCH 01/19] fix(script): keep current Clippy checks passing --- .changeset/current-clippy-baseline.md | 5 +++++ crates/google-workspace-cli/src/helpers/script.rs | 8 +------- 2 files changed, 6 insertions(+), 7 deletions(-) create mode 100644 .changeset/current-clippy-baseline.md diff --git a/.changeset/current-clippy-baseline.md b/.changeset/current-clippy-baseline.md new file mode 100644 index 000000000..a14e86242 --- /dev/null +++ b/.changeset/current-clippy-baseline.md @@ -0,0 +1,5 @@ +--- +"@googleworkspace/cli": patch +--- + +Keep Apps Script file selection compatible with the current Clippy checks. diff --git a/crates/google-workspace-cli/src/helpers/script.rs b/crates/google-workspace-cli/src/helpers/script.rs index 11bcdebec..4b31db62d 100644 --- a/crates/google-workspace-cli/src/helpers/script.rs +++ b/crates/google-workspace-cli/src/helpers/script.rs @@ -169,13 +169,7 @@ fn process_file(path: &Path) -> Result, GwsError> { filename.trim_end_matches(".js").trim_end_matches(".gs"), ), "html" => ("HTML", filename.trim_end_matches(".html")), - "json" => { - if filename == "appsscript.json" { - ("JSON", "appsscript") - } else { - return Ok(None); - } - } + "json" if filename == "appsscript.json" => ("JSON", "appsscript"), _ => return Ok(None), }; From f4c9d6ba24c8ac616fac37481d7c103f1cefae5e Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:04:16 -0300 Subject: [PATCH 02/19] feat(cli): allow unknown request fields with explicit opt-in Add a method-scoped validation policy for raw JSON requests, preserving strict helper calls and validation of known fields. Cover recursive preview fields, dry runs, request serialization, and CLI scope with regression tests. --- .changeset/allow-unknown-fields.md | 8 + README.md | 38 ++ crates/google-workspace-cli/src/commands.rs | 114 ++++- crates/google-workspace-cli/src/executor.rs | 484 +++++++++++++++++++- crates/google-workspace-cli/src/main.rs | 59 ++- 5 files changed, 676 insertions(+), 27 deletions(-) create mode 100644 .changeset/allow-unknown-fields.md diff --git a/.changeset/allow-unknown-fields.md b/.changeset/allow-unknown-fields.md new file mode 100644 index 000000000..cf9db0878 --- /dev/null +++ b/.changeset/allow-unknown-fields.md @@ -0,0 +1,8 @@ +--- +"@googleworkspace/cli": minor +--- + +Add `--allow-unknown-fields` to raw API methods with JSON request bodies. Explicitly +allow fields absent from Discovery recursively, including in dry runs, while +preserving validation of known fields, required fields, JSON, URLs and file paths. +Handwritten helpers retain strict validation. diff --git a/README.md b/README.md index 04c532d0a..1570db19b 100644 --- a/README.md +++ b/README.md @@ -104,6 +104,44 @@ gws schema drive.files.list gws drive files list --params '{"pageSize": 100}' --page-all | jq -r '.files[].name' ``` +### Fields absent from Discovery + +Raw API methods with a request body accept `--allow-unknown-fields` alongside +`--json`. Use it explicitly when an API supports fields that its public Discovery +document does not yet describe. It allows unknown properties recursively, +including nested objects and array elements, and forwards their values unchanged. +JSON is still parsed and serialized normally; whitespace and key order may change. + +Validation remains strict by default. With the flag, known-field types, enums and +required fields are still checked, as are JSON syntax, required URL parameters and +file paths. It does not allow new enum values on a known field. The flag is local +to raw methods and does not apply to handwritten `+` helpers. + +For example, Docs suggestions and comments require a Cloud project enrolled in the +[Google Workspace Developer Preview Program](https://developers.google.com/workspace/preview). +Google still enforces API availability, OAuth scopes, document permissions and +server-side validation. This flag grants no additional access. + +```bash +# Preview a suggested insertion (Docs Developer Preview). +gws docs documents batchUpdate \ + --params '{"documentId":"DOCUMENT_ID"}' \ + --json '{"requests":[{"insertText":{"location":{"index":1},"text":"Suggested text"}}],"writeControl":{"writeMode":"SUGGEST"}}' \ + --allow-unknown-fields --dry-run + +# Preview a comment anchored to existing text; adjust the range for your document. +gws docs documents batchUpdate \ + --params '{"documentId":"DOCUMENT_ID"}' \ + --json '{"requests":[{"insertComment":{"content":"Please review this text.","range":{"startIndex":1,"endIndex":5}}}]}' \ + --allow-unknown-fields --dry-run +``` + +`--dry-run` uses the same validation policy and shows the request without sending +it. It cannot verify preview enrollment or server acceptance. Remove `--dry-run` +to submit a request. See the Docs +[request reference](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents/request#InsertCommentRequest) +for preview field requirements. + ## Authentication The CLI supports multiple auth workflows so it works on your laptop, in CI, and on a server. diff --git a/crates/google-workspace-cli/src/commands.rs b/crates/google-workspace-cli/src/commands.rs index 27324e42b..e559ebe10 100644 --- a/crates/google-workspace-cli/src/commands.rs +++ b/crates/google-workspace-cli/src/commands.rs @@ -112,12 +112,20 @@ fn build_resource_command(name: &str, resource: &RestResource) -> Option, body_json: Option<&str>, is_media_upload: bool, + validation_policy: BodyValidationPolicy, ) -> Result { let params: Map = if let Some(p) = params_json { serde_json::from_str(p) @@ -112,7 +120,7 @@ fn parse_and_validate_inputs( if let Some(ref req_ref) = method.request { if let Some(ref schema_name) = req_ref.schema_ref { - validate_body_against_schema(&val, schema_name, doc)?; + validate_body_against_schema(&val, schema_name, doc, validation_policy)?; } } @@ -411,7 +419,54 @@ pub async fn execute_method( output_format: &crate::formatter::OutputFormat, capture_output: bool, ) -> Result, GwsError> { - let input = parse_and_validate_inputs(doc, method, params_json, body_json, upload.is_some())?; + execute_method_with_policy( + doc, + method, + params_json, + body_json, + token, + auth_method, + output_path, + upload, + dry_run, + pagination, + sanitize_template, + sanitize_mode, + output_format, + capture_output, + BodyValidationPolicy::Strict, + ) + .await +} + +/// Executes a raw API method with an explicit request-body validation policy. +/// Handwritten helpers use [`execute_method`] to retain strict validation. +#[allow(clippy::too_many_arguments)] +pub async fn execute_method_with_policy( + doc: &RestDescription, + method: &RestMethod, + params_json: Option<&str>, + body_json: Option<&str>, + token: Option<&str>, + auth_method: AuthMethod, + output_path: Option<&str>, + upload: Option>, + dry_run: bool, + pagination: &PaginationConfig, + sanitize_template: Option<&str>, + sanitize_mode: &crate::helpers::modelarmor::SanitizeMode, + output_format: &crate::formatter::OutputFormat, + capture_output: bool, + validation_policy: BodyValidationPolicy, +) -> Result, GwsError> { + let input = parse_and_validate_inputs( + doc, + method, + params_json, + body_json, + upload.is_some(), + validation_policy, + )?; if dry_run { let dry_run_info = json!({ @@ -996,9 +1051,10 @@ fn validate_body_against_schema( body: &Value, schema_name: &str, doc: &RestDescription, + validation_policy: BodyValidationPolicy, ) -> Result<(), GwsError> { let mut errors = Vec::new(); - validate_value(body, schema_name, doc, "$", &mut errors); + validate_value(body, schema_name, doc, "$", &mut errors, validation_policy); if !errors.is_empty() { return Err(GwsError::Validation(format!( @@ -1016,6 +1072,7 @@ fn validate_value( doc: &RestDescription, path: &str, errors: &mut Vec, + validation_policy: BodyValidationPolicy, ) { let schema = match doc.schemas.get(schema_ref_name) { Some(s) => s, @@ -1028,7 +1085,15 @@ fn validate_value( // If the top-level schema is an object if schema.schema_type.as_deref() == Some("object") || !schema.properties.is_empty() { if let Value::Object(obj) = value { - validate_properties(obj, &schema.properties, &schema.required, doc, path, errors); + validate_properties( + obj, + &schema.properties, + &schema.required, + doc, + path, + errors, + validation_policy, + ); } else { errors.push(format!("{path}: Expected object")); } @@ -1042,6 +1107,7 @@ fn validate_properties( doc: &RestDescription, path: &str, errors: &mut Vec, + validation_policy: BodyValidationPolicy, ) { let valid_keys: std::collections::HashSet<&String> = properties.keys().collect(); @@ -1060,15 +1126,24 @@ fn validate_properties( }; if !valid_keys.contains(key) { - errors.push(format!( - "{current_path}: Unknown property. Valid properties: {:?}", - valid_keys.iter().map(|k| k.as_str()).collect::>() - )); + if validation_policy == BodyValidationPolicy::Strict { + errors.push(format!( + "{current_path}: Unknown property. Valid properties: {:?}", + valid_keys.iter().map(|k| k.as_str()).collect::>() + )); + } continue; } let prop_schema = &properties[key]; - validate_property(val, prop_schema, doc, ¤t_path, errors); + validate_property( + val, + prop_schema, + doc, + ¤t_path, + errors, + validation_policy, + ); } } @@ -1078,10 +1153,11 @@ fn validate_property( doc: &RestDescription, path: &str, errors: &mut Vec, + validation_policy: BodyValidationPolicy, ) { // 1. Resolve $ref if present if let Some(ref_name) = &prop_schema.schema_ref { - validate_value(value, ref_name, doc, path, errors); + validate_value(value, ref_name, doc, path, errors, validation_policy); return; } @@ -1113,7 +1189,14 @@ fn validate_property( if let Value::Array(arr) = value { for (i, item) in arr.iter().enumerate() { let item_path = format!("{path}[{i}]"); - validate_property(item, items_schema, doc, &item_path, errors); + validate_property( + item, + items_schema, + doc, + &item_path, + errors, + validation_policy, + ); } } } @@ -1122,7 +1205,15 @@ fn validate_property( // 4. Object properties validation if prop_schema.prop_type.as_deref() == Some("object") && !prop_schema.properties.is_empty() { if let Value::Object(obj) = value { - validate_properties(obj, &prop_schema.properties, &[], doc, path, errors); + validate_properties( + obj, + &prop_schema.properties, + &[], + doc, + path, + errors, + validation_policy, + ); } } @@ -1186,6 +1277,332 @@ pub fn mime_to_extension(mime: &str) -> &str { } } +#[cfg(test)] +mod preview_fields_tests { + use super::*; + + fn fixture() -> (RestDescription, RestMethod) { + let doc = serde_json::from_value(json!({ + "name": "docs", + "version": "v1", + "rootUrl": "https://example.invalid/", + "servicePath": "v1/", + "schemas": { + "Body": { + "type": "object", + "required": ["name"], + "properties": { + "name": {"type": "string"}, + "mode": {"type": "string", "enum": ["ACTIVE"]}, + "count": {"type": "integer"}, + "tags": {"type": "array", "items": {"type": "string"}}, + "writeControl": { + "type": "object", + "properties": {"requiredRevisionId": {"type": "string"}} + }, + "child": {"$ref": "Child"}, + "requests": {"type": "array", "items": {"$ref": "Request"}}, + "children": { + "type": "array", + "items": {"type": "object", "properties": {"id": {"type": "string"}}} + } + } + }, + "Child": { + "type": "object", "required": ["id"], + "properties": {"id": {"type": "string"}} + }, + "Request": { + "type": "object", + "properties": { + "insertText": { + "type": "object", "properties": {"text": {"type": "string"}} + } + } + } + } + })) + .unwrap(); + let method = serde_json::from_value(json!({ + "httpMethod": "POST", + "path": "documents/{+documentId}:batchUpdate", + "parameterOrder": ["documentId"], + "parameters": { + "documentId": {"type": "string", "location": "path", "required": true}, + "view": {"type": "string", "location": "query", "required": true} + }, + "request": {"$ref": "Body"} + })) + .unwrap(); + (doc, method) + } + + const PARAMS: &str = r#"{"documentId":"test-document","view":"preview"}"#; + // Canonical JSON lets the request test assert exact emitted bytes as well as values. + const PREVIEW_BODY: &str = r#"{"name":"demo","preview":[null,true,1.25,9223372036854775807,{"text":"café\n\"quoted\""}],"requests":[{"insertComment":{"content":"Review"}}],"writeControl":{"writeMode":"SUGGEST"}}"#; + + #[test] + fn unknown_properties_require_opt_in_at_every_depth() { + let (doc, _) = fixture(); + for (body, path) in [ + (json!({"name": "demo", "preview": true}), "preview"), + ( + json!({"name": "demo", "writeControl": {"writeMode": "SUGGEST"}}), + "writeControl.writeMode", + ), + ( + json!({"name": "demo", "child": {"id": "one", "preview": null}}), + "child.preview", + ), + ( + json!({"name": "demo", "requests": [{"insertComment": {"content": "Review"}}]}), + "requests[0].insertComment", + ), + ( + json!({"name": "demo", "children": [{"id": "one", "preview": [1, true]}]}), + "children[0].preview", + ), + ] { + let err = + validate_body_against_schema(&body, "Body", &doc, BodyValidationPolicy::Strict) + .unwrap_err(); + assert!(err + .to_string() + .contains(&format!("{path}: Unknown property"))); + let result = validate_body_against_schema( + &body, + "Body", + &doc, + BodyValidationPolicy::AllowUnknownFields, + ); + assert!(result.is_ok(), "{path}: {result:?}"); + } + } + + #[test] + fn opt_in_preserves_known_field_and_required_validation() { + let (doc, _) = fixture(); + for (body, expected) in [ + ( + json!({"name": 42, "preview": true}), + "name: Expected type 'string'", + ), + ( + json!({"name": "demo", "count": 1.5, "preview": true}), + "count: Expected type 'integer'", + ), + ( + json!({"name": "demo", "mode": "PREVIEW", "preview": true}), + "not a valid enum member", + ), + (json!({"preview": true}), "Missing required property 'name'"), + ( + json!({"name": "demo", "child": {"preview": true}}), + "child: Missing required property 'id'", + ), + ( + json!({"name": "demo", "child": []}), + "child: Expected object", + ), + ( + json!({"name": "demo", "writeControl": {"requiredRevisionId": 42, "preview": true}}), + "writeControl.requiredRevisionId: Expected type 'string'", + ), + ( + json!({"name": "demo", "requests": [{"insertText": {"text": 42}, "preview": true}]}), + "requests[0].insertText.text: Expected type 'string'", + ), + ( + json!({"name": "demo", "children": [{"id": 42, "preview": true}]}), + "children[0].id: Expected type 'string'", + ), + ( + json!({"name": "demo", "tags": [true], "preview": true}), + "tags[0]: Expected type 'string'", + ), + ( + json!({"name": "demo", "requests": {}, "preview": true}), + "requests: Expected type 'array'", + ), + (json!([]), "$: Expected object"), + ] { + for policy in [ + BodyValidationPolicy::Strict, + BodyValidationPolicy::AllowUnknownFields, + ] { + let err = validate_body_against_schema(&body, "Body", &doc, policy).unwrap_err(); + assert!(err.to_string().contains(expected), "{policy:?}: {err}"); + } + } + } + + #[test] + fn opt_in_preserves_missing_schema_errors() { + let (doc, _) = fixture(); + let err = validate_body_against_schema( + &json!({}), + "Missing", + &doc, + BodyValidationPolicy::AllowUnknownFields, + ) + .unwrap_err(); + assert!(err.to_string().contains("Schema 'Missing' not found")); + } + + #[tokio::test] + async fn opt_in_dry_run_preserves_preview_body() { + let (doc, method) = fixture(); + let output = execute_method_with_policy( + &doc, + &method, + Some(PARAMS), + Some(PREVIEW_BODY), + None, + AuthMethod::None, + None, + None, + true, + &PaginationConfig::default(), + None, + &crate::helpers::modelarmor::SanitizeMode::Warn, + &crate::formatter::OutputFormat::Json, + true, + BodyValidationPolicy::AllowUnknownFields, + ) + .await + .unwrap() + .unwrap(); + assert_eq!( + output, + json!({ + "dry_run": true, + "url": "https://example.invalid/v1/documents/test%2Ddocument:batchUpdate", + "method": "POST", + "query_params": [["view", "preview"]], + "body": { + "name": "demo", + "preview": [null, true, 1.25, 9223372036854775807_i64, {"text": "café\n\"quoted\""}], + "requests": [{"insertComment": {"content": "Review"}}], + "writeControl": {"writeMode": "SUGGEST"} + }, + "is_multipart_upload": false + }) + ); + } + + #[tokio::test] + async fn helper_executor_entry_point_remains_strict() { + let (doc, method) = fixture(); + let err = execute_method( + &doc, + &method, + Some(PARAMS), + Some(PREVIEW_BODY), + None, + AuthMethod::None, + None, + None, + true, + &PaginationConfig::default(), + None, + &crate::helpers::modelarmor::SanitizeMode::Warn, + &crate::formatter::OutputFormat::Json, + true, + ) + .await + .unwrap_err(); + assert!(err.to_string().contains("Unknown property")); + } + + #[tokio::test] + #[serial_test::serial] + async fn opt_in_preserves_request_body_bytes() { + let (doc, method) = fixture(); + let input = parse_and_validate_inputs( + &doc, + &method, + Some(PARAMS), + Some(PREVIEW_BODY), + false, + BodyValidationPolicy::AllowUnknownFields, + ) + .unwrap(); + // Avoid native roots and all credential lookup. Build only; never send. + let client = reqwest::Client::builder() + .tls_built_in_root_certs(false) + .build() + .unwrap(); + let previous_project = std::env::var_os("GOOGLE_WORKSPACE_PROJECT_ID"); + std::env::set_var("GOOGLE_WORKSPACE_PROJECT_ID", "test-project"); + let request = build_http_request( + &client, + &method, + &input, + None, + &AuthMethod::None, + None, + 0, + &None, + ) + .await; + match previous_project { + Some(value) => std::env::set_var("GOOGLE_WORKSPACE_PROJECT_ID", value), + None => std::env::remove_var("GOOGLE_WORKSPACE_PROJECT_ID"), + } + let request = request.unwrap().build().unwrap(); + assert_eq!( + request.body().unwrap().as_bytes().unwrap(), + PREVIEW_BODY.as_bytes() + ); + assert_eq!(request.method(), reqwest::Method::POST); + assert_eq!( + request.url().as_str(), + "https://example.invalid/v1/documents/test%2Ddocument:batchUpdate?view=preview" + ); + assert!(!request.headers().contains_key("authorization")); + } + + #[test] + fn opt_in_preserves_json_parameter_and_url_errors() { + let (doc, method) = fixture(); + for (params, body, expected) in [ + (Some(PARAMS), "{", "Invalid --json body"), + (Some("{"), PREVIEW_BODY, "Invalid --params JSON"), + (Some("[]"), PREVIEW_BODY, "Invalid --params JSON"), + (None, PREVIEW_BODY, "Required path parameter documentId"), + ( + Some(r#"{"documentId":"test-document"}"#), + PREVIEW_BODY, + "Required parameter 'view'", + ), + ( + Some(r#"{"documentId":"../secret","view":"preview"}"#), + PREVIEW_BODY, + "path traversal", + ), + ( + Some(r#"{"documentId":"document?injected=true","view":"preview"}"#), + PREVIEW_BODY, + "must not contain '?'", + ), + ] { + let result = parse_and_validate_inputs( + &doc, + &method, + params, + Some(body), + false, + BodyValidationPolicy::AllowUnknownFields, + ); + let err = result.err().expect("unsafe input must fail"); + assert!( + err.to_string().contains(expected), + "expected {expected}: {err}" + ); + } + } +} + #[cfg(test)] mod tests { use super::*; @@ -1253,7 +1670,9 @@ mod tests { }; let body = json!({ "name": "My File" }); - assert!(validate_body_against_schema(&body, "File", &doc).is_ok()); + assert!( + validate_body_against_schema(&body, "File", &doc, BodyValidationPolicy::Strict).is_ok() + ); } #[test] @@ -1283,7 +1702,8 @@ mod tests { }; let body = json!({ "name": "My File", "invalidField": 123 }); - let result = validate_body_against_schema(&body, "File", &doc); + let result = + validate_body_against_schema(&body, "File", &doc, BodyValidationPolicy::Strict); assert!(result.is_err()); assert!(result.unwrap_err().to_string().contains("Unknown property")); } @@ -1373,23 +1793,39 @@ mod tests { "tags": ["one", "two"], "parent": { "id": "123" } }); - assert!(validate_body_against_schema(&body, "File", &doc).is_ok()); + assert!( + validate_body_against_schema(&body, "File", &doc, BodyValidationPolicy::Strict).is_ok() + ); // Missing Required Field let body_missing = json!({ "name": "My File" }); - let err = validate_body_against_schema(&body_missing, "File", &doc).unwrap_err(); + let err = + validate_body_against_schema(&body_missing, "File", &doc, BodyValidationPolicy::Strict) + .unwrap_err(); assert!(err .to_string() .contains("Missing required property 'status'")); // Invalid Enum Value let body_bad_enum = json!({ "name": "My File", "status": "UNKNOWN" }); - let err = validate_body_against_schema(&body_bad_enum, "File", &doc).unwrap_err(); + let err = validate_body_against_schema( + &body_bad_enum, + "File", + &doc, + BodyValidationPolicy::Strict, + ) + .unwrap_err(); assert!(err.to_string().contains("not a valid enum member")); // Invalid Type let body_bad_type = json!({ "name": "My File", "status": "ACTIVE", "count": "10" }); - let err = validate_body_against_schema(&body_bad_type, "File", &doc).unwrap_err(); + let err = validate_body_against_schema( + &body_bad_type, + "File", + &doc, + BodyValidationPolicy::Strict, + ) + .unwrap_err(); assert!(err .to_string() .contains("Expected type 'integer', found string")); @@ -1400,12 +1836,20 @@ mod tests { "status": "ACTIVE", "parent": { "invalidField": "123" } }); - let err = validate_body_against_schema(&body_bad_ref, "File", &doc).unwrap_err(); + let err = + validate_body_against_schema(&body_bad_ref, "File", &doc, BodyValidationPolicy::Strict) + .unwrap_err(); assert!(err.to_string().contains("Unknown property")); // Expected Object Type Failure let body_not_object = json!([]); - let err = validate_body_against_schema(&body_not_object, "File", &doc).unwrap_err(); + let err = validate_body_against_schema( + &body_not_object, + "File", + &doc, + BodyValidationPolicy::Strict, + ) + .unwrap_err(); assert!(err.to_string().contains("Expected object")); } #[tokio::test] diff --git a/crates/google-workspace-cli/src/main.rs b/crates/google-workspace-cli/src/main.rs index 41dcc1e1f..d634e4730 100644 --- a/crates/google-workspace-cli/src/main.rs +++ b/crates/google-workspace-cli/src/main.rs @@ -279,7 +279,7 @@ async fn run() -> Result<(), GwsError> { }; // Execute - executor::execute_method( + executor::execute_method_with_policy( &doc, method, params_json, @@ -294,11 +294,26 @@ async fn run() -> Result<(), GwsError> { &sanitize_config.mode, &output_format, false, + parse_body_validation_policy(matched_args), ) .await .map(|_| ()) } +fn parse_body_validation_policy(matches: &clap::ArgMatches) -> executor::BodyValidationPolicy { + if matches + .try_get_one::("allow-unknown-fields") + .ok() + .flatten() + .copied() + .unwrap_or(false) + { + executor::BodyValidationPolicy::AllowUnknownFields + } else { + executor::BodyValidationPolicy::Strict + } +} + /// Select the best scope from a method's scope list. /// /// Discovery Documents list method scopes as alternatives — any single scope @@ -525,6 +540,48 @@ fn is_version_flag(arg: &str) -> bool { mod tests { use super::*; + #[test] + fn test_parse_body_validation_policy_from_raw_method_flags() { + let doc: discovery::RestDescription = serde_json::from_value(serde_json::json!({ + "name": "test", + "version": "v1", + "rootUrl": "https://example.invalid/", + "servicePath": "", + "resources": {"files": {"methods": { + "create": {"path": "files", "httpMethod": "POST", "request": {"$ref": "File"}}, + "list": {"path": "files", "httpMethod": "GET"} + }}} + })) + .unwrap(); + for (args, expected) in [ + ( + vec!["gws", "files", "list"], + executor::BodyValidationPolicy::Strict, + ), + ( + vec!["gws", "files", "create", "--json", "{}"], + executor::BodyValidationPolicy::Strict, + ), + ( + vec![ + "gws", + "files", + "create", + "--json", + "{}", + "--allow-unknown-fields", + ], + executor::BodyValidationPolicy::AllowUnknownFields, + ), + ] { + let matches = commands::build_cli(&doc) + .try_get_matches_from(args) + .unwrap(); + let (_, method_args) = resolve_method_from_matches(&doc, &matches).unwrap(); + assert_eq!(parse_body_validation_policy(method_args), expected); + } + } + #[test] fn test_parse_pagination_config_defaults() { let matches = clap::Command::new("test") From 7588c6a5e3cfe6dfbac2683a9d513ad7848b8793 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:06:25 -0300 Subject: [PATCH 03/19] fix(auth): preserve credentials when decryption fails --- .changeset/preserve-credentials.md | 8 + README.md | 20 ++ crates/google-workspace-cli/src/auth.rs | 298 +++++++++++++++++------- 3 files changed, 245 insertions(+), 81 deletions(-) create mode 100644 .changeset/preserve-credentials.md diff --git a/.changeset/preserve-credentials.md b/.changeset/preserve-credentials.md new file mode 100644 index 000000000..4b7bafe37 --- /dev/null +++ b/.changeset/preserve-credentials.md @@ -0,0 +1,8 @@ +--- +"@googleworkspace/cli": patch +--- + +Preserve saved encrypted credentials and token caches when credential loading, +decryption, or keyring access fails. Report recovery guidance and stop authentication +instead of silently selecting plaintext credentials or another account through ADC. +Explicit token and credentials-file overrides and intentional logout remain unchanged. diff --git a/README.md b/README.md index 1570db19b..a0df710c8 100644 --- a/README.md +++ b/README.md @@ -253,6 +253,26 @@ export GOOGLE_WORKSPACE_CLI_TOKEN=$(gcloud auth print-access-token) Environment variables can also live in a `.env` file. +### Troubleshooting saved credentials + +If `gws` cannot read or decrypt `credentials.enc` (including a keyring access +failure), it returns an authentication error and preserves that file, +`token_cache.json`, and `sa_token_cache.json`. It does not silently switch to +plaintext credentials or Application Default Credentials (ADC). This applies to +the default configuration directory and `GOOGLE_WORKSPACE_CLI_CONFIG_DIR`. + +Check that you are using the original configuration directory and can access its +original OS keyring or encryption key. Back up the configuration before changing +key storage or replacing credentials. Preservation does not recover a lost key. +If you intentionally want to discard saved credentials and sign in again, use +`gws auth logout` followed by `gws auth login`; logout still removes saved +credentials and token caches. + +An explicit `GOOGLE_WORKSPACE_CLI_TOKEN` or +`GOOGLE_WORKSPACE_CLI_CREDENTIALS_FILE` still takes precedence. A missing or invalid +explicit credentials file is an error. When no encrypted credentials file exists, +the usual plaintext and ADC fallback remains available. + ## AI Agent Skills The repo ships 100+ Agent Skills (`SKILL.md` files) — one for every supported API, plus higher-level helpers for common workflows and 50 curated recipes for Gmail, Drive, Docs, Calendar, and Sheets. See the full [Skills Index](docs/skills.md) for the complete list. diff --git a/crates/google-workspace-cli/src/auth.rs b/crates/google-workspace-cli/src/auth.rs index 9d8847e4b..6b27e3f92 100644 --- a/crates/google-workspace-cli/src/auth.rs +++ b/crates/google-workspace-cli/src/auth.rs @@ -336,6 +336,22 @@ async fn load_credentials_inner( env_file: Option<&str>, enc_path: &std::path::Path, default_path: &std::path::Path, +) -> anyhow::Result { + load_credentials_with_loader( + env_file, + enc_path, + default_path, + credential_store::load_encrypted_from_path, + ) + .await +} + +// Keep credential selection testable without accessing the OS keyring. +async fn load_credentials_with_loader( + env_file: Option<&str>, + enc_path: &std::path::Path, + default_path: &std::path::Path, + load_encrypted: impl FnOnce(&std::path::Path) -> anyhow::Result, ) -> anyhow::Result { // 1. Explicit env var — plaintext file (User or Service Account) if let Some(path) = env_file { @@ -353,40 +369,21 @@ async fn load_credentials_inner( // 2. Encrypted credentials if enc_path.exists() { - match credential_store::load_encrypted_from_path(enc_path) { - Ok(json_str) => { - return parse_credential_file(enc_path, &json_str).await; - } - Err(e) => { - // Decryption failed — the encryption key likely changed (e.g. after - // an upgrade that migrated keys between keyring and file storage). - // Remove the stale file so the next `gws auth login` starts fresh, - // and fall through to other credential sources (plaintext, ADC). - eprintln!( - "Warning: removing undecryptable credentials file ({}): {e:#}", - enc_path.display() - ); - if let Err(err) = tokio::fs::remove_file(enc_path).await { - eprintln!( - "Warning: failed to remove stale credentials file '{}': {err}", - enc_path.display() - ); - } - // Also remove stale token caches that used the old key. - for cache_file in ["token_cache.json", "sa_token_cache.json"] { - let path = enc_path.with_file_name(cache_file); - if let Err(err) = tokio::fs::remove_file(&path).await { - if err.kind() != std::io::ErrorKind::NotFound { - eprintln!( - "Warning: failed to remove stale token cache '{}': {err}", - path.display() - ); - } - } - } - // Fall through to remaining credential sources below. - } - } + // A read, decryption, or keyring failure does not mean the files are + // disposable. Stop here so a retry cannot silently select another account. + // Do not render backend error details, which may contain sensitive data. + let json_str = load_encrypted(enc_path).map_err(|_| { + anyhow::anyhow!( + "Failed to read or decrypt saved credentials at {}. \ + Check access to the original OS keyring or encryption key and verify \ + GOOGLE_WORKSPACE_CLI_CONFIG_DIR. Credentials and token caches have been \ + preserved; no fallback credentials were used. Back up the configuration \ + before intentionally replacing credentials with `gws auth logout` and \ + `gws auth login`.", + crate::output::sanitize_for_terminal(&enc_path.display().to_string()) + ) + })?; + return parse_credential_file(enc_path, &json_str).await; } // 3. Plaintext credentials at default path (AuthorizedUser) @@ -825,75 +822,214 @@ mod tests { #[tokio::test] #[serial_test::serial] - async fn test_load_credentials_corrupt_encrypted_file_is_removed() { - // When credentials.enc cannot be decrypted, the file should be removed - // automatically and the function should fall through to other sources. - let tmp = tempfile::tempdir().unwrap(); - let _home_guard = EnvVarGuard::set("HOME", tmp.path()); - let _adc_guard = EnvVarGuard::remove("GOOGLE_APPLICATION_CREDENTIALS"); + async fn test_load_credentials_preserves_failed_encrypted_credentials_and_caches() { + let dir = tempfile::tempdir().unwrap(); + let enc_path = dir.path().join("credentials.enc"); + let token_path = dir.path().join("token_cache.json"); + let service_token_path = dir.path().join("sa_token_cache.json"); + let absent_path = dir.path().join("missing.json"); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &absent_path); + + // A short invalid payload fails before accessing any OS keyring. + std::fs::write(&enc_path, b"bad").unwrap(); + std::fs::write(&token_path, b"synthetic-user-cache").unwrap(); + std::fs::write(&service_token_path, b"synthetic-service-cache").unwrap(); + + for _ in 0..2 { + let result = load_credentials_inner(None, &enc_path, &absent_path).await; + + assert!(result.is_err()); + assert!( + enc_path.exists(), + "Authentication failure must preserve saved encrypted credentials" + ); + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); + assert_eq!(std::fs::read(&token_path).unwrap(), b"synthetic-user-cache"); + assert_eq!( + std::fs::read(&service_token_path).unwrap(), + b"synthetic-service-cache" + ); + } + } + #[tokio::test] + #[serial_test::serial] + async fn test_load_credentials_corrupt_encrypted_reports_safe_remediation() { let dir = tempfile::tempdir().unwrap(); let enc_path = dir.path().join("credentials.enc"); + let absent_path = dir.path().join("missing.json"); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &absent_path); + std::fs::write(&enc_path, b"bad").unwrap(); - // Write garbage data that cannot be decrypted. - tokio::fs::write(&enc_path, b"not-valid-encrypted-data-at-all-1234567890") + let err = load_credentials_inner(None, &enc_path, &absent_path) .await - .unwrap(); - assert!(enc_path.exists()); - - let result = - load_credentials_inner(None, &enc_path, &PathBuf::from("/does/not/exist")).await; + .unwrap_err(); + let msg = format!("{err:#}"); + + assert!(msg.contains("decrypt"), "{msg}"); + assert!(msg.contains("keyring"), "{msg}"); + assert!(msg.contains("preserved"), "{msg}"); + assert!(msg.contains("Back up"), "{msg}"); + assert!(!msg.contains("No credentials found"), "{msg}"); + assert!(!msg.contains("bad"), "{msg}"); + } - // Should fall through to "No credentials found" (not a decryption error). - assert!(result.is_err()); - let msg = result.unwrap_err().to_string(); - assert!( - msg.contains("No credentials found"), - "Should fall through to final error, got: {msg}" - ); - assert!( - !enc_path.exists(), - "Stale credentials.enc must be removed after decryption failure" - ); + #[tokio::test] + #[serial_test::serial] + async fn test_load_credentials_corrupt_encrypted_blocks_plaintext_and_adc() { + // Exercise both a default-style layout and a configured directory, + // without changing HOME or touching the actual default directory. + let dir = tempfile::tempdir().unwrap(); + let fallback_json = r#"{ + "client_id": "different-account", + "client_secret": "synthetic-secret", + "refresh_token": "synthetic-refresh", + "type": "authorized_user" + }"#; + let adc_path = dir.path().join("adc.json"); + std::fs::write(&adc_path, fallback_json).unwrap(); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &adc_path); + + for layout in [".config/gws", "custom-config"] { + let config = dir.path().join(layout); + std::fs::create_dir_all(&config).unwrap(); + let enc_path = config.join("credentials.enc"); + let plain_path = config.join("credentials.json"); + std::fs::write(&enc_path, b"bad").unwrap(); + std::fs::write(&plain_path, fallback_json).unwrap(); + + for fallback in [&plain_path, &config.join("missing.json")] { + let err = load_credentials_inner(None, &enc_path, fallback) + .await + .expect_err("Broken encrypted credentials must block another account"); + assert!(err.to_string().contains("decrypt")); + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); + assert_eq!(std::fs::read_to_string(&plain_path).unwrap(), fallback_json); + assert_eq!(std::fs::read_to_string(&adc_path).unwrap(), fallback_json); + } + } } #[tokio::test] #[serial_test::serial] - async fn test_load_credentials_corrupt_encrypted_falls_through_to_plaintext() { - // When credentials.enc is corrupt but a valid plaintext file exists, - // the function should fall through and use the plaintext credentials. + async fn test_load_credentials_keyring_failure_preserves_files_and_blocks_adc() { let dir = tempfile::tempdir().unwrap(); let enc_path = dir.path().join("credentials.enc"); let plain_path = dir.path().join("credentials.json"); + let adc_path = dir.path().join("adc.json"); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &adc_path); + let sentinels: &[(&str, &[u8])] = &[ + ("credentials.enc", b"synthetic-encrypted-credentials"), + ("token_cache.json", b"synthetic-user-cache"), + ("sa_token_cache.json", b"synthetic-service-cache"), + (".encryption_key", b"synthetic-key"), + ]; + for (name, bytes) in sentinels { + std::fs::write(dir.path().join(name), bytes).unwrap(); + } + std::fs::write( + &adc_path, + r#"{"type":"authorized_user","client_id":"other","client_secret":"secret","refresh_token":"refresh"}"#, + ) + .unwrap(); - // Write garbage encrypted data. - tokio::fs::write(&enc_path, b"not-valid-encrypted-data-at-all-1234567890") + for _ in 0..2 { + let err = load_credentials_with_loader(None, &enc_path, &plain_path, |_| { + anyhow::bail!("OS keyring unavailable: synthetic-sensitive-detail") + }) .await - .unwrap(); + .expect_err("Key acquisition failure must stop credential selection"); + for (name, bytes) in sentinels { + assert_eq!(std::fs::read(dir.path().join(name)).unwrap(), *bytes); + } + let msg = format!("{err:#}"); + assert!(msg.contains("keyring"), "{msg}"); + assert!(msg.contains("preserved"), "{msg}"); + assert!(!msg.contains("synthetic-sensitive-detail"), "{msg}"); + assert!(!msg.contains("synthetic-key"), "{msg}"); + } + } - // Write valid plaintext credentials. - let plain_json = r#"{ - "client_id": "fallback_id", - "client_secret": "fallback_secret", - "refresh_token": "fallback_refresh", - "type": "authorized_user" - }"#; - tokio::fs::write(&plain_path, plain_json).await.unwrap(); + #[tokio::test] + #[serial_test::serial] + async fn test_load_credentials_explicit_file_precedes_broken_encrypted_credentials() { + let dir = tempfile::tempdir().unwrap(); + let enc_path = dir.path().join("credentials.enc"); + let explicit_path = dir.path().join("explicit.json"); + let missing_path = dir.path().join("missing.json"); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &missing_path); + std::fs::write(&enc_path, b"bad").unwrap(); + std::fs::write( + &explicit_path, + r#"{"type":"authorized_user","client_id":"explicit","client_secret":"secret","refresh_token":"refresh"}"#, + ) + .unwrap(); - let res = load_credentials_inner(None, &enc_path, &plain_path) + let creds = load_credentials_inner(explicit_path.to_str(), &enc_path, &missing_path) .await .unwrap(); + match creds { + Credential::AuthorizedUser(secret) => assert_eq!(secret.client_id, "explicit"), + _ => panic!("Expected explicitly selected account"), + } + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); - match res { - Credential::AuthorizedUser(secret) => { - assert_eq!( - secret.client_id, "fallback_id", - "Should fall through to plaintext credentials" - ); + // A missing or malformed explicit file must not select another account. + for contents in [None, Some("invalid-json")] { + if let Some(contents) = contents { + std::fs::write(&missing_path, contents).unwrap(); } - _ => panic!("Expected AuthorizedUser from plaintext fallback"), + let err = load_credentials_inner(missing_path.to_str(), &enc_path, &explicit_path) + .await + .unwrap_err(); + assert!(err.to_string().contains(if contents.is_some() { + "Failed to parse" + } else { + "does not exist" + })); + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); + } + } + + #[tokio::test] + #[serial_test::serial] + async fn test_get_token_preserves_configured_credentials_until_explicit_logout() { + let dir = tempfile::tempdir().unwrap(); + let enc_path = dir.path().join("credentials.enc"); + let missing_path = dir.path().join("missing.json"); + let _config_guard = EnvVarGuard::set("GOOGLE_WORKSPACE_CLI_CONFIG_DIR", dir.path()); + let _adc_guard = EnvVarGuard::set("GOOGLE_APPLICATION_CREDENTIALS", &missing_path); + let _file_guard = EnvVarGuard::remove("GOOGLE_WORKSPACE_CLI_CREDENTIALS_FILE"); + let _token_guard = EnvVarGuard::set("GOOGLE_WORKSPACE_CLI_TOKEN", ""); + let names = [ + "credentials.enc", + "credentials.json", + "token_cache.json", + "sa_token_cache.json", + ]; + for name in names { + std::fs::write(dir.path().join(name), b"bad").unwrap(); + } + + let err = get_token(&[]).await.unwrap_err(); + assert!(err.to_string().contains("decrypt"), "{err}"); + for name in names { + assert_eq!(std::fs::read(dir.path().join(name)).unwrap(), b"bad"); + } + + // An explicit token still takes precedence over all credential files. + { + let _token_guard = EnvVarGuard::set("GOOGLE_WORKSPACE_CLI_TOKEN", "synthetic-token"); + assert_eq!(get_token(&[]).await.unwrap(), "synthetic-token"); + assert_eq!(std::fs::read(&enc_path).unwrap(), b"bad"); + } + + crate::auth_commands::handle_auth_command(&["logout".into()]) + .await + .unwrap(); + for name in names { + assert!(!dir.path().join(name).exists(), "Logout must remove {name}"); } - assert!(!enc_path.exists(), "Stale credentials.enc must be removed"); } #[tokio::test] From fd79f1726d50bf1dc368ccb9e2c0f0250948b407 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:00:19 -0300 Subject: [PATCH 04/19] feat(docs): add structured document reader --- .changeset/docs-structured-read.md | 9 + .../google-workspace-cli/src/helpers/docs.rs | 13 +- .../src/helpers/docs/read.rs | 536 ++++++++++++++++ .../src/helpers/docs/read_tests.rs | 593 ++++++++++++++++++ docs/docs-read.md | 84 +++ 5 files changed, 1234 insertions(+), 1 deletion(-) create mode 100644 .changeset/docs-structured-read.md create mode 100644 crates/google-workspace-cli/src/helpers/docs/read.rs create mode 100644 crates/google-workspace-cli/src/helpers/docs/read_tests.rs create mode 100644 docs/docs-read.md diff --git a/.changeset/docs-structured-read.md b/.changeset/docs-structured-read.md new file mode 100644 index 000000000..41b070876 --- /dev/null +++ b/.changeset/docs-structured-read.md @@ -0,0 +1,9 @@ +--- +"@googleworkspace/cli": minor +--- + +Add `gws docs +read` to translate documents into compact structured content with +recursive tabs, headings and an outline, styled text, suggestions, nested tables, +figure metadata, and reference markers. Preserve API indices and revisions, +reject partial field masks, and support the existing formatters, sanitization, +and credential-free dry-run. diff --git a/crates/google-workspace-cli/src/helpers/docs.rs b/crates/google-workspace-cli/src/helpers/docs.rs index d3ef7fa21..6492cb7c9 100644 --- a/crates/google-workspace-cli/src/helpers/docs.rs +++ b/crates/google-workspace-cli/src/helpers/docs.rs @@ -21,14 +21,21 @@ use serde_json::json; use std::future::Future; use std::pin::Pin; +mod read; + pub struct DocsHelper; +#[cfg(test)] +#[path = "docs/read_tests.rs"] +mod read_tests; + impl Helper for DocsHelper { fn inject_commands( &self, mut cmd: Command, _doc: &crate::discovery::RestDescription, ) -> Command { + cmd = cmd.subcommand(read::command()); cmd = cmd.subcommand( Command::new("+write") .about("[Helper] Append text to a document") @@ -63,9 +70,13 @@ TIPS: &'a self, doc: &'a crate::discovery::RestDescription, matches: &'a ArgMatches, - _sanitize_config: &'a crate::helpers::modelarmor::SanitizeConfig, + sanitize_config: &'a crate::helpers::modelarmor::SanitizeConfig, ) -> Pin> + Send + 'a>> { Box::pin(async move { + if let Some(matches) = matches.subcommand_matches("+read") { + read::handle(doc, matches, sanitize_config).await?; + return Ok(true); + } if let Some(matches) = matches.subcommand_matches("+write") { let (params_str, body_str, scopes) = build_write_request(matches, doc)?; diff --git a/crates/google-workspace-cli/src/helpers/docs/read.rs b/crates/google-workspace-cli/src/helpers/docs/read.rs new file mode 100644 index 000000000..db9bcc9b7 --- /dev/null +++ b/crates/google-workspace-cli/src/helpers/docs/read.rs @@ -0,0 +1,536 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Translate Docs structure, rather than render its visual layout. Keep API +//! indices and metadata; never calculate edit offsets from extracted text. + +use crate::discovery::RestDescription; +use crate::error::GwsError; +use crate::executor::{self, AuthMethod, PaginationConfig}; +use crate::formatter::{format_value, OutputFormat}; +use crate::helpers::modelarmor::SanitizeConfig; +use clap::{Arg, ArgMatches, Command}; +use serde_json::{json, Map, Value}; +use std::future::Future; + +pub(super) fn command() -> Command { + Command::new("+read") + .about("[Helper] Read a document as compact structured content") + .arg(Arg::new("document").long("document").help("Document ID").required(true).value_name("ID")) + .arg(Arg::new("params").long("params").help("Additional documents.get API parameters as JSON").value_name("JSON")) + .after_help( + "\ +EXAMPLES: + gws docs +read --document DOC_ID + gws docs +read --document DOC_ID --format yaml + gws docs +read --document DOC_ID --params '{\"fields\":\"*\"}' --dry-run + gws docs +read --document DOC_ID | jq '.outline' + gws docs +read --document DOC_ID | jq '.. | objects | select(.paragraphStyle?.headingId? == \"HEADING_ID\")' + +TIPS: + Requests all tabs with includeTabsContent=true and suggestionsViewMode=SUGGESTIONS_INLINE. + Only those tab/suggestion options are supported; fields must be absent or exactly \"*\". + Other documents.get options pass through --params; alt must be json. $fields is rejected. + JSON/YAML preserve the structured view; table/CSV use the global formatter's array summary. + tabs[].blocks and childTabs keep API order; outline lists headings with tab IDs and JSON Pointer paths. + Paragraph text concatenates text runs only. elements retain styles, links, suggestion IDs and reference markers. + Tables contain rows[].cells[].blocks recursively. Headers, footers and footnotes have separate blocks. + figures contain image/drawing metadata, including alt text and URIs when returned; images are never downloaded. + Unknown blocks, inline elements and tab types retain type=unknown markers and raw data. + startIndex/endIndex are the API's UTF-16 offsets, scoped to each tab/segment; never offsets into extracted text. + revisionId and suggestionsViewMode are retained when returned. Missing revisionId is not synthesized. + source=legacyBody indicates a fallback response without populated tabs; all-tab coverage cannot be confirmed. + This is a content view, not a layout renderer or lossless API round trip. Inherited styles are not resolved. + Suggestions remain inline, including proposed deletions; this helper does not accept or reject suggestions. + Use raw documents get for unsupported views or field masks. Missing body content produces an error. + --dry-run validates and prints a request plan without acquiring credentials or fetching document content. + --sanitize uses the existing Model Armor policy before normalization and retains _sanitization metadata.", + ) +} + +pub(super) async fn handle( + doc: &RestDescription, + matches: &ArgMatches, + sanitize: &SanitizeConfig, +) -> Result<(), GwsError> { + let output = run(doc, matches, sanitize, async { + crate::auth::get_token(&["https://www.googleapis.com/auth/documents.readonly"]) + .await + .map_err(|e| GwsError::Auth(format!("Docs auth failed: {e}"))) + }) + .await?; + println!("{output}"); + Ok(()) +} + +/// Token acquisition is lazy so local validation and dry-run never read +/// credentials. The executor remains responsible for HTTP and sanitization. +pub(super) async fn run( + doc: &RestDescription, + matches: &ArgMatches, + sanitize: &SanitizeConfig, + token: impl Future>, +) -> Result { + let params = build_params( + matches.get_one::("document").unwrap(), + matches.get_one::("params").map(String::as_str), + )?; + let method = doc + .resources + .get("documents") + .and_then(|r| r.methods.get("get")) + .ok_or_else(|| GwsError::Discovery("Method 'documents.get' not found".into()))?; + let dry_run = matches.get_flag("dry-run"); + let token = if dry_run { None } else { Some(token.await?) }; + let format = matches + .get_one::("format") + .map(|f| OutputFormat::from_str(f)) + .unwrap_or_default(); + let result = executor::execute_method( + doc, + method, + Some(¶ms.to_string()), + None, + token.as_deref(), + if token.is_some() { + AuthMethod::OAuth + } else { + AuthMethod::None + }, + None, + None, + dry_run, + &PaginationConfig::default(), + sanitize.template.as_deref(), + &sanitize.mode, + &format, + true, + ) + .await? + .ok_or_else(|| invalid_content("expected a JSON document response"))?; + let output = if dry_run { result } else { normalize(&result)? }; + Ok(format_value(&output, &format)) +} + +pub(super) fn build_params(document: &str, params: Option<&str>) -> Result { + crate::validate::validate_resource_name(document)?; + let mut params: Map = match params { + Some(raw) => serde_json::from_str(raw) + .map_err(|e| GwsError::Validation(format!("Invalid --params JSON object: {e}")))?, + None => Map::new(), + }; + // A complete response is required for normalization. A narrow allowlist is + // deliberate: parsing arbitrary nested masks cannot prove completeness as + // the Docs API grows. Reject the system-parameter alias as well. + if params.contains_key("$fields") || params.get("fields").is_some_and(|v| v != "*") { + return Err(GwsError::Validation( + "docs +read requires all content: omit fields or use \"*\"; $fields is unsupported" + .into(), + )); + } + for (key, required) in [ + ("documentId", json!(document)), + ("includeTabsContent", json!(true)), + ("suggestionsViewMode", json!("SUGGESTIONS_INLINE")), + ] { + if params.get(key).is_some_and(|v| v != &required) { + return Err(GwsError::Validation(format!( + "docs +read requires {key}={required}" + ))); + } + params.insert(key.into(), required); + } + if params.get("alt").is_some_and(|v| v != "json") { + return Err(GwsError::Validation("docs +read requires alt=json".into())); + } + Ok(Value::Object(params)) +} + +fn invalid_content(detail: &str) -> GwsError { + GwsError::Validation(format!( + "Incomplete or invalid Docs response: {detail}; use raw documents get to inspect it" + )) +} + +fn object(value: &Value) -> Result, GwsError> { + value + .as_object() + .cloned() + .ok_or_else(|| invalid_content("expected an object")) +} + +fn array<'a>(value: &'a Value, field: &str) -> Result<&'a [Value], GwsError> { + value + .get(field) + .and_then(Value::as_array) + .map(Vec::as_slice) + .ok_or_else(|| invalid_content(&format!("missing or invalid {field} array"))) +} + +const TAB_CONTENT: &[&str] = &[ + "body", + "headers", + "footers", + "footnotes", + "inlineObjects", + "positionedObjects", + "lists", + "namedStyles", + "namedRanges", + "suggestedNamedStylesChanges", +]; + +pub(super) fn normalize(document: &Value) -> Result { + let mut result = object(document)?; + let mut outline = Vec::new(); + let tabs = match document.get("tabs") { + Some(_) => array(document, "tabs")?, + None => &[], + }; + let (source, tabs) = if tabs.is_empty() { + let mut tab = Map::from_iter([ + ("tabId".into(), Value::Null), + ("parentTabId".into(), Value::Null), + ("childTabs".into(), json!([])), + ]); + if let Some(title) = document.get("title") { + tab.insert("title".into(), title.clone()); + } + let content: Map = TAB_CONTENT + .iter() + .filter_map(|key| document.get(key).map(|v| ((*key).into(), v.clone()))) + .collect(); + tab.extend(contents( + &Value::Object(content), + &Value::Null, + "/tabs/0", + &mut outline, + )?); + ("legacyBody", vec![Value::Object(tab)]) + } else { + let tabs = tabs + .iter() + .enumerate() + .map(|(i, tab)| normalize_tab(tab, &Value::Null, &format!("/tabs/{i}"), &mut outline)) + .collect::, _>>()?; + ("tabs", tabs) + }; + for field in TAB_CONTENT { + result.remove(*field); + } + result.insert("source".into(), json!(source)); + result.insert("tabs".into(), json!(tabs)); + result.insert("outline".into(), json!(outline)); + Ok(Value::Object(result)) +} + +fn normalize_tab( + tab: &Value, + parent: &Value, + path: &str, + outline: &mut Vec, +) -> Result { + let mut result = tab + .get("tabProperties") + .map(object) + .transpose()? + .unwrap_or_default(); + result + .entry("parentTabId") + .or_insert_with(|| parent.clone()); + let id = result.get("tabId").cloned().unwrap_or(Value::Null); + if let Some(content) = tab.get("documentTab") { + result.extend(contents(content, &id, path, outline)?); + let mut extra = object(tab)?; + for key in ["tabProperties", "documentTab", "childTabs"] { + extra.remove(key); + } + if !extra.is_empty() { + result.insert("metadata".into(), Value::Object(extra)); + } + } else { + result.insert("type".into(), json!("unknown")); + result.insert("data".into(), tab.clone()); + } + let children = if tab.get("childTabs").is_some() { + array(tab, "childTabs")? + } else { + &[] + }; + let children = children + .iter() + .enumerate() + .map(|(i, tab)| normalize_tab(tab, &id, &format!("{path}/childTabs/{i}"), outline)) + .collect::, _>>()?; + result.insert("childTabs".into(), json!(children)); + Ok(Value::Object(result)) +} + +fn contents( + content: &Value, + tab_id: &Value, + path: &str, + outline: &mut Vec, +) -> Result, GwsError> { + let mut result = object(content)?; + let body = result + .remove("body") + .ok_or_else(|| invalid_content("missing body"))?; + result.insert( + "blocks".into(), + blocks( + array(&body, "content")?, + tab_id, + &format!("{path}/blocks"), + outline, + )?, + ); + let mut body_metadata = object(&body)?; + body_metadata.remove("content"); + if !body_metadata.is_empty() { + result.insert("bodyMetadata".into(), Value::Object(body_metadata)); + } + for kind in ["headers", "footers", "footnotes"] { + if let Some(segments) = result.get_mut(kind) { + let mut normalized = Map::new(); + for (id, segment) in object(segments)? { + let mut segment_result = object(&segment)?; + // JSON Pointer escaping, not URL escaping. + let pointer_id = id.replace('~', "~0").replace('/', "~1"); + segment_result.insert( + "blocks".into(), + blocks( + array(&segment, "content")?, + tab_id, + &format!("{path}/{kind}/{pointer_id}/blocks"), + outline, + )?, + ); + segment_result.remove("content"); + normalized.insert(id, Value::Object(segment_result)); + } + *segments = Value::Object(normalized); + } + } + let mut figures = Map::new(); + for (field, properties, placement) in [ + ("inlineObjects", "inlineObjectProperties", "inline"), + ( + "positionedObjects", + "positionedObjectProperties", + "positioned", + ), + ] { + if let Some(objects) = result.remove(field) { + for (id, value) in object(&objects)? { + let mut figure = object(&value)?; + if let Some(properties) = figure.remove(properties) { + figure.extend(object(&properties)?); + } + let kind = if figure + .get("embeddedObject") + .and_then(|v| v.get("imageProperties")) + .is_some() + { + "image" + } else if figure + .get("embeddedObject") + .and_then(|v| v.get("embeddedDrawingProperties")) + .is_some() + { + "drawing" + } else { + "unknown" + }; + figure.insert("type".into(), json!(kind)); + figure.insert("placement".into(), json!(placement)); + figure.entry("objectId").or_insert_with(|| json!(id)); + figures.insert(id, Value::Object(figure)); + } + } + } + if !figures.is_empty() { + result.insert("figures".into(), Value::Object(figures)); + } + Ok(result) +} + +/// Flatten a known union arm, retaining styles, suggestions, source indices and +/// future metadata fields. Unrecognized union arms are retained as raw markers. +fn payload(value: &Value, key: &str, kind: &str) -> Result, GwsError> { + let mut outer = object(value)?; + let mut inner = object( + &outer + .remove(key) + .ok_or_else(|| invalid_content("missing element"))?, + )?; + inner.extend(outer); + inner.insert("type".into(), json!(kind)); + Ok(inner) +} + +fn unknown(value: &Value) -> Value { + let mut marker = json!({"type": "unknown", "data": value}); + for key in ["startIndex", "endIndex"] { + if let Some(index) = value.get(key) { + marker[key] = index.clone(); + } + } + marker +} + +fn element(value: &Value) -> Result { + for (key, kind) in [ + ("textRun", "text"), + ("inlineObjectElement", "figure"), + ("footnoteReference", "footnoteReference"), + ("horizontalRule", "horizontalRule"), + ("pageBreak", "pageBreak"), + ("columnBreak", "columnBreak"), + ("equation", "equation"), + ("autoText", "autoText"), + ] { + if value.get(key).is_some() { + let mut result = payload(value, key, kind)?; + if key == "textRun" { + let text = result + .remove("content") + .filter(Value::is_string) + .ok_or_else(|| invalid_content("textRun missing content"))?; + result.insert("text".into(), text); + } else if key == "inlineObjectElement" { + if let Some(id) = result.remove("inlineObjectId") { + result.insert("objectId".into(), id); + } + } + return Ok(Value::Object(result)); + } + } + Ok(unknown(value)) +} + +fn blocks( + content: &[Value], + tab_id: &Value, + path: &str, + outline: &mut Vec, +) -> Result { + content + .iter() + .enumerate() + .map(|(i, block)| { + let path = format!("{path}/{i}"); + if let Some(paragraph) = block.get("paragraph") { + let mut result = payload(block, "paragraph", "paragraph")?; + let elements = array(paragraph, "elements")? + .iter() + .map(element) + .collect::, _>>()?; + let text: String = elements + .iter() + .filter_map(|e| e.get("text").and_then(Value::as_str)) + .collect(); + result.insert("text".into(), json!(text)); + result.insert("elements".into(), json!(elements)); + if let Some(style) = paragraph.get("paragraphStyle") { + if let Some(level) = + style + .get("namedStyleType") + .and_then(Value::as_str) + .filter(|s| { + matches!( + *s, + "TITLE" + | "SUBTITLE" + | "HEADING_1" + | "HEADING_2" + | "HEADING_3" + | "HEADING_4" + | "HEADING_5" + | "HEADING_6" + ) + }) + { + let mut heading = + json!({"tabId": tab_id, "level": level, "text": text, "path": path}); + for key in ["startIndex", "endIndex"] { + if let Some(value) = block.get(key) { + heading[key] = value.clone(); + } + } + if let Some(id) = style.get("headingId") { + heading["headingId"] = id.clone(); + } + outline.push(heading); + } + } + Ok(Value::Object(result)) + } else if let Some(table) = block.get("table") { + let mut result = payload(block, "table", "table")?; + let rows = array(table, "tableRows")? + .iter() + .enumerate() + .map(|(r, row)| { + let mut normalized = object(row)?; + let cells = array(row, "tableCells")? + .iter() + .enumerate() + .map(|(c, cell)| { + let mut normalized = object(cell)?; + normalized.insert( + "blocks".into(), + blocks( + array(cell, "content")?, + tab_id, + &format!("{path}/rows/{r}/cells/{c}/blocks"), + outline, + )?, + ); + normalized.remove("content"); + Ok(Value::Object(normalized)) + }) + .collect::, GwsError>>()?; + normalized.remove("tableCells"); + normalized.insert("cells".into(), json!(cells)); + Ok(Value::Object(normalized)) + }) + .collect::, GwsError>>()?; + if let Some(count) = result.remove("rows") { + result.insert("rowCount".into(), count); + } + result.remove("tableRows"); + result.insert("rows".into(), json!(rows)); + Ok(Value::Object(result)) + } else if let Some(toc) = block.get("tableOfContents") { + let mut result = payload(block, "tableOfContents", "tableOfContents")?; + result.insert( + "blocks".into(), + blocks( + array(toc, "content")?, + tab_id, + &format!("{path}/blocks"), + outline, + )?, + ); + result.remove("content"); + Ok(Value::Object(result)) + } else if block.get("sectionBreak").is_some() { + payload(block, "sectionBreak", "sectionBreak").map(Value::Object) + } else { + Ok(unknown(block)) + } + }) + .collect::, _>>() + .map(Value::Array) +} diff --git a/crates/google-workspace-cli/src/helpers/docs/read_tests.rs b/crates/google-workspace-cli/src/helpers/docs/read_tests.rs new file mode 100644 index 000000000..0debe7083 --- /dev/null +++ b/crates/google-workspace-cli/src/helpers/docs/read_tests.rs @@ -0,0 +1,593 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use crate::commands::build_cli; +use crate::discovery::RestDescription; +use crate::error::GwsError; +use crate::helpers::modelarmor::{SanitizeConfig, SanitizeMode}; +use serde_json::{json, Value}; + +use super::read; + +fn discovery() -> RestDescription { + serde_json::from_value(serde_json::json!({ + "name": "docs", "version": "v1", "rootUrl": "https://docs.example.invalid/", + "servicePath": "v1/", + "resources": {"documents": {"methods": {"get": { + "id": "docs.documents.get", "httpMethod": "GET", + "path": "documents/{documentId}", "parameterOrder": ["documentId"], + "parameters": {"documentId": {"type": "string", "location": "path", "required": true}}, + "scopes": ["https://www.googleapis.com/auth/documents.readonly"] + }}}} + })) + .unwrap() +} + +#[test] +fn registers_read_alongside_write_with_required_document() { + let cli = build_cli(&discovery()); + assert!(cli.find_subcommand("+write").is_some()); + assert!( + cli.find_subcommand("+read").is_some(), + "missing structured reader" + ); + let error = cli + .clone() + .try_get_matches_from(["gws", "+read"]) + .unwrap_err(); + assert_eq!( + error.kind(), + clap::error::ErrorKind::MissingRequiredArgument + ); + assert!(cli + .try_get_matches_from([ + "gws", + "+read", + "--document", + "synthetic", + "--params", + "{}", + "--format", + "yaml", + "--dry-run" + ]) + .is_ok()); +} + +fn matches(args: &[&str]) -> clap::ArgMatches { + build_cli(&discovery()).try_get_matches_from(args).unwrap() +} + +fn paragraph(text: &str) -> Value { + json!({"startIndex": 1, "endIndex": 5, "paragraph": { + "elements": [{"startIndex": 1, "endIndex": 5, "textRun": {"content": text}}] + }}) +} + +fn legacy() -> Value { + json!({ + "documentId": "synthetic", "title": "Example", "revisionId": "rev-1", + "suggestionsViewMode": "SUGGESTIONS_INLINE", + "body": {"content": [paragraph("Hi😀")] } + }) +} + +#[test] +fn request_requires_full_inline_tabs_and_preserves_other_params() { + let params = + read::build_params("synthetic", Some(r#"{"fields":"*","prettyPrint":false}"#)).unwrap(); + assert_eq!( + params, + json!({ + "documentId": "synthetic", "fields": "*", "prettyPrint": false, + "includeTabsContent": true, "suggestionsViewMode": "SUGGESTIONS_INLINE" + }) + ); + assert_eq!( + read::build_params("id", None).unwrap()["includeTabsContent"], + true + ); + assert!(read::build_params("id", Some( + r#"{"includeTabsContent":true,"suggestionsViewMode":"SUGGESTIONS_INLINE","documentId":"id"}"# + )).is_ok()); +} + +#[test] +fn request_rejects_partial_masks_lossy_views_and_parameter_bypasses() { + for params in [ + r#"{"fields":"title"}"#, + r#"{"fields":"tabs(documentTab/body/content)"}"#, + r#"{"fields":""}"#, + r#"{"fields":null}"#, + r#"{"fields": ["*"]}"#, + r#"{"includeTabsContent":false}"#, + r#"{"includeTabsContent":"true"}"#, + r#"{"suggestionsViewMode":"PREVIEW_WITHOUT_SUGGESTIONS"}"#, + r#"{"suggestionsViewMode":"DEFAULT_FOR_CURRENT_ACCESS"}"#, + r#"{"documentId":"other"}"#, + r#"{"alt":"media"}"#, + r#"{"$fields":"title"}"#, + "[]", + "null", + "{", + ] { + assert!(read::build_params("id", Some(params)).is_err(), "{params}"); + } + for id in [ + "", + "../../secret", + "id?fields=title", + "id#fragment", + "id\n", + "%2e%2e", + ] { + assert!(read::build_params(id, None).is_err(), "{id:?}"); + } +} + +#[test] +fn legacy_body_keeps_source_revision_text_and_utf16_indices() { + let output = read::normalize(&legacy()).unwrap(); + assert_eq!(output["documentId"], "synthetic"); + assert_eq!(output["revisionId"], "rev-1"); + assert_eq!(output["suggestionsViewMode"], "SUGGESTIONS_INLINE"); + assert_eq!(output["source"], "legacyBody"); + assert_eq!(output["tabs"][0]["tabId"], Value::Null); + let block = &output["tabs"][0]["blocks"][0]; + assert_eq!(block["type"], "paragraph"); + assert_eq!(block["text"], "Hi😀"); + assert_eq!(block["endIndex"], 5); + assert_eq!(block["elements"][0]["endIndex"], 5); + assert_eq!(block["elements"][0]["text"], "Hi😀"); +} + +#[test] +fn recursively_reads_tabs_and_child_tabs_in_api_order_without_duplicate_legacy_body() { + let mut input = legacy(); + input["tabs"] = json!([ + {"tabProperties": {"tabId": "a", "title": "First", "index": 0}, + "documentTab": {"body": {"content": [paragraph("first")]}}, + "childTabs": [{"tabProperties": {"tabId": "b", "title": "Child", "parentTabId": "a"}, + "documentTab": {"body": {"content": [paragraph("child")]}}, + "childTabs": [{"tabProperties": {"tabId": "c", "title": "Grandchild"}, + "documentTab": {"body": {"content": [paragraph("grandchild")]}}}]}]}, + {"tabProperties": {"tabId": "d", "title": "Last", "index": 1}, + "documentTab": {"body": {"content": [paragraph("last")]}}} + ]); + let output = read::normalize(&input).unwrap(); + assert_eq!(output["source"], "tabs"); + assert_eq!(output["tabs"].as_array().unwrap().len(), 2); + assert_eq!(output["tabs"][0]["blocks"][0]["text"], "first"); + assert_eq!(output["tabs"][0]["childTabs"][0]["parentTabId"], "a"); + assert_eq!( + output["tabs"][0]["childTabs"][0]["childTabs"][0]["parentTabId"], + "b" + ); + assert_eq!(output["tabs"][1]["tabId"], "d"); + input["tabs"] = json!([]); + assert_eq!(read::normalize(&input).unwrap()["source"], "legacyBody"); +} + +#[test] +fn preserves_styled_link_runs_and_inline_suggestion_metadata_in_outline_order() { + let input = json!({"documentId": "id", "title": "Styled", "body": {"content": [{ + "startIndex": 1, "endIndex": 9, "paragraph": { + "paragraphStyle": {"namedStyleType": "HEADING_2", "headingId": "h1"}, + "suggestedParagraphStyleChanges": {"s1": {"paragraphStyle": {"namedStyleType": "HEADING_1"}}}, + "elements": [ + {"startIndex": 1, "endIndex": 5, "textRun": { + "content": "Look", "textStyle": {"bold": true, "link": {"url": "https://example.invalid"}}, + "suggestedInsertionIds": ["s1"], + "suggestedTextStyleChanges": {"s2": {"textStyle": {"italic": true}}}}}, + {"startIndex": 5, "endIndex": 9, "textRun": { + "content": "here", "textStyle": {"italic": true, "link": {"heading": {"id": "h2", "tabId": "t2"}}}, + "suggestedDeletionIds": ["s3"]}} + ] + } + }, {"startIndex": 9, "endIndex": 14, "paragraph": { + "paragraphStyle": {"namedStyleType": "TITLE"}, "elements": [] + }}]}}); + let output = read::normalize(&input).unwrap(); + let block = &output["tabs"][0]["blocks"][0]; + assert_eq!(block["text"], "Lookhere"); + assert_eq!(block["paragraphStyle"]["headingId"], "h1"); + assert!(block["suggestedParagraphStyleChanges"]["s1"].is_object()); + assert_eq!(block["elements"][0]["textStyle"]["bold"], true); + assert_eq!( + block["elements"][0]["textStyle"]["link"]["url"], + "https://example.invalid" + ); + assert_eq!(block["elements"][0]["suggestedInsertionIds"], json!(["s1"])); + assert_eq!(block["elements"][1]["suggestedDeletionIds"], json!(["s3"])); + assert_eq!( + block["elements"][1]["textStyle"]["link"]["heading"]["tabId"], + "t2" + ); + assert!(block["elements"][0]["suggestedTextStyleChanges"]["s2"].is_object()); + assert_eq!( + output["outline"][0], + json!({ + "tabId": null, "level": "HEADING_2", "headingId": "h1", "text": "Lookhere", + "startIndex": 1, "endIndex": 9, "path": "/tabs/0/blocks/0" + }) + ); + assert_eq!(output["outline"][1]["level"], "TITLE"); +} + +#[test] +fn nested_tables_keep_row_cell_order_indices_styles_and_suggestions() { + let input = json!({"documentId": "id", "body": {"content": [{ + "startIndex": 10, "endIndex": 30, "table": {"rows": 1, "columns": 2, + "tableRows": [{"startIndex": 11, "endIndex": 29, "tableCells": [ + {"startIndex": 12, "endIndex": 25, "tableCellStyle": {"rowSpan": 1, "columnSpan": 1}, + "suggestedInsertionIds": ["cell-s"], "content": [ + paragraph("cell"), + {"startIndex": 17, "endIndex": 24, "table": {"rows": 1, "columns": 1, + "tableRows": [{"tableCells": [{"content": [paragraph("nested")]}]}]}} + ]}, + {"content": [paragraph("second")]} + ]}] + } + }, paragraph("after")]}}); + let output = read::normalize(&input).unwrap(); + let table = &output["tabs"][0]["blocks"][0]; + assert_eq!(table["type"], "table"); + assert_eq!(table["startIndex"], 10); + assert_eq!(table["rowCount"], 1); + assert_eq!(table["columns"], 2); + assert_eq!(table["rows"][0]["endIndex"], 29); + let cell = &table["rows"][0]["cells"][0]; + assert_eq!(cell["startIndex"], 12); + assert_eq!(cell["suggestedInsertionIds"], json!(["cell-s"])); + assert_eq!(cell["tableCellStyle"]["columnSpan"], 1); + assert_eq!(cell["blocks"][0]["text"], "cell"); + assert_eq!( + cell["blocks"][1]["rows"][0]["cells"][0]["blocks"][0]["text"], + "nested" + ); + assert_eq!(table["rows"][0]["cells"][1]["blocks"][0]["text"], "second"); + assert_eq!(output["tabs"][0]["blocks"][1]["text"], "after"); +} + +#[test] +fn figures_keep_references_and_metadata_even_without_content_uri() { + let mut input = legacy(); + input["body"]["content"][0]["paragraph"]["elements"] = json!([ + {"startIndex": 1, "endIndex": 2, "inlineObjectElement": { + "inlineObjectId": "image", "suggestedInsertionIds": ["s1"], "textStyle": {"baselineOffset": "SUPERSCRIPT"}}}, + {"startIndex": 2, "endIndex": 3, "inlineObjectElement": {"inlineObjectId": "missing"}} + ]); + input["body"]["content"][0]["paragraph"]["positionedObjectIds"] = json!(["drawing"]); + input["inlineObjects"] = json!({"image": { + "objectId": "image", "inlineObjectProperties": {"embeddedObject": { + "title": "Alt title", "description": "Alt text", "size": {"width": {"magnitude": 42, "unit": "PT"}}, + "imageProperties": {"sourceUri": "https://example.invalid/image.png"} + }}, "suggestedDeletionIds": ["s2"] + }}); + input["positionedObjects"] = json!({"drawing": { + "objectId": "drawing", "positionedObjectProperties": {"embeddedObject": {"embeddedDrawingProperties": {}}} + }}); + let output = read::normalize(&input).unwrap(); + let tab = &output["tabs"][0]; + assert_eq!(tab["blocks"][0]["elements"][0]["type"], "figure"); + assert_eq!(tab["blocks"][0]["elements"][0]["objectId"], "image"); + assert_eq!( + tab["blocks"][0]["elements"][0]["suggestedInsertionIds"], + json!(["s1"]) + ); + assert_eq!(tab["blocks"][0]["elements"][1]["objectId"], "missing"); + assert_eq!(tab["blocks"][0]["positionedObjectIds"], json!(["drawing"])); + assert_eq!(tab["figures"]["image"]["type"], "image"); + assert_eq!( + tab["figures"]["image"]["embeddedObject"]["description"], + "Alt text" + ); + assert!(tab["figures"]["image"]["embeddedObject"]["imageProperties"] + .get("contentUri") + .is_none()); + assert_eq!( + tab["figures"]["image"]["suggestedDeletionIds"], + json!(["s2"]) + ); + assert_eq!(tab["figures"]["drawing"]["placement"], "positioned"); +} + +#[test] +fn preserves_reference_markers_segments_unknown_blocks_and_unknown_inline_elements() { + let mut input = legacy(); + input["body"]["content"] = json!([ + {"endIndex": 1, "sectionBreak": {"sectionStyle": {"columnSeparatorStyle": "NONE"}}}, + {"startIndex": 1, "endIndex": 4, "paragraph": {"elements": [ + {"startIndex": 1, "endIndex": 2, "footnoteReference": {"footnoteId": "f1", "footnoteNumber": "1"}}, + {"startIndex": 2, "endIndex": 3, "person": {"personId": "p1"}}, + {"startIndex": 3, "endIndex": 4, "futureInline": {"label": "unrecognized"}} + ]}}, + {"startIndex": 4, "endIndex": 8, "futureBlock": {"content": "keep me"}}, + {"tableOfContents": {"content": [paragraph("toc")]}} + ]); + input["headers"] = json!({"h1": {"headerId": "h1", "content": [paragraph("header")]}}); + input["footers"] = json!({"f2": {"footerId": "f2", "content": [paragraph("footer")]}}); + input["footnotes"] = json!({"f1": {"footnoteId": "f1", "content": [paragraph("note")]}}); + input["namedStyles"] = json!({"styles": [{"namedStyleType": "NORMAL_TEXT"}]}); + let output = read::normalize(&input).unwrap(); + let tab = &output["tabs"][0]; + assert_eq!(tab["blocks"][0]["type"], "sectionBreak"); + assert!(tab["blocks"][0].get("startIndex").is_none()); + assert_eq!(tab["blocks"][1]["elements"][0]["type"], "footnoteReference"); + assert_eq!(tab["blocks"][1]["elements"][0]["footnoteId"], "f1"); + assert_eq!(tab["blocks"][1]["elements"][1]["type"], "unknown"); + assert_eq!( + tab["blocks"][1]["elements"][1]["data"]["person"]["personId"], + "p1" + ); + assert_eq!( + tab["blocks"][1]["elements"][2]["data"]["futureInline"]["label"], + "unrecognized" + ); + assert_eq!(tab["blocks"][2]["type"], "unknown"); + assert_eq!(tab["blocks"][2]["startIndex"], 4); + assert_eq!( + tab["blocks"][2]["data"]["futureBlock"]["content"], + "keep me" + ); + assert_eq!(tab["blocks"][3]["blocks"][0]["text"], "toc"); + assert_eq!(tab["headers"]["h1"]["blocks"][0]["text"], "header"); + assert_eq!(tab["footers"]["f2"]["blocks"][0]["text"], "footer"); + assert_eq!(tab["footnotes"]["f1"]["blocks"][0]["text"], "note"); + assert_eq!( + tab["namedStyles"]["styles"][0]["namedStyleType"], + "NORMAL_TEXT" + ); +} + +#[test] +fn rejects_missing_content_instead_of_claiming_empty_document() { + for input in [ + json!({"documentId": "id", "title": "metadata only"}), + json!({"body": {}}), + json!({"tabs": [{"tabProperties": {"tabId": "t"}, "documentTab": {}}]}), + json!({"tabs": [{"documentTab": {"body": {"content": "not an array"}}}]}), + json!({"tabs": "not an array", "body": {"content": []}}), + json!({"body": {"content": [{"table": {"rows": 1}}]}}), + ] { + assert!(read::normalize(&input).is_err(), "{input}"); + } + assert!(read::normalize(&json!({"body": {"content": []}})).is_ok()); +} + +#[test] +fn preserves_unknown_tab_and_sanitization_annotation() { + let output = read::normalize(&json!({ + "documentId": "id", "_sanitization": {"filterMatchState": "NO_MATCH_FOUND"}, + "tabs": [{"tabProperties": {"tabId": "future", "title": "Future"}, + "futureTab": {"content": "opaque"}}] + })) + .unwrap(); + assert_eq!( + output["_sanitization"]["filterMatchState"], + "NO_MATCH_FOUND" + ); + assert_eq!(output["tabs"][0]["type"], "unknown"); + assert_eq!(output["tabs"][0]["data"]["futureTab"]["content"], "opaque"); +} + +#[tokio::test] +async fn dry_run_uses_executor_plan_without_polling_auth_or_sanitize() { + let args = matches(&[ + "gws", + "+read", + "--document", + "a/b c", + "--dry-run", + "--params", + r#"{"prettyPrint":false}"#, + ]); + let result = read::run( + &discovery(), + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig { + template: Some("never-call".into()), + mode: SanitizeMode::Block, + }, + async { panic!("dry-run must not poll authentication") }, + ) + .await + .unwrap(); + let output: Value = serde_json::from_str(&result).unwrap(); + assert_eq!(output["dry_run"], true); + assert_eq!(output["method"], "GET"); + assert_eq!( + output["url"], + "https://docs.example.invalid/v1/documents/a%2Fb%20c" + ); + let query = output["query_params"].as_array().unwrap(); + assert!(query.contains(&json!(["includeTabsContent", "true"]))); + assert!(query.contains(&json!(["suggestionsViewMode", "SUGGESTIONS_INLINE"]))); + assert!(query.contains(&json!(["prettyPrint", "false"]))); + assert!(output.get("tabs").is_none()); +} + +#[tokio::test] +async fn rejects_partial_mask_before_polling_authentication() { + let args = matches(&[ + "gws", + "+read", + "--document", + "id", + "--params", + r#"{"fields":"title"}"#, + ]); + let result = read::run( + &discovery(), + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { panic!("invalid request must fail before authentication") }, + ) + .await; + assert!(matches!(result, Err(GwsError::Validation(_)))); +} + +#[tokio::test] +async fn propagates_auth_and_discovery_failures() { + let args = matches(&["gws", "+read", "--document", "id"]); + let result = read::run( + &discovery(), + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { Err(GwsError::Auth("synthetic failure".into())) }, + ) + .await; + assert!(matches!(result, Err(GwsError::Auth(_)))); + let result = read::run( + &RestDescription::default(), + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { panic!("missing method must fail before authentication") }, + ) + .await; + assert!(matches!(result, Err(GwsError::Discovery(_)))); +} + +// The transport is the only fake: real executor, request building, capture, +// normalization and formatting run against a loopback server with synthetic auth. +async fn serve(status: &str, body: String) -> (RestDescription, tokio::task::JoinHandle) { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let mut doc = discovery(); + doc.root_url = format!("http://{}/", listener.local_addr().unwrap()); + let response = format!( + "HTTP/1.1 {status}\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", + body.len() + ); + let task = tokio::spawn(async move { + let (mut stream, _) = listener.accept().await.unwrap(); + let mut request = Vec::new(); + loop { + let mut buffer = [0; 1024]; + let len = stream.read(&mut buffer).await.unwrap(); + assert_ne!(len, 0); + request.extend_from_slice(&buffer[..len]); + if request.windows(4).any(|w| w == b"\r\n\r\n") { + break; + } + } + stream.write_all(response.as_bytes()).await.unwrap(); + String::from_utf8(request).unwrap() + }); + (doc, task) +} + +// Stop the existing executor's quota lookup before it can read host config/ADC. +struct SyntheticQuota(Option); +impl SyntheticQuota { + fn new() -> Self { + let old = std::env::var_os("GOOGLE_WORKSPACE_PROJECT_ID"); + std::env::set_var("GOOGLE_WORKSPACE_PROJECT_ID", "synthetic-project"); + Self(old) + } +} +impl Drop for SyntheticQuota { + fn drop(&mut self) { + if let Some(old) = &self.0 { + std::env::set_var("GOOGLE_WORKSPACE_PROJECT_ID", old); + } else { + std::env::remove_var("GOOGLE_WORKSPACE_PROJECT_ID"); + } + } +} + +#[tokio::test] +#[serial_test::serial] +async fn executor_fetches_full_content_with_auth_and_honors_all_global_formats() { + let _quota = SyntheticQuota::new(); + for format in ["json", "yaml", "table", "csv"] { + let (doc, request) = serve("200 OK", legacy().to_string()).await; + let args = matches(&[ + "gws", + "+read", + "--document", + "synthetic", + "--format", + format, + ]); + let rendered = read::run( + &doc, + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { Ok("synthetic-token".into()) }, + ) + .await + .unwrap(); + let request = request.await.unwrap(); + assert!(request.starts_with("GET /v1/documents/synthetic?")); + assert!(request.contains("includeTabsContent=true")); + assert!(request.contains("suggestionsViewMode=SUGGESTIONS_INLINE")); + assert!(request.contains("authorization: Bearer synthetic-token\r\n")); + match format { + "json" => assert_eq!( + serde_json::from_str::(&rendered).unwrap()["tabs"][0]["blocks"][0]["text"], + "Hi😀" + ), + "yaml" => assert!(rendered.contains("documentId: \"synthetic\"")), + "table" => assert!(rendered.contains("─") && rendered.contains("blocks")), + "csv" => assert!( + rendered.lines().next().unwrap().contains("blocks,") + && rendered.contains("\"\"text\"\"") + ), + _ => unreachable!(), + } + } +} + +#[tokio::test] +#[serial_test::serial] +async fn executor_propagates_server_errors_and_rejects_non_document_responses() { + let _quota = SyntheticQuota::new(); + for status in ["403 Forbidden", "500 Internal Server Error"] { + let (doc, request) = serve( + status, + json!({"error": {"message": "synthetic denied"}}).to_string(), + ) + .await; + let args = matches(&["gws", "+read", "--document", "synthetic"]); + let result = read::run( + &doc, + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { Ok("synthetic-token".into()) }, + ) + .await; + match result.unwrap_err() { + GwsError::Api { code, message, .. } => { + assert_eq!(code, if status.starts_with("403") { 403 } else { 500 }); + assert_eq!(message, "synthetic denied"); + } + other => panic!("unexpected error: {other:?}"), + } + request.await.unwrap(); + } + for body in [r#"{"documentId":"id","title":"partial"}"#, "invalid JSON"] { + let (doc, request) = serve("200 OK", body.into()).await; + let args = matches(&["gws", "+read", "--document", "synthetic"]); + assert!(read::run( + &doc, + args.subcommand_matches("+read").unwrap(), + &SanitizeConfig::default(), + async { Ok("synthetic-token".into()) } + ) + .await + .is_err()); + request.await.unwrap(); + } +} diff --git a/docs/docs-read.md b/docs/docs-read.md new file mode 100644 index 000000000..d741684a6 --- /dev/null +++ b/docs/docs-read.md @@ -0,0 +1,84 @@ +# Read structured Google Docs content + +```bash +gws docs +read --document DOC_ID +gws docs +read --document DOC_ID --format yaml +gws docs +read --document DOC_ID --dry-run +gws docs +read --document DOC_ID | jq '.outline' +``` + +`+read` translates the Docs API's nested structural elements into ordered +blocks. It uses the existing authentication, request executor, Model Armor +sanitization, and output formatters. `+write` is unchanged. + +The request uses `includeTabsContent=true` and +`suggestionsViewMode=SUGGESTIONS_INLINE`. Pass other API options through +`--params`. To prevent omitted content from looking like an empty document, +`fields` must be absent or exactly `"*"`. `$fields`, other tab/suggestion views, +conflicting document IDs, and non-JSON `alt` responses are rejected before +authentication. Dry-run prints the executor's request plan without acquiring +credentials, fetching document content, or invoking Model Armor. + +## Output + +- The root retains `documentId`, `title`, `revisionId`, `suggestionsViewMode`, + and `_sanitization` when returned. No revision is invented when absent. +- `tabs` and recursive `childTabs` preserve API order, IDs, titles, and parents. + Each tab has ordered `blocks`. `source: "legacyBody"` means the response had + no populated tabs; that fallback cannot confirm coverage of other tabs. +- Paragraph blocks have `text`, ordered `elements`, and `paragraphStyle`. + Text elements retain separate runs, text styles, links (including tab-aware + internal links), and suggested insertion/deletion/style changes. Other + returned paragraph metadata, such as bullets and positioned object IDs, + stays on the block. +- Table blocks have `rowCount`, `columns`, and + `rows[].cells[].blocks`, including nested tables. Row/cell styles and + suggestion metadata are retained. +- Each tab's `figures` map contains inline and positioned object metadata. + `embeddedObject` retains alt text, dimensions, and image properties when + available. Figure elements reference `objectId`. Missing metadata or + `contentUri` does not remove the reference. +- Headers, footers, and footnotes remain separate maps with their own `blocks`. + Footnote references and structural markers remain in content order. + Unsupported elements use `type: "unknown"` with their original `data`. +- `outline` contains titles, subtitles, and headings with text, style level, + tab ID, heading ID when present, source indices, and a JSON Pointer `path` + to the normalized paragraph, including headings in tables and segments. + +For example, select a heading by its returned ID: + +```bash +gws docs +read --document DOC_ID | + jq '.. | objects | select(.paragraphStyle?.headingId? == "HEADING_ID")' +``` + +To select the blocks between two top-level headings in a particular tab: + +```bash +gws docs +read --document DOC_ID | + jq --arg tab TAB_ID --argjson start 10 --argjson end 50 \ + '.. | objects | select(.tabId? == $tab and has("blocks")) | + .blocks[] | select(.startIndex >= $start and .startIndex < $end)' +``` + +Use indices actually returned for that tab. `startIndex` and `endIndex` are +UTF-16 offsets in the API's tab/segment, **not** byte or character offsets into +the extracted `text`. The text convenience field concatenates text runs only; +figures and other markers remain in `elements`. + +## Limits + +This is a structured content view, not a visual layout renderer or a lossless +API round trip. It does not resolve inherited styles, render drawings or +equations, download images, or accept/reject suggestions. Proposed deletions +remain inline. Image URIs are included only when returned and may expire. +Use raw `gws docs documents get` for other views or partial field masks. + +JSON is the default and retains the entire normalized tree. YAML uses the +existing serializer. Table and CSV use the existing formatter's first +nonempty-array summary (typically the outline, otherwise tabs); table cells +can be truncated. Use JSON for complete downstream processing. + +Usage and limitations also live in the command's help, the source consumed by +`gws generate-skills`. Generated skill files are maintained by the repository's +Generate Skills workflow. From 78e6e571d2ca377a2ad48f32d411c723ae8fc630 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:14:37 -0300 Subject: [PATCH 05/19] fix(docs): preserve automatic text subtypes --- .../src/helpers/docs/read.rs | 5 ++++ .../src/helpers/docs/read_tests.rs | 25 +++++++++++++++++++ docs/docs-read.md | 3 ++- 3 files changed, 32 insertions(+), 1 deletion(-) diff --git a/crates/google-workspace-cli/src/helpers/docs/read.rs b/crates/google-workspace-cli/src/helpers/docs/read.rs index db9bcc9b7..4e2e8625c 100644 --- a/crates/google-workspace-cli/src/helpers/docs/read.rs +++ b/crates/google-workspace-cli/src/helpers/docs/read.rs @@ -413,6 +413,11 @@ fn element(value: &Value) -> Result { if let Some(id) = result.remove("inlineObjectId") { result.insert("objectId".into(), id); } + } else if key == "autoText" { + // The source's `type` is content, distinct from our union tag. + if let Some(subtype) = value[key].get("type") { + result.insert("autoTextType".into(), subtype.clone()); + } } return Ok(Value::Object(result)); } diff --git a/crates/google-workspace-cli/src/helpers/docs/read_tests.rs b/crates/google-workspace-cli/src/helpers/docs/read_tests.rs index 0debe7083..95a545ffb 100644 --- a/crates/google-workspace-cli/src/helpers/docs/read_tests.rs +++ b/crates/google-workspace-cli/src/helpers/docs/read_tests.rs @@ -83,6 +83,31 @@ fn legacy() -> Value { }) } +#[test] +fn auto_text_preserves_page_number_and_count_with_indices_and_styles() { + let mut outputs = Vec::new(); + for subtype in ["PAGE_NUMBER", "PAGE_COUNT"] { + let mut input = legacy(); + input["body"]["content"][0]["paragraph"]["elements"] = json!([{ + "startIndex": 0, "endIndex": 1, + "autoText": {"type": subtype, "textStyle": {"bold": true}, + "suggestedInsertionIds": ["s1"]} + }]); + let output = read::normalize(&input).unwrap(); + let element = &output["tabs"][0]["blocks"][0]["elements"][0]; + assert_eq!( + element, + &json!({ + "type": "autoText", "autoTextType": subtype, + "startIndex": 0, "endIndex": 1, + "textStyle": {"bold": true}, "suggestedInsertionIds": ["s1"] + }) + ); + outputs.push(output); + } + assert_ne!(outputs[0], outputs[1]); +} + #[test] fn request_requires_full_inline_tabs_and_preserves_other_params() { let params = diff --git a/docs/docs-read.md b/docs/docs-read.md index d741684a6..b95b8a66c 100644 --- a/docs/docs-read.md +++ b/docs/docs-read.md @@ -30,7 +30,8 @@ credentials, fetching document content, or invoking Model Armor. Text elements retain separate runs, text styles, links (including tab-aware internal links), and suggested insertion/deletion/style changes. Other returned paragraph metadata, such as bullets and positioned object IDs, - stays on the block. + stays on the block. Automatic-text markers use `type: "autoText"` and retain + their source subtype (`PAGE_NUMBER` or `PAGE_COUNT`) as `autoTextType`. - Table blocks have `rowCount`, `columns`, and `rows[].cells[].blocks`, including nested tables. Row/cell styles and suggestion metadata are retained. From a362bce447fd270f9eb2ca3cf921e44f3228ee11 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:17:17 -0300 Subject: [PATCH 06/19] fix(formatter): separate empty YAML collection values --- .changeset/yaml-empty-collections.md | 7 ++++ crates/google-workspace-cli/src/formatter.rs | 23 +++++++++++- .../src/helpers/docs/read_tests.rs | 37 ++++++++++++++++++- 3 files changed, 64 insertions(+), 3 deletions(-) create mode 100644 .changeset/yaml-empty-collections.md diff --git a/.changeset/yaml-empty-collections.md b/.changeset/yaml-empty-collections.md new file mode 100644 index 000000000..2ea985ea5 --- /dev/null +++ b/.changeset/yaml-empty-collections.md @@ -0,0 +1,7 @@ +--- +"@googleworkspace/cli": patch +--- + +Fix YAML mapping values containing empty arrays or objects by separating their +inline collection syntax from the mapping colon. This also fixes structured +Docs reader output with empty outlines, child tabs, or style maps. diff --git a/crates/google-workspace-cli/src/formatter.rs b/crates/google-workspace-cli/src/formatter.rs index 08d4d287a..57ae406af 100644 --- a/crates/google-workspace-cli/src/formatter.rs +++ b/crates/google-workspace-cli/src/formatter.rs @@ -319,7 +319,10 @@ fn json_to_yaml(value: &Value, indent: usize) -> String { match val { Value::Object(_) | Value::Array(_) => { let val_str = json_to_yaml(val, indent + 1); - let _ = write!(out, "\n{prefix}{key}:{val_str}"); + // Empty collections use inline flow syntax and need a + // space after the colon; block collections start a line. + let separator = if val_str.starts_with('\n') { "" } else { " " }; + let _ = write!(out, "\n{prefix}{key}:{separator}{val_str}"); } _ => { let val_str = json_to_yaml(val, indent); @@ -637,6 +640,24 @@ mod tests { assert!(output.contains("count: 42")); } + #[test] + fn test_format_yaml_empty_collections_as_mapping_values() { + let value = json!({"array": [], "object": {}, "tail": true}); + assert_eq!( + format_value(&value, &OutputFormat::Yaml), + "\narray: []\nobject: {}\ntail: true" + ); + } + + #[test] + fn test_format_yaml_empty_collections_nested_in_sequences() { + let value = json!({"items": [[], {}, {"array": [], "object": {}}, "tail"]}); + assert_eq!( + format_value(&value, &OutputFormat::Yaml), + "\nitems:\n - []\n - {}\n - \n array: []\n object: {}\n - \"tail\"" + ); + } + #[test] fn test_format_table_empty_array() { let val = json!({"files": []}); diff --git a/crates/google-workspace-cli/src/helpers/docs/read_tests.rs b/crates/google-workspace-cli/src/helpers/docs/read_tests.rs index 95a545ffb..d0152bf9f 100644 --- a/crates/google-workspace-cli/src/helpers/docs/read_tests.rs +++ b/crates/google-workspace-cli/src/helpers/docs/read_tests.rs @@ -537,7 +537,9 @@ impl Drop for SyntheticQuota { async fn executor_fetches_full_content_with_auth_and_honors_all_global_formats() { let _quota = SyntheticQuota::new(); for format in ["json", "yaml", "table", "csv"] { - let (doc, request) = serve("200 OK", legacy().to_string()).await; + let mut input = legacy(); + input["body"]["content"][0]["paragraph"]["elements"][0]["textRun"]["textStyle"] = json!({}); + let (doc, request) = serve("200 OK", input.to_string()).await; let args = matches(&[ "gws", "+read", @@ -564,7 +566,38 @@ async fn executor_fetches_full_content_with_auth_and_honors_all_global_formats() serde_json::from_str::(&rendered).unwrap()["tabs"][0]["blocks"][0]["text"], "Hi😀" ), - "yaml" => assert!(rendered.contains("documentId: \"synthetic\"")), + // Complete consumer-visible YAML, including the empty map and + // arrays that previously lacked a mapping-value separator. + "yaml" => assert_eq!( + rendered, + concat!( + "\ndocumentId: \"synthetic\"", + "\noutline: []", + "\nrevisionId: \"rev-1\"", + "\nsource: \"legacyBody\"", + "\nsuggestionsViewMode: \"SUGGESTIONS_INLINE\"", + "\ntabs:", + "\n - ", + "\n blocks:", + "\n - ", + "\n elements:", + "\n - ", + "\n endIndex: 5", + "\n startIndex: 1", + "\n text: \"Hi😀\"", + "\n textStyle: {}", + "\n type: \"text\"", + "\n endIndex: 5", + "\n startIndex: 1", + "\n text: \"Hi😀\"", + "\n type: \"paragraph\"", + "\n childTabs: []", + "\n parentTabId: null", + "\n tabId: null", + "\n title: \"Example\"", + "\ntitle: \"Example\"" + ) + ), "table" => assert!(rendered.contains("─") && rendered.contains("blocks")), "csv" => assert!( rendered.lines().next().unwrap().contains("blocks,") From d9693807040b14b07104daca8b3cfaa51a618b07 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:03:33 -0300 Subject: [PATCH 07/19] fix(auth): skip authentication for request dry-runs --- .changeset/offline-dry-run.md | 9 + README.md | 16 + .../google-workspace-cli/src/helpers/docs.rs | 15 +- crates/google-workspace-cli/src/main.rs | 31 +- crates/google-workspace-cli/tests/dry_run.rs | 533 ++++++++++++++++++ 5 files changed, 586 insertions(+), 18 deletions(-) create mode 100644 .changeset/offline-dry-run.md create mode 100644 crates/google-workspace-cli/tests/dry_run.rs diff --git a/.changeset/offline-dry-run.md b/.changeset/offline-dry-run.md new file mode 100644 index 000000000..ca86d5669 --- /dev/null +++ b/.changeset/offline-dry-run.md @@ -0,0 +1,9 @@ +--- +"@googleworkspace/cli": patch +--- + +Skip authentication for Discovery-generated API and `docs +write` dry-runs. +Validate and preview requests without accessing the keyring or reading, changing, +or deleting stored credentials and token caches. Dry-runs work offline with a +fresh cached Discovery schema; schema fetching on first use or cache expiry is +unchanged. Real requests retain their existing authentication and error handling. diff --git a/README.md b/README.md index a0df710c8..12544f5eb 100644 --- a/README.md +++ b/README.md @@ -104,6 +104,21 @@ gws schema drive.files.list gws drive files list --params '{"pageSize": 100}' --page-all | jq -r '.files[].name' ``` +Discovery-generated API commands and `gws docs +write` support credential-free +`--dry-run`: they validate inputs and display the request without obtaining a +token, accessing the keyring, reading or changing stored credentials, or sending +the API request. + +```bash +# Preview a Docs append without signing in +gws docs +write --document DOC_ID --text 'Hello, world!' --dry-run +``` + +These previews work offline with a fresh cached Discovery schema (24-hour TTL). +First use or an expired cache can still fetch the schema over the network. +Other helpers may need authenticated reads to prepare their plans; this guarantee +applies to raw API commands and `docs +write`. + ### Fields absent from Discovery Raw API methods with a request body accept `--allow-unknown-fields` alongside @@ -142,6 +157,7 @@ to submit a request. See the Docs [request reference](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents/request#InsertCommentRequest) for preview field requirements. + ## Authentication The CLI supports multiple auth workflows so it works on your laptop, in CI, and on a server. diff --git a/crates/google-workspace-cli/src/helpers/docs.rs b/crates/google-workspace-cli/src/helpers/docs.rs index 6492cb7c9..c42c02feb 100644 --- a/crates/google-workspace-cli/src/helpers/docs.rs +++ b/crates/google-workspace-cli/src/helpers/docs.rs @@ -81,10 +81,15 @@ TIPS: let (params_str, body_str, scopes) = build_write_request(matches, doc)?; let scope_strs: Vec<&str> = scopes.iter().map(|s| s.as_str()).collect(); - let (token, auth_method) = match auth::get_token(&scope_strs).await { - Ok(t) => (Some(t), executor::AuthMethod::OAuth), - Err(_) if matches.get_flag("dry-run") => (None, executor::AuthMethod::None), - Err(e) => return Err(GwsError::Auth(format!("Docs auth failed: {e}"))), + let dry_run = matches.get_flag("dry-run"); + // Skip auth entirely: even failed auth can mutate stored credentials. + let (token, auth_method) = if dry_run { + (None, executor::AuthMethod::None) + } else { + match auth::get_token(&scope_strs).await { + Ok(t) => (Some(t), executor::AuthMethod::OAuth), + Err(e) => return Err(GwsError::Auth(format!("Docs auth failed: {e}"))), + } }; // Method: documents.batchUpdate @@ -111,7 +116,7 @@ TIPS: auth_method, None, None, - matches.get_flag("dry-run"), + dry_run, &pagination, None, &crate::helpers::modelarmor::SanitizeMode::Warn, diff --git a/crates/google-workspace-cli/src/main.rs b/crates/google-workspace-cli/src/main.rs index d634e4730..a0e9241a7 100644 --- a/crates/google-workspace-cli/src/main.rs +++ b/crates/google-workspace-cli/src/main.rs @@ -261,19 +261,24 @@ async fn run() -> Result<(), GwsError> { // to avoid restrictive scopes like gmail.metadata that block query parameters. let scopes: Vec<&str> = select_scope(&method.scopes).into_iter().collect(); - // Authenticate: try OAuth, fail with error if credentials exist but are broken - let (token, auth_method) = match auth::get_token(&scopes).await { - Ok(t) => (Some(t), executor::AuthMethod::OAuth), - Err(e) => { - // If credentials were found but failed (e.g. decryption error, invalid token), - // propagate the error instead of silently falling back to unauthenticated. - // Only fall back to None if no credentials exist at all. - let err_msg = format!("{e:#}"); - // NB: matches the bail!() message in auth::load_credentials_inner - if err_msg.starts_with("No credentials found") { - (None, executor::AuthMethod::None) - } else { - return Err(GwsError::Auth(format!("Authentication failed: {err_msg}"))); + // Dry-runs only need the schema and inputs. Do not load credentials: + // authentication may access the keyring or remove corrupt credential files. + let (token, auth_method) = if dry_run { + (None, executor::AuthMethod::None) + } else { + match auth::get_token(&scopes).await { + Ok(t) => (Some(t), executor::AuthMethod::OAuth), + Err(e) => { + // If credentials were found but failed (e.g. decryption error, invalid token), + // propagate the error instead of silently falling back to unauthenticated. + // Only fall back to None if no credentials exist at all. + let err_msg = format!("{e:#}"); + // NB: matches the bail!() message in auth::load_credentials_inner + if err_msg.starts_with("No credentials found") { + (None, executor::AuthMethod::None) + } else { + return Err(GwsError::Auth(format!("Authentication failed: {err_msg}"))); + } } } }; diff --git a/crates/google-workspace-cli/tests/dry_run.rs b/crates/google-workspace-cli/tests/dry_run.rs new file mode 100644 index 000000000..3f03a3496 --- /dev/null +++ b/crates/google-workspace-cli/tests/dry_run.rs @@ -0,0 +1,533 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use serde_json::{json, Value}; +use std::collections::BTreeMap; +use std::fs; +use std::io::{Read, Write}; +use std::net::TcpListener; +use std::path::{Path, PathBuf}; +use std::process::{Command, Output, Stdio}; +use std::sync::{mpsc, Arc, Mutex}; +use std::thread; +use std::time::{Duration, Instant, SystemTime}; +use tempfile::TempDir; + +const BODY: &str = + r#"{"requests":[{"insertText":{"text":"hello","endOfSegmentLocation":{"segmentId":""}}}]}"#; +const RAW: &[&str] = &[ + "docs", + "documents", + "batchUpdate", + "--params", + r#"{"documentId":"doc /?#","fields":"documentId"}"#, + "--json", + BODY, +]; +const WRITE: &[&str] = &["docs", "+write", "--document", "doc /?#", "--text", "hello"]; + +// Observe every API/proxy connection without contacting Google. Returning an +// error also lets real-request tests verify that API failures stay failures. +struct NetworkTrap { + url: String, + requests: Arc>>, + stop: mpsc::Sender<()>, + worker: Option>, +} + +impl NetworkTrap { + fn new() -> Self { + let listener = TcpListener::bind("127.0.0.1:0").unwrap(); + listener.set_nonblocking(true).unwrap(); + let url = format!("http://{}/", listener.local_addr().unwrap()); + let requests = Arc::new(Mutex::new(Vec::new())); + let captured = Arc::clone(&requests); + let (stop, stopping) = mpsc::channel(); + let worker = thread::spawn(move || loop { + match listener.accept() { + Ok((mut stream, _)) => { + // Record the connection even if the client fails before + // sending HTTP (for example, during TLS/proxy setup). + let mut requests = captured.lock().unwrap(); + requests.push(String::new()); + stream + .set_read_timeout(Some(Duration::from_secs(1))) + .unwrap(); + let mut header = Vec::new(); + let mut buffer = [0; 1024]; + while header.len() < 8192 && !header.windows(4).any(|w| w == b"\r\n\r\n") { + match stream.read(&mut buffer) { + Ok(0) | Err(_) => break, + Ok(size) => header.extend_from_slice(&buffer[..size]), + } + } + *requests.last_mut().unwrap() = String::from_utf8_lossy(&header).into_owned(); + let body = r#"{"error":{"code":403,"message":"Synthetic denial","errors":[{"reason":"forbidden"}]}}"#; + let _ = write!( + stream, + "HTTP/1.1 403 Forbidden\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", + body.len() + ); + } + Err(error) if error.kind() == std::io::ErrorKind::WouldBlock => { + if stopping.recv_timeout(Duration::from_millis(5)) + != Err(mpsc::RecvTimeoutError::Timeout) + { + break; + } + } + Err(error) => panic!("Network trap failed: {error}"), + } + }); + Self { + url, + requests, + stop, + worker: Some(worker), + } + } +} + +impl Drop for NetworkTrap { + fn drop(&mut self) { + let _ = self.stop.send(()); + self.worker.take().unwrap().join().unwrap(); + } +} + +type Snapshot = BTreeMap, SystemTime)>; + +fn snapshot(root: &Path) -> Snapshot { + let mut files = BTreeMap::new(); + for entry in fs::read_dir(root).unwrap() { + let path = entry.unwrap().path(); + if path.is_dir() { + files.extend(snapshot(&path)); + } else { + files.insert( + path.clone(), + ( + fs::read(&path).unwrap(), + fs::metadata(path).unwrap().modified().unwrap(), + ), + ); + } + } + files +} + +struct Fixture { + dir: TempDir, + config: PathBuf, + network: NetworkTrap, +} + +impl Fixture { + fn new(corrupt_credentials: bool) -> Self { + let dir = tempfile::tempdir().unwrap(); + let config = dir.path().join("config"); + fs::create_dir_all(config.join("cache")).unwrap(); + // Stop dotenvy's ancestor search and isolate all credential sources. + fs::write(dir.path().join(".env"), "").unwrap(); + fs::create_dir(dir.path().join("home")).unwrap(); + let fixture = Self { + dir, + config, + network: NetworkTrap::new(), + }; + fixture.write_discovery(&fixture.discovery()); + if corrupt_credentials { + for name in [ + "credentials.enc", + "credentials.json", + ".encryption_key", + "token_cache.json", + "sa_token_cache.json", + ] { + fs::write( + fixture.config.join(name), + format!("corrupt synthetic sentinel: {name}"), + ) + .unwrap(); + } + } + fixture + } + + fn discovery(&self) -> Value { + json!({ + "name": "docs", + "version": "v1", + "rootUrl": self.network.url, + "servicePath": "v1/", + "resources": { + "documents": { + "methods": { + "batchUpdate": { + "id": "docs.documents.batchUpdate", + "httpMethod": "POST", + "path": "documents/{documentId}:batchUpdate", + "parameterOrder": ["documentId"], + "parameters": { + "documentId": {"type": "string", "location": "path", "required": true}, + "fields": {"type": "string", "location": "query"} + }, + "request": {"$ref": "BatchUpdateDocumentRequest"}, + "scopes": ["https://www.googleapis.com/auth/documents"] + } + } + } + }, + "schemas": { + "BatchUpdateDocumentRequest": { + "type": "object", + "required": ["requests"], + "properties": { + "requests": {"type": "array", "items": {"$ref": "Request"}} + } + }, + "Request": { + "type": "object", + "properties": { + "insertText": { + "type": "object", + "properties": { + "text": {"type": "string"}, + "endOfSegmentLocation": { + "type": "object", + "properties": {"segmentId": {"type": "string"}} + } + } + } + } + } + } + }) + } + + fn write_discovery(&self, discovery: &Value) { + // Exercise the real config override, cache filename and freshness check. + fs::write( + self.config.join("cache/docs_v1.json"), + serde_json::to_vec(discovery).unwrap(), + ) + .unwrap(); + } + + fn run(&self, args: &[&str], token: Option<&str>) -> Output { + let mut command = Command::new(env!("CARGO_BIN_EXE_gws")); + command + .args(args) + .current_dir(self.dir.path()) + .env_clear() + .env("HOME", self.dir.path().join("home")) + .env("USERPROFILE", self.dir.path().join("home")) + .env("APPDATA", self.dir.path().join("home")) + .env("XDG_CONFIG_HOME", self.dir.path().join("home")) + .env("USER", "gws-dry-run-test") + .env("USERNAME", "gws-dry-run-test") + .env("GOOGLE_WORKSPACE_CLI_CONFIG_DIR", &self.config) + // Never query an actual OS account, even when testing broken auth. + .env("GOOGLE_WORKSPACE_CLI_KEYRING_BACKEND", "file") + .env("HTTP_PROXY", &self.network.url) + .env("HTTPS_PROXY", &self.network.url) + .env("ALL_PROXY", &self.network.url) + .stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + if let Some(system_root) = std::env::var_os("SystemRoot") { + command.env("SystemRoot", system_root); + } + if let Some(token) = token { + command.env("GOOGLE_WORKSPACE_CLI_TOKEN", token); + } + let mut child = command.spawn().unwrap(); + let deadline = Instant::now() + Duration::from_secs(15); + while child.try_wait().unwrap().is_none() { + if Instant::now() >= deadline { + child.kill().unwrap(); + let output = child.wait_with_output().unwrap(); + panic!("CLI timed out: {output:?}"); + } + thread::sleep(Duration::from_millis(10)); + } + child.wait_with_output().unwrap() + } + + fn dry_run(&self, args: &[&str], exit_code: i32) -> Value { + let before = snapshot(self.dir.path()); + let mut args = args.to_vec(); + args.push("--dry-run"); + let output = self.run(&args, None); + assert_eq!( + output.status.code(), + Some(exit_code), + "stdout: {}\nstderr: {}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + let after = snapshot(self.dir.path()); + assert_eq!( + after.keys().collect::>(), + before.keys().collect::>(), + "Dry-run created or deleted fixture files" + ); + for (path, expected) in &before { + assert!( + after.get(path) == Some(expected), + "Dry-run changed contents or mtime: {path:?}" + ); + } + assert!( + self.network.requests.lock().unwrap().is_empty(), + "Dry-run attempted an API, auth or Discovery connection" + ); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!( + !stderr.contains("keyring") && !stderr.contains("credentials"), + "Dry-run unexpectedly used auth: {stderr}" + ); + serde_json::from_slice(&output.stdout).unwrap() + } + + fn assert_preview(&self, preview: &Value, query: Value) { + assert_eq!( + preview, + &json!({ + "dry_run": true, + "url": format!("{}v1/documents/doc%20%2F%3F%23:batchUpdate", self.network.url), + "method": "POST", + "query_params": query, + "body": { + "requests": [{ + "insertText": { + "text": "hello", + "endOfSegmentLocation": {"segmentId": ""} + } + }] + }, + "is_multipart_upload": false + }) + ); + } +} + +#[test] +fn raw_dry_run_preserves_corrupt_credentials() { + let fixture = Fixture::new(true); + fixture.assert_preview(&fixture.dry_run(RAW, 0), json!([["fields", "documentId"]])); +} + +#[test] +fn raw_dry_run_needs_no_credentials() { + let fixture = Fixture::new(false); + fixture.assert_preview(&fixture.dry_run(RAW, 0), json!([["fields", "documentId"]])); +} + +#[test] +fn docs_write_dry_run_preserves_corrupt_credentials() { + let fixture = Fixture::new(true); + fixture.assert_preview(&fixture.dry_run(WRITE, 0), json!([])); +} + +#[test] +fn docs_write_dry_run_needs_no_credentials() { + let fixture = Fixture::new(false); + fixture.assert_preview(&fixture.dry_run(WRITE, 0), json!([])); +} + +fn assert_validation(error: &Value, message: &str) { + assert_eq!(error["error"]["reason"], "validationError"); + assert!( + error["error"]["message"] + .as_str() + .unwrap() + .contains(message), + "{error}" + ); +} + +#[test] +fn raw_dry_run_rejects_malformed_body_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + *args.last_mut().unwrap() = "{"; + assert_validation(&fixture.dry_run(&args, 3), "Invalid --json body"); +} + +#[test] +fn raw_dry_run_validates_nested_body_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + *args.last_mut().unwrap() = r#"{"requests":[{"insertText":{"text":42}}]}"#; + assert_validation(&fixture.dry_run(&args, 3), "Expected type 'string'"); +} + +#[test] +fn raw_dry_run_rejects_malformed_params_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + args[4] = "{"; + assert_validation(&fixture.dry_run(&args, 3), "Invalid --params JSON"); +} + +#[test] +fn raw_dry_run_requires_path_parameter_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + args[4] = "{}"; + assert_validation(&fixture.dry_run(&args, 3), "documentId is missing"); +} + +#[test] +fn raw_dry_run_requires_query_parameter_without_auth() { + let fixture = Fixture::new(true); + let mut doc = fixture.discovery(); + doc["resources"]["documents"]["methods"]["batchUpdate"]["parameters"]["revision"] = + json!({"type": "string", "location": "query", "required": true}); + fixture.write_discovery(&doc); + assert_validation(&fixture.dry_run(RAW, 3), "'revision' is missing"); +} + +#[test] +fn raw_dry_run_rejects_resource_traversal_without_auth() { + let fixture = Fixture::new(true); + let mut doc = fixture.discovery(); + doc["resources"]["documents"]["methods"]["batchUpdate"]["path"] = + json!("documents/{+documentId}:batchUpdate"); + fixture.write_discovery(&doc); + let mut args = RAW.to_vec(); + args[4] = r#"{"documentId":"../outside"}"#; + assert_validation(&fixture.dry_run(&args, 3), "traversal"); +} + +#[test] +fn raw_dry_run_rejects_output_traversal_without_auth() { + let fixture = Fixture::new(true); + let mut args = RAW.to_vec(); + args.extend(["--output", "../outside"]); + assert_validation(&fixture.dry_run(&args, 3), "outside the current directory"); +} + +#[test] +fn docs_write_dry_run_requires_document() { + let fixture = Fixture::new(true); + assert_validation( + &fixture.dry_run(&["docs", "+write", "--text", "hello"], 3), + "--document", + ); +} + +#[test] +fn docs_write_dry_run_requires_text() { + let fixture = Fixture::new(true); + assert_validation( + &fixture.dry_run(&["docs", "+write", "--document", "doc"], 3), + "--text", + ); +} + +#[test] +fn docs_write_dry_run_validates_generated_body_without_auth() { + let fixture = Fixture::new(true); + let mut doc = fixture.discovery(); + doc["schemas"]["BatchUpdateDocumentRequest"]["required"] = json!(["requests", "title"]); + fixture.write_discovery(&doc); + assert_validation( + &fixture.dry_run(WRITE, 3), + "Missing required property 'title'", + ); +} + +#[test] +fn docs_write_dry_run_preserves_discovery_errors() { + let fixture = Fixture::new(true); + let mut doc = fixture.discovery(); + doc["resources"]["documents"]["methods"] = json!({}); + fixture.write_discovery(&doc); + let error = fixture.dry_run(WRITE, 4); + assert_eq!(error["error"]["reason"], "discoveryError"); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("batchUpdate")); +} + +fn assert_auth_failure(args: &[&str]) { + let fixture = Fixture::new(true); + let output = fixture.run(args, None); + assert_eq!(output.status.code(), Some(2), "{output:?}"); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert_eq!(error["error"]["reason"], "authError"); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("credentials")); + // Prove this is the real auth path: existing corrupt-credential cleanup + // still runs, and the broken plaintext fallback remains a hard error. + assert!(!fixture.config.join("credentials.enc").exists()); + assert!(!fixture.config.join("token_cache.json").exists()); + assert!(!fixture.config.join("sa_token_cache.json").exists()); + assert!(fixture.network.requests.lock().unwrap().is_empty()); +} + +#[test] +fn raw_real_request_still_fails_on_broken_credentials() { + assert_auth_failure(RAW); +} + +#[test] +fn docs_write_real_request_still_fails_on_broken_credentials() { + assert_auth_failure(WRITE); +} + +#[test] +fn raw_real_request_without_credentials_preserves_access_denied() { + let fixture = Fixture::new(false); + let output = fixture.run(RAW, None); + assert_eq!(output.status.code(), Some(2), "{output:?}"); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert_eq!(error["error"]["reason"], "authError"); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("No credentials provided")); + let requests = fixture.network.requests.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert!(!requests[0].to_lowercase().contains("authorization:")); +} + +fn assert_authenticated_api_failure(args: &[&str]) { + let fixture = Fixture::new(false); + let output = fixture.run(args, Some("synthetic-test-token")); + assert_eq!(output.status.code(), Some(1), "{output:?}"); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert_eq!(error["error"]["code"], 403); + assert_eq!(error["error"]["message"], "Synthetic denial"); + let requests = fixture.network.requests.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert!(requests[0] + .to_lowercase() + .contains("authorization: bearer synthetic-test-token")); +} + +#[test] +fn raw_real_request_uses_token_and_preserves_api_failure() { + assert_authenticated_api_failure(RAW); +} + +#[test] +fn docs_write_real_request_uses_token_and_preserves_api_failure() { + assert_authenticated_api_failure(WRITE); +} From eda58cac355ad10926e432dd436ffdc75f8f8a36 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:19:40 -0300 Subject: [PATCH 08/19] test(auth): isolate dry-run fixtures from profile credentials --- crates/google-workspace-cli/tests/dry_run.rs | 70 ++++++++++++++++++-- 1 file changed, 63 insertions(+), 7 deletions(-) diff --git a/crates/google-workspace-cli/tests/dry_run.rs b/crates/google-workspace-cli/tests/dry_run.rs index 3f03a3496..bcd578e8e 100644 --- a/crates/google-workspace-cli/tests/dry_run.rs +++ b/crates/google-workspace-cli/tests/dry_run.rs @@ -225,7 +225,7 @@ impl Fixture { .unwrap(); } - fn run(&self, args: &[&str], token: Option<&str>) -> Output { + fn command(&self, args: &[&str], token: Option<&str>) -> Command { let mut command = Command::new(env!("CARGO_BIN_EXE_gws")); command .args(args) @@ -238,6 +238,12 @@ impl Fixture { .env("USER", "gws-dry-run-test") .env("USERNAME", "gws-dry-run-test") .env("GOOGLE_WORKSPACE_CLI_CONFIG_DIR", &self.config) + // Windows known-folder lookup ignores the home overrides above. + // Pin both token loading and quota-project lookup to the fixture. + .env( + "GOOGLE_APPLICATION_CREDENTIALS", + self.dir.path().join("missing-adc.json"), + ) // Never query an actual OS account, even when testing broken auth. .env("GOOGLE_WORKSPACE_CLI_KEYRING_BACKEND", "file") .env("HTTP_PROXY", &self.network.url) @@ -252,6 +258,14 @@ impl Fixture { if let Some(token) = token { command.env("GOOGLE_WORKSPACE_CLI_TOKEN", token); } + command + } + + fn run(&self, args: &[&str], token: Option<&str>) -> Output { + Self::run_command(self.command(args, token)) + } + + fn run_command(mut command: Command) -> Output { let mut child = command.spawn().unwrap(); let deadline = Instant::now() + Duration::from_secs(15); while child.try_wait().unwrap().is_none() { @@ -474,11 +488,6 @@ fn assert_auth_failure(args: &[&str]) { .as_str() .unwrap() .contains("credentials")); - // Prove this is the real auth path: existing corrupt-credential cleanup - // still runs, and the broken plaintext fallback remains a hard error. - assert!(!fixture.config.join("credentials.enc").exists()); - assert!(!fixture.config.join("token_cache.json").exists()); - assert!(!fixture.config.join("sa_token_cache.json").exists()); assert!(fixture.network.requests.lock().unwrap().is_empty()); } @@ -493,12 +502,45 @@ fn docs_write_real_request_still_fails_on_broken_credentials() { } #[test] -fn raw_real_request_without_credentials_preserves_access_denied() { +fn raw_real_request_rejects_fixture_adc_without_profile_fallback() { let fixture = Fixture::new(false); let output = fixture.run(RAW, None); assert_eq!(output.status.code(), Some(2), "{output:?}"); let error: Value = serde_json::from_slice(&output.stdout).unwrap(); assert_eq!(error["error"]["reason"], "authError"); + let message = error["error"]["message"].as_str().unwrap(); + assert!( + message.contains("GOOGLE_APPLICATION_CREDENTIALS points to"), + "{error}" + ); + assert!( + message.contains( + fixture + .dir + .path() + .join("missing-adc.json") + .to_str() + .unwrap() + ), + "{error}" + ); + assert!(message.contains("file does not exist"), "{error}"); + assert!(fixture.network.requests.lock().unwrap().is_empty()); +} + +// This case must leave ADC unset to exercise the real no-credentials fallback. +// Only Unix dirs::home_dir() respects our HOME isolation; Windows uses the +// actual profile's known folder, so this case must not run there. +#[cfg(unix)] +#[test] +fn raw_real_request_without_credentials_preserves_access_denied() { + let fixture = Fixture::new(false); + let mut command = fixture.command(RAW, None); + command.env_remove("GOOGLE_APPLICATION_CREDENTIALS"); + let output = Fixture::run_command(command); + assert_eq!(output.status.code(), Some(2), "{output:?}"); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert_eq!(error["error"]["reason"], "authError"); assert!(error["error"]["message"] .as_str() .unwrap() @@ -510,6 +552,16 @@ fn raw_real_request_without_credentials_preserves_access_denied() { fn assert_authenticated_api_failure(args: &[&str]) { let fixture = Fixture::new(false); + // Make a mistaken profile fallback observable on Unix without using a real + // profile. Windows known-folder lookup ignores these home overrides, so the + // fixture must explicitly redirect ADC there as well. + let adc_dir = fixture.dir.path().join("home/.config/gcloud"); + fs::create_dir_all(&adc_dir).unwrap(); + fs::write( + adc_dir.join("application_default_credentials.json"), + r#"{"quota_project_id":"synthetic-profile-must-not-be-read"}"#, + ) + .unwrap(); let output = fixture.run(args, Some("synthetic-test-token")); assert_eq!(output.status.code(), Some(1), "{output:?}"); let error: Value = serde_json::from_slice(&output.stdout).unwrap(); @@ -520,6 +572,10 @@ fn assert_authenticated_api_failure(args: &[&str]) { assert!(requests[0] .to_lowercase() .contains("authorization: bearer synthetic-test-token")); + assert!( + !requests[0].to_lowercase().contains("x-goog-user-project:"), + "Token-authenticated requests must not read quota attribution from profile ADC" + ); } #[test] From e88fbe30736d8912f9c5b18c807351c8d0c09f33 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 11:59:44 -0300 Subject: [PATCH 09/19] feat(files): add a trusted root for output and upload paths --- .changeset/scoped-file-roots.md | 9 + AGENTS.md | 2 + README.md | 31 ++ .../google-workspace-cli/tests/file_roots.rs | 210 ++++++++ crates/google-workspace/src/validate.rs | 458 ++++++++++++++++-- 5 files changed, 680 insertions(+), 30 deletions(-) create mode 100644 .changeset/scoped-file-roots.md create mode 100644 crates/google-workspace-cli/tests/file_roots.rs diff --git a/.changeset/scoped-file-roots.md b/.changeset/scoped-file-roots.md new file mode 100644 index 000000000..3db02fb8e --- /dev/null +++ b/.changeset/scoped-file-roots.md @@ -0,0 +1,9 @@ +--- +"@googleworkspace/cli": minor +--- + +Allow operators to set `GOOGLE_WORKSPACE_CLI_FILE_ROOT` to an existing directory +for `--output` and `--upload` paths while keeping CWD confinement by default. +Relative CLI paths remain CWD-relative. Reject invalid roots, parent traversal +with an explicit root, control characters, and symlink escapes, including +dangling symlinks. Directory flags retain their existing boundaries. diff --git a/AGENTS.md b/AGENTS.md index 722112264..196fc4fda 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -112,6 +112,7 @@ When adding new helpers or CLI flags that accept file paths, **always validate** | -------------------------------------- | ---------------------------------------- | -------------------------------------------------------------------- | | File path for writing (`--output-dir`) | `validate::validate_safe_output_dir()` | Absolute paths, `../` traversal, symlinks outside CWD, control chars | | File path for reading (`--dir`) | `validate::validate_safe_dir_path()` | Absolute paths, `../` traversal, symlinks outside CWD, control chars | +| File path (`--output`, `--upload`) | `validate::validate_safe_file_path()` | Paths outside CWD or the trusted `GOOGLE_WORKSPACE_CLI_FILE_ROOT`, control chars, symlink escapes; `..` components with an explicit root | | Enum/allowlist values (`--msg-format`) | clap `value_parser` (see `gmail/mod.rs`) | Any value not in the allowlist | ```rust @@ -208,6 +209,7 @@ See [`src/helpers/README.md`](crates/google-workspace-cli/src/helpers/README.md) | Variable | Description | |---|---| | `GOOGLE_WORKSPACE_CLI_CONFIG_DIR` | Override the config directory (default: `~/.config/gws`) | +| `GOOGLE_WORKSPACE_CLI_FILE_ROOT` | Trusted boundary for `--output` / `--upload` files (default: CWD). Must be an existing directory; canonicalized. Relative CLI paths remain CWD-relative. Does not expand directory validators. | ### OAuth Client diff --git a/README.md b/README.md index 12544f5eb..3f3dfb58f 100644 --- a/README.md +++ b/README.md @@ -340,6 +340,36 @@ Installing this extension gives your Gemini CLI agent direct access to all `gws` gws drive files create --json '{"name": "report.pdf"}' --upload ./report.pdf ``` +### Output and upload file roots + +`--output` and `--upload` accept paths within the current working directory +(CWD) by default, including absolute paths that resolve inside CWD. To allow +files elsewhere, set a trusted operator environment variable to an existing +directory: + +```bash +mkdir -p /tmp/gws-files +export GOOGLE_WORKSPACE_CLI_FILE_ROOT=/tmp/gws-files +gws drive files get --params '{"fileId":"FILE_ID","alt":"media"}' \ + --output /tmp/gws-files/report.pdf +gws drive files create --json '{"name":"report.pdf"}' \ + --upload /tmp/gws-files/report.pdf +``` + +The root replaces the allowed file boundary; relative CLI paths still resolve +from CWD. For example, `--output report.pdf` is rejected if CWD is outside the +configured root. The root is canonicalized and must exist as a directory; an +empty or invalid value fails validation. Relative root settings resolve from +CWD too. With an explicit root, CLI paths containing `..` components are +rejected. Control characters and symlinks escaping the boundary are rejected; +symlinks resolving inside it are allowed, but dangling symlinks are rejected. + +This setting affects only these file flags, not `--dir` or `--output-dir`. +It does not create parent directories or change the default download filename +when `--output` is omitted. Validation cannot prevent another local process +from replacing a path component between validation and I/O; choose a root +whose directories you control. Unset the variable to restore the CWD boundary. + ### Pagination | Flag | Description | Default | @@ -456,6 +486,7 @@ All variables are optional. See [`.env.example`](.env.example) for a copy-paste | `GOOGLE_WORKSPACE_CLI_CLIENT_ID` | OAuth client ID (alternative to `client_secret.json`) | | `GOOGLE_WORKSPACE_CLI_CLIENT_SECRET` | OAuth client secret (paired with `CLIENT_ID`) | | `GOOGLE_WORKSPACE_CLI_CONFIG_DIR` | Override config directory (default: `~/.config/gws`) | +| `GOOGLE_WORKSPACE_CLI_FILE_ROOT` | Existing directory allowed for `--output` / `--upload` paths (default: CWD); relative CLI paths remain CWD-relative | | `GOOGLE_WORKSPACE_CLI_SANITIZE_TEMPLATE` | Default Model Armor template | | `GOOGLE_WORKSPACE_CLI_SANITIZE_MODE` | `warn` (default) or `block` | | `GOOGLE_WORKSPACE_CLI_LOG` | Log level for stderr (e.g., `gws=debug`). Off by default. | diff --git a/crates/google-workspace-cli/tests/file_roots.rs b/crates/google-workspace-cli/tests/file_roots.rs new file mode 100644 index 000000000..614256605 --- /dev/null +++ b/crates/google-workspace-cli/tests/file_roots.rs @@ -0,0 +1,210 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Exercise the real CLI with synthetic cached Discovery and child-only env. + +use std::fs; +use std::io::{Read, Write}; +use std::net::TcpListener; +use std::path::{Path, PathBuf}; +use std::process::{Command, Output}; +use std::time::{Duration, Instant}; + +use serde_json::{json, Value}; +use tempfile::{tempdir, TempDir}; + +struct Fixture { + _temp: TempDir, + cwd: PathBuf, + root: PathBuf, + config: PathBuf, +} + +impl Fixture { + fn new(root_url: &str) -> Self { + let temp = tempdir().unwrap(); + let base = temp.path().canonicalize().unwrap(); + let cwd = base.join("working"); + let root = base.join("files"); + let config = base.join("config"); + fs::create_dir(&cwd).unwrap(); + fs::create_dir(&root).unwrap(); + fs::create_dir_all(config.join("cache")).unwrap(); + // Stop dotenv from searching parent directories for real configuration. + fs::write(cwd.join(".env"), "").unwrap(); + fs::write(config.join("cache/drive_v3.json"), json!({ + "name": "drive", "version": "v3", "rootUrl": root_url, + "resources": {"files": {"methods": { + "get": {"httpMethod": "GET", "path": "files/synthetic"}, + "create": {"httpMethod": "POST", "path": "files", "supportsMediaUpload": true, + "mediaUpload": {"protocols": {"simple": {"path": "/upload/files", "multipart": true}}}} + }}} + }).to_string()).unwrap(); + Self { + _temp: temp, + cwd, + root, + config, + } + } + + fn command(&self, configured: bool) -> Command { + let mut command = Command::new(env!("CARGO_BIN_EXE_gws")); + command + .env_clear() + .current_dir(&self.cwd) + .env("HOME", &self.cwd) + .env("USERPROFILE", &self.cwd) + .env("GOOGLE_WORKSPACE_CLI_CONFIG_DIR", &self.config) + .env("GOOGLE_WORKSPACE_CLI_TOKEN", "synthetic-test-token") + .env("GOOGLE_WORKSPACE_CLI_KEYRING_BACKEND", "file") + .env("NO_COLOR", "1") + // A cache regression must fail locally, never fetch real Discovery. + .env("HTTPS_PROXY", "http://127.0.0.1:1") + .env("HTTP_PROXY", "http://127.0.0.1:1") + .env("NO_PROXY", "127.0.0.1,localhost"); + if configured { + command.env("GOOGLE_WORKSPACE_CLI_FILE_ROOT", &self.root); + } + command + } + + fn dry_run(&self, flag: &str, path: &Path, configured: bool) -> Output { + let method = if flag == "--upload" { "create" } else { "get" }; + self.command(configured) + .args(["drive", "files", method, "--dry-run", flag]) + .arg(path) + .output() + .unwrap() + } +} + +fn successful_json(output: &Output) -> Value { + assert!( + output.status.success(), + "status: {:?}\nstdout: {}\nstderr: {}", + output.status.code(), + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + serde_json::from_slice(&output.stdout).unwrap() +} + +#[test] +fn file_root_cli_dry_run_accepts_external_output_and_upload() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + fs::write(fixture.root.join("upload.txt"), "synthetic upload").unwrap(); + for (flag, name) in [("--output", "new.bin"), ("--upload", "upload.txt")] { + let output = fixture.dry_run(flag, &fixture.root.join(name), true); + let result = successful_json(&output); + assert_eq!(result["dry_run"], true); + assert_eq!(result["is_multipart_upload"], flag == "--upload"); + } + assert!(!fixture.root.join("new.bin").exists()); +} + +#[test] +fn file_root_cli_default_rejects_external_paths_but_keeps_local_output() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + let output = fixture.dry_run("--output", &fixture.root.join("new.bin"), false); + assert_eq!(output.status.code(), Some(3)); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("current directory")); + assert_eq!( + successful_json(&fixture.dry_run("--output", Path::new("new.bin"), false))["dry_run"], + true + ); +} + +#[test] +fn file_root_cli_rejects_cwd_relative_path_outside_configured_root() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + let output = fixture.dry_run("--output", Path::new("new.bin"), true); + assert_eq!( + output.status.code(), + Some(3), + "{}", + String::from_utf8_lossy(&output.stdout) + ); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + assert!(error["error"]["message"] + .as_str() + .unwrap() + .contains("GOOGLE_WORKSPACE_CLI_FILE_ROOT")); +} + +#[test] +fn file_root_cli_download_propagates_canonical_output() { + let listener = TcpListener::bind("127.0.0.1:0").unwrap(); + listener.set_nonblocking(true).unwrap(); + let fixture = Fixture::new(&format!("http://{}/", listener.local_addr().unwrap())); + let output_path = fixture.root.join("./output.bin"); + let mut child = fixture + .command(true) + .args(["drive", "files", "get", "--output"]) + .arg(&output_path) + .stdout(std::process::Stdio::piped()) + .stderr(std::process::Stdio::piped()) + .spawn() + .unwrap(); + let deadline = Instant::now() + Duration::from_secs(15); + loop { + match listener.accept() { + Ok((mut stream, _)) => { + stream + .set_read_timeout(Some(Duration::from_secs(5))) + .unwrap(); + let mut request = Vec::new(); + let mut buffer = [0; 1024]; + while !request.windows(4).any(|part| part == b"\r\n\r\n") { + let read = stream.read(&mut buffer).unwrap(); + assert!(read > 0, "request ended before headers"); + request.extend_from_slice(&buffer[..read]); + } + assert!(request.starts_with(b"GET /files/synthetic HTTP/1.1\r\n")); + stream.write_all(b"HTTP/1.1 200 OK\r\nContent-Type: application/octet-stream\r\nContent-Length: 9\r\nConnection: close\r\n\r\nsynthetic").unwrap(); + break; + } + Err(err) if err.kind() == std::io::ErrorKind::WouldBlock => { + if child.try_wait().unwrap().is_some() { + break; + } + if Instant::now() >= deadline { + child.kill().unwrap(); + let output = child.wait_with_output().unwrap(); + panic!( + "CLI did not contact local fixture: {}", + String::from_utf8_lossy(&output.stderr) + ); + } + std::thread::sleep(Duration::from_millis(10)); + } + Err(err) => panic!("local fixture failed: {err}"), + } + } + let result = successful_json(&child.wait_with_output().unwrap()); + assert_eq!( + result["saved_file"], + fixture.root.join("output.bin").to_str().unwrap() + ); + assert_eq!(result["bytes"], 9); + assert_eq!( + fs::read(fixture.root.join("output.bin")).unwrap(), + b"synthetic" + ); + assert!(!fixture.cwd.join("output.bin").exists()); +} diff --git a/crates/google-workspace/src/validate.rs b/crates/google-workspace/src/validate.rs index 32ef200f9..491c5d33b 100644 --- a/crates/google-workspace/src/validate.rs +++ b/crates/google-workspace/src/validate.rs @@ -161,11 +161,13 @@ pub fn validate_safe_dir_path(dir: &str) -> Result { /// Validates that a file path (e.g. `--upload` or `--output`) is safe. /// -/// Rejects paths that escape above CWD via `..` traversal, contain -/// control characters, or follow symlinks to locations outside CWD. -/// Absolute paths are allowed (reading an existing file from a known -/// location is legitimate) but the resolved target must still live -/// under CWD. +/// By default, the resolved target must live under CWD. The trusted operator +/// environment variable `GOOGLE_WORKSPACE_CLI_FILE_ROOT` can select a different +/// boundary: an existing directory, canonicalized before validation. With an +/// explicit root, CLI paths must not contain `..` components. Relative CLI paths +/// always resolve from CWD, not from the configured root. Absolute paths within +/// the boundary are allowed. Control characters and symlink escapes are rejected. +/// Directory validators do not use this setting. /// /// # TOCTOU caveat /// @@ -175,48 +177,127 @@ pub fn validate_safe_dir_path(dir: &str) -> Result { /// TOCTOU would require `openat(O_NOFOLLOW)` on each path component, /// which is tracked as a follow-up for Unix platforms. pub fn validate_safe_file_path(path_str: &str, flag_name: &str) -> Result { - reject_dangerous_chars(path_str, flag_name)?; - - let path = Path::new(path_str); let cwd = std::env::current_dir() .map_err(|e| GwsError::Validation(format!("Failed to determine current directory: {e}")))?; + let file_root = std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT"); + validate_file_path_with_root( + path_str, + flag_name, + &cwd, + file_root.as_deref().map(Path::new), + ) +} - let resolved = if path.is_absolute() { - path.to_path_buf() - } else { - cwd.join(path) - }; +/// Explicit policy keeps filesystem validation independent of process-global env. +fn validate_file_path_with_root( + path_str: &str, + flag_name: &str, + cwd: &Path, + file_root: Option<&Path>, +) -> Result { + reject_dangerous_chars(path_str, flag_name)?; - // For existing files, canonicalize to resolve symlinks. - // For non-existing files, get the prefix canonicalized then normalize - // the remaining components to resolve any `..` or `.` segments. - let canonical = if resolved.exists() { - resolved.canonicalize().map_err(|e| { - GwsError::Validation(format!("Failed to resolve {flag_name} '{}': {e}", path_str)) + let canonical_root = if let Some(root) = file_root { + if root.as_os_str().is_empty() { + return Err(GwsError::Validation( + "GOOGLE_WORKSPACE_CLI_FILE_ROOT must name an existing directory; got an empty value" + .to_string(), + )); + } + // Environment is trusted: relative roots (including `..`) are valid. + let canonical = cwd.join(root).canonicalize().map_err(|e| { + GwsError::Validation(format!( + "GOOGLE_WORKSPACE_CLI_FILE_ROOT {root:?} must name an existing directory: {e}" + )) + })?; + if !canonical.is_dir() { + return Err(GwsError::Validation(format!( + "GOOGLE_WORKSPACE_CLI_FILE_ROOT {root:?} must name an existing directory" + ))); + } + canonical + } else { + cwd.canonicalize().map_err(|e| { + GwsError::Validation(format!("Failed to canonicalize current directory: {e}")) })? + }; + let boundary = if file_root.is_some() { + format!("GOOGLE_WORKSPACE_CLI_FILE_ROOT directory {canonical_root:?}") } else { - let raw = normalize_non_existing(&resolved)?; - // normalize_non_existing does NOT resolve `..` in the non-existent - // suffix. We must resolve them here to prevent bypass via paths like - // `non_existent/../../etc/passwd`. - normalize_dotdot(&raw) + format!("current directory {canonical_root:?}") }; - let canonical_cwd = cwd.canonicalize().map_err(|e| { - GwsError::Validation(format!("Failed to canonicalize current directory: {e}")) + let path = Path::new(path_str); + if file_root.is_some() + && path + .components() + .any(|component| component == std::path::Component::ParentDir) + { + return Err(GwsError::Validation(format!( + "{flag_name} must not contain parent traversal ('..') components within the {boundary}; use a path without '..'" + ))); + } + + // Path::join preserves absolute arguments; relative arguments stay CWD-relative. + let resolved = cwd.join(path); + let canonical = canonicalize_file_path(&resolved).map_err(|e| { + GwsError::Validation(format!( + "Failed to resolve {flag_name} {path_str:?} within the {boundary}: {e}" + )) })?; + // Preserve default handling of paths that normalize safely within CWD. + let canonical = normalize_dotdot(&canonical); - if !canonical.starts_with(&canonical_cwd) { + if !canonical.starts_with(&canonical_root) { return Err(GwsError::Validation(format!( - "{flag_name} '{}' resolves to '{}' which is outside the current directory", - path_str, - canonical.display() + "{flag_name} {path_str:?} resolves to {canonical:?} which is outside the {boundary}; set GOOGLE_WORKSPACE_CLI_FILE_ROOT to an existing directory containing the intended file" ))); } Ok(canonical) } +/// Canonicalize the existing file or nearest existing parent, then append the +/// missing suffix. Unlike `exists()`, symlink_metadata does not mistake dangling +/// symlinks for missing files. Keep this stricter resolver local to file flags. +fn canonicalize_file_path(path: &Path) -> std::io::Result { + let mut current = path; + let mut remaining = Vec::new(); + loop { + match std::fs::symlink_metadata(current) { + Ok(_) => { + let mut canonical = current.canonicalize()?; + if !remaining.is_empty() && !canonical.is_dir() { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "existing file parent must be a directory", + )); + } + for component in remaining.into_iter().rev() { + canonical.push(component); + } + return Ok(canonical); + } + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + let name = current.file_name().ok_or_else(|| { + std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "cannot resolve an existing directory prefix", + ) + })?; + remaining.push(name); + current = current.parent().ok_or_else(|| { + std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "cannot resolve a file parent", + ) + })?; + } + Err(error) => return Err(error), + } + } +} + /// Resolve `.` and `..` components in a path without touching the filesystem. fn normalize_dotdot(path: &Path) -> PathBuf { let mut out = PathBuf::new(); @@ -835,4 +916,321 @@ mod tests { "traversal via non-existent prefix should be rejected" ); } + + // Scoped file-root acceptance tests. Only directory-scope checks mutate env; + // policy tests pass CWD and the trusted root explicitly. + struct FilePathEnvironment { + cwd: PathBuf, + root: Option, + } + + impl FilePathEnvironment { + fn set(cwd: &Path, root: Option<&Path>) -> Self { + let saved = Self { + cwd: std::env::current_dir().unwrap(), + root: std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), + }; + std::env::set_current_dir(cwd).unwrap(); + match root { + Some(root) => std::env::set_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT", root), + None => std::env::remove_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), + } + saved + } + } + + impl Drop for FilePathEnvironment { + fn drop(&mut self) { + std::env::set_current_dir(&self.cwd).unwrap(); + match &self.root { + Some(root) => std::env::set_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT", root), + None => std::env::remove_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), + } + } + } + + fn file_path_under_root( + path: &Path, + flag: &str, + cwd: &Path, + root: Option<&Path>, + ) -> Result { + validate_file_path_with_root(path.to_str().unwrap(), flag, cwd, root) + } + + #[test] + fn file_root_default_preserves_cwd_boundary_and_resolution() { + let dir = tempdir().unwrap(); + let cwd = dir.path().canonicalize().unwrap(); + fs::create_dir(cwd.join("nested")).unwrap(); + fs::write(cwd.join("upload.txt"), "synthetic upload").unwrap(); + for path in [cwd.join("upload.txt"), PathBuf::from("upload.txt")] { + assert_eq!( + file_path_under_root(&path, "--upload", &cwd, None).unwrap(), + cwd.join("upload.txt") + ); + } + assert_eq!( + file_path_under_root(Path::new("nested/../new.txt"), "--output", &cwd, None).unwrap(), + cwd.join("new.txt") + ); + for path in [ + cwd.parent().unwrap().join("outside.txt"), + PathBuf::from("../outside.txt"), + ] { + let err = file_path_under_root(&path, "--output", &cwd, None) + .unwrap_err() + .to_string(); + assert!(err.contains("outside the current directory"), "{err}"); + assert!(err.contains("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), "{err}"); + } + assert!(file_path_under_root( + Path::new("missing/../../outside.txt"), + "--output", + &cwd, + None + ) + .is_err()); + } + + #[test] + fn file_root_accepts_absolute_output_and_existing_upload() { + let cwd = tempdir().unwrap(); + let root = tempdir().unwrap(); + let canonical_root = root.path().canonicalize().unwrap(); + fs::write(root.path().join("upload.txt"), "synthetic upload").unwrap(); + for (name, flag) in [("new.txt", "--output"), ("upload.txt", "--upload")] { + assert_eq!( + file_path_under_root(&root.path().join(name), flag, cwd.path(), Some(root.path())) + .unwrap(), + canonical_root.join(name) + ); + } + assert!(!root.path().join("new.txt").exists()); + } + + #[test] + fn file_root_rejects_sibling_even_with_shared_name_prefix() { + let dir = tempdir().unwrap(); + let root = dir.path().join("allowed"); + let sibling = dir.path().join("allowed-sibling"); + fs::create_dir(&root).unwrap(); + fs::create_dir(&sibling).unwrap(); + let err = file_path_under_root( + &sibling.join("new.txt"), + "--output", + dir.path(), + Some(&root), + ) + .unwrap_err() + .to_string(); + assert!(err.contains("outside"), "{err}"); + assert!(err.contains("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), "{err}"); + assert!(err.contains(root.to_str().unwrap()), "{err}"); + } + + #[test] + fn file_root_rejects_parent_components_even_inside_boundary() { + let root = tempdir().unwrap(); + fs::create_dir(root.path().join("nested")).unwrap(); + for path in ["nested/../new.txt", "missing/../new.txt", "../outside.txt"] { + assert!( + file_path_under_root(Path::new(path), "--output", root.path(), Some(root.path())) + .is_err(), + "accepted {path}" + ); + } + } + + #[test] + fn file_root_rejects_control_and_dangerous_unicode_arguments() { + let root = tempdir().unwrap(); + for path in [ + "bad\0.txt", + "bad\n.txt", + "bad\u{202e}.txt", + "bad\u{200b}.txt", + ] { + assert!( + file_path_under_root(Path::new(path), "--output", root.path(), Some(root.path())) + .is_err(), + "accepted {path:?}" + ); + } + } + + #[test] + fn file_root_rejects_invalid_roots_without_falling_back_to_cwd() { + let cwd = tempdir().unwrap(); + let file = cwd.path().join("file.txt"); + fs::write(&file, "synthetic file").unwrap(); + for root in [PathBuf::new(), cwd.path().join("missing"), file] { + let err = + file_path_under_root(Path::new("new.txt"), "--output", cwd.path(), Some(&root)) + .unwrap_err() + .to_string(); + assert!(err.contains("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), "{err}"); + assert!(err.contains("directory"), "{err}"); + } + } + + #[test] + fn file_root_keeps_relative_arguments_cwd_relative() { + let root = tempdir().unwrap(); + let cwd = root.path().join("working"); + fs::create_dir(&cwd).unwrap(); + assert_eq!( + file_path_under_root(Path::new("new.txt"), "--output", &cwd, Some(root.path())) + .unwrap(), + cwd.canonicalize().unwrap().join("new.txt") + ); + let unrelated = tempdir().unwrap(); + assert!(file_path_under_root( + Path::new("new.txt"), + "--output", + unrelated.path(), + Some(root.path()) + ) + .is_err()); + } + + #[test] + fn file_root_canonicalizes_trusted_relative_root_with_parent_components() { + let root = tempdir().unwrap(); + let cwd = root.path().join("working"); + fs::create_dir(&cwd).unwrap(); + assert_eq!( + file_path_under_root( + Path::new("new.txt"), + "--output", + &cwd, + Some(Path::new("..")) + ) + .unwrap(), + cwd.canonicalize().unwrap().join("new.txt") + ); + } + + #[test] + fn file_root_validates_nested_new_file_parents_without_creating_them() { + let cwd = tempdir().unwrap(); + let root = tempdir().unwrap(); + let output = root.path().join("new/nested/output.bin"); + assert_eq!( + file_path_under_root(&output, "--output", cwd.path(), Some(root.path())).unwrap(), + root.path() + .canonicalize() + .unwrap() + .join("new/nested/output.bin") + ); + assert!(!root.path().join("new").exists()); + let file = root.path().join("file.txt"); + fs::write(&file, "synthetic file").unwrap(); + assert!(file_path_under_root( + &file.join("output.bin"), + "--output", + cwd.path(), + Some(root.path()) + ) + .is_err()); + } + + #[test] + #[serial] + fn file_root_does_not_expand_directory_validators() { + let cwd = tempdir().unwrap(); + let root = tempdir().unwrap(); + let _environment = FilePathEnvironment::set(cwd.path(), Some(root.path())); + assert!(validate_safe_output_dir(root.path().to_str().unwrap()).is_err()); + assert!(validate_safe_dir_path(root.path().to_str().unwrap()).is_err()); + assert_eq!( + validate_safe_output_dir("new").unwrap(), + cwd.path().canonicalize().unwrap().join("new") + ); + assert!(validate_safe_dir_path(".").is_ok()); + } + + #[test] + #[serial] + fn file_root_environment_is_restored_on_unwind() { + let cwd = std::env::current_dir().unwrap(); + let root = std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT"); + let dir = tempdir().unwrap(); + let result = std::panic::catch_unwind(|| { + let _environment = FilePathEnvironment::set(dir.path(), Some(dir.path())); + panic!("exercise restoration"); + }); + assert!(result.is_err()); + assert_eq!(std::env::current_dir().unwrap(), cwd); + assert_eq!(std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), root); + } + + #[cfg(unix)] + #[test] + fn file_root_resolves_inside_symlinks_and_rejects_escapes() { + use std::os::unix::fs::symlink; + let root = tempdir().unwrap(); + let outside = tempdir().unwrap(); + fs::create_dir(root.path().join("inside")).unwrap(); + fs::write(root.path().join("inside/upload.txt"), "inside").unwrap(); + fs::write(outside.path().join("upload.txt"), "outside").unwrap(); + symlink(root.path().join("inside"), root.path().join("safe")).unwrap(); + symlink(outside.path(), root.path().join("escape")).unwrap(); + for (suffix, flag) in [("upload.txt", "--upload"), ("new/output.bin", "--output")] { + assert_eq!( + file_path_under_root( + &root.path().join("safe").join(suffix), + flag, + root.path(), + Some(root.path()) + ) + .unwrap(), + root.path() + .canonicalize() + .unwrap() + .join("inside") + .join(suffix) + ); + assert!(file_path_under_root( + &root.path().join("escape").join(suffix), + flag, + root.path(), + Some(root.path()) + ) + .is_err()); + } + // A trusted root may itself be a symlink to an existing directory. + assert_eq!( + file_path_under_root( + &root.path().join("safe/upload.txt"), + "--upload", + outside.path(), + Some(&root.path().join("safe")) + ) + .unwrap(), + root.path() + .canonicalize() + .unwrap() + .join("inside/upload.txt") + ); + } + + #[cfg(unix)] + #[test] + fn file_root_rejects_dangling_symlinks_and_loops() { + use std::os::unix::fs::symlink; + let root = tempdir().unwrap(); + let outside = tempdir().unwrap(); + symlink(outside.path().join("new.txt"), root.path().join("dangling")).unwrap(); + symlink("loop", root.path().join("loop")).unwrap(); + for root_policy in [None, Some(root.path())] { + for name in ["dangling", "dangling/new.txt", "loop", "loop/new.txt"] { + assert!( + file_path_under_root(Path::new(name), "--output", root.path(), root_policy) + .is_err(), + "accepted {name}" + ); + } + } + } } From 9b50a02dd39168ce6b4b53a381163465b565ca6d Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:21:07 -0300 Subject: [PATCH 10/19] fix(files): reject unsupported canonical path encodings --- .changeset/scoped-file-roots.md | 4 + README.md | 3 + .../src/helpers/gmail/mod.rs | 38 ++++++++++ crates/google-workspace-cli/src/main.rs | 73 ++++++++++++++++++- .../google-workspace-cli/tests/file_roots.rs | 64 ++++++++++++++++ crates/google-workspace/src/validate.rs | 36 ++++----- 6 files changed, 198 insertions(+), 20 deletions(-) diff --git a/.changeset/scoped-file-roots.md b/.changeset/scoped-file-roots.md index 3db02fb8e..6e17edcc3 100644 --- a/.changeset/scoped-file-roots.md +++ b/.changeset/scoped-file-roots.md @@ -7,3 +7,7 @@ for `--output` and `--upload` paths while keeping CWD confinement by default. Relative CLI paths remain CWD-relative. Reject invalid roots, parent traversal with an explicit root, control characters, and symlink escapes, including dangling symlinks. Directory flags retain their existing boundaries. + +Reject canonical file paths that cannot be represented as UTF-8 at the CLI +string boundary, so explicit output/upload paths cannot silently become omitted +arguments. diff --git a/README.md b/README.md index 3f3dfb58f..0e549f66a 100644 --- a/README.md +++ b/README.md @@ -363,6 +363,9 @@ empty or invalid value fails validation. Relative root settings resolve from CWD too. With an explicit root, CLI paths containing `..` components are rejected. Control characters and symlinks escaping the boundary are rejected; symlinks resolving inside it are allowed, but dangling symlinks are rejected. +These CLI file flags require a UTF-8 canonical path. If a symlink resolves to a +path with unsupported encoding, the command returns a validation error rather +than dropping the upload or selecting the default output file. This setting affects only these file flags, not `--dir` or `--output-dir`. It does not create parent directories or change the default download filename diff --git a/crates/google-workspace-cli/src/helpers/gmail/mod.rs b/crates/google-workspace-cli/src/helpers/gmail/mod.rs index caeb8b6b0..27de3eb9c 100644 --- a/crates/google-workspace-cli/src/helpers/gmail/mod.rs +++ b/crates/google-workspace-cli/src/helpers/gmail/mod.rs @@ -3008,6 +3008,28 @@ mod tests { // --- Attachment tests --- + // Attachment parsing calls the public file validator. Default-policy tests + // must ignore and restore an inherited operator root, including on panic. + // All users of this guard are serialized with the other environment tests. + struct DefaultFileRoot(Option); + + impl DefaultFileRoot { + fn unset() -> Self { + let saved = Self(std::env::var_os("GOOGLE_WORKSPACE_CLI_FILE_ROOT")); + std::env::remove_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT"); + saved + } + } + + impl Drop for DefaultFileRoot { + fn drop(&mut self) { + match &self.0 { + Some(root) => std::env::set_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT", root), + None => std::env::remove_var("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), + } + } + } + fn make_attach_matches(args: &[&str]) -> ArgMatches { let cmd = Command::new("test").arg( Arg::new("attach") @@ -3095,14 +3117,18 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_rejects_control_chars() { + let _root = DefaultFileRoot::unset(); let matches = make_attach_matches(&["test", "-a", "file\0name.pdf"]); let err = parse_attachments(&matches).unwrap_err(); assert!(err.to_string().contains("control characters")); } #[test] + #[serial_test::serial] fn test_parse_attachments_rejects_directory() { + let _root = DefaultFileRoot::unset(); // Use a relative directory that exists in CWD let matches = make_attach_matches(&["test", "-a", "src"]); let err = parse_attachments(&matches).unwrap_err(); @@ -3110,14 +3136,18 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_empty_returns_empty_vec() { + let _root = DefaultFileRoot::unset(); let matches = make_attach_matches(&["test"]); let attachments = parse_attachments(&matches).unwrap(); assert!(attachments.is_empty()); } #[test] + #[serial_test::serial] fn test_parse_attachments_reads_real_file() { + let _root = DefaultFileRoot::unset(); use std::io::Write; let cwd = std::env::current_dir().unwrap().canonicalize().unwrap(); let dir = tempfile::tempdir_in(&cwd).unwrap(); @@ -3137,7 +3167,9 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_nonexistent_file() { + let _root = DefaultFileRoot::unset(); let matches = make_attach_matches(&["test", "-a", "nonexistent_file.pdf"]); let err = parse_attachments(&matches).unwrap_err(); assert!( @@ -3148,7 +3180,9 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_unknown_extension_falls_back_to_octet_stream() { + let _root = DefaultFileRoot::unset(); use std::io::Write; let cwd = std::env::current_dir().unwrap().canonicalize().unwrap(); let dir = tempfile::tempdir_in(&cwd).unwrap(); @@ -3165,7 +3199,9 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_size_limit_accumulates() { + let _root = DefaultFileRoot::unset(); let cwd = std::env::current_dir().unwrap().canonicalize().unwrap(); let dir = tempfile::tempdir_in(&cwd).unwrap(); @@ -3193,7 +3229,9 @@ mod tests { } #[test] + #[serial_test::serial] fn test_parse_attachments_rejects_empty_file() { + let _root = DefaultFileRoot::unset(); let cwd = std::env::current_dir().unwrap().canonicalize().unwrap(); let dir = tempfile::tempdir_in(&cwd).unwrap(); let file_path = dir.path().join("empty.txt"); diff --git a/crates/google-workspace-cli/src/main.rs b/crates/google-workspace-cli/src/main.rs index a0e9241a7..ea53a93b8 100644 --- a/crates/google-workspace-cli/src/main.rs +++ b/crates/google-workspace-cli/src/main.rs @@ -225,7 +225,7 @@ async fn run() -> Result<(), GwsError> { // Validate file paths against traversal before any I/O. // Use the returned canonical paths so the validated path is the one - // actually used for I/O (closes TOCTOU gap). + // actually used for I/O. Local path-replacement races still apply. let upload_path_buf = if let Some(p) = upload_path { Some(crate::validate::validate_safe_file_path(p, "--upload")?) } else { @@ -236,8 +236,8 @@ async fn run() -> Result<(), GwsError> { } else { None }; - let upload_path = upload_path_buf.as_deref().and_then(|p| p.to_str()); - let output_path = output_path_buf.as_deref().and_then(|p| p.to_str()); + let upload_path = optional_file_path_as_str(upload_path_buf.as_deref(), "--upload")?; + let output_path = optional_file_path_as_str(output_path_buf.as_deref(), "--output")?; let upload = { let upload_content_type = matched_args @@ -319,6 +319,22 @@ fn parse_body_validation_policy(matches: &clap::ArgMatches) -> executor::BodyVal } } +// The executor takes strings. An explicit path must never become an omitted +// argument just because canonicalization found a non-UTF-8 component. +fn optional_file_path_as_str<'a>( + path: Option<&'a std::path::Path>, + flag_name: &str, +) -> Result, GwsError> { + path.map(|path| { + path.to_str().ok_or_else(|| { + GwsError::Validation(format!( + "{flag_name} resolves to a path that is not valid UTF-8; choose a path whose canonical components are valid UTF-8" + )) + }) + }) + .transpose() +} + /// Select the best scope from a method's scope list. /// /// Discovery Documents list method scopes as alternatives — any single scope @@ -587,6 +603,57 @@ mod tests { } } + #[test] + fn file_root_path_encoding_preserves_present_and_absent_paths() { + for flag in ["--output", "--upload"] { + assert_eq!(optional_file_path_as_str(None, flag).unwrap(), None); + assert_eq!( + optional_file_path_as_str(Some(std::path::Path::new("résumé.pdf")), flag).unwrap(), + Some("résumé.pdf") + ); + } + } + + #[cfg(any(unix, windows))] + fn non_utf8_canonical_path() -> std::path::PathBuf { + // Construct an OS path in memory: no filesystem support is required. + #[cfg(unix)] + { + use std::os::unix::ffi::OsStringExt; + std::ffi::OsString::from_vec(b"/files/bytes-\xff/report.pdf".to_vec()).into() + } + #[cfg(windows)] + { + use std::os::windows::ffi::OsStringExt; + let mut units: Vec = r"C:\files\bytes-".encode_utf16().collect(); + units.push(0xD800); // unpaired surrogate + units.extend(r"\report.pdf".encode_utf16()); + std::ffi::OsString::from_wide(&units).into() + } + } + + #[cfg(any(unix, windows))] + #[test] + fn file_root_path_encoding_rejects_explicit_output_instead_of_fallback() { + let path = non_utf8_canonical_path(); + let error = optional_file_path_as_str(Some(&path), "--output").unwrap_err(); + assert!(matches!(error, GwsError::Validation(_))); + let message = error.to_string(); + assert!(message.contains("--output"), "{message}"); + assert!(message.contains("UTF-8"), "{message}"); + } + + #[cfg(any(unix, windows))] + #[test] + fn file_root_path_encoding_rejects_explicit_upload_instead_of_omitting_it() { + let path = non_utf8_canonical_path(); + let error = optional_file_path_as_str(Some(&path), "--upload").unwrap_err(); + assert!(matches!(error, GwsError::Validation(_))); + let message = error.to_string(); + assert!(message.contains("--upload"), "{message}"); + assert!(message.contains("UTF-8"), "{message}"); + } + #[test] fn test_parse_pagination_config_defaults() { let matches = clap::Command::new("test") diff --git a/crates/google-workspace-cli/tests/file_roots.rs b/crates/google-workspace-cli/tests/file_roots.rs index 614256605..9dc516821 100644 --- a/crates/google-workspace-cli/tests/file_roots.rs +++ b/crates/google-workspace-cli/tests/file_roots.rs @@ -208,3 +208,67 @@ fn file_root_cli_download_propagates_canonical_output() { ); assert!(!fixture.cwd.join("output.bin").exists()); } + +// macOS filesystems commonly reject invalid UTF-8 directory names. The CLI +// conversion itself is covered without filesystem access by main.rs unit tests; +// these Linux regressions exercise the complete canonical symlink handoff. +#[cfg(target_os = "linux")] +fn non_utf8_alias(fixture: &Fixture, filename: &str) -> PathBuf { + use std::os::unix::{ffi::OsStringExt, fs::symlink}; + let target = fixture + .root + .join(std::ffi::OsString::from_vec(b"bytes-\xff".to_vec())); + fs::create_dir(&target).unwrap(); + fs::write(target.join("upload.txt"), b"synthetic upload").unwrap(); + let alias = fixture.root.join("alias"); + symlink(&target, &alias).unwrap(); + alias.join(filename) +} + +#[cfg(target_os = "linux")] +#[test] +fn file_root_cli_rejects_non_utf8_output_without_fallback_write() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + let output_path = non_utf8_alias(&fixture, "new.bin"); + let fallback = fixture.cwd.join("download.bin"); + fs::write(&fallback, b"keep existing download").unwrap(); + // A non-dry run must fail at validation, before HTTP or the default output + // can be selected. No live service or credentials are involved. + let output = fixture + .command(true) + .args(["drive", "files", "get", "--output"]) + .arg(&output_path) + .output() + .unwrap(); + assert_eq!(fs::read(&fallback).unwrap(), b"keep existing download"); + assert!(!output_path.exists()); + assert_eq!( + output.status.code(), + Some(3), + "{}", + String::from_utf8_lossy(&output.stdout) + ); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + let message = error["error"]["message"].as_str().unwrap(); + assert!(message.contains("--output"), "{message}"); + assert!(message.contains("UTF-8"), "{message}"); +} + +#[cfg(target_os = "linux")] +#[test] +fn file_root_cli_rejects_non_utf8_upload_instead_of_omitting_it() { + let fixture = Fixture::new("http://127.0.0.1:1/"); + let upload_path = non_utf8_alias(&fixture, "upload.txt"); + let output = fixture.dry_run("--upload", &upload_path, true); + assert_eq!( + output.status.code(), + Some(3), + "{}", + String::from_utf8_lossy(&output.stdout) + ); + let error: Value = serde_json::from_slice(&output.stdout).unwrap(); + let message = error["error"]["message"].as_str().unwrap(); + assert!(message.contains("--upload"), "{message}"); + assert!(message.contains("UTF-8"), "{message}"); + assert_eq!(fs::read(&upload_path).unwrap(), b"synthetic upload"); +} diff --git a/crates/google-workspace/src/validate.rs b/crates/google-workspace/src/validate.rs index 491c5d33b..a951cae56 100644 --- a/crates/google-workspace/src/validate.rs +++ b/crates/google-workspace/src/validate.rs @@ -844,11 +844,9 @@ mod tests { let canonical_dir = dir.path().canonicalize().unwrap(); fs::write(canonical_dir.join("test.txt"), "data").unwrap(); - let saved_cwd = std::env::current_dir().unwrap(); - std::env::set_current_dir(&canonical_dir).unwrap(); + let _environment = FilePathEnvironment::set(&canonical_dir, None); let result = validate_safe_file_path("test.txt", "--upload"); - std::env::set_current_dir(&saved_cwd).unwrap(); assert!(result.is_ok(), "expected Ok, got: {result:?}"); } @@ -859,11 +857,9 @@ mod tests { let dir = tempdir().unwrap(); let canonical_dir = dir.path().canonicalize().unwrap(); - let saved_cwd = std::env::current_dir().unwrap(); - std::env::set_current_dir(&canonical_dir).unwrap(); + let _environment = FilePathEnvironment::set(&canonical_dir, None); let result = validate_safe_file_path("../../etc/passwd", "--upload"); - std::env::set_current_dir(&saved_cwd).unwrap(); assert!(result.is_err(), "path traversal should be rejected"); assert!( @@ -873,7 +869,10 @@ mod tests { } #[test] + #[serial] fn test_file_path_rejects_control_chars() { + let dir = tempdir().unwrap(); + let _environment = FilePathEnvironment::set(dir.path(), None); let result = validate_safe_file_path("file\x00.txt", "--output"); assert!(result.is_err(), "null bytes should be rejected"); } @@ -889,11 +888,9 @@ mod tests { let link_path = canonical_dir.join("escape"); std::os::unix::fs::symlink("/tmp", &link_path).unwrap(); - let saved_cwd = std::env::current_dir().unwrap(); - std::env::set_current_dir(&canonical_dir).unwrap(); + let _environment = FilePathEnvironment::set(&canonical_dir, None); let result = validate_safe_file_path("escape/secret.txt", "--output"); - std::env::set_current_dir(&saved_cwd).unwrap(); assert!(result.is_err(), "symlink escape should be rejected"); } @@ -905,11 +902,9 @@ mod tests { let dir = tempdir().unwrap(); let canonical_dir = dir.path().canonicalize().unwrap(); - let saved_cwd = std::env::current_dir().unwrap(); - std::env::set_current_dir(&canonical_dir).unwrap(); + let _environment = FilePathEnvironment::set(&canonical_dir, None); let result = validate_safe_file_path("doesnt_exist/../../etc/passwd", "--output"); - std::env::set_current_dir(&saved_cwd).unwrap(); assert!( result.is_err(), @@ -917,8 +912,8 @@ mod tests { ); } - // Scoped file-root acceptance tests. Only directory-scope checks mutate env; - // policy tests pass CWD and the trusted root explicitly. + // Default public-validator and directory-scope tests isolate global state; + // scoped file-policy tests pass CWD and the trusted root explicitly. struct FilePathEnvironment { cwd: PathBuf, root: Option, @@ -1012,8 +1007,12 @@ mod tests { #[test] fn file_root_rejects_sibling_even_with_shared_name_prefix() { let dir = tempdir().unwrap(); - let root = dir.path().join("allowed"); - let sibling = dir.path().join("allowed-sibling"); + // A literal backslash on Unix, a separator on Windows; both must be + // compared using the escaped canonical representation in diagnostics. + let parent = dir.path().join(r"back\slash"); + fs::create_dir_all(&parent).unwrap(); + let root = parent.join("allowed"); + let sibling = parent.join("allowed-sibling"); fs::create_dir(&root).unwrap(); fs::create_dir(&sibling).unwrap(); let err = file_path_under_root( @@ -1026,7 +1025,10 @@ mod tests { .to_string(); assert!(err.contains("outside"), "{err}"); assert!(err.contains("GOOGLE_WORKSPACE_CLI_FILE_ROOT"), "{err}"); - assert!(err.contains(root.to_str().unwrap()), "{err}"); + assert!( + err.contains(&format!("{:?}", root.canonicalize().unwrap())), + "{err}" + ); } #[test] From 5685fe5c236cc16f92ccd0e967d6e18cdae38875 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:02:08 -0300 Subject: [PATCH 11/19] feat(docs): add revision-bound reviewed text patch workflow --- .changeset/docs-review-workflow.md | 7 + .github/workflows/ci.yml | 15 + examples/docs-review/README.md | 161 ++++++ examples/docs-review/docs_review.py | 595 ++++++++++++++++++++++ examples/docs-review/test_docs_review.py | 608 +++++++++++++++++++++++ 5 files changed, 1386 insertions(+) create mode 100644 .changeset/docs-review-workflow.md create mode 100644 examples/docs-review/README.md create mode 100755 examples/docs-review/docs_review.py create mode 100644 examples/docs-review/test_docs_review.py diff --git a/.changeset/docs-review-workflow.md b/.changeset/docs-review-workflow.md new file mode 100644 index 000000000..0c3060012 --- /dev/null +++ b/.changeset/docs-review-workflow.md @@ -0,0 +1,7 @@ +--- +"@googleworkspace/cli": minor +--- + +Add a standalone Python Docs review example that plans one literal text replacement, +binds it to a source revision and tab, applies it through existing gws commands, +and verifies the result without retrying ambiguous writes. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 1eedc9972..f9417abd9 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -30,6 +30,21 @@ env: SCCACHE_IGNORE_SERVER_IO_ERROR: "true" jobs: + docs-review: + name: Docs Review Python Example + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [ubuntu-latest, macos-latest] + env: + PYTHONDONTWRITEBYTECODE: "1" + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4 + - name: Test with synthetic fixtures and stub gws + run: | + python3 --version + python3 -m unittest discover -s examples/docs-review -p 'test_*.py' -v + changes: name: Detect Changes runs-on: ubuntu-latest diff --git a/examples/docs-review/README.md b/examples/docs-review/README.md new file mode 100644 index 000000000..61d595e76 --- /dev/null +++ b/examples/docs-review/README.md @@ -0,0 +1,161 @@ +# Review one Google Docs text patch + +`docs_review.py` is a standalone Python standard-library companion to `gws`. +It creates a reviewable plan, checks that the source has not changed, applies +one literal replacement, and verifies the result. It adds no authentication +implementation or Rust commands and needs no other examples. + +Requirements: Python 3.10+ on a POSIX system supporting `O_NOFOLLOW` and +directory-relative file access (Linux/macOS), with `gws` on `PATH` for live +reads/apply. Existing `gws` authentication and Model Armor environment settings +are inherited. Offline preview and tests need neither credentials nor `gws`. + +## Workflow + +All file arguments must be relative to your current working directory, with +existing parent directories. Absolute paths, `..`, symlinks (including inward +symlinks), devices and overwriting an existing plan are rejected. Plan files +are created with mode `0600`. Keep both the plan and its digest from review. + +These commands use synthetic placeholder IDs/text. The `plan` command performs +one document read, including every tab and inline suggestions. It never sends +a document mutation. Replace the placeholder IDs only when working on a +document you are authorized to edit. + +```sh +# From the repository root. printf deliberately does not append a newline. +printf '%s' 'world' > find.txt +printf '%s' 'reader' > replacement.txt + +python3 examples/docs-review/docs_review.py plan \ + --document SYNTHETIC_DOCUMENT_ID --tab t.synthetic \ + --find find.txt --replacement replacement.txt --out reviewed-plan.json + +# Inspect the full JSON plan: IDs, revision, literal text, diff, target, digest. +cat reviewed-plan.json + +# Local request/diff preview: validates the plan, calls no gws, writes no files. +python3 examples/docs-review/docs_review.py apply \ + --plan reviewed-plan.json --dry-run + +# After review: reread, compare, submit once, reread and verify. +python3 examples/docs-review/docs_review.py apply --plan reviewed-plan.json +``` + +Omit `--tab` only for a document containing exactly one tab. Child tabs count; +titles are not selectors. Use a new `--out` path when regenerating a plan. +An empty replacement file deletes the matched text. Empty finds and no-op +replacements are rejected. Do not use `echo` to prepare text files: it commonly +adds a newline, which this workflow deliberately rejects. + +The versioned plan contains the document/tab IDs, exact source revision, source +fingerprint, find/replacement text, expected count `1`, UTF-16 target offsets, +diff, and deterministic SHA-256 digest. It does not contain document titles, +the complete source, surrounding text, credentials, or command strings. +The diff is the exact removed/inserted text, not a full-document preview. +Plans still contain sensitive text if your inputs do; treat them accordingly. + +The digest detects accidental edits and binds the review fields together; it +is **not a signature or an authorization mechanism**. Anyone who can replace +the entire plan can recompute it. Protect the reviewed file and compare its +digest with your separately retained review record before applying. + +## Supported text and verification + +- Exactly one case-sensitive literal occurrence in one tab. Regex is disabled. + Overlapping occurrences also count as ambiguous. +- The match must fit inside one top-level body paragraph. It may cross text + style runs. Unicode offsets use UTF-16 code units, including emoji. +- Tables, images, headers, footers, footnotes, other paragraphs and other tabs + remain in the document. Text in non-body regions still counts toward + uniqueness: a duplicate in a header or table causes refusal. +- Apply reconstructs the plan from the fresh source and compares every field. + A changed revision, source, target or digest stops before submission. +- Exactly one `replaceAllText` request is sent with + `writeControl.requiredRevisionId` and `tabsCriteria.tabIds`. No block deletion, + full-document reupload, or arbitrary request from a plan is allowed. +- Success requires `occurrencesChanged == 1`, a new returned revision, and a + reread at that revision. The resulting text, paragraph/structure metadata, + shifted body indices, image metadata, other tabs, and untouched character + formatting must match the expected result. + +Google controls formatting inheritance **inside the replaced span**. This +example does not set or promise the replacement's character styles. It checks +the replacement text and all formatting outside that span, accepting changes +in text-run splitting that leave character formatting unchanged. Temporary +image `contentUri` values and gws `_sanitization` annotations are excluded from +fingerprints; other image/style metadata remains checked. The verification is +an API snapshot, not a visual rendering or a guarantee against later edits. + +## Deliberate limitations + +This is a text patch workflow, not a structural document editor. Paragraph +breaks, tabs/control characters, private-use characters and object markers +are rejected in find/replacement files. Moving blocks, inserting tables/images, +editing across images, and targets inside tables, tables of contents, headers, +footers or footnotes are unsupported. Structural requests cannot be supplied +through the plan. + +The selected tab must have no unresolved suggestions, named ranges or +bookmarks. Unknown tab regions, unknown structural blocks, equations, +automatic text, rich links and smart chips in the selected tab are refused. +These strict limits avoid interpreting inaccessible text, unstable anchors, +or unsupported element boundaries as a safe replacement. Other tabs are +fingerprinted and verified, not edited. New API structures may require an +explicit compatibility update. + +Each text input is limited to 16 KiB, the plan to 1 MiB, and each gws response +to 16 MiB. Source verification additionally limits total text to 250,000 +characters and its estimated expanded style representation to 16 MiB. +Oversized or malformed inputs fail closed. `--timeout SECONDS` bounds each +gws subprocess (default 60, maximum 600); there are no mutation retries. + +## Outcomes and recovery + +Commands emit JSON. `plan`, offline preview, and a verified apply exit `0` +with status `planned`, `preview`, or `applied`. Preview only validates local +data: it cannot establish that a live revision is still current. + +Exit `2` / `refused` means the companion did not submit a document write. +Correct the input or reread the source and review a **new** plan. + +Exit `3` / `ambiguous` means a write **may have applied**, or its result could +not be verified. This includes subprocess timeouts, nonzero write exits +(including a concurrent API 400), malformed responses, missing revisions, +unexpected reply counts, and failed/mismatching rereads. The original plan +stays unchanged. Inspect the document and revision through your normal tools; +do not blindly rerun apply. Keep the failed plan as your review record and +create a new plan if further work is required. The companion does not persist +a separate attempt journal or prevent an operator from manually rerunning it. + +All subprocesses use argument arrays, checked exit codes and timeouts. +Raw gws output is not echoed into errors. Diagnostics produce a fixed, +redacted notice; inspect your gws/Model Armor configuration when it appears. +Model Armor block outcomes stop the workflow; settings are never disabled. +This example uses raw `gws docs documents` methods, whose executor at the +accompanying source revision does not retry requests. + +## Tests + +```sh +python3 -m unittest discover -s examples/docs-review -p 'test_*.py' -v +``` + +The suite executes the real companion CLI with a temporary stub `gws` and +synthetic documents. It passes an isolated environment, never reads real +credentials, and never contacts Google. CI runs it independently of Rust +changes on Linux and macOS. Tests cover the request contract, no-write +planning/preview, revision/source checks, Unicode/tabs, paths and plan +tampering, structures/styles, reply counts, ambiguity and plan retention. + +## API references + +- [Documents.get](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents/get): + `includeTabsContent` and `suggestionsViewMode`. +- [Document structure](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents): + indices are measured in UTF-16 code units; revisions are opaque. +- [ReplaceAllTextRequest](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents/request#ReplaceAllTextRequest): + literal matching and tab criteria. +- [BatchUpdate / WriteControl](https://developers.google.com/workspace/docs/api/reference/rest/v1/documents/batchUpdate): + stale `requiredRevisionId` requests are rejected with HTTP 400. A required + revision in the response identifies the revision after application. diff --git a/examples/docs-review/docs_review.py b/examples/docs-review/docs_review.py new file mode 100755 index 000000000..27a1700f6 --- /dev/null +++ b/examples/docs-review/docs_review.py @@ -0,0 +1,595 @@ +#!/usr/bin/env python3 +# Copyright 2026 Google LLC +# SPDX-License-Identifier: Apache-2.0 +"""Review and apply one revision-bound Google Docs text replacement (stdlib). + +Only the gws executable communicates with Google. Plans contain data, never +commands. See README.md for the intentionally limited structural support. +""" + +import argparse +from contextlib import contextmanager +import difflib +import hashlib +import hmac +import json +import math +import os +import re +import stat +import subprocess +import sys +import tempfile + + +MAX_PLAN = 1024 * 1024 +MAX_TEXT = 16 * 1024 +MAX_DOCUMENT = 16 * 1024 * 1024 +PLAN_KEYS = { + "version", "document_id", "tab_id", "revision_id", "source_sha256", + "find", "replacement", "expected_occurrences", "target", "diff", "digest", +} + + +class Refusal(Exception): + """A fixed, safe message; never construct one from subprocess output.""" + + +class Ambiguous(Refusal): + """A submission may have applied; never retry automatically.""" + + +def require(condition, message): + if not condition: + raise Refusal(message) + + +def canonical(value): + return json.dumps(value, sort_keys=True, separators=(",", ":"), + ensure_ascii=True, allow_nan=False).encode("utf-8") + + +def sha256(value): + return hashlib.sha256(canonical(value)).hexdigest() + + +def utf16(text): + return len(text.encode("utf-16-le")) // 2 + + +def unique_object(pairs): + result = {} + for key, value in pairs: + require(key not in result, "Duplicate JSON keys are not supported.") + result[key] = value + return result + + +def parse_json(data): + try: + return json.loads(data, object_pairs_hook=unique_object, + parse_constant=lambda _: invalid_json()) + except (ValueError, UnicodeError, RecursionError): + raise Refusal("Invalid UTF-8 JSON; regenerate the plan or check gws.") from None + + +def invalid_json(): + raise Refusal("Non-finite JSON numbers are not supported.") + + +def identifier(value, tab=False): + pattern = r"[A-Za-z0-9_.-]{1,200}" if tab else r"[A-Za-z0-9_-]{1,200}" + require(isinstance(value, str) and re.fullmatch(pattern, value) is not None + and ".." not in value, "Invalid document or tab ID; supply an ID, not a URL.") + return value + + +def revision(value): + require(isinstance(value, str) and 0 < len(value) <= 4096 + and not any(ord(c) < 32 or 127 <= ord(c) <= 159 for c in value), + "Missing or invalid revision; read an editable document again.") + return value + + +def text_input(value, allow_empty=False): + require(isinstance(value, str), "Find and replacement must be UTF-8 text.") + require(allow_empty or bool(value), "Find text must not be empty.") + # Docs can strip these or treat them as structural edits. Reject rather + # than silently submit a replacement different from the reviewed text. + require(not any( + ord(c) < 32 or 127 <= ord(c) <= 159 + or 0xD800 <= ord(c) <= 0xF8FF or 0xFFF9 <= ord(c) <= 0xFFFF + or ord(c) in (0x2028, 0x2029) + or 0xF0000 <= ord(c) <= 0xFFFFD + or 0x100000 <= ord(c) <= 0x10FFFD + for c in value + ), "Unsupported structural/control character; use single-paragraph plain text.") + require(len(value.encode("utf-8")) <= MAX_TEXT, "Text input exceeds 16 KiB.") + return value + + +@contextmanager +def confined_parent(path): + """Hold directory descriptors so symlink swaps cannot redirect file I/O. + + All symlinks (even inward ones) are rejected. POSIX openat/O_NOFOLLOW is + required; fail closed on platforms without it. + """ + require(isinstance(path, str) and path and not path.startswith("/") + and "\\" not in path and ":" not in path + and not any(ord(c) < 32 or 127 <= ord(c) <= 159 for c in path), + "Use a relative path within the current directory.") + parts = path.split("/") + require(all(p not in ("", ".", "..") for p in parts), + "Path traversal and empty path components are not allowed.") + require(hasattr(os, "O_NOFOLLOW") and os.open in os.supports_dir_fd, + "Safe file access requires a POSIX platform with O_NOFOLLOW.") + descriptors = [] + try: + flags = os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW + descriptors.append(os.open(".", flags)) + for part in parts[:-1]: + descriptors.append(os.open(part, flags, dir_fd=descriptors[-1])) + yield descriptors[-1], parts[-1] + except OSError: + raise Refusal( + "Cannot access path safely; check parents, symlinks and permissions." + ) from None + finally: + for descriptor in reversed(descriptors): + os.close(descriptor) + + +def read_file(path, limit): + with confined_parent(path) as (parent, name): + descriptor = os.open(name, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK, + dir_fd=parent) + with os.fdopen(descriptor, "rb") as stream: + require(stat.S_ISREG(os.fstat(stream.fileno()).st_mode), + "Input must be a regular file.") + data = stream.read(limit + 1) + require(len(data) <= limit, "Input file exceeds the supported size limit.") + try: + return data.decode("utf-8") + except UnicodeError: + raise Refusal("Input file must be valid UTF-8.") from None + + +def unused_output(parent, name): + try: + os.stat(name, dir_fd=parent, follow_symlinks=False) + except FileNotFoundError: + return + raise Refusal("Output already exists; choose a new plan path.") + + +def write_plan(parent, name, plan): + data = json.dumps(plan, ensure_ascii=True, sort_keys=True, indent=2) + "\n" + require(len(data.encode()) <= MAX_PLAN, "Plan exceeds 1 MiB; use a smaller patch.") + # Exclusive creation prevents overwrites/hardlink attacks; plans are private. + descriptor = os.open(name, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, dir_fd=parent) + with os.fdopen(descriptor, "w", encoding="utf-8") as stream: + stream.write(data) + stream.flush() + os.fsync(stream.fileno()) + + +class Gws: + def __init__(self, timeout): + self.timeout = timeout + self.diagnostics = False + + def call(self, method, document_id, request=None): + params = {"documentId": document_id} + if method == "get": + params.update(includeTabsContent=True, suggestionsViewMode="SUGGESTIONS_INLINE") + args = ["gws", "docs", "documents", method, + "--params", canonical(params).decode(), "--format", "json"] + if request is not None: + args += ["--json", canonical(request).decode()] + try: + # Inherit gws auth/Model Armor settings. Raw diagnostics never reach + # the terminal (they may contain document text or credentials). + with tempfile.TemporaryFile() as output, tempfile.TemporaryFile() as diagnostics: + result = subprocess.run( + args, stdin=subprocess.DEVNULL, stdout=output, + stderr=diagnostics, timeout=self.timeout, check=False, + ) + self.diagnostics |= diagnostics.tell() > 0 + require(result.returncode == 0, + "gws failed; check access, revision and Model Armor settings.") + output.seek(0) + data = output.read(MAX_DOCUMENT + 1) + require(len(data) <= MAX_DOCUMENT, "gws response exceeds 16 MiB.") + value = parse_json(data) + require(isinstance(value, dict) and "error" not in value, + "gws did not return a successful JSON object.") + return value + except subprocess.TimeoutExpired: + raise Refusal("gws timed out; check connectivity and timeout settings.") from None + except OSError: + raise Refusal("Cannot execute gws; check installation and PATH.") from None + + def get(self, document_id): + return self.call("get", document_id) + + +def walk(value, path=()): + if isinstance(value, dict): + yield path, value + for key, item in value.items(): + yield from walk(item, path + (key,)) + elif isinstance(value, list): + for index, item in enumerate(value): + yield from walk(item, path + (index,)) + + +def at(value, path): + for key in path: + value = value[key] + return value + + +def select_tab(document, document_id, tab_id): + require(isinstance(document, dict) and document.get("documentId") == document_id, + "Document identity mismatch.") + revision(document.get("revisionId")) + require(document.get("suggestionsViewMode") == "SUGGESTIONS_INLINE", + "Expected suggestions-inline source; refusing an incomplete view.") + require(isinstance(document.get("tabs"), list) and document["tabs"], + "Missing all-tabs content; check gws Docs discovery support.") + tabs = [] + + def visit(items, path): + require(isinstance(items, list), "Malformed document tabs.") + for i, item in enumerate(items): + require(isinstance(item, dict) + and isinstance(item.get("tabProperties"), dict) + and isinstance(item.get("documentTab"), dict), "Unsupported tab shape.") + identity = identifier(item["tabProperties"].get("tabId"), tab=True) + tabs.append((identity, path + (i, "documentTab"))) + if "childTabs" in item: + visit(item["childTabs"], path + (i, "childTabs")) + + visit(document["tabs"], ("tabs",)) + require(len({identity for identity, _ in tabs}) == len(tabs), "Duplicate tab IDs.") + if tab_id is None: + require(len(tabs) == 1, "Multiple tabs; select exactly one with --tab ID.") + tab_id = tabs[0][0] + matches = [path for identity, path in tabs if identity == tab_id] + require(len(matches) == 1, "Selected tab does not exist.") + return tab_id, matches[0] + + +def check_supported(tab): + require(isinstance(tab.get("body"), dict) + and isinstance(tab["body"].get("content"), list), "Tab body is missing.") + require(set(tab) <= { + "body", "headers", "footers", "footnotes", "documentStyle", "namedStyles", + "lists", "namedRanges", "inlineObjects", "positionedObjects", "bookmarks", + "suggestedDocumentStyleChanges", "suggestedNamedStylesChanges", + }, "Unsupported tab region; this example requires a known document structure.") + for _, obj in walk(tab): + require(not any(k.startswith("suggested") and v for k, v in obj.items()), + "Suggested content in selected tab is unsupported; resolve suggestions first.") + require(not obj.get("namedRanges") and not obj.get("bookmarks"), + "Named ranges and bookmarks in the selected tab are unsupported.") + if "content" in obj and isinstance(obj["content"], list): + for block in obj["content"]: + require(isinstance(block, dict) + and len(set(block) - {"startIndex", "endIndex"}) == 1 + and len(set(block) & { + "paragraph", "sectionBreak", "table", "tableOfContents"}) == 1, + "Unsupported document structure.") + if "paragraph" not in obj: + continue + paragraph = obj["paragraph"] + require(isinstance(paragraph, dict) + and isinstance(paragraph.get("elements"), list), "Malformed paragraph.") + cursor = obj.get("startIndex", 0) + require(type(cursor) is int and cursor >= 0, "Invalid paragraph index.") + for element in paragraph["elements"]: + require(isinstance(element, dict) + and type(element.get("startIndex", 0)) is int + and type(element.get("endIndex")) is int + and element.get("startIndex", 0) == cursor + and element["endIndex"] > cursor, "Non-contiguous paragraph indices.") + kinds = set(element) - {"startIndex", "endIndex"} + require(len(kinds) == 1 and kinds <= { + "textRun", "inlineObjectElement", "footnoteReference", + "horizontalRule", "pageBreak", "columnBreak"}, + "Unsupported paragraph element; rich links, equations and chips are excluded.") + if "textRun" in element: + run = element["textRun"] + require(isinstance(run, dict) and isinstance(run.get("content"), str), + "Malformed text run.") + require(utf16(run["content"]) == element["endIndex"] - cursor, + "Text run does not match its UTF-16 indices.") + cursor = element["endIndex"] + require(obj.get("endIndex") == cursor, "Paragraph end index mismatch.") + + +def check_document_budget(document): + """Bound character/style expansion before constructing canonical atoms.""" + expanded_bytes = 0 + characters = 0 + for _, obj in walk(document): + if "textRun" not in obj: + continue + run = obj["textRun"] + require(isinstance(run, dict) and isinstance(run.get("content"), str), + "Malformed text run.") + length = len(run["content"]) + characters += length + metadata = {k: v for k, v in run.items() if k != "content"} + expanded_bytes += length * (len(canonical(metadata)) + 64) + require(characters <= 250000 and expanded_bytes <= MAX_DOCUMENT, + "Document is too large for safe text/style verification.") + + +def normalized(value, path=()): + """Canonical semantic shape; text-run splitting is not a style change.""" + if isinstance(value, list): + return [normalized(v, path + (i,)) for i, v in enumerate(value)] + if not isinstance(value, dict): + return value + result = {} + for key, item in value.items(): + if not path and key in ("revisionId", "_sanitization"): + continue + # Google refreshes this temporary URL on reads; keep all other image + # metadata, including sourceUri, object IDs, dimensions and crop data. + if key == "contentUri" and path and path[-1] == "imageProperties": + continue + if key == "elements" and path and path[-1] == "paragraph": + elements = [] + for element in item: + if "textRun" in element: + run = element["textRun"] + metadata = {k: v for k, v in run.items() if k != "content"} + metadata.setdefault("textStyle", {}) + for char in run["content"]: + elements.append({"text": char, "format": metadata}) + else: + elements.append(normalized(element, path + (key,))) + result[key] = elements + else: + result[key] = normalized(item, path + (key,)) + return result + + +def locate(document, tab_path, find): + """Count overlapping occurrences in every text segment of the chosen tab.""" + tab = at(document, tab_path) + matches = [] + for path, block in walk(tab): + if "paragraph" not in block: + continue + # Only top-level body paragraphs are editable. Tables, TOCs, headers, + # footers and footnotes still participate in ambiguity detection. + supported = (len(path) == 3 and path[:2] == ("body", "content") + and isinstance(path[2], int)) + cursor = block.get("startIndex", 0) + tokens = [] + indices = [] + for element in block["paragraph"]["elements"]: + if "textRun" in element: + for char in element["textRun"]["content"]: + tokens.append(char) + indices.append(cursor) + cursor += utf16(char) + else: + tokens.append("\ufffc") + indices.append(cursor) + cursor = element["endIndex"] + text = "".join(tokens) + offset = text.find(find) + while offset != -1: + matches.append((supported, tab_path + path + ("paragraph", "elements"), + offset, indices[offset])) + offset = text.find(find, offset + 1) + require(len(matches) == 1, + "Expected exactly one match in the selected tab; found zero or multiple.") + supported, path, offset, start = matches[0] + require(supported, "Target is outside a supported body paragraph.") + return path, offset, start + + +def review_diff(find, replacement): + return "".join(difflib.unified_diff( + [find + "\n"], [replacement + "\n"], fromfile="before", tofile="after", + )) + + +def build_plan(document, document_id, tab_id, find, replacement): + identifier(document_id) + text_input(find) + text_input(replacement, allow_empty=True) + require(find != replacement, "No-op replacement; choose different text.") + tab_id, tab_path = select_tab(document, document_id, tab_id) + check_supported(at(document, tab_path)) + check_document_budget(document) + _, _, start = locate(document, tab_path, find) + result = { + "version": 1, "document_id": document_id, "tab_id": tab_id, + "revision_id": document["revisionId"], "source_sha256": sha256(normalized(document)), + "find": find, "replacement": replacement, "expected_occurrences": 1, + "target": {"start_index": start, "end_index": start + utf16(find)}, + "diff": review_diff(find, replacement), + } + result["digest"] = sha256(result) + return result + + +def validate_plan(plan): + require(isinstance(plan, dict) and set(plan) == PLAN_KEYS, "Unsupported plan schema.") + require(type(plan["version"]) is int and plan["version"] == 1, + "Unsupported plan version.") + require(type(plan["expected_occurrences"]) is int and plan["expected_occurrences"] == 1, + "Plan must specify exactly one replacement.") + identifier(plan["document_id"]) + identifier(plan["tab_id"], tab=True) + revision(plan["revision_id"]) + text_input(plan["find"]) + text_input(plan["replacement"], allow_empty=True) + require(plan["find"] != plan["replacement"], "No-op replacement.") + target = plan["target"] + require(isinstance(target, dict) and set(target) == {"start_index", "end_index"} + and all(type(v) is int and v >= 1 for v in target.values()) + and target["end_index"] - target["start_index"] == utf16(plan["find"]), + "Invalid UTF-16 target range.") + require(plan["diff"] == review_diff(plan["find"], plan["replacement"]), + "Plan diff does not match its replacement.") + for field in ("digest", "source_sha256"): + require(isinstance(plan[field], str) and re.fullmatch(r"[0-9a-f]{64}", plan[field]), + "Invalid SHA-256 field.") + payload = {k: v for k, v in plan.items() if k != "digest"} + require(hmac.compare_digest(plan["digest"], sha256(payload)), + "Plan digest mismatch; regenerate and review a fresh plan.") + return plan + + +def request_body(plan): + return { + "writeControl": {"requiredRevisionId": plan["revision_id"]}, + "requests": [{"replaceAllText": { + "containsText": {"text": plan["find"], "matchCase": True, "searchByRegex": False}, + "replaceText": plan["replacement"], + "tabsCriteria": {"tabIds": [plan["tab_id"]]}, + }}], + } + + +def verify(before, after, plan, write_revision): + _, tab_path = select_tab(before, plan["document_id"], plan["tab_id"]) + select_tab(after, plan["document_id"], plan["tab_id"]) + require(after["revisionId"] == write_revision, "Post-write revision mismatch.") + check_supported(at(after, tab_path)) + check_document_budget(after) + path, offset, _ = locate(before, tab_path, plan["find"]) + expected, actual = normalized(before), normalized(after) + end = plan["target"]["end_index"] + delta = utf16(plan["replacement"]) - utf16(plan["find"]) + # Indices in the body shift; headers/footers/footnotes have separate indices. + for _, obj in walk(at(expected, tab_path)["body"]): + for key in ("startIndex", "endIndex"): + if key in obj and obj[key] >= end: + obj[key] += delta + replacement = [{"text": c} for c in plan["replacement"]] + expected_elements = at(expected, path) + expected_elements[offset:offset + len(plan["find"])] = replacement + actual_elements = at(actual, path) + # Formatting within newly inserted text is Google-controlled. Verify its + # exact characters but compare formatting of every untouched character. + for i in range(offset, offset + len(replacement)): + require(i < len(actual_elements) and "text" in actual_elements[i], + "Replacement missing from expected location.") + actual_elements[i] = {"text": actual_elements[i]["text"]} + require(actual == expected, "Post-write text, structure or untouched style mismatch.") + + +def apply_plan(plan, gws): + before = gws.get(plan["document_id"]) + rebuilt = build_plan(before, plan["document_id"], plan["tab_id"], + plan["find"], plan["replacement"]) + require(rebuilt == plan, + "Source revision or content changed; regenerate and review a new plan.") + try: + reply = gws.call("batchUpdate", plan["document_id"], request_body(plan)) + require(reply.get("documentId") == plan["document_id"], "Write identity mismatch.") + replies = reply.get("replies") + require(isinstance(replies, list) and len(replies) == 1 + and isinstance(replies[0], dict), "Missing replacement reply.") + change = replies[0].get("replaceAllText") + require(isinstance(change, dict) and type(change.get("occurrencesChanged")) is int + and change["occurrencesChanged"] == 1, "Replacement count was not exactly one.") + control = reply.get("writeControl") + require(isinstance(control, dict), "Missing write revision.") + write_revision = revision(control.get("requiredRevisionId")) + require(write_revision != plan["revision_id"], "Write revision did not advance.") + after = gws.get(plan["document_id"]) + verify(before, after, plan, write_revision) + except (Refusal, OSError, ValueError, KeyError, TypeError, IndexError, RecursionError, + KeyboardInterrupt): + raise Ambiguous( + "Write may have applied; verification did not establish success. " + "Keep the original plan, inspect the document and revision, and do not " + "blindly retry. A concurrent revision rejection requires a newly reviewed plan." + ) from None + return {"status": "applied", "digest": plan["digest"], "revision_id": write_revision} + + +class Parser(argparse.ArgumentParser): + def error(self, message): + raise Refusal("Invalid arguments; use --help for supported options.") + + +def main(argv=None): + parser = Parser(description=__doc__) + commands = parser.add_subparsers(dest="command", required=True) + plan_parser = commands.add_parser("plan", help="Read a document and create a reviewable plan") + plan_parser.add_argument("--document", required=True) + plan_parser.add_argument( + "--tab", help="Exact tab ID (required when there is more than one tab)") + plan_parser.add_argument("--find", required=True, help="Relative UTF-8 input file") + plan_parser.add_argument("--replacement", required=True, help="Relative UTF-8 input file") + plan_parser.add_argument("--out", required=True, help="New relative plan file") + apply_parser = commands.add_parser("apply", help="Validate, apply once, and verify") + apply_parser.add_argument("--plan", required=True) + apply_parser.add_argument("--dry-run", action="store_true", + help="Offline plan validation and request preview; no gws calls") + for command in (plan_parser, apply_parser): + command.add_argument("--timeout", type=float, default=60, + help="Per-gws-call timeout in seconds (default: 60)") + try: + args = parser.parse_args(argv) + require(math.isfinite(args.timeout) and 0 < args.timeout <= 600, + "Timeout must be greater than zero and at most 600 seconds.") + gws = Gws(args.timeout) + if args.command == "plan": + identifier(args.document) + if args.tab is not None: + identifier(args.tab, tab=True) + find = text_input(read_file(args.find, MAX_TEXT)) + replacement = text_input(read_file(args.replacement, MAX_TEXT), allow_empty=True) + with confined_parent(args.out) as (parent, name): + unused_output(parent, name) + plan = build_plan( + gws.get(args.document), args.document, args.tab, find, replacement) + write_plan(parent, name, plan) + result = {"status": "planned", "digest": plan["digest"], + "message": "Review the plan and digest before applying."} + else: + plan = validate_plan(parse_json(read_file(args.plan, MAX_PLAN))) + if args.dry_run: + result = {"status": "preview", "digest": plan["digest"], + "diff": plan["diff"], "request": request_body(plan)} + else: + result = apply_plan(plan, gws) + if gws.diagnostics: + result["gws_diagnostics"] = ( + "gws reported diagnostics; check gws and Model Armor settings. " + "Raw output was suppressed to protect document content." + ) + print(json.dumps(result, ensure_ascii=True, sort_keys=True)) + return 0 + except Ambiguous as error: + print(json.dumps({"status": "ambiguous", "message": str(error)}), file=sys.stderr) + return 3 + except Refusal as error: + print(json.dumps({"status": "refused", "message": str(error)}), file=sys.stderr) + return 2 + except (OSError, ValueError, KeyError, TypeError, IndexError, RecursionError): + print(json.dumps({"status": "refused", "message": + "Malformed source or inaccessible file; check inputs and regenerate."}), + file=sys.stderr) + return 2 + except KeyboardInterrupt: + print(json.dumps({"status": "refused", "message": "Interrupted before submission."}), + file=sys.stderr) + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/examples/docs-review/test_docs_review.py b/examples/docs-review/test_docs_review.py new file mode 100644 index 000000000..59c5916a3 --- /dev/null +++ b/examples/docs-review/test_docs_review.py @@ -0,0 +1,608 @@ +#!/usr/bin/env python3 +# Copyright 2026 Google LLC +# SPDX-License-Identifier: Apache-2.0 +"""Behavior tests: all gws calls go to an isolated executable, never Google.""" + +import copy +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + + +SCRIPT = Path(__file__).with_name("docs_review.py") + + +def paragraph(parts, start=1): + elements = [] + cursor = start + for text, style in parts: + end = cursor + len(text.encode("utf-16-le")) // 2 + elements.append({ + "startIndex": cursor, "endIndex": end, + "textRun": {"content": text, "textStyle": style}, + }) + cursor = end + return { + "startIndex": start, "endIndex": cursor, + "paragraph": { + "elements": elements, + "paragraphStyle": {"namedStyleType": "NORMAL_TEXT"}, + }, + } + + +def document(text="Hello world.\n", revision="rev-1"): + return { + "documentId": "synthetic-doc", "revisionId": revision, + "title": "Synthetic review fixture", + "suggestionsViewMode": "SUGGESTIONS_INLINE", + "tabs": [{ + "tabProperties": {"tabId": "t.main", "title": "Main", "index": 0}, + "documentTab": { + "body": {"content": [ + {"endIndex": 1, "sectionBreak": { + "sectionStyle": {"sectionType": "CONTINUOUS"}}}, + paragraph([(text, {})]), + ]}, + }, + }], + } + + +def body(doc): + return doc["tabs"][0]["documentTab"]["body"]["content"] + + +def resign(plan): + payload = {k: v for k, v in plan.items() if k != "digest"} + plan["digest"] = hashlib.sha256(json.dumps( + payload, sort_keys=True, separators=(",", ":"), ensure_ascii=True, + allow_nan=False, + ).encode()).hexdigest() + return plan + + +# The executable checks the actual argv contract and simulates the remote +# boundary only. Its output fixtures are independent of production helpers. +STUB = r''' +import json, os, pathlib, sys, time +root = pathlib.Path(os.environ["STUB_ROOT"]) +args = sys.argv[1:] +with (root / "calls.jsonl").open("a") as f: + f.write(json.dumps(args) + "\n") +assert args[:2] == ["docs", "documents"], args +assert args[args.index("--format") + 1] == "json", args +params = json.loads(args[args.index("--params") + 1]) +assert params["documentId"] == "synthetic-doc", params +method = args[2] +mode = (root / "mode").read_text() +if method == "get": + assert params["includeTabsContent"] is True, params + assert params["suggestionsViewMode"] == "SUGGESTIONS_INLINE", params + name = "after.json" if (root / "submitted").exists() else "before.json" + if mode == "read-error" or (mode == "verify-error" and name == "after.json"): + print("secret-token \x1b[31m remote private content", file=sys.stderr) + sys.exit(1) + if mode == "armor-block": + print(json.dumps({"error": "Content blocked by Model Armor"})) + sys.exit(1) + if mode == "armor-warn": + assert os.environ["GOOGLE_WORKSPACE_CLI_SANITIZE_TEMPLATE"] == "synthetic-template" + print("secret-token \x1b[31m Model Armor warning", file=sys.stderr) + print((root / name).read_text()) +elif method == "batchUpdate": + request = json.loads(args[args.index("--json") + 1]) + (root / "request.json").write_text(json.dumps(request)) + (root / "submitted").touch() + if mode == "timeout": + time.sleep(5) + if mode in ("conflict", "write-error"): + print(json.dumps({"error": {"code": 400 if mode == "conflict" else 503, + "message": "secret-token \x1b[31m private response"}})) + sys.exit(1) + if mode == "bad-response": + print("secret-token malformed") + else: + result = {"documentId": "synthetic-doc", + "replies": [{"replaceAllText": {"occurrencesChanged": 1}}], + "writeControl": {"requiredRevisionId": "rev-2"}} + if mode == "zero": + result["replies"][0]["replaceAllText"]["occurrencesChanged"] = 0 + if mode == "two": + result["replies"][0]["replaceAllText"]["occurrencesChanged"] = 2 + if mode == "missing-reply": + result["replies"] = [] + if mode == "missing-write-revision": + result.pop("writeControl") + print(json.dumps(result)) +else: + raise AssertionError(args) +''' + + +class ReviewCliTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.bin = self.root / "bin" + self.bin.mkdir() + stub = self.bin / "gws" + stub.write_text("#!" + sys.executable + "\n" + STUB) + stub.chmod(0o700) + # Do not pass real credentials/configuration into any child process. + self.env = { + "PATH": str(self.bin) + os.pathsep + os.defpath, + "STUB_ROOT": str(self.root), + "PYTHONDONTWRITEBYTECODE": "1", + "GOOGLE_WORKSPACE_CLI_CONFIG_DIR": str(self.root / "config"), + } + self.put("mode", "ok") + self.put("find.txt", "world") + self.put("replacement.txt", "reader") + self.fixture(document(), document("Hello reader.\n", "rev-2")) + + def put(self, name, value): + (self.root / name).write_text(value, encoding="utf-8") + + def fixture(self, before, after=None): + self.put("before.json", json.dumps(before)) + self.put("after.json", json.dumps(after or before)) + + def cli(self, *args): + return subprocess.run( + [sys.executable, str(SCRIPT), *args], cwd=self.root, + env=self.env, text=True, capture_output=True, timeout=10, + ) + + def plan(self, *extra): + return self.cli("plan", "--document", "synthetic-doc", + "--find", "find.txt", "--replacement", "replacement.txt", + "--out", "plan.json", *extra) + + def apply(self, *extra): + return self.cli("apply", "--plan", "plan.json", *extra) + + def load_plan(self): + return json.loads((self.root / "plan.json").read_text()) + + def calls(self): + path = self.root / "calls.jsonl" + return [json.loads(x) for x in path.read_text().splitlines()] if path.exists() else [] + + def success(self, result): + self.assertEqual(result.returncode, 0, result.stderr) + return json.loads(result.stdout) + + def refused(self, result, status="refused"): + self.assertNotEqual(result.returncode, 0) + self.assertNotIn("secret-token", result.stderr + result.stdout) + self.assertNotIn("\x1b", result.stderr + result.stdout) + try: + payload = json.loads(result.stderr) + except ValueError: + self.fail("Expected a structured refusal, got: " + result.stderr[:200]) + self.assertEqual(payload["status"], status) + + def test_plan_is_deterministic_reviewable_and_reads_only_once(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + plan = self.load_plan() + self.assertEqual(plan["version"], 1) + self.assertEqual(plan["revision_id"], "rev-1") + self.assertEqual(plan["tab_id"], "t.main") + self.assertEqual(plan["expected_occurrences"], 1) + self.assertEqual(plan["target"], {"start_index": 7, "end_index": 12}) + self.assertIn("-world", plan["diff"]) + self.assertIn("+reader", plan["diff"]) + self.assertNotIn("Synthetic review fixture", original.decode()) + self.assertNotIn("Hello", original.decode()) + self.assertEqual(resign(copy.deepcopy(plan)), plan) + self.assertEqual([c[2] for c in self.calls()], ["get"]) + (self.root / "plan.json").unlink() + self.success(self.plan()) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_apply_submits_only_reviewed_revision_and_tab_then_verifies(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + result = self.success(self.apply()) + self.assertEqual(result["status"], "applied") + self.assertEqual(json.loads((self.root / "request.json").read_text()), { + "writeControl": {"requiredRevisionId": "rev-1"}, + "requests": [{"replaceAllText": { + "containsText": {"text": "world", "matchCase": True, + "searchByRegex": False}, + "replaceText": "reader", "tabsCriteria": {"tabIds": ["t.main"]}, + }}], + }) + self.assertEqual([c[2] for c in self.calls()], + ["get", "get", "batchUpdate", "get"]) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_zero_multiple_and_overlapping_matches_refuse_without_write(self): + for text, find in [("Nothing.\n", "world"), + ("world world\n", "world"), ("aaa\n", "aa")]: + with self.subTest(text=text): + self.fixture(document(text)) + self.put("find.txt", find) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + self.assertTrue(all(c[2] == "get" for c in self.calls())) + + def test_unicode_utf16_and_nested_tab_scope(self): + before = document("😀 café world.\n") + before["tabs"][0]["childTabs"] = [{ + "tabProperties": {"tabId": "t.child", "title": "Main", "index": 0}, + "documentTab": {"body": {"content": [paragraph([("world\n", {})])]}} + }] + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + body(after)[1] = paragraph([("😀 café reader.\n", {})]) + self.fixture(before, after) + self.refused(self.plan()) # No silent first-tab selection. + self.success(self.plan("--tab", "t.main")) + self.assertEqual(self.load_plan()["target"], + {"start_index": 9, "end_index": 14}) + self.success(self.apply()) + request = json.loads((self.root / "request.json").read_text()) + self.assertEqual(request["requests"][0]["replaceAllText"]["tabsCriteria"], + {"tabIds": ["t.main"]}) + + def test_replaces_unicode_in_selected_child_without_touching_parent(self): + before = document("😀targetZ\n") + child = copy.deepcopy(before["tabs"][0]) + child["tabProperties"]["tabId"] = "t.child" + before["tabs"][0]["childTabs"] = [child] + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + # Preserve Z/newline styles while allowing Google to style inserted text. + after["tabs"][0]["childTabs"][0]["documentTab"]["body"]["content"][1] = ( + paragraph([("🛰️", {"italic": True}), ("Z\n", {})])) + self.fixture(before, after) + self.put("find.txt", "😀target") + self.put("replacement.txt", "🛰️") + self.success(self.plan("--tab", "t.child")) + self.assertEqual(self.load_plan()["target"], + {"start_index": 1, "end_index": 9}) + self.success(self.apply()) + + def test_literal_matching_does_not_enable_regex_or_shell(self): + literal = "$(touch injected);.*" + self.fixture(document(literal + "\n"), document("done\n", "rev-2")) + self.put("find.txt", literal) + self.put("replacement.txt", "done") + self.success(self.plan()) + self.success(self.apply()) + self.assertFalse((self.root / "injected").exists()) + + def test_preview_is_offline_and_does_not_submit(self): + self.success(self.plan()) + calls = self.calls() + before = (self.root / "plan.json").read_bytes() + self.success(self.apply("--dry-run")) + self.assertEqual(self.calls(), calls) + self.assertEqual((self.root / "plan.json").read_bytes(), before) + + def test_source_revision_missing_or_changed_refuses(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + for doc in [document(revision="rev-new"), document("Different world.\n"), + {k: v for k, v in document().items() if k != "revisionId"}]: + with self.subTest(doc=doc): + self.fixture(doc) + self.refused(self.apply()) + self.assertFalse((self.root / "submitted").exists()) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_plan_without_revision_refuses(self): + doc = document() + doc.pop("revisionId") + self.fixture(doc) + self.refused(self.plan()) + + def test_malformed_tampered_and_oversized_plan_refuse_offline(self): + self.success(self.plan()) + original = self.load_plan() + tampered = copy.deepcopy(original) + tampered["replacement"] = "unreviewed" + unknown = resign(dict(original, command="touch injected")) + wrong_target = copy.deepcopy(original) + wrong_target["target"]["start_index"] = -1 + wrong_version = resign(dict(original, version=True)) + for value in ["{", "[]", '{"version":1,"version":2}', + json.dumps(tampered), json.dumps(unknown), + json.dumps(resign(wrong_target)), json.dumps(wrong_version), + " " * (1024 * 1024 + 1)]: + with self.subTest(value=value[:90]): + self.put("plan.json", value) + before = (self.root / "plan.json").read_bytes() + calls = self.calls() + self.refused(self.apply()) + self.assertEqual(self.calls(), calls) + self.assertEqual((self.root / "plan.json").read_bytes(), before) + + def test_resigned_semantic_tampering_is_reconstructed_before_write(self): + self.success(self.plan()) + plan = self.load_plan() + plan["target"]["start_index"] = 8 + self.put("plan.json", json.dumps(resign(plan))) + self.refused(self.apply()) + self.assertFalse((self.root / "submitted").exists()) + + def test_unsupported_structural_edits_and_noops_refuse(self): + for find, replacement in [("", "new"), ("world", "world"), + ("world", "one\ntwo"), ("world", "\ufffc"), + ("world.\n", "new"), ("world", "\x00")]: + with self.subTest(find=find, replacement=replacement): + self.put("find.txt", find) + self.put("replacement.txt", replacement) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + + def test_table_header_suggestion_and_image_crossing_targets_refuse(self): + table = document() + body(table)[1] = {"startIndex": 1, "endIndex": 14, "table": { + "rows": 1, "columns": 1, "tableRows": [{"tableCells": [{ + "content": [paragraph([("world\n", {})], 3)]}]}]}} + header = document("Other.\n") + header["tabs"][0]["documentTab"]["headers"] = { + "h.1": {"content": [paragraph([("world\n", {})])]}} + suggested = document() + body(suggested)[1]["paragraph"]["elements"][0]["textRun"][ + "suggestedInsertionIds"] = ["suggestion-1"] + image = document() + body(image)[1] = paragraph([("wor", {}), ("ld\n", {})]) + elements = body(image)[1]["paragraph"]["elements"] + elements.insert(1, {"startIndex": 4, "endIndex": 5, + "inlineObjectElement": {"inlineObjectId": "img-1"}}) + elements[2]["startIndex"] += 1 + elements[2]["endIndex"] += 1 + body(image)[1]["endIndex"] += 1 + for doc in [table, header, suggested, image]: + with self.subTest(doc=doc): + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "submitted").exists()) + + def test_duplicate_in_table_or_header_also_refuses(self): + doc = document() + doc["tabs"][0]["documentTab"]["footers"] = { + "f.1": {"content": [paragraph([("world\n", {})])]}} + self.fixture(doc) + self.refused(self.plan()) + + def test_style_run_splitting_and_deletion_are_supported(self): + before = document() + body(before)[1] = paragraph([ + ("Hello wo", {"bold": True}), ("rld", {"italic": True}), (".\n", {})]) + after = document(revision="rev-2") + body(after)[1] = paragraph([("Hello ", {"bold": True}), (".\n", {})]) + self.fixture(before, after) + self.put("replacement.txt", "") + self.success(self.plan()) + self.success(self.apply()) + + def test_preserves_tables_images_styles_and_checks_untouched_content(self): + before = document() + body(before).append({"startIndex": 14, "endIndex": 15, "paragraph": { + "elements": [{"startIndex": 14, "endIndex": 15, + "inlineObjectElement": {"inlineObjectId": "img-1"}}]}}) + before["tabs"][0]["documentTab"]["inlineObjects"] = { + "img-1": {"inlineObjectProperties": {"embeddedObject": { + "imageProperties": {"contentUri": "https://example.invalid/temporary"}, + "size": {"width": {"magnitude": 50, "unit": "PT"}}}}}} + body(before).append({"startIndex": 15, "endIndex": 25, "table": { + "rows": 1, "columns": 1, "tableRows": [{"tableCells": [{ + "content": [paragraph([("cell\n", {"bold": True})], 18)]}]}]}}) + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + body(after)[1] = paragraph([("Hello reader.\n", {})]) + # One extra UTF-16 unit shifts following structures, not their content. + def shift(value): + if isinstance(value, dict): + for k, v in value.items(): + if k in ("startIndex", "endIndex"): + value[k] = v + 1 + else: + shift(v) + elif isinstance(value, list): + for v in value: + shift(v) + shift(body(after)[2:]) + after["tabs"][0]["documentTab"]["inlineObjects"]["img-1"][ + "inlineObjectProperties"]["embeddedObject"]["imageProperties"][ + "contentUri"] = "https://example.invalid/refreshed" + self.fixture(before, after) + self.success(self.plan()) + self.success(self.apply()) + for kind in ["table", "image-id", "image-size"]: + with self.subTest(kind=kind): + changed = copy.deepcopy(after) + if kind == "table": + table_paragraph = body(changed)[3]["table"]["tableRows"][0][ + "tableCells"][0]["content"][0] + table_paragraph["paragraph"]["elements"][0]["textRun"]["content"] = "sell\n" + elif kind == "image-id": + body(changed)[2]["paragraph"]["elements"][0][ + "inlineObjectElement"]["inlineObjectId"] = "different-image" + else: + changed["tabs"][0]["documentTab"]["inlineObjects"]["img-1"][ + "inlineObjectProperties"]["embeddedObject"]["size"]["width"][ + "magnitude"] = 51 + (self.root / "submitted").unlink() + self.fixture(before, changed) + self.refused(self.apply(), "ambiguous") + + def test_post_write_mismatch_and_missing_revision_are_not_success(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + missing = document("Hello reader.\n") + missing.pop("revisionId") + changed_style = document("Hello reader.\n", "rev-2") + body(changed_style)[1]["paragraph"]["elements"][0]["textRun"][ + "textStyle"] = {"bold": True} + for doc in [document("Hello incorrect.\n", "rev-2"), missing, + changed_style, document("Hello reader.\n", "rev-unexpected")]: + with self.subTest(doc=doc): + (self.root / "submitted").unlink(missing_ok=True) + self.fixture(document(), doc) + self.refused(self.apply(), "ambiguous") + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_failed_write_or_verification_never_retries_and_keeps_plan(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + for mode in ["conflict", "write-error", "zero", "two", "missing-reply", + "bad-response", "missing-write-revision", "verify-error"]: + with self.subTest(mode=mode): + (self.root / "submitted").unlink(missing_ok=True) + self.put("mode", mode) + calls = len(self.calls()) + self.refused(self.apply(), "ambiguous") + writes = [c for c in self.calls()[calls:] if c[2] == "batchUpdate"] + self.assertEqual(len(writes), 1) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_timeout_after_submission_reports_possible_application(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + self.put("mode", "timeout") + result = self.apply("--timeout", "0.3") + self.refused(result, "ambiguous") + self.assertIn("may have", json.loads(result.stderr)["message"]) + self.assertEqual(len([c for c in self.calls() if c[2] == "batchUpdate"]), 1) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_paths_are_relative_confined_and_never_overwrite(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + self.refused(self.plan()) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + for path in ["../escape", "/tmp/escape", "sub/../../escape", "bad\nname"]: + with self.subTest(path=path): + self.refused(self.cli("apply", "--plan", path)) + self.refused(self.cli("plan", "--document", "synthetic-doc", + "--find", path, "--replacement", "replacement.txt", + "--out", "new-plan.json")) + self.refused(self.cli("plan", "--document", "synthetic-doc", + "--find", "find.txt", "--replacement", "replacement.txt", + "--out", path)) + with tempfile.TemporaryDirectory() as outside: + (self.root / "escape").symlink_to(outside, target_is_directory=True) + Path(outside, "plan.json").write_bytes(original) + self.refused(self.cli("apply", "--plan", "escape/plan.json")) + self.refused(self.cli("plan", "--document", "synthetic-doc", + "--find", "find.txt", "--replacement", "replacement.txt", + "--out", "escape/new.json")) + self.assertFalse(Path(outside, "new.json").exists()) + + def test_invalid_document_id_and_gws_failure_are_redacted(self): + result = self.cli("plan", "--document", "../secret?token", + "--find", "find.txt", "--replacement", "replacement.txt", + "--out", "plan.json") + self.refused(result) + self.assertEqual(self.calls(), []) + self.put("mode", "read-error") + self.refused(self.plan()) + + def test_model_armor_diagnostics_remain_visible_without_leaking_content(self): + self.env["GOOGLE_WORKSPACE_CLI_SANITIZE_TEMPLATE"] = "synthetic-template" + self.put("mode", "armor-block") + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + self.put("mode", "armor-warn") + result = self.success(self.plan()) + self.assertTrue(result.get("gws_diagnostics")) + self.assertNotIn("secret-token", json.dumps(result)) + self.assertNotIn("\x1b", json.dumps(result)) + + def test_unknown_regions_and_malformed_structures_fail_before_plan(self): + unknown = document() + unknown["tabs"][0]["documentTab"]["futureRegion"] = {"text": "world"} + malformed = document() + body(malformed)[0]["futureBlock"] = {} + for doc in [unknown, malformed]: + with self.subTest(doc=doc): + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + + def test_omitted_empty_text_style_does_not_fail_verification(self): + before = document() + after = document("Hello reader.\n", "rev-2") + body(after)[1]["paragraph"]["elements"][0]["textRun"].pop("textStyle") + self.fixture(before, after) + self.success(self.plan()) + self.success(self.apply()) + + def test_c1_control_characters_in_paths_are_rejected_before_read(self): + self.put("unsafe\u0085.txt", "world") + self.refused(self.cli( + "plan", "--document", "synthetic-doc", "--find", "unsafe\u0085.txt", + "--replacement", "replacement.txt", "--out", "plan.json", + )) + self.assertEqual(self.calls(), []) + + def test_nonregular_files_and_leaf_symlinks_refuse(self): + os.mkfifo(self.root / "pipe") + (self.root / "link.txt").symlink_to(self.root / "find.txt") + for path in ["pipe", "link.txt"]: + with self.subTest(path=path): + self.refused(self.cli( + "plan", "--document", "synthetic-doc", "--find", path, + "--replacement", "replacement.txt", "--out", "plan.json", + )) + self.assertEqual(self.calls(), []) + + def test_invalid_utf8_oversized_text_and_timeout_values_refuse(self): + for data in [b"\xff", b"x" * (16 * 1024 + 1)]: + with self.subTest(size=len(data)): + (self.root / "find.txt").write_bytes(data) + self.refused(self.plan()) + for timeout in ["nan", "inf", "0", "-1", "601"]: + with self.subTest(timeout=timeout): + self.refused(self.plan("--timeout", timeout)) + self.assertEqual(self.calls(), []) + + def test_malformed_source_and_unsupported_anchors_refuse(self): + for value in ["[]", "{", '{"documentId":NaN}']: + with self.subTest(value=value): + self.put("before.json", value) + self.refused(self.plan()) + for key in ["namedRanges", "bookmarks"]: + doc = document() + doc["tabs"][0]["documentTab"][key] = {"synthetic-anchor": {}} + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "submitted").exists()) + + def test_large_expanded_style_payload_refuses_before_plan(self): + doc = document("world" + "a" * 20000 + "\n") + body(doc)[1]["paragraph"]["elements"][0]["textRun"]["textStyle"] = { + "link": {"url": "https://example.invalid/" + "x" * 2000}} + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + + def test_default_zero_indices_in_untouched_header_are_supported(self): + before = document() + header = paragraph([("Header\n", {})], start=0) + del header["startIndex"] + del header["paragraph"]["elements"][0]["startIndex"] + before["tabs"][0]["documentTab"]["headers"] = {"h.1": {"content": [header]}} + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + body(after)[1] = paragraph([("Hello reader.\n", {})]) + self.fixture(before, after) + self.success(self.plan()) + self.success(self.apply()) + + +if __name__ == "__main__": + unittest.main() From af8043a96725210a23bf0146a79f8e216a77bf7a Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:21:50 -0300 Subject: [PATCH 12/19] fix(docs): preserve outcomes and validate every tab --- .changeset/docs-review-workflow.md | 4 +- examples/docs-review/README.md | 27 ++- examples/docs-review/docs_review.py | 215 +++++++++++++++++------ examples/docs-review/test_docs_review.py | 171 +++++++++++++++++- 4 files changed, 355 insertions(+), 62 deletions(-) diff --git a/.changeset/docs-review-workflow.md b/.changeset/docs-review-workflow.md index 0c3060012..eee647198 100644 --- a/.changeset/docs-review-workflow.md +++ b/.changeset/docs-review-workflow.md @@ -4,4 +4,6 @@ Add a standalone Python Docs review example that plans one literal text replacement, binds it to a source revision and tab, applies it through existing gws commands, -and verifies the result without retrying ambiguous writes. +and verifies the result without retrying ambiguous writes. Validate structures +and text ranges across all tabs before normalization, and preserve attempted or +confirmed mutation outcomes through final output failures and interruptions. diff --git a/examples/docs-review/README.md b/examples/docs-review/README.md index 61d595e76..404b80f6f 100644 --- a/examples/docs-review/README.md +++ b/examples/docs-review/README.md @@ -101,8 +101,14 @@ bookmarks. Unknown tab regions, unknown structural blocks, equations, automatic text, rich links and smart chips in the selected tab are refused. These strict limits avoid interpreting inaccessible text, unstable anchors, or unsupported element boundaries as a safe replacement. Other tabs are -fingerprinted and verified, not edited. New API structures may require an -explicit compatibility update. +fingerprinted and verified, not edited. Before normalization, every tab's +paragraph elements must have valid UTF-16 lengths and contiguous ranges. +Recognized regions and structural blocks must have valid object/list shapes, +including table rows and cell content. Malformed snapshots in any tab are +refused before submission or reported as ambiguous after submission. +Unknown metadata is retained for comparison; unknown structural blocks and +extra text-element fields that would be discarded are refused. New API +structures may require an explicit compatibility update. Each text input is limited to 16 KiB, the plan to 1 MiB, and each gws response to 16 MiB. Source verification additionally limits total text to 250,000 @@ -120,11 +126,18 @@ Exit `2` / `refused` means the companion did not submit a document write. Correct the input or reread the source and review a **new** plan. Exit `3` / `ambiguous` means a write **may have applied**, or its result could -not be verified. This includes subprocess timeouts, nonzero write exits -(including a concurrent API 400), malformed responses, missing revisions, -unexpected reply counts, and failed/mismatching rereads. The original plan -stays unchanged. Inspect the document and revision through your normal tools; -do not blindly rerun apply. Keep the failed plan as your review record and +not be verified or reported successfully. This includes subprocess timeouts, +nonzero write exits (including a concurrent API 400), malformed responses, missing revisions, +unexpected reply counts, failed/mismatching rereads, and interruptions or +output failures after submission. The diagnostic JSON on stderr includes +`mutation_state: "attempted"` or `"confirmed"`. A confirmed mutation followed +by a closed stdout consumer or failed final reporting still exits `3`; it +does not claim that the write was refused or never submitted. Stdout is +flushed before success is returned, so buffered output failures are handled +inside the same outcome check. + +The original plan stays unchanged. Inspect the document and revision through +your normal tools; do not blindly rerun apply. Keep the failed plan as your review record and create a new plan if further work is required. The companion does not persist a separate attempt journal or prevent an operator from manually rerunning it. diff --git a/examples/docs-review/docs_review.py b/examples/docs-review/docs_review.py index 27a1700f6..1f122daaf 100755 --- a/examples/docs-review/docs_review.py +++ b/examples/docs-review/docs_review.py @@ -179,6 +179,7 @@ class Gws: def __init__(self, timeout): self.timeout = timeout self.diagnostics = False + self.mutation_state = "not_attempted" def call(self, method, document_id, request=None): params = {"documentId": document_id} @@ -275,39 +276,109 @@ def check_supported(tab): "Suggested content in selected tab is unsupported; resolve suggestions first.") require(not obj.get("namedRanges") and not obj.get("bookmarks"), "Named ranges and bookmarks in the selected tab are unsupported.") - if "content" in obj and isinstance(obj["content"], list): - for block in obj["content"]: - require(isinstance(block, dict) - and len(set(block) - {"startIndex", "endIndex"}) == 1 - and len(set(block) & { - "paragraph", "sectionBreak", "table", "tableOfContents"}) == 1, - "Unsupported document structure.") if "paragraph" not in obj: continue - paragraph = obj["paragraph"] - require(isinstance(paragraph, dict) - and isinstance(paragraph.get("elements"), list), "Malformed paragraph.") - cursor = obj.get("startIndex", 0) - require(type(cursor) is int and cursor >= 0, "Invalid paragraph index.") - for element in paragraph["elements"]: - require(isinstance(element, dict) - and type(element.get("startIndex", 0)) is int - and type(element.get("endIndex")) is int - and element.get("startIndex", 0) == cursor - and element["endIndex"] > cursor, "Non-contiguous paragraph indices.") + for element in obj["paragraph"]["elements"]: kinds = set(element) - {"startIndex", "endIndex"} require(len(kinds) == 1 and kinds <= { "textRun", "inlineObjectElement", "footnoteReference", "horizontalRule", "pageBreak", "columnBreak"}, "Unsupported paragraph element; rich links, equations and chips are excluded.") - if "textRun" in element: - run = element["textRun"] - require(isinstance(run, dict) and isinstance(run.get("content"), str), - "Malformed text run.") - require(utf16(run["content"]) == element["endIndex"] - cursor, - "Text run does not match its UTF-16 indices.") - cursor = element["endIndex"] - require(obj.get("endIndex") == cursor, "Paragraph end index mismatch.") + + +def index_range(value): + start, end = value.get("startIndex", 0), value.get("endIndex") + require(type(start) is int and type(end) is int and 0 <= start < end, + "Invalid structural or paragraph element indices.") + return start, end + + +def check_region(region): + require(isinstance(region, dict) and isinstance(region.get("content"), list), + "Malformed document region; expected structural content.") + + +def check_content(content): + for block in content: + require(isinstance(block, dict), "Malformed structural block.") + kinds = set(block) - {"startIndex", "endIndex"} + require(len(kinds) == 1 and kinds <= { + "paragraph", "sectionBreak", "table", "tableOfContents"}, + "Unsupported document structure.") + require(isinstance(block[next(iter(kinds))], dict), "Malformed structural block value.") + index_range(block) + + +def check_table(table): + require(isinstance(table, dict), "Malformed table.") + require(type(table.get("rows")) is int and table["rows"] > 0 + and type(table.get("columns")) is int and table["columns"] > 0, + "Malformed table dimensions.") + rows = table.get("tableRows") + require(isinstance(rows, list) and len(rows) == table["rows"], "Malformed table rows.") + for row in rows: + require(isinstance(row, dict) and isinstance(row.get("tableCells"), list) + and 0 < len(row["tableCells"]) <= table["columns"], "Malformed table row.") + for cell in row["tableCells"]: + check_region(cell) + + +def check_paragraph(block): + paragraph = block["paragraph"] + require(isinstance(paragraph, dict) + and isinstance(paragraph.get("elements"), list) + and paragraph["elements"], "Malformed paragraph.") + require(isinstance(paragraph.get("paragraphStyle", {}), dict), + "Malformed paragraph style.") + cursor, paragraph_end = index_range(block) + for element in paragraph["elements"]: + require(isinstance(element, dict), "Malformed paragraph element.") + start, end = index_range(element) + require(start == cursor, "Non-contiguous paragraph indices.") + kinds = set(element) - {"startIndex", "endIndex"} + # Extra siblings of textRun would otherwise disappear in normalization. + # Unrecognized non-text unions are retained whole in unselected tabs. + require(len(kinds) == 1 and isinstance(element[next(iter(kinds))], dict), + "Malformed or unsupported paragraph element fields.") + if "textRun" in element: + run = element["textRun"] + require(isinstance(run.get("content"), str) + and isinstance(run.get("textStyle", {}), dict), "Malformed text run.") + require(utf16(run["content"]) == end - start, + "Text run does not match its UTF-16 indices.") + cursor = end + require(paragraph_end == cursor, "Paragraph end index mismatch.") + + +def check_document_structure(document): + """Validate every region/range the normalizer interprets, in every tab. + + Unknown metadata is retained unchanged. Unknown structural blocks or + text-element siblings that cannot be retained safely are refused. + """ + for _, obj in walk(document): + if "documentTab" in obj: + require(isinstance(obj["documentTab"], dict) and "body" in obj["documentTab"], + "Missing document tab body.") + if "body" in obj: + check_region(obj["body"]) + for group in ("headers", "footers", "footnotes"): + if group in obj: + require(isinstance(obj[group], dict), "Malformed document region map.") + for region in obj[group].values(): + check_region(region) + if isinstance(obj.get("content"), list): + check_content(obj["content"]) + if "paragraph" in obj: + check_paragraph(obj) + if "table" in obj: + check_table(obj["table"]) + if "tableOfContents" in obj: + check_region(obj["tableOfContents"]) + if "sectionBreak" in obj: + require(isinstance(obj["sectionBreak"], dict) + and isinstance(obj["sectionBreak"].get("sectionStyle", {}), dict), + "Malformed section break.") def check_document_budget(document): @@ -328,10 +399,17 @@ def check_document_budget(document): "Document is too large for safe text/style verification.") -def normalized(value, path=()): +def normalized(document): + # This gate is part of normalization itself, so no caller can accidentally + # compare discarded text-run indices before validating the complete source. + check_document_structure(document) + return _normalized(document) + + +def _normalized(value, path=()): """Canonical semantic shape; text-run splitting is not a style change.""" if isinstance(value, list): - return [normalized(v, path + (i,)) for i, v in enumerate(value)] + return [_normalized(v, path + (i,)) for i, v in enumerate(value)] if not isinstance(value, dict): return value result = {} @@ -352,10 +430,10 @@ def normalized(value, path=()): for char in run["content"]: elements.append({"text": char, "format": metadata}) else: - elements.append(normalized(element, path + (key,))) + elements.append(_normalized(element, path + (key,))) result[key] = elements else: - result[key] = normalized(item, path + (key,)) + result[key] = _normalized(item, path + (key,)) return result @@ -408,12 +486,13 @@ def build_plan(document, document_id, tab_id, find, replacement): text_input(replacement, allow_empty=True) require(find != replacement, "No-op replacement; choose different text.") tab_id, tab_path = select_tab(document, document_id, tab_id) - check_supported(at(document, tab_path)) check_document_budget(document) + source = normalized(document) + check_supported(at(document, tab_path)) _, _, start = locate(document, tab_path, find) result = { "version": 1, "document_id": document_id, "tab_id": tab_id, - "revision_id": document["revisionId"], "source_sha256": sha256(normalized(document)), + "revision_id": document["revisionId"], "source_sha256": sha256(source), "find": find, "replacement": replacement, "expected_occurrences": 1, "target": {"start_index": start, "end_index": start + utf16(find)}, "diff": review_diff(find, replacement), @@ -465,10 +544,10 @@ def verify(before, after, plan, write_revision): _, tab_path = select_tab(before, plan["document_id"], plan["tab_id"]) select_tab(after, plan["document_id"], plan["tab_id"]) require(after["revisionId"] == write_revision, "Post-write revision mismatch.") - check_supported(at(after, tab_path)) check_document_budget(after) - path, offset, _ = locate(before, tab_path, plan["find"]) expected, actual = normalized(before), normalized(after) + check_supported(at(after, tab_path)) + path, offset, _ = locate(before, tab_path, plan["find"]) end = plan["target"]["end_index"] delta = utf16(plan["replacement"]) - utf16(plan["find"]) # Indices in the body shift; headers/footers/footnotes have separate indices. @@ -496,6 +575,7 @@ def apply_plan(plan, gws): require(rebuilt == plan, "Source revision or content changed; regenerate and review a new plan.") try: + gws.mutation_state = "attempted" reply = gws.call("batchUpdate", plan["document_id"], request_body(plan)) require(reply.get("documentId") == plan["document_id"], "Write identity mismatch.") replies = reply.get("replies") @@ -510,6 +590,7 @@ def apply_plan(plan, gws): require(write_revision != plan["revision_id"], "Write revision did not advance.") after = gws.get(plan["document_id"]) verify(before, after, plan, write_revision) + gws.mutation_state = "confirmed" except (Refusal, OSError, ValueError, KeyError, TypeError, IndexError, RecursionError, KeyboardInterrupt): raise Ambiguous( @@ -525,6 +606,46 @@ def error(self, message): raise Refusal("Invalid arguments; use --help for supported options.") +def silence_failed_stdout(): + """Prevent a failed buffered write from being retried at interpreter exit.""" + try: + with open(os.devnull, "w") as sink: + os.dup2(sink.fileno(), sys.stdout.fileno()) + except (OSError, ValueError, AttributeError): + pass + + +def report_failure(error, mutation_state): + if mutation_state != "not_attempted": + message = ( + "Write was confirmed, but final reporting failed. " + if mutation_state == "confirmed" else + "Write may have applied; verification did not establish success. " + ) + outcome = { + "status": "ambiguous", "mutation_state": mutation_state, + "message": message + "Keep the original plan, inspect the document and revision, " + "and do not blindly retry. Further changes require a newly reviewed plan.", + } + code = 3 + else: + if isinstance(error, KeyboardInterrupt): + message = "Interrupted before submission." + elif isinstance(error, Refusal): + message = str(error) + else: + message = "Malformed source or inaccessible file; check inputs and regenerate." + outcome = {"status": "refused", "message": message} + code = 2 + try: + print(json.dumps(outcome), file=sys.stderr, flush=True) + except (OSError, KeyboardInterrupt): + # Neither an unavailable diagnostic channel nor another interruption + # changes whether a mutation was attempted. + pass + return code + + def main(argv=None): parser = Parser(description=__doc__) commands = parser.add_subparsers(dest="command", required=True) @@ -542,6 +663,8 @@ def main(argv=None): for command in (plan_parser, apply_parser): command.add_argument("--timeout", type=float, default=60, help="Per-gws-call timeout in seconds (default: 60)") + gws = None + reporting = False try: args = parser.parse_args(argv) require(math.isfinite(args.timeout) and 0 < args.timeout <= 600, @@ -572,23 +695,15 @@ def main(argv=None): "gws reported diagnostics; check gws and Model Armor settings. " "Raw output was suppressed to protect document content." ) - print(json.dumps(result, ensure_ascii=True, sort_keys=True)) + reporting = True + print(json.dumps(result, ensure_ascii=True, sort_keys=True), flush=True) return 0 - except Ambiguous as error: - print(json.dumps({"status": "ambiguous", "message": str(error)}), file=sys.stderr) - return 3 - except Refusal as error: - print(json.dumps({"status": "refused", "message": str(error)}), file=sys.stderr) - return 2 - except (OSError, ValueError, KeyError, TypeError, IndexError, RecursionError): - print(json.dumps({"status": "refused", "message": - "Malformed source or inaccessible file; check inputs and regenerate."}), - file=sys.stderr) - return 2 - except KeyboardInterrupt: - print(json.dumps({"status": "refused", "message": "Interrupted before submission."}), - file=sys.stderr) - return 2 + except (Refusal, OSError, ValueError, KeyError, TypeError, IndexError, + RecursionError, KeyboardInterrupt) as error: + if reporting: + silence_failed_stdout() + mutation_state = gws.mutation_state if gws is not None else "not_attempted" + return report_failure(error, mutation_state) if __name__ == "__main__": diff --git a/examples/docs-review/test_docs_review.py b/examples/docs-review/test_docs_review.py index 59c5916a3..03f5af589 100644 --- a/examples/docs-review/test_docs_review.py +++ b/examples/docs-review/test_docs_review.py @@ -225,6 +225,86 @@ def test_apply_submits_only_reviewed_revision_and_tab_then_verifies(self): ["get", "get", "batchUpdate", "get"]) self.assertEqual((self.root / "plan.json").read_bytes(), original) + def test_closed_stdout_after_apply_keeps_confirmed_mutation_state(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + for python_flags in [[], ["-u"]]: + with self.subTest(python_flags=python_flags): + (self.root / "submitted").unlink(missing_ok=True) + previous_calls = len(self.calls()) + read_fd, write_fd = os.pipe() + os.close(read_fd) + try: + result = subprocess.run( + [sys.executable, *python_flags, str(SCRIPT), + "apply", "--plan", "plan.json"], + cwd=self.root, env=self.env, stdout=write_fd, + stderr=subprocess.PIPE, text=True, timeout=10, + ) + finally: + os.close(write_fd) + self.assertEqual(result.returncode, 3, result.stderr) + outcome = json.loads(result.stderr) + self.assertEqual(outcome["status"], "ambiguous") + self.assertEqual(outcome["mutation_state"], "confirmed") + self.assertIn("inspect", outcome["message"]) + self.assertIn("do not blindly retry", outcome["message"]) + self.assertEqual([c[2] for c in self.calls()[previous_calls:]], + ["get", "batchUpdate", "get"]) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_interruption_after_apply_returns_keeps_confirmed_mutation_state(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + # Inject only the interruption at the return boundary. The actual + # apply implementation still performs every read/write/verification. + wrapper = """ +import runpy, sys +namespace = runpy.run_path(sys.argv[1]) +real_apply = namespace["apply_plan"] +def interrupted_return(*args, **kwargs): + real_apply(*args, **kwargs) + raise KeyboardInterrupt +namespace["main"].__globals__["apply_plan"] = interrupted_return +sys.exit(namespace["main"](["apply", "--plan", "plan.json"])) +""" + result = subprocess.run( + [sys.executable, "-c", wrapper, str(SCRIPT)], + cwd=self.root, env=self.env, capture_output=True, text=True, timeout=10, + ) + self.assertEqual(result.returncode, 3, result.stderr) + self.refused(result, "ambiguous") + outcome = json.loads(result.stderr) + self.assertEqual(outcome["mutation_state"], "confirmed") + self.assertIn("do not blindly retry", outcome["message"]) + self.assertNotIn("before submission", outcome["message"]) + self.assertEqual([c[2] for c in self.calls()], + ["get", "get", "batchUpdate", "get"]) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_final_serialization_failure_keeps_confirmed_mutation_state(self): + self.success(self.plan()) + original = (self.root / "plan.json").read_bytes() + wrapper = """ +import json, runpy, sys +namespace = runpy.run_path(sys.argv[1]) +real_dumps = json.dumps +def failed_result(value, *args, **kwargs): + if isinstance(value, dict) and value.get("status") == "applied": + raise ValueError("secret-token final output failure") + return real_dumps(value, *args, **kwargs) +json.dumps = failed_result +sys.exit(namespace["main"](["apply", "--plan", "plan.json"])) +""" + result = subprocess.run( + [sys.executable, "-c", wrapper, str(SCRIPT)], + cwd=self.root, env=self.env, capture_output=True, text=True, timeout=10, + ) + self.assertEqual(result.returncode, 3, result.stderr) + self.refused(result, "ambiguous") + self.assertEqual(json.loads(result.stderr)["mutation_state"], "confirmed") + self.assertEqual((self.root / "plan.json").read_bytes(), original) + def test_zero_multiple_and_overlapping_matches_refuse_without_write(self): for text, find in [("Nothing.\n", "world"), ("world world\n", "world"), ("aaa\n", "aa")]: @@ -330,10 +410,18 @@ def test_malformed_tampered_and_oversized_plan_refuse_offline(self): def test_resigned_semantic_tampering_is_reconstructed_before_write(self): self.success(self.plan()) plan = self.load_plan() - plan["target"]["start_index"] = 8 - self.put("plan.json", json.dumps(resign(plan))) - self.refused(self.apply()) - self.assertFalse((self.root / "submitted").exists()) + altered_target = copy.deepcopy(plan) + altered_target["target"] = {"start_index": 8, "end_index": 13} + altered_fingerprint = dict(plan, source_sha256="0" * 64) + for altered in [altered_target, altered_fingerprint]: + with self.subTest(altered=altered): + self.put("plan.json", json.dumps(resign(altered))) + original = (self.root / "plan.json").read_bytes() + previous_calls = len(self.calls()) + self.refused(self.apply()) + self.assertEqual([c[2] for c in self.calls()[previous_calls:]], ["get"]) + self.assertFalse((self.root / "submitted").exists()) + self.assertEqual((self.root / "plan.json").read_bytes(), original) def test_unsupported_structural_edits_and_noops_refuse(self): for find, replacement in [("", "new"), ("world", "world"), @@ -533,6 +621,81 @@ def test_unknown_regions_and_malformed_structures_fail_before_plan(self): self.refused(self.plan()) self.assertFalse((self.root / "plan.json").exists()) + def test_null_scalar_and_incomplete_structures_refuse_before_plan(self): + cases = [] + for value in [None, 3, {}, {"rows": 1, "columns": 1, "tableRows": None}, + {"rows": 1, "columns": 1, "tableRows": [None]}, + {"rows": 1, "columns": 1, "tableRows": [{"tableCells": [None]}]}, + {"rows": 1, "columns": 1, "tableRows": [ + {"tableCells": [{"content": "not structural content"}]}]}]: + doc = document() + body(doc).append({"startIndex": 14, "endIndex": 15, "table": value}) + cases.append(doc) + for region in ["headers", "footers", "footnotes"]: + for value in [None, [], {"segment-1": None}, + {"segment-1": {"content": "not structural content"}}]: + doc = document() + doc["tabs"][0]["documentTab"][region] = value + cases.append(doc) + for kind in ["sectionBreak", "tableOfContents"]: + doc = document() + body(doc).append({"startIndex": 14, "endIndex": 15, kind: None}) + cases.append(doc) + for doc in cases: + with self.subTest(document=doc): + (self.root / "plan.json").unlink(missing_ok=True) + self.fixture(doc) + self.refused(self.plan()) + self.assertFalse((self.root / "plan.json").exists()) + self.assertFalse((self.root / "submitted").exists()) + + def test_unselected_tab_malformed_ranges_refuse_preflight_and_postwrite(self): + before = document() + child = copy.deepcopy(document("😀 untouched\n")["tabs"][0]) + child["tabProperties"]["tabId"] = "t.other" + before["tabs"].append(child) + after = copy.deepcopy(before) + after["revisionId"] = "rev-2" + body(after)[1] = paragraph([("Hello reader.\n", {})]) + self.fixture(before, after) + self.success(self.plan("--tab", "t.main")) + original = (self.root / "plan.json").read_bytes() + for phase in ["preflight", "postwrite"]: + for field, value in [("endIndex", 999), ("startIndex", 2), + ("endIndex", True)]: + with self.subTest(phase=phase, field=field, value=value): + (self.root / "submitted").unlink(missing_ok=True) + bad = copy.deepcopy(before if phase == "preflight" else after) + bad["tabs"][1]["documentTab"]["body"]["content"][1][ + "paragraph"]["elements"][0][field] = value + self.fixture(bad if phase == "preflight" else before, + bad if phase == "postwrite" else after) + previous_calls = len(self.calls()) + self.refused(self.apply(), + "refused" if phase == "preflight" else "ambiguous") + expected_calls = (["get"] if phase == "preflight" else + ["get", "batchUpdate", "get"]) + self.assertEqual([c[2] for c in self.calls()[previous_calls:]], + expected_calls) + self.assertEqual((self.root / "plan.json").read_bytes(), original) + + def test_unselected_text_run_shape_is_checked_before_discarding_fields(self): + for extra in [{"textRun": None}, {"textRun": {"content": "extra\n", + "textStyle": "invalid"}}, + {"futureElementMetadata": {"value": "must not disappear"}}]: + with self.subTest(extra=extra): + (self.root / "plan.json").unlink(missing_ok=True) + doc = document() + other = copy.deepcopy(document("extra\n")["tabs"][0]) + other["tabProperties"]["tabId"] = "t.other" + other["documentTab"]["body"]["content"][1]["paragraph"]["elements"][0].update( + extra) + doc["tabs"].append(other) + self.fixture(doc) + self.refused(self.plan("--tab", "t.main")) + self.assertFalse((self.root / "plan.json").exists()) + self.assertFalse((self.root / "submitted").exists()) + def test_omitted_empty_text_style_does_not_fail_verification(self): before = document() after = document("Hello reader.\n", "rev-2") From fcdd6311c9b6ec413eca8fc0f7a23a3b3aafaeb7 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:04:09 -0300 Subject: [PATCH 13/19] feat(docs): add visual review bundle companion --- .changeset/docs-review-bundle.md | 7 + .github/workflows/ci.yml | 11 + examples/docs-review-bundle/README.md | 182 ++++ .../docs-review-bundle/docs_review_bundle.py | 749 ++++++++++++++++ .../test_docs_review_bundle.py | 813 ++++++++++++++++++ 5 files changed, 1762 insertions(+) create mode 100644 .changeset/docs-review-bundle.md create mode 100644 examples/docs-review-bundle/README.md create mode 100755 examples/docs-review-bundle/docs_review_bundle.py create mode 100644 examples/docs-review-bundle/test_docs_review_bundle.py diff --git a/.changeset/docs-review-bundle.md b/.changeset/docs-review-bundle.md new file mode 100644 index 000000000..bac3f4983 --- /dev/null +++ b/.changeset/docs-review-bundle.md @@ -0,0 +1,7 @@ +--- +"@googleworkspace/cli": minor +--- + +Add a standalone Python companion for visual Google Docs review bundles with +native exports, safe DOCX raster extraction, local HTML, optional PDF page +previews, revision observations, and an offline fixture workflow. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f9417abd9..797c27d85 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -45,6 +45,17 @@ jobs: python3 --version python3 -m unittest discover -s examples/docs-review -p 'test_*.py' -v + docs-review-bundle: + name: Docs Review Bundle (Python) + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [ubuntu-latest, macos-latest] + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4 + - name: Test stdlib companion with synthetic fixtures + run: python3 -B -m unittest discover -s examples/docs-review-bundle -p 'test_*.py' -v + changes: name: Detect Changes runs-on: ubuntu-latest diff --git a/examples/docs-review-bundle/README.md b/examples/docs-review-bundle/README.md new file mode 100644 index 000000000..fcc7f58ca --- /dev/null +++ b/examples/docs-review-bundle/README.md @@ -0,0 +1,182 @@ +# Visual Docs review bundle + +This standalone companion combines native Docs JSON, Drive PDF/DOCX/Markdown +exports, DOCX raster assets, and a local HTML review index. It uses only the +Python standard library and existing `gws` commands. It does not change `gws`, +authenticate separately, or download document hyperlinks or image `contentUri`s. + +## Run + +Requires Python 3.10+ on Linux or macOS and an authenticated `gws` executable +with read access to the document through both Docs and Drive. + +From the repository root: + +```sh +python3 examples/docs-review-bundle/docs_review_bundle.py \ + --document-id DOCUMENT_ID review-bundle +``` + +`review-bundle` must be a **new relative directory inside the current working +directory**. Existing directories, absolute paths, `..`, symlink components, +control characters, and names outside the portable ASCII subset are refused. +Nested paths work when their parents already exist. The final directory is +created with mode `0700`; keep its parents under your control. + +Optional orchestration flags: + +```sh +python3 examples/docs-review-bundle/docs_review_bundle.py \ + --document-id DOCUMENT_ID --include-comments --render-pages \ + --timeout 120 --gws /trusted/path/to/gws another-review-bundle +``` + +- `--include-comments`: collect every returned Drive comments page. The entire + optional artifact is omitted and marked unavailable if retrieval is incomplete + or fails. Comments are a separate observation, not revision-bound or mapped to + PDF coordinates. Deleted comments are not requested. +- `--render-pages`: use a trusted `pdftoppm` on `PATH` to generate 96 DPI PNGs. + Rendering is opt-in. Missing tools, failures, timeouts, invalid output and + noncontiguous page numbers retain the PDF and report no available raster + previews. Partial page output is discarded. +- `--timeout`: positive finite seconds per subprocess; default 60. This is not + a total workflow deadline. +- `--gws`: trusted executable, resolved before changing subprocess working + directories. Existing `gws` authentication and Model Armor environment + settings are inherited. No credential values or raw process diagnostics + are inserted into HTML or error messages. + +The companion requests `docs documents get` with `includeTabsContent: true` +and uses `drive files export` with `--format json` and fixed relative +`--output` filenames. Every `gws` subprocess runs inside the new bundle. +Export success, MIME type, destination and byte count are checked against the +written artifact. This requires the existing `gws` binary-export receipt format. + +Open `index.html` locally. The index has no JavaScript or remote dependencies. +It contains a sandboxed PDF frame, optional page images, a paragraph/table +outline grouped by native tabs, DOCX figures and nearby text, native object +metadata, and escaped Markdown source. Some browsers block local PDF frames; +use the PDF artifact link in that case. Markdown is readable escaped source, +not rendered Markdown. + +## Artifacts and completion + +| File | Meaning | +| --- | --- | +| `source.json` | Original Docs JSON response, including all returned tabs | +| `revision-after.json` | Final revision observation, when available | +| `document.pdf`, `document.docx`, `document.md` | Required native Drive exports | +| `comments.json` | Optional fully retrieved comments result | +| `assets/` | Recognized DOCX PNG, JPEG, GIF and WebP media | +| `pages/` | Optional successful local page-rendering output | +| `index.html` | Local review index | +| `manifest.json` | State, versions, capabilities, limitations, mappings and hashes | + +The manifest begins as `in-progress` and becomes `complete` only after all +required exports, validation, index generation and artifact hashing succeed. +`complete` means the required bundle files were produced; it does **not** mean +an atomic snapshot, successful optional rendering, or verified export tab +coverage. Check `revisions`, `comments`, and `rendering` independently. + +Revision status is `mixed` for differing observed Docs revision IDs, `unchanged` +for equal nonempty IDs, and `unknown` if either is missing. Even `unchanged` +does not prove atomicity or that every export represents the same revision. +The manifest always records `atomic_snapshot: false`. + +Required failures return exit code 1 and leave a `failed` manifest with a safe +error code and stage. If writing the failure state also fails (for example, +a full disk), the earlier `in-progress` state can remain. Process termination +can also leave that state. No such bundle should be treated as complete. +Usage errors return 2. Optional failures and mixed/unknown revisions return 0 +when the required bundle completes. Existing output directories are never +overwritten; retries need a new name. + +SHA-256 and byte counts cover each produced artifact, including the index and +assets. The manifest does not hash itself and is not an authenticity signature. + +## Scope and safety limits + +- Native JSON traversal includes nested tabs, body paragraphs, tables and + tables of contents. The outline identifies paragraph styles but does not + recreate full layout. Headers, footers, notes, lists, equations, charts, + suggestions and other document features are not fully represented. +- Google determines the PDF/DOCX/Markdown export layout and tab coverage. + This companion cannot verify that every tab appears in those formats or + associate a PDF page/DOCX figure with an exact native tab. +- DOCX figure order comes from main-document drawing relationships, including + drawings in tables. Alt text and nearby paragraphs are context, **not exact + captions**. Native object IDs are recorded separately; the companion never + fabricates a matching native ID or a source tab ID for a DOCX image. +- Media basenames are replaced with distinct generated local filenames. + External, missing, traversing and unsupported relationships remain visibly + unavailable. Unreferenced recognized rasters are retained as artifacts. +- ZIP validation rejects traversal, absolute/Windows/control-character paths, + duplicate names (case-insensitive), symlinks and other special files, + encryption and unsupported compression. It never calls `extractall`. + Limits are 2,000 members, 20 MiB per member and 100 MiB total uncompressed. + ZIP paths are restricted to printable ASCII. Only stored/deflate compression + is accepted. +- Every XML/relationship member is parsed after rejecting DTDs, entities, + UTF-16/32 and non-UTF-8 encodings. XML trees are limited to 100 levels and + 100,000 nodes per part. The supported DOCX vocabulary is transitional OOXML + main-document drawings; unsupported content can remain unmapped. +- Each source/export/JSON artifact is capped at 20 MiB. Page output is limited + to 500 files and 100 MiB total. PDF header/EOF and raster signatures are + checked; these are **not full format validation**. Renderer exit success and + contiguous numbering do not independently prove the original PDF page count. +- Subprocess capture is file-backed, with size checks after exit. Timeouts and + post-render limits do not enforce disk or memory quotas on external tools. + `pdftoppm`, `gws`, local viewers and the operator-controlled parent directory + are trusted. This is not a sandbox against a concurrent local attacker. +- Display text is HTML/attribute escaped; URI references use generated local + allowlisted names. The index has a restrictive content-security policy and + no scripts. Source URLs and recognizable bearer/token strings are redacted + from display text, but this is not a general secret scanner or Model Armor + replacement. Existing `gws` sanitization behavior is preserved; binary + exports are not made safe by JSON sanitization. +- Raw artifacts deliberately retain original content, including any temporary + URLs or sensitive text. Treat the whole directory as sensitive. Review + external links and active content in native viewers separately; the + companion never follows them automatically. + +## Offline fixtures and visual QA + +Offline mode performs no `gws` calls. Supply a relative directory containing +`source.json`, `document.pdf`, `document.docx`, and UTF-8 `document.md`. +`revision-after.json` and `comments.json` are optional; a missing revision +observation remains unknown. Files and path components must not be symlinks. + +Generate entirely synthetic fixtures using the test utility, then create a +review bundle suitable for inspecting the HTML: + +```sh +python3 -B - <<'PY' +import sys +from pathlib import Path +sys.path.insert(0, "examples/docs-review-bundle") +from test_docs_review_bundle import fixtures +fixtures(Path("synthetic-docs-fixture")) +PY + +python3 -B examples/docs-review-bundle/docs_review_bundle.py \ + --from-fixture synthetic-docs-fixture --render-pages synthetic-docs-review +``` + +These generated fixtures intentionally contain HTML injection strings, an +untrusted synthetic URL, duplicate image basenames, missing image URIs, +nested tabs and table content. They are hand-built test data, not an actual +Google export or proof of cross-format fidelity. No real documents, +credentials or network access are needed. Omit `--render-pages` to avoid +running external software. + +## Tests + +```sh +python3 -B -m unittest discover \ + -s examples/docs-review-bundle -p 'test_*.py' -v +``` + +Tests use generated JSON/PDF/DOCX files and explicit `gws`/renderer executables +as stubs. Their subprocess environment omits real authentication settings; +no installed `gws`, real document, or network request is needed. CI runs this +same command in a dedicated Linux/macOS job. diff --git a/examples/docs-review-bundle/docs_review_bundle.py b/examples/docs-review-bundle/docs_review_bundle.py new file mode 100755 index 000000000..fd54477d4 --- /dev/null +++ b/examples/docs-review-bundle/docs_review_bundle.py @@ -0,0 +1,749 @@ +#!/usr/bin/env python3 +# Copyright 2026 Google LLC +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy at https://www.apache.org/licenses/LICENSE-2.0 +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Build a local visual review bundle using Python's standard library and gws.""" + +import argparse +import hashlib +import html +import io +import json +import math +import os +from pathlib import Path, PurePosixPath +import re +import shutil +import stat +import subprocess +import sys +import tempfile +import xml.etree.ElementTree as ET +import zipfile + + +VERSION = "1.0" +MIB = 1024 * 1024 +FILE_LIMIT = 20 * MIB +TOTAL_LIMIT = 100 * MIB +MEMBER_COUNT = 2000 +MAX_PAGES = 500 +EXPORTS = { + "document.pdf": "application/pdf", + "document.docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document", + "document.md": "text/markdown", +} +W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}" +A = "{http://schemas.openxmlformats.org/drawingml/2006/main}" +WP = "{http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing}" +R = "{http://schemas.openxmlformats.org/officeDocument/2006/relationships}" +REL = "{http://schemas.openxmlformats.org/package/2006/relationships}" +LIMITATIONS = [ + "Exports are sequential, not an atomic snapshot; unchanged revisions are observations only.", + "Native outline includes body paragraphs, tables and nested tabs, not full layout or styling.", + "Drive export tab coverage is not verified; pages and DOCX figures have no reliable tab mapping.", + "DOCX relationships provide document order and nearby text, " + "not exact captions or native Docs IDs.", + "Only recognized PNG, JPEG, GIF and WebP media are previewed; " + "signatures are not full validation.", + "Headers, footers, notes, charts, vectors and unsupported drawings " + "may be absent from the outline or figures.", + "Content URIs and document hyperlinks are never fetched; URLs are redacted from display text.", + "Raw exports and JSON are sensitive, unsanitized source artifacts " + "and may contain temporary URLs.", + "The companion is not a document sanitizer or a sandbox for PDF/image viewers or pdftoppm.", +] + + +class BundleError(Exception): + """A fixed, non-sensitive failure code safe for manifests and terminals.""" + + +def relative_parts(value): + """Conservative portable path subset; do not normalize away traversal.""" + parts = value.split("/") + if not parts or any( + part in ("", ".", "..") or not re.fullmatch(r"[A-Za-z0-9_. -]+", part) + for part in parts + ): + raise BundleError("unsafe-relative-path") + return parts + + +def directory(value, *, create=False): + """Walk existing parents with no-follow descriptors (Linux/macOS).""" + parts = relative_parts(value) + flags = os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW + fd = os.open(".", flags) + try: + for index, part in enumerate(parts): + if create and index == len(parts) - 1: + os.mkdir(part, mode=0o700, dir_fd=fd) + next_fd = os.open(part, flags, dir_fd=fd) + os.close(fd) + fd = next_fd + except OSError: + raise BundleError("directory-exists-or-unsafe") from None + finally: + os.close(fd) + return Path.cwd().joinpath(*parts) + + +def read_bytes(path, limit=FILE_LIMIT): + """Bound reads and reject special files, including symlinks.""" + try: + fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + with os.fdopen(fd, "rb") as stream: + info = os.fstat(stream.fileno()) + if not stat.S_ISREG(info.st_mode) or info.st_size > limit: + raise BundleError("file-type-or-size-limit") + data = stream.read(limit + 1) + except OSError: + raise BundleError("missing-or-unsafe-file") from None + if len(data) > limit: + raise BundleError("file-size-limit") + return data + + +def write_bytes(path, data): + # Files are new, inside the newly created private bundle directory. + with path.open("xb") as stream: + stream.write(data) + + +def json_bytes(value): + return (json.dumps(value, ensure_ascii=True, indent=2) + "\n").encode("utf-8") + + +def parse_json(data): + try: + value = json.loads(data) + except (ValueError, UnicodeError): + raise BundleError("invalid-json") from None + if not isinstance(value, dict) or "error" in value: + raise BundleError("invalid-json-response") + return value + + +def publish_manifest(bundle, manifest): + temporary = bundle / ".manifest.tmp" + write_bytes(temporary, json_bytes(manifest)) + os.replace(temporary, bundle / "manifest.json") + + +def local_uri(value): + """HTML references are generated file names, never document-supplied URIs.""" + if not re.fullmatch(r"[A-Za-z0-9_-]+(?:[./][A-Za-z0-9_-]+)*", value): + raise BundleError("unsafe-local-uri") + return value + + +def display(value): + text = str(value or "") + text = re.sub(r"(?i)\b(?:https?|ftp|file|data|javascript):[^\s<>\"']+", "[URL omitted]", text) + text = re.sub(r"(?i)\bBearer\s+\S+", "[credential omitted]", text) + text = re.sub( + r"(?i)\b(?:access_token|authorization|token)\s*[:=]\s*[^\s<>\"']+", + "[credential omitted]", + text, + ) + text = "".join(char for char in text if char in "\n\t" or ord(char) >= 32) + return html.escape(text, quote=True) + + +def safe_xml(data): + # Reject UTF-16/32 (including declaration smuggling via NUL bytes), DTDs and + # entities before giving XML to the stdlib parser. DOCX normally uses UTF-8. + if b"\x00" in data or re.search(br"]*encoding=['\"]([^'\"]+)", text, re.I) + if declaration and declaration[1].lower() not in ("utf-8", "utf8", "us-ascii"): + raise BundleError("unsupported-xml-encoding") + root = ET.fromstring(text) + except (ET.ParseError, UnicodeError): + raise BundleError("invalid-xml") from None + pending = [(root, 0)] + count = 0 + while pending: + node, depth = pending.pop() + count += 1 + if depth > 100 or count > 100_000: + raise BundleError("xml-complexity-limit") + pending.extend((child, depth + 1) for child in node) + return root + + +def raster_extension(data): + if data.startswith(b"\x89PNG\r\n\x1a\n"): + return "png" + if data.startswith(b"\xff\xd8\xff"): + return "jpg" + if data.startswith((b"GIF87a", b"GIF89a")): + return "gif" + if data.startswith(b"RIFF") and data[8:12] == b"WEBP": + return "webp" + return None + + +def zip_members(source, member_limit, total_limit, member_count): + """Read a bounded archive without ever extracting its member paths.""" + try: + with zipfile.ZipFile(io.BytesIO(read_bytes(source))) as archive: + infos = archive.infolist() + if len(infos) > member_count: + raise BundleError("zip-member-count-limit") + seen = set() + declared_total = 0 + for info in infos: + name = info.orig_filename + parts = name.rstrip("/").split("/") + kind = stat.S_IFMT(info.external_attr >> 16) + if ( + name != info.filename + or name.startswith("/") + or "\\" in name + or ":" in name + or any(part in ("", ".", "..") for part in parts) + or any(ord(char) < 32 or ord(char) > 126 for char in name) + or kind not in (0, stat.S_IFREG, stat.S_IFDIR) + or info.flag_bits & 1 + or info.compress_type not in (zipfile.ZIP_STORED, zipfile.ZIP_DEFLATED) + or name.casefold() in seen + ): + raise BundleError("unsafe-zip-member") + seen.add(name.casefold()) + declared_total += info.file_size + if info.file_size > member_limit or declared_total > total_limit: + raise BundleError("zip-size-limit") + members = {} + actual_total = 0 + for info in infos: + if info.is_dir(): + continue + with archive.open(info) as stream: + data = stream.read(member_limit + 1) + actual_total += len(data) + if ( + len(data) > member_limit + or actual_total > total_limit + or len(data) != info.file_size + ): + raise BundleError("zip-size-limit") + members[info.filename] = data + return members + except (zipfile.BadZipFile, RuntimeError, NotImplementedError, OSError, EOFError): + raise BundleError("invalid-zip") from None + + +def extract_docx( + source, output, *, + member_limit=FILE_LIMIT, total_limit=TOTAL_LIMIT, member_count=MEMBER_COUNT, +): + members = zip_members(source, member_limit, total_limit, member_count) + # Validate even unused XML before any asset writes. + trees = { + name: safe_xml(data) + for name, data in members.items() + if name.lower().endswith((".xml", ".rels")) + } + document = trees.get("word/document.xml") + if document is None or document.tag != W + "document": + raise BundleError("missing-docx-document") + relationships = {} + rels = trees.get("word/_rels/document.xml.rels") + if rels is not None: + for rel in rels.findall(REL + "Relationship"): + identity = rel.get("Id") + if not identity or identity in relationships: + raise BundleError("ambiguous-docx-relationship") + relationships[identity] = rel.attrib + output.mkdir(mode=0o700) + assets = {} + for name, data in members.items(): + extension = raster_extension(data) + if name.startswith("word/media/") and extension: + filename = f"image-{len(assets) + 1}.{extension}" + write_bytes(output / filename, data) + assets[name] = local_uri(f"{output.name}/{filename}") + paragraphs = list(document.iter(W + "p")) + texts = [ + "".join(text.text or "" for text in paragraph.iter(W + "t")) + for paragraph in paragraphs + ] + figures = [] + for index, paragraph in enumerate(paragraphs): + nearby = texts[index] or " ".join( + texts[max(0, index - 1):index] + texts[index + 1:index + 2] + ) + for drawing in paragraph.iter(W + "drawing"): + properties = drawing.find(".//" + WP + "docPr") + alt = "" + if properties is not None: + alt = " ".join(properties.get(field, "") for field in ("title", "descr")).strip() + for blip in drawing.iter(A + "blip"): + identity = blip.get(R + "embed") or blip.get(R + "link") + relationship = relationships.get(identity, {}) + asset = None + availability = "missing-or-unsupported" + if relationship.get("TargetMode", "").lower() == "external": + availability = "external-not-fetched" + elif relationship.get("Type") == R[1:-1] + "/image": + target = relationship.get("Target", "") + # Allow only relative media targets in the document's part. + if ( + target.startswith("media/") + and not any(p in ("", ".", "..") for p in target.split("/")) + and not any(char in target for char in "\\:%?#") + ): + asset = assets.get(str(PurePosixPath("word") / target)) + if asset: + availability = "available" + figures.append({ + "order": len(figures) + 1, + "relationship_id": identity, + "asset": asset, + "alt": alt, + "nearby_text": nearby[:1000], + "availability": availability, + "mapping_confidence": "docx-relationship-only", + "native_object_id": None, + "source_tab_id": None, + }) + return {"figures": figures, "assets": list(assets.values())} + + +def native_view(source): + if not isinstance(source.get("body"), dict) and not isinstance(source.get("tabs"), list): + raise BundleError("missing-document-content") + tabs = [] + images = [] + + def add_tab(tab, depth): + if depth > 50 or len(tabs) >= 1000: + raise BundleError("native-tab-limit") + properties = tab.get("tabProperties", {}) + document = tab.get("documentTab", {}) + tab_id = properties.get("tabId") + tabs.append({ + "id": tab_id, + "title": properties.get("title", "Document"), + "depth": depth, + "blocks": document.get("body", {}).get("content", []), + }) + for collection, property_name in ( + ("inlineObjects", "inlineObjectProperties"), + ("positionedObjects", "positionedObjectProperties"), + ): + for object_id, obj in document.get(collection, {}).items(): + embedded = obj.get(property_name, {}).get("embeddedObject", {}) + images.append({ + "object_id": object_id, + "tab_id": tab_id, + "title": embedded.get("title", ""), + "description": embedded.get("description", ""), + "content_uri_available": bool( + embedded.get("imageProperties", {}).get("contentUri") + ), + "asset": None, + "mapping_confidence": "unmapped", + }) + for child in tab.get("childTabs", []): + add_tab(child, depth + 1) + + if source.get("tabs"): + for tab in source["tabs"]: + add_tab(tab, 0) + else: + add_tab({"documentTab": source}, 0) + return tabs, images + + +def outline_html(blocks, depth=0): + if depth > 50: + raise BundleError("native-outline-depth-limit") + parts = [] + for block in blocks: + if "paragraph" in block: + paragraph = block["paragraph"] + text = "".join( + element.get("textRun", {}).get("content", "") + for element in paragraph.get("elements", []) + ) + style = paragraph.get("paragraphStyle", {}).get("namedStyleType", "NORMAL_TEXT") + parts.append(f"

{display(style)} {display(text)}

") + elif "table" in block: + parts.append("") + for row in block["table"].get("tableRows", []): + parts.append("") + for cell in row.get("tableCells", []): + parts.append( + "" + ) + parts.append("") + parts.append("
" + outline_html(cell.get("content", []), depth + 1) + "
") + elif "tableOfContents" in block: + parts.append(outline_html(block["tableOfContents"].get("content", []), depth + 1)) + return "".join(parts) + + +def index_html(source, markdown, tabs, manifest, artifacts): + parts = [ + '', + '', + '", + "Docs review bundle", + "", + f"

{display(source.get('title', 'Docs review bundle'))}

", + f"

Revision observation: {display(manifest['revisions']['status'])}. " + "Sequential exports; not an atomic snapshot.

", + "

Artifacts

    ", + ] + for name in [*artifacts, "manifest.json"]: + parts.append(f'
  • {display(name)}
  • ') + parts.extend( + [ + "

Raw files may contain sensitive content and temporary URLs.

", + "

Native PDF

", + '', + "

If your browser blocks the embedded viewer, open the local PDF artifact.

", + f"

Raster previews: {display(manifest['rendering']['status'])}; " + f"{display(manifest['rendering'].get('reason', ''))}

", + ] + ) + for page in manifest["rendering"]["pages"]: + parts.append(f'{display(page)}') + parts.append("

Native outline and tabs

") + for tab in tabs: + parts.append( + f"

{display(tab['title'])}

" + f"

Tab: {display(tab['id'] or 'legacy body')}; " + f"depth: {tab['depth']}

{outline_html(tab['blocks'])}" + ) + parts.append("

DOCX figures

") + for figure in manifest["figures"]: + parts.append(f"

Figure {figure['order']}

") + if figure["asset"]: + parts.append( + f'' + ) + parts.append( + f"
{display(figure['alt'])}
" + f"

Availability: {display(figure['availability'])}

" + f"

Nearby text (not an exact caption): {display(figure['nearby_text'])}

" + "

DOCX relationship only; native object and source tab mapping unknown.

" + ) + parts.append("

Native image metadata

") + for image in manifest["native_images"]: + parts.append( + f"

Object {display(image['object_id'])}, tab {display(image['tab_id'])}: " + f"{display(image['title'])} {display(image['description'])}. " + f"contentUri available: {image['content_uri_available']}; asset mapping: unmapped.

" + ) + parts.append( + "

Readable Markdown source

" + f"
{display(markdown)}
" + ) + parts.append( + f"

Comments

{display(manifest['comments']['status'])}

" + ) + parts.append("

Capabilities and limitations

    ") + parts.extend(f"
  • {display(item)}
  • " for item in LIMITATIONS) + parts.append("
") + return "".join(parts).encode("utf-8") + + +def run_process(argv, bundle, timeout): + # File-backed capture bounds memory. No shell and no untrusted diagnostics + # echoed to the terminal or copied into the manifest/display HTML. + with tempfile.TemporaryFile(dir=bundle) as output: + try: + process = subprocess.run( + argv, + cwd=bundle, + stdout=output, + stderr=subprocess.DEVNULL, + stdin=subprocess.DEVNULL, + timeout=timeout, + check=False, + ) + except subprocess.TimeoutExpired: + raise BundleError("timeout") from None + except OSError: + raise BundleError("process-unavailable") from None + if process.returncode: + raise BundleError("process-failed") + if output.tell() > FILE_LIMIT: + raise BundleError("process-output-limit") + output.seek(0) + return output.read(FILE_LIMIT + 1) + + +def gws_json(executable, bundle, timeout, command, params, output=None): + argv = [executable, *command, "--params", json.dumps(params), "--format", "json"] + if output: + argv.extend(["--output", local_uri(output)]) + data = run_process(argv, bundle, timeout) + return data, parse_json(data) + + +def collect_comments(executable, bundle, timeout, document_id): + comments = [] + seen = set() + token = None + total = 0 + for _ in range(100): + params = {"fileId": document_id, "fields": "nextPageToken,comments", "pageSize": 100} + if token: + params["pageToken"] = token + raw, page = gws_json(executable, bundle, timeout, ["drive", "comments", "list"], params) + total += len(raw) + if total > FILE_LIMIT or not isinstance(page.get("comments", []), list): + raise BundleError("invalid-or-oversized-comments") + comments.extend(page.get("comments", [])) + token = page.get("nextPageToken") + if not token: + return json_bytes({"comments": comments}) + if not isinstance(token, str) or token in seen: + raise BundleError("incomplete-comments") + seen.add(token) + raise BundleError("incomplete-comments") + + +def render_pages(bundle, timeout, requested): + result = {"status": "not-requested", "pages": []} + if not requested: + return result + renderer = shutil.which("pdftoppm") + if not renderer: + return {"status": "unavailable", "reason": "pdftoppm-not-found; PDF retained", "pages": []} + try: + with tempfile.TemporaryDirectory(prefix=".render-", dir=bundle) as temp: + staging = Path(temp) + prefix = str(staging.relative_to(bundle) / "page") + run_process([renderer, "-png", "-r", "96", "document.pdf", prefix], bundle, timeout) + pages = {} + total = 0 + for path in staging.iterdir(): + match = re.fullmatch(r"page-([0-9]+)\.png", path.name) + if not match: + raise BundleError("invalid-render") + number = int(match[1]) + data = read_bytes(path) + total += len(data) + if number in pages or raster_extension(data) != "png" or total > TOTAL_LIMIT: + raise BundleError("invalid-render") + pages[number] = path.name + if ( + not pages + or len(pages) > MAX_PAGES + or sorted(pages) != list(range(1, len(pages) + 1)) + ): + raise BundleError("invalid-render") + staging.rename(bundle / "pages") + return { + "status": "complete", + "pages": [local_uri("pages/" + pages[number]) for number in sorted(pages)], + } + except (BundleError, OSError) as error: + return { + "status": "failed", "pages": [], + "reason": "timeout" if str(error) == "timeout" else "invalid-or-failed-render", + } + + +def revision_observation(before, after): + before_id = before.get("revisionId") + after_id = after.get("revisionId") + before_id = before_id if isinstance(before_id, str) and before_id else None + after_id = after_id if isinstance(after_id, str) and after_id else None + status = "unknown" + if before_id and after_id: + status = "unchanged" if before_id == after_id else "mixed" + return {"before": before_id, "after": after_id, "status": status, "atomic_snapshot": False} + + +def build_bundle(args, bundle, manifest): + artifacts = ["source.json", *EXPORTS] + manifest["stage"] = "exports" + if args.from_fixture: + fixture = directory(args.from_fixture) + for name in artifacts: + write_bytes(bundle / name, read_bytes(fixture / name)) + source = parse_json(read_bytes(bundle / "source.json")) + after_path = fixture / "revision-after.json" + after = parse_json(read_bytes(after_path)) if after_path.exists() else {} + if after_path.exists(): + write_bytes(bundle / "revision-after.json", read_bytes(after_path)) + artifacts.append("revision-after.json") + executable = None + else: + executable = shutil.which(args.gws) + if not executable: + raise BundleError("gws-not-found") + executable = str(Path(executable).resolve()) + try: + version = run_process([executable, "--version"], bundle, args.timeout).decode("utf-8") + match = re.fullmatch(r"gws ([0-9][A-Za-z0-9.+-]*)\s*", version) + manifest["versions"]["gws"] = match[1] if match else "unknown" + except BundleError: + manifest["versions"]["gws"] = "unknown" + params = {"documentId": args.document_id, "includeTabsContent": True} + raw, source = gws_json( + executable, bundle, args.timeout, ["docs", "documents", "get"], params + ) + if source.get("documentId") != args.document_id: + raise BundleError("document-id-mismatch") + write_bytes(bundle / "source.json", raw) + for name, mime in EXPORTS.items(): + _, receipt = gws_json( + executable, bundle, args.timeout, ["drive", "files", "export"], + {"fileId": args.document_id, "mimeType": mime}, output=name, + ) + data = read_bytes(bundle / name) + if ( + receipt.get("status") != "success" + or receipt.get("saved_file") != name + or str(receipt.get("mimeType", "")).split(";")[0].strip().lower() != mime + or type(receipt.get("bytes")) is not int + or receipt["bytes"] != len(data) + ): + raise BundleError("invalid-export-receipt") + raw, after = gws_json( + executable, bundle, args.timeout, ["docs", "documents", "get"], + {**params, "fields": "revisionId"}, + ) + write_bytes(bundle / "revision-after.json", raw) + artifacts.append("revision-after.json") + + manifest["revisions"] = revision_observation(source, after) + manifest["comments"] = {"status": "not-requested"} + if args.include_comments: + try: + if executable: + comments = collect_comments(executable, bundle, args.timeout, args.document_id) + else: + comments = read_bytes(fixture / "comments.json") + value = parse_json(comments) + if not isinstance(value.get("comments"), list) or value.get("nextPageToken"): + raise BundleError("incomplete-comments") + write_bytes(bundle / "comments.json", comments) + artifacts.append("comments.json") + manifest["comments"] = {"status": "available"} + except (BundleError, OSError): + (bundle / "comments.json").unlink(missing_ok=True) + manifest["comments"] = { + "status": "unavailable", + "reason": "retrieval-incomplete-or-failed", + } + + manifest["stage"] = "validate-and-extract" + pdf = read_bytes(bundle / "document.pdf") + if not pdf.startswith(b"%PDF-") or b"%%EOF" not in pdf[-1024:]: + raise BundleError("invalid-or-truncated-pdf") + markdown = read_bytes(bundle / "document.md").decode("utf-8") + tabs, native_images = native_view(source) + extracted = extract_docx(bundle / "document.docx", bundle / "assets") + manifest["figures"] = extracted["figures"] + manifest["native_images"] = native_images + manifest["tabs"] = [{k: v for k, v in tab.items() if k != "blocks"} for tab in tabs] + artifacts.extend(extracted["assets"]) + manifest["stage"] = "render" + manifest["rendering"] = render_pages(bundle, args.timeout, args.render_pages) + artifacts.extend(manifest["rendering"]["pages"]) + manifest["stage"] = "index" + write_bytes(bundle / "index.html", index_html(source, markdown, tabs, manifest, artifacts)) + artifacts.append("index.html") + manifest["artifacts"] = {} + for name in artifacts: + data = read_bytes(bundle / local_uri(name)) + manifest["artifacts"][name] = { + "bytes": len(data), + "sha256": hashlib.sha256(data).hexdigest(), + } + manifest["status"] = "complete" + manifest["stage"] = "complete" + publish_manifest(bundle, manifest) + + +def positive_timeout(value): + number = float(value) + if not math.isfinite(number) or number <= 0: + raise argparse.ArgumentTypeError("timeout must be a positive finite number") + return number + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + source = parser.add_mutually_exclusive_group(required=True) + source.add_argument("--document-id", help="Google Docs ID (not a URL)") + source.add_argument( + "--from-fixture", help="Relative directory containing captured export files" + ) + parser.add_argument("output_dir", help="New relative directory within CWD; parents must exist") + parser.add_argument("--include-comments", action="store_true") + parser.add_argument( + "--render-pages", action="store_true", help="Opt in to local pdftoppm rendering" + ) + parser.add_argument( + "--timeout", type=positive_timeout, default=60.0, + help="Seconds per process (default: 60)", + ) + parser.add_argument( + "--gws", default="gws", help="Trusted gws executable (default: PATH lookup)" + ) + args = parser.parse_args(argv) + if args.document_id and not re.fullmatch(r"[A-Za-z0-9_-]+", args.document_id): + parser.error("document-id must be a Google Docs ID, not a URL or path") + bundle = None + manifest = { + "schema_version": 1, "status": "in-progress", "stage": "initialize", + "mode": "offline" if args.from_fixture else "gws", + "versions": {"companion": VERSION, "python": sys.version.split()[0], "gws": None}, + "capabilities": { + "all_native_tabs": True, "native_pdf": True, "docx_raster_assets": True, + "native_markdown": True, "network_image_fetch": False, + "exact_native_asset_mapping": False, "atomic_snapshot": False, + }, + "limitations": LIMITATIONS, + } + try: + bundle = directory(args.output_dir, create=True) + publish_manifest(bundle, manifest) + build_bundle(args, bundle, manifest) + except (Exception, KeyboardInterrupt) as error: + code = str(error) if isinstance(error, BundleError) else "invalid-or-incomplete-bundle" + if bundle is not None: + manifest["status"] = "failed" + manifest["error"] = code + try: + publish_manifest(bundle, manifest) + except OSError: + # An earlier in-progress manifest remains non-complete if storage fails. + pass + print(f"Review bundle failed: {code}.", file=sys.stderr) + return 1 + print("Review bundle complete. Open index.html in the new output directory.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/examples/docs-review-bundle/test_docs_review_bundle.py b/examples/docs-review-bundle/test_docs_review_bundle.py new file mode 100644 index 000000000..56ede7b74 --- /dev/null +++ b/examples/docs-review-bundle/test_docs_review_bundle.py @@ -0,0 +1,813 @@ +#!/usr/bin/env python3 +# Copyright 2026 Google LLC +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy at https://www.apache.org/licenses/LICENSE-2.0 +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Hermetic behavior tests: generated documents and executable process stubs only.""" + +import base64 +import contextlib +import hashlib +from html.parser import HTMLParser +import importlib.util +import io +import json +import os +from pathlib import Path +import stat +import subprocess +import sys +import tempfile +import unittest +from unittest import mock +import zipfile + + +SCRIPT = Path(__file__).with_name("docs_review_bundle.py") +PNG = base64.b64decode( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8" + "/x8AAwMCAO+jRZkAAAAASUVORK5CYII=" +) +W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" +R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +A = "http://schemas.openxmlformats.org/drawingml/2006/main" +WP = "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing" +REL = "http://schemas.openxmlformats.org/package/2006/relationships" +REMOTE = "https://untrusted.invalid/image?token=SIGNED_SECRET" + + +def paragraph(text, style="NORMAL_TEXT"): + return { + "paragraph": { + "paragraphStyle": {"namedStyleType": style}, + "elements": [ + {"textRun": {"content": text, "textStyle": {"link": {"url": REMOTE}}}} + ], + } + } + + +def native_document(): + return { + "documentId": "synthetic-doc", + "title": 'Review ', + "revisionId": "revision-one", + "tabs": [ + { + "tabProperties": {"tabId": "tab-main", "title": "Main"}, + "documentTab": { + "body": { + "content": [ + paragraph("Heading ", "HEADING_1"), + { + "table": { + "tableRows": [ + { + "tableCells": [ + {"content": [paragraph("Table cell")]} + ] + } + ] + } + }, + { + "paragraph": { + "elements": [ + { + "inlineObjectElement": { + "inlineObjectId": "native-image" + } + } + ] + } + }, + ] + }, + "inlineObjects": { + "native-image": { + "inlineObjectProperties": { + "embeddedObject": { + "title": "Native figure", + "description": "No exact DOCX mapping", + "imageProperties": {"contentUri": REMOTE}, + } + } + }, + "missing-uri": { + "inlineObjectProperties": { + "embeddedObject": { + "description": "No downloadable URI", + "imageProperties": {}, + } + } + }, + }, + }, + "childTabs": [ + { + "tabProperties": {"tabId": "tab-child", "title": "Child"}, + "documentTab": { + "body": {"content": [paragraph("Nested tab paragraph")]} + }, + } + ], + } + ], + } + + +def pdf_bytes(): + """A complete synthetic one-page PDF, also usable by real pdftoppm.""" + content = ( + b"BT /F1 22 Tf 48 720 Td (Synthetic Docs review) Tj ET\n" + b"BT /F1 12 Tf 48 687 Td (Local fixture - no Google document) Tj ET\n" + b"0.85 0.92 1 rg 48 435 516 210 re f\n" + b"0.08 0.25 0.5 rg 72 459 120 162 re f\n" + b"0.12 0.45 0.6 rg 216 459 120 105 re f\n" + b"0.1 0.6 0.45 rg 360 459 120 140 re f\n" + b"0 0 0 rg BT /F1 12 Tf 48 402 Td (Synthetic figure and table context) Tj ET\n" + b"0.5 G 48 270 516 90 re S 48 315 m 564 315 l S\n" + b"306 270 m 306 360 l S\n" + b"BT /F1 12 Tf 60 333 Td (Column A) Tj 258 0 Td (Column B) Tj ET\n" + b"BT /F1 12 Tf 60 288 Td (Cell one) Tj 258 0 Td (Cell two) Tj ET\n" + ) + objects = [ + b"<< /Type /Catalog /Pages 2 0 R >>", + b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] " + b"/Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>", + f"<< /Length {len(content)} >>\nstream\n".encode() + content + b"endstream", + b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", + ] + data = b"%PDF-1.4\n" + offsets = [0] + for index, obj in enumerate(objects, 1): + offsets.append(len(data)) + data += f"{index} 0 obj\n".encode() + obj + b"\nendobj\n" + startxref = len(data) + data += b"xref\n0 6\n0000000000 65535 f \n" + for offset in offsets[1:]: + data += f"{offset:010d} 00000 n \n".encode() + data += ( + f"trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n{startxref}\n%%EOF\n" + ).encode() + return data + + +def docx_members(): + xml = f""" + + + +Nearby <script>bad()</script> + + +Table image context + + + + + +""" + rels = f""" + + +""" + return { + "word/document.xml": xml.encode(), + "word/_rels/document.xml.rels": rels.encode(), + "word/styles.xml": f''.encode(), + "word/media/image1.png": PNG, + "word/media/nested/image1.png": PNG + b"different", + } + + +def write_docx(path, members=None): + with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as archive: + for name, data in (members if members is not None else docx_members()).items(): + archive.writestr(name, data) + + +def fixtures(path): + path.mkdir() + (path / "source.json").write_text(json.dumps(native_document()), encoding="utf-8") + (path / "revision-after.json").write_text( + '{"revisionId":"revision-one"}', encoding="utf-8" + ) + (path / "document.pdf").write_bytes(pdf_bytes()) + (path / "document.md").write_text( + "# Markdown\n\n![remote](" + REMOTE + ")\n", + encoding="utf-8", + ) + write_docx(path / "document.docx") + return path + + +GWS_STUB = r''' +import json +import os +from pathlib import Path +import shutil +import sys +import time + +args = sys.argv[1:] +mode = os.environ.get("STUB_MODE", "") +fixture = Path(os.environ["STUB_FIXTURES"]) +with open(os.environ["STUB_LOG"], "a", encoding="utf-8") as log: + log.write(json.dumps({"args": args, "cwd": os.getcwd(), + "sanitize": os.environ.get("GOOGLE_WORKSPACE_CLI_SANITIZE_MODE")}) + "\n") +if args == ["--version"]: + print("gws 0.0.0-synthetic") + sys.exit(0) +params = json.loads(args[args.index("--params") + 1]) +assert args[args.index("--format") + 1] == "json" +if mode == "timeout": + time.sleep(10) +if args[:3] == ["docs", "documents", "get"]: + assert params["documentId"] == "synthetic-doc" + assert params["includeTabsContent"] is True + if params.get("fields") == "revisionId": + source = {"revisionId": "revision-two" if mode == "mixed" else "revision-one"} + if mode == "missing-revision": + source = {} + else: + source = json.loads((fixture / "source.json").read_text()) + print(json.dumps(source)) +elif args[:3] == ["drive", "files", "export"]: + assert params["fileId"] == "synthetic-doc" + expected = { + "application/pdf": "document.pdf", + "application/vnd.openxmlformats-officedocument.wordprocessingml.document": "document.docx", + "text/markdown": "document.md", + } + name = args[args.index("--output") + 1] + assert expected[params["mimeType"]] == name + assert "/" not in name and "\\" not in name + if mode == "failed-export" and name == "document.docx": + Path(name).write_bytes(b"partial") + print("Bearer PRIVATE_TOKEN " + "\x1b[31m", file=sys.stderr) + sys.exit(7) + if mode != "no-export-file": + shutil.copyfile(fixture / name, name) + print(json.dumps({ + "status": "error" if mode == "bad-export-status" else "success", + "saved_file": name, + "mimeType": params["mimeType"], + "bytes": 1 if mode == "wrong-export-size" else (fixture / name).stat().st_size, + })) +elif args[:3] == ["drive", "comments", "list"]: + assert params["fileId"] == "synthetic-doc" + assert "nextPageToken" in params["fields"] + if mode == "failed-comments" and params.get("pageToken"): + sys.exit(9) + if params.get("pageToken"): + print(json.dumps({"comments": [{"id": "two", "content": "Second comment"}]})) + else: + print(json.dumps({"nextPageToken": "page-two", "comments": [{"id": "one"}]})) +else: + raise AssertionError(args) +''' + +RENDER_STUB = r''' +import base64 +import os +from pathlib import Path +import sys +import time +assert sys.argv[1:4] == ["-png", "-r", "96"] +assert sys.argv[-2] == "document.pdf" +prefix = Path(sys.argv[-1]) +png = base64.b64decode(os.environ["STUB_PNG"]) +mode = os.environ.get("RENDER_MODE", "") +Path(str(prefix) + "-1.png").write_bytes(png) +if mode == "failed": + sys.exit(2) +if mode == "timeout": + time.sleep(10) +if mode == "gap": + Path(str(prefix) + "-3.png").write_bytes(png) +''' + + +class HTMLInspection(HTMLParser): + def __init__(self, text): + super().__init__() + self.tags = [] + self.references = [] + self.feed(text) + + def handle_starttag(self, tag, attrs): + self.tags.append((tag, dict(attrs))) + for name, value in attrs: + if name in ("src", "href", "data"): + self.references.append(value) + + +class BundleTestCase(unittest.TestCase): + def setUp(self): + # All generated files stay inside this assigned worktree and are removed. + self.temp = tempfile.TemporaryDirectory(dir=SCRIPT.parent) + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name).resolve() + self.fixture = fixtures(self.root / "fixtures") + self.bin = self.root / "bin" + self.bin.mkdir() + self.env = { + "PATH": str(self.bin), + "PYTHONDONTWRITEBYTECODE": "1", + "STUB_FIXTURES": str(self.fixture), + "STUB_LOG": str(self.root / "gws.log"), + "STUB_PNG": base64.b64encode(PNG).decode(), + "GOOGLE_WORKSPACE_CLI_CONFIG_DIR": str(self.root / "unused-config"), + "GOOGLE_WORKSPACE_CLI_SANITIZE_MODE": "block", + } + + def api(self): + self.assertTrue(SCRIPT.is_file(), "Missing executable review bundle companion") + if not hasattr(self, "_api"): + spec = importlib.util.spec_from_file_location("docs_review_bundle", SCRIPT) + self._api = importlib.util.module_from_spec(spec) + spec.loader.exec_module(self._api) + return self._api + + def executable(self, name, source): + path = self.bin / name + path.write_text(f"#!{sys.executable}\n" + source, encoding="utf-8") + path.chmod(0o700) + return path + + def run_bundle(self, *args, live=False, **env): + self.assertTrue(SCRIPT.is_file(), "Missing executable review bundle companion") + if live: + self.executable("gws", GWS_STUB) + source = ["--document-id", "synthetic-doc"] + else: + source = ["--from-fixture", "fixtures"] + return subprocess.run( + [sys.executable, "-B", str(SCRIPT), *source, *args], + cwd=self.root, + env={**self.env, **env}, + text=True, + capture_output=True, + timeout=15, + ) + + def manifest(self, directory="review"): + return json.loads((self.root / directory / "manifest.json").read_text()) + + +class ExtractionTests(BundleTestCase): + def extract(self, **limits): + return self.api().extract_docx( + self.fixture / "document.docx", self.root / "assets", **limits + ) + + def test_relationships_preserve_order_alt_text_and_table_context(self): + result = self.extract() + figures = result["figures"] + self.assertEqual([f["order"] for f in figures], [1, 2, 3]) + self.assertIn('Figure " onerror="bad()', figures[0]["alt"]) + self.assertIn("Table image context", figures[1]["nearby_text"]) + self.assertIsNone(figures[2]["asset"]) + self.assertEqual(figures[2]["availability"], "external-not-fetched") + self.assertIsNone(figures[0]["native_object_id"]) + self.assertEqual(figures[0]["mapping_confidence"], "docx-relationship-only") + + def test_duplicate_basenames_get_distinct_local_assets(self): + result = self.extract() + paths = [f["asset"] for f in result["figures"][:2]] + self.assertNotEqual(*paths) + self.assertEqual((self.root / paths[0]).read_bytes(), PNG) + self.assertEqual((self.root / paths[1]).read_bytes(), PNG + b"different") + + def test_zip_traversal_absolute_windows_and_control_paths_are_rejected(self): + api = self.api() + for index, name in enumerate( + ["../escape", "/escape", "C:/escape", r"..\escape", "word/../escape", "bad\x01"] + ): + with self.subTest(name=name): + members = {**docx_members(), name: b"bad"} + write_docx(self.fixture / "document.docx", members) + with self.assertRaises(api.BundleError): + api.extract_docx( + self.fixture / "document.docx", self.root / f"assets-{index}" + ) + self.assertFalse((self.root / "escape").exists()) + + def test_zip_symlink_rejected(self): + api = self.api() + with zipfile.ZipFile(self.fixture / "document.docx", "a") as archive: + link = zipfile.ZipInfo("word/media/link.png") + link.create_system = 3 + link.external_attr = (stat.S_IFLNK | 0o777) << 16 + archive.writestr(link, "../../../escape") + with self.assertRaises(api.BundleError): + self.extract() + + def test_member_total_and_count_limits_reject_before_writing_assets(self): + api = self.api() + for limits in ( + {"member_limit": 32}, + {"total_limit": 32}, + {"member_count": 2}, + ): + with self.subTest(limits=limits), self.assertRaises(api.BundleError): + self.extract(**limits) + self.assertFalse((self.root / "assets").exists()) + + def test_duplicate_archive_member_rejected(self): + api = self.api() + with zipfile.ZipFile(self.fixture / "document.docx", "a") as archive: + archive.writestr("WORD/MEDIA/IMAGE1.PNG", PNG) + with self.assertRaises(api.BundleError): + self.extract() + + def test_doctype_entities_and_utf16_are_rejected_even_in_unused_xml(self): + api = self.api() + for data in ( + b']>&x;', + b'', + ']>&x;'.encode("utf-16"), + ): + with self.subTest(data=data[:30]): + members = {**docx_members(), "word/unused.xml": data} + write_docx(self.fixture / "document.docx", members) + with self.assertRaises(api.BundleError): + self.extract() + self.assertFalse((self.root / "assets").exists()) + + def test_relationship_traversal_and_svg_are_not_local_html_assets(self): + members = docx_members() + members["word/_rels/document.xml.rels"] = members[ + "word/_rels/document.xml.rels" + ].replace(b"media/image1.png", b"../outside.png") + members["word/media/nested/image1.png"] = b"" + write_docx(self.fixture / "document.docx", members) + figures = self.extract()["figures"] + self.assertIsNone(figures[0]["asset"]) + self.assertIsNone(figures[1]["asset"]) + # The safe, now unreferenced raster is still preserved; the SVG is not. + files = list((self.root / "assets").iterdir()) + self.assertEqual(len(files), 1) + self.assertEqual(files[0].read_bytes(), PNG) + + def test_malformed_xml_and_corrupt_zip_fail_closed(self): + api = self.api() + for data in (b"not a zip", None): + if data is None: + write_docx( + self.fixture / "document.docx", + {**docx_members(), "word/document.xml": b"", html) + self.assertIn("<script>", html) + self.assertNotIn("SIGNED_SECRET", html) + self.assertNotIn("untrusted.invalid", html) + self.assertTrue(any(tag == "iframe" for tag, _ in inspection.tags)) + for tag, attrs in inspection.tags: + self.assertNotEqual(tag, "script") + self.assertFalse(any(name.startswith("on") for name in attrs)) + if tag == "iframe": + self.assertIn("sandbox", attrs) + for ref in inspection.references: + self.assertNotIn(":", ref) + self.assertNotIn("..", ref) + self.assertNotIn("%", ref) + self.assertTrue((self.root / "review" / ref).is_file(), ref) + + def test_native_outline_includes_tables_child_tabs_styles_and_uncertain_images(self): + result = self.run_bundle("review") + self.assertEqual(result.returncode, 0, result.stderr) + html = (self.root / "review/index.html").read_text() + for text in ("Table cell", "Nested tab paragraph", "HEADING_1", "tab-child"): + self.assertIn(text, html) + self.assertIn(" Date: Fri, 11 Sep 2026 12:23:54 -0300 Subject: [PATCH 14/19] fix(docs): correct bundle mapping and optional output states --- .changeset/docs-review-bundle.md | 5 +- examples/docs-review-bundle/README.md | 32 ++-- .../docs-review-bundle/docs_review_bundle.py | 128 +++++++++------ .../test_docs_review_bundle.py | 146 ++++++++++++++++-- 4 files changed, 245 insertions(+), 66 deletions(-) diff --git a/.changeset/docs-review-bundle.md b/.changeset/docs-review-bundle.md index bac3f4983..e8504e8ce 100644 --- a/.changeset/docs-review-bundle.md +++ b/.changeset/docs-review-bundle.md @@ -4,4 +4,7 @@ Add a standalone Python companion for visual Google Docs review bundles with native exports, safe DOCX raster extraction, local HTML, optional PDF page -previews, revision observations, and an offline fixture workflow. +previews with explicitly unverified page coverage, revision observations, +and an offline fixture workflow. Preserve nested image occurrences and +legitimate asset reuse, and keep oversized optional comments from failing +the required bundle. diff --git a/examples/docs-review-bundle/README.md b/examples/docs-review-bundle/README.md index fcc7f58ca..1943ac148 100644 --- a/examples/docs-review-bundle/README.md +++ b/examples/docs-review-bundle/README.md @@ -33,12 +33,17 @@ python3 examples/docs-review-bundle/docs_review_bundle.py \ - `--include-comments`: collect every returned Drive comments page. The entire optional artifact is omitted and marked unavailable if retrieval is incomplete - or fails. Comments are a separate observation, not revision-bound or mapped to + or fails, or if the final serialized artifact exceeds 20 MiB. Size is checked + before publication, so an oversized optional result does not fail the required + bundle. Comments are a separate observation, not revision-bound or mapped to PDF coordinates. Deleted comments are not requested. - `--render-pages`: use a trusted `pdftoppm` on `PATH` to generate 96 DPI PNGs. - Rendering is opt-in. Missing tools, failures, timeouts, invalid output and - noncontiguous page numbers retain the PDF and report no available raster - previews. Partial page output is discarded. + Its path is resolved before running inside the bundle, including relative + `PATH` entries. Rendering is opt-in. Missing tools, failures, timeouts, invalid + output and noncontiguous page numbers retain the PDF and report no available + raster previews. Outputs from these detected failures are discarded. After a + successful process exit and validation, previews are labeled `available` + with `coverage: "unverified"`; a missing page suffix cannot be detected. - `--timeout`: positive finite seconds per subprocess; default 60. This is not a total workflow deadline. - `--gws`: trusted executable, resolved before changing subprocess working @@ -78,6 +83,12 @@ required exports, validation, index generation and artifact hashing succeed. an atomic snapshot, successful optional rendering, or verified export tab coverage. Check `revisions`, `comments`, and `rendering` independently. +Page previews are never labeled `complete`. `rendering.status: "available"` +means that local PNG files passed validation, while +`rendering.coverage: "unverified"` means the original PDF page count was not +independently checked. Even a contiguous list beginning with page 1 may omit +later pages. The HTML displays this same coverage limitation. + Revision status is `mixed` for differing observed Docs revision IDs, `unchanged` for equal nonempty IDs, and `unknown` if either is missing. Even `unchanged` does not prove atomicity or that every export represents the same revision. @@ -103,10 +114,12 @@ assets. The manifest does not hash itself and is not an authenticity signature. - Google determines the PDF/DOCX/Markdown export layout and tab coverage. This companion cannot verify that every tab appears in those formats or associate a PDF page/DOCX figure with an exact native tab. -- DOCX figure order comes from main-document drawing relationships, including - drawings in tables. Alt text and nearby paragraphs are context, **not exact - captions**. Native object IDs are recorded separately; the companion never - fabricates a matching native ID or a source tab ID for a DOCX image. +- DOCX figure order follows individual image occurrences in the main document, + including drawings in tables and nested text boxes. Each occurrence uses its + nearest drawing's alt text and nearest paragraph's text; legitimate repeated + uses of an asset remain separate figures. Alt text and nearby paragraphs are + context, **not exact captions**. Native object IDs are recorded separately; + the companion never fabricates a matching native ID or source tab ID. - Media basenames are replaced with distinct generated local filenames. External, missing, traversing and unsupported relationships remain visibly unavailable. Unreferenced recognized rasters are retained as artifacts. @@ -123,7 +136,8 @@ assets. The manifest does not hash itself and is not an authenticity signature. - Each source/export/JSON artifact is capped at 20 MiB. Page output is limited to 500 files and 100 MiB total. PDF header/EOF and raster signatures are checked; these are **not full format validation**. Renderer exit success and - contiguous numbering do not independently prove the original PDF page count. + contiguous numbering do not independently prove the original PDF page count, + so preview coverage is always labeled unverified. - Subprocess capture is file-backed, with size checks after exit. Timeouts and post-render limits do not enforce disk or memory quotas on external tools. `pdftoppm`, `gws`, local viewers and the operator-controlled parent directory diff --git a/examples/docs-review-bundle/docs_review_bundle.py b/examples/docs-review-bundle/docs_review_bundle.py index fd54477d4..841dbe2d3 100755 --- a/examples/docs-review-bundle/docs_review_bundle.py +++ b/examples/docs-review-bundle/docs_review_bundle.py @@ -49,6 +49,7 @@ "Exports are sequential, not an atomic snapshot; unchanged revisions are observations only.", "Native outline includes body paragraphs, tables and nested tabs, not full layout or styling.", "Drive export tab coverage is not verified; pages and DOCX figures have no reliable tab mapping.", + "Page preview coverage is unverified; available previews may omit pages.", "DOCX relationships provide document order and nearby text, " "not exact captions or native Docs IDs.", "Only recognized PNG, JPEG, GIF and WebP media are previewed; " @@ -165,7 +166,7 @@ def safe_xml(data): raise BundleError("unsafe-xml") try: text = data.decode("utf-8-sig") - declaration = re.search(r"<\?xml[^>]*encoding=['\"]([^'\"]+)", text, re.I) + declaration = re.search(r"<\?xml[^>]*\bencoding\s*=\s*['\"]([^'\"]+)", text, re.I) if declaration and declaration[1].lower() not in ("utf-8", "utf8", "us-ascii"): raise BundleError("unsupported-xml-encoding") root = ET.fromstring(text) @@ -244,6 +245,40 @@ def zip_members(source, member_limit, total_limit, member_count): raise BundleError("invalid-zip") from None +def docx_image_occurrences(document): + """Visit each blip once, retaining its nearest paragraph and drawing.""" + paragraph_text = {} + paragraph_index = {} + drawing_alt = {} + occurrences = [] + pending = [(document, None, None)] + while pending: + node, paragraph, drawing = pending.pop() + if node.tag == W + "p": + paragraph = node + paragraph_index[node] = len(paragraph_text) + paragraph_text[node] = [] + elif node.tag == W + "drawing": + drawing = node + elif node.tag == W + "t" and paragraph is not None: + paragraph_text[paragraph].append(node.text or "") + elif node.tag == WP + "docPr" and drawing is not None: + drawing_alt.setdefault( + drawing, " ".join(node.get(field, "") for field in ("title", "descr")).strip() + ) + elif node.tag == A + "blip" and paragraph is not None and drawing is not None: + occurrences.append((node, paragraph, drawing)) + pending.extend((child, paragraph, drawing) for child in reversed(node)) + + texts = ["".join(parts) for parts in paragraph_text.values()] + for blip, paragraph, drawing in occurrences: + index = paragraph_index[paragraph] + nearby = texts[index] or " ".join( + texts[max(0, index - 1):index] + texts[index + 1:index + 2] + ) + yield blip, drawing_alt.get(drawing, ""), nearby + + def extract_docx( source, output, *, member_limit=FILE_LIMIT, total_limit=TOTAL_LIMIT, member_count=MEMBER_COUNT, @@ -274,50 +309,36 @@ def extract_docx( filename = f"image-{len(assets) + 1}.{extension}" write_bytes(output / filename, data) assets[name] = local_uri(f"{output.name}/{filename}") - paragraphs = list(document.iter(W + "p")) - texts = [ - "".join(text.text or "" for text in paragraph.iter(W + "t")) - for paragraph in paragraphs - ] figures = [] - for index, paragraph in enumerate(paragraphs): - nearby = texts[index] or " ".join( - texts[max(0, index - 1):index] + texts[index + 1:index + 2] - ) - for drawing in paragraph.iter(W + "drawing"): - properties = drawing.find(".//" + WP + "docPr") - alt = "" - if properties is not None: - alt = " ".join(properties.get(field, "") for field in ("title", "descr")).strip() - for blip in drawing.iter(A + "blip"): - identity = blip.get(R + "embed") or blip.get(R + "link") - relationship = relationships.get(identity, {}) - asset = None - availability = "missing-or-unsupported" - if relationship.get("TargetMode", "").lower() == "external": - availability = "external-not-fetched" - elif relationship.get("Type") == R[1:-1] + "/image": - target = relationship.get("Target", "") - # Allow only relative media targets in the document's part. - if ( - target.startswith("media/") - and not any(p in ("", ".", "..") for p in target.split("/")) - and not any(char in target for char in "\\:%?#") - ): - asset = assets.get(str(PurePosixPath("word") / target)) - if asset: - availability = "available" - figures.append({ - "order": len(figures) + 1, - "relationship_id": identity, - "asset": asset, - "alt": alt, - "nearby_text": nearby[:1000], - "availability": availability, - "mapping_confidence": "docx-relationship-only", - "native_object_id": None, - "source_tab_id": None, - }) + for blip, alt, nearby in docx_image_occurrences(document): + identity = blip.get(R + "embed") or blip.get(R + "link") + relationship = relationships.get(identity, {}) + asset = None + availability = "missing-or-unsupported" + if relationship.get("TargetMode", "").lower() == "external": + availability = "external-not-fetched" + elif relationship.get("Type") == R[1:-1] + "/image": + target = relationship.get("Target", "") + # Allow only relative media targets in the document's part. + if ( + target.startswith("media/") + and not any(p in ("", ".", "..") for p in target.split("/")) + and not any(char in target for char in "\\:%?#") + ): + asset = assets.get(str(PurePosixPath("word") / target)) + if asset: + availability = "available" + figures.append({ + "order": len(figures) + 1, + "relationship_id": identity, + "asset": asset, + "alt": alt, + "nearby_text": nearby[:1000], + "availability": availability, + "mapping_confidence": "docx-relationship-only", + "native_object_id": None, + "source_tab_id": None, + }) return {"figures": figures, "assets": list(assets.values())} @@ -428,6 +449,11 @@ def index_html(source, markdown, tabs, manifest, artifacts): f"{display(manifest['rendering'].get('reason', ''))}

", ] ) + if manifest["rendering"]["pages"]: + parts.append( + f"

Page coverage: {display(manifest['rendering']['coverage'])}. " + "The PDF page count has not been verified; previews may omit pages.

" + ) for page in manifest["rendering"]["pages"]: parts.append(f'{display(page)}') parts.append("

Native outline and tabs

") @@ -536,6 +562,7 @@ def render_pages(bundle, timeout, requested): if not renderer: return {"status": "unavailable", "reason": "pdftoppm-not-found; PDF retained", "pages": []} try: + renderer = str(Path(renderer).resolve()) with tempfile.TemporaryDirectory(prefix=".render-", dir=bundle) as temp: staging = Path(temp) prefix = str(staging.relative_to(bundle) / "page") @@ -560,7 +587,8 @@ def render_pages(bundle, timeout, requested): raise BundleError("invalid-render") staging.rename(bundle / "pages") return { - "status": "complete", + "status": "available", + "coverage": "unverified", "pages": [local_uri("pages/" + pages[number]) for number in sorted(pages)], } except (BundleError, OSError) as error: @@ -645,14 +673,20 @@ def build_bundle(args, bundle, manifest): value = parse_json(comments) if not isinstance(value.get("comments"), list) or value.get("nextPageToken"): raise BundleError("incomplete-comments") + if len(comments) > FILE_LIMIT: + raise BundleError("comments-size-limit") write_bytes(bundle / "comments.json", comments) artifacts.append("comments.json") manifest["comments"] = {"status": "available"} - except (BundleError, OSError): + except (BundleError, OSError) as error: (bundle / "comments.json").unlink(missing_ok=True) manifest["comments"] = { "status": "unavailable", - "reason": "retrieval-incomplete-or-failed", + "reason": ( + "comments-size-limit" + if isinstance(error, BundleError) and str(error) == "comments-size-limit" + else "retrieval-incomplete-or-failed" + ), } manifest["stage"] = "validate-and-extract" diff --git a/examples/docs-review-bundle/test_docs_review_bundle.py b/examples/docs-review-bundle/test_docs_review_bundle.py index 56ede7b74..84e9a6e0e 100644 --- a/examples/docs-review-bundle/test_docs_review_bundle.py +++ b/examples/docs-review-bundle/test_docs_review_bundle.py @@ -122,8 +122,8 @@ def native_document(): } -def pdf_bytes(): - """A complete synthetic one-page PDF, also usable by real pdftoppm.""" +def pdf_bytes(page_count=1): + """A complete synthetic PDF, also usable by real pdftoppm.""" content = ( b"BT /F1 22 Tf 48 720 Td (Synthetic Docs review) Tj ET\n" b"BT /F1 12 Tf 48 687 Td (Local fixture - no Google document) Tj ET\n" @@ -137,25 +137,36 @@ def pdf_bytes(): b"BT /F1 12 Tf 60 333 Td (Column A) Tj 258 0 Td (Column B) Tj ET\n" b"BT /F1 12 Tf 60 288 Td (Cell one) Tj 258 0 Td (Cell two) Tj ET\n" ) + content_id = page_count + 3 + font_id = page_count + 4 + kids = " ".join(f"{number} 0 R" for number in range(3, page_count + 3)) objects = [ b"<< /Type /Catalog /Pages 2 0 R >>", - b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>", - b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] " - b"/Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>", + f"<< /Type /Pages /Kids [{kids}] /Count {page_count} >>".encode(), + ] + objects.extend( + ( + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] " + f"/Resources << /Font << /F1 {font_id} 0 R >> >> /Contents {content_id} 0 R >>" + ).encode() + for _ in range(page_count) + ) + objects.extend([ f"<< /Length {len(content)} >>\nstream\n".encode() + content + b"endstream", b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", - ] + ]) data = b"%PDF-1.4\n" offsets = [0] for index, obj in enumerate(objects, 1): offsets.append(len(data)) data += f"{index} 0 obj\n".encode() + obj + b"\nendobj\n" startxref = len(data) - data += b"xref\n0 6\n0000000000 65535 f \n" + object_count = len(objects) + 1 + data += f"xref\n0 {object_count}\n0000000000 65535 f \n".encode() for offset in offsets[1:]: data += f"{offset:010d} 00000 n \n".encode() data += ( - f"trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n{startxref}\n%%EOF\n" + f"trailer\n<< /Size {object_count} /Root 1 0 R >>\nstartxref\n{startxref}\n%%EOF\n" ).encode() return data @@ -190,6 +201,22 @@ def docx_members(): } +def nested_docx_members(*, outer_image=False): + members = docx_members() + outer_blip = '' if outer_image else "" + members["word/document.xml"] = f""" + +Outer paragraph + +Inner paragraph + + + +{outer_blip} +""".encode() + return members + + def write_docx(path, members=None): with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as archive: for name, data in (members if members is not None else docx_members()).items(): @@ -267,6 +294,13 @@ def fixtures(path): elif args[:3] == ["drive", "comments", "list"]: assert params["fileId"] == "synthetic-doc" assert "nextPageToken" in params["fields"] + if mode == "expanded-comments": + # ~8 MiB on the wire; >24 MiB after ensure_ascii=True serialization. + print(json.dumps( + {"comments": [{"id": "large", "content": "é" * (4 * 1024 * 1024)}]}, + ensure_ascii=False, separators=(",", ":"), + )) + sys.exit(0) if mode == "failed-comments" and params.get("pageToken"): sys.exit(9) if params.get("pageToken"): @@ -389,6 +423,39 @@ def test_duplicate_basenames_get_distinct_local_assets(self): self.assertEqual((self.root / paths[0]).read_bytes(), PNG) self.assertEqual((self.root / paths[1]).read_bytes(), PNG + b"different") + def test_nested_text_box_image_has_one_occurrence_with_inner_ownership(self): + write_docx(self.fixture / "document.docx", nested_docx_members()) + figures = self.extract()["figures"] + self.assertEqual(len(figures), 1) + self.assertEqual(figures[0]["order"], 1) + self.assertEqual(figures[0]["alt"], "Inner image") + self.assertEqual(figures[0]["nearby_text"], "Inner paragraph") + self.assertEqual((self.root / figures[0]["asset"]).read_bytes(), PNG) + + def test_nested_image_order_and_context_follow_nearest_owners(self): + write_docx(self.fixture / "document.docx", nested_docx_members(outer_image=True)) + figures = self.extract()["figures"] + self.assertEqual([f["relationship_id"] for f in figures], ["rId1", "rId2"]) + self.assertEqual([f["alt"] for f in figures], ["Inner image", "Outer text box"]) + self.assertEqual( + [f["nearby_text"] for f in figures], ["Inner paragraph", "Outer paragraph"] + ) + self.assertEqual([f["order"] for f in figures], [1, 2]) + + def test_repeated_asset_uses_remain_distinct_figure_occurrences(self): + members = docx_members() + members["word/document.xml"] = members["word/document.xml"].replace( + b'r:embed="rId2"', b'r:embed="rId1"' + ) + write_docx(self.fixture / "document.docx", members) + figures = self.extract()["figures"] + self.assertEqual(len(figures), 3) + self.assertEqual([f["order"] for f in figures], [1, 2, 3]) + self.assertEqual(figures[0]["asset"], figures[1]["asset"]) + self.assertEqual([f["relationship_id"] for f in figures[:2]], ["rId1", "rId1"]) + self.assertNotEqual(figures[0]["alt"], figures[1]["alt"]) + self.assertEqual(figures[1]["nearby_text"], "Table image context") + def test_zip_traversal_absolute_windows_and_control_paths_are_rejected(self): api = self.api() for index, name in enumerate( @@ -445,6 +512,26 @@ def test_doctype_entities_and_utf16_are_rejected_even_in_unused_xml(self): self.extract() self.assertFalse((self.root / "assets").exists()) + def test_xml_encoding_allowlist_handles_declaration_whitespace(self): + api = self.api() + for encoding in ("UTF-7", "UTF-16", "UTF-32", "ISO-8859-1"): + for assignment in (f' = "{encoding}"', f"= '{encoding}'", f'\t=\n"{encoding}"'): + with self.subTest(encoding=encoding, assignment=assignment): + data = f''.encode("utf-8") + with self.assertRaises(api.BundleError) as raised: + api.safe_xml(data) + self.assertEqual(str(raised.exception), "unsupported-xml-encoding") + + def test_xml_utf8_bom_is_accepted_and_utf16_utf32_are_rejected(self): + api = self.api() + data = 'café'.encode("utf-8-sig") + root = api.safe_xml(data) + self.assertEqual(root.tag, "x") + self.assertEqual(root.text, "café") + for encoding in ("utf-16", "utf-32"): + with self.subTest(encoding=encoding), self.assertRaises(api.BundleError): + api.safe_xml("".encode(encoding)) + def test_relationship_traversal_and_svg_are_not_local_html_assets(self): members = docx_members() members["word/_rels/document.xml.rels"] = members[ @@ -715,10 +802,38 @@ def test_successful_renderer_publishes_local_page_previews(self): self.executable("pdftoppm", RENDER_STUB) result = self.run_bundle("--render-pages", "review") self.assertEqual(result.returncode, 0, result.stderr) - self.assertEqual(self.manifest()["rendering"]["status"], "complete") + self.assertEqual(self.manifest()["rendering"]["status"], "available") + self.assertEqual(self.manifest()["rendering"]["coverage"], "unverified") self.assertEqual(self.manifest()["rendering"]["pages"], ["pages/page-1.png"]) self.assertIn("pages/page-1.png", self.manifest()["artifacts"]) + def test_zero_exit_page_prefix_has_unverified_coverage_not_complete(self): + (self.fixture / "document.pdf").write_bytes(pdf_bytes(page_count=2)) + self.executable("pdftoppm", RENDER_STUB) + result = self.run_bundle("--render-pages", "review") + self.assertEqual(result.returncode, 0, result.stderr) + manifest = self.manifest() + self.assertEqual(manifest["status"], "complete") + self.assertEqual(manifest["rendering"]["status"], "available") + self.assertEqual(manifest["rendering"]["coverage"], "unverified") + self.assertEqual(manifest["rendering"]["pages"], ["pages/page-1.png"]) + html = (self.root / "review/index.html").read_text() + self.assertIn("Page coverage: unverified", html) + self.assertIn("previews may omit pages", html) + self.assertTrue((self.root / "review/document.pdf").is_file()) + + def test_renderer_discovered_on_relative_path_runs_in_bundle(self): + self.executable("pdftoppm", RENDER_STUB) + for output, search_path in (("relative-path", "bin"), ("absolute-path", str(self.bin))): + with self.subTest(search_path=search_path): + result = self.run_bundle("--render-pages", output, PATH=search_path) + self.assertEqual(result.returncode, 0, result.stderr) + rendering = self.manifest(output)["rendering"] + self.assertEqual(rendering["status"], "available") + self.assertEqual(rendering["coverage"], "unverified") + self.assertEqual(rendering["pages"], ["pages/page-1.png"]) + self.assertEqual((self.root / output / "pages/page-1.png").read_bytes(), PNG) + def test_renderer_failure_timeout_and_gap_publish_no_partial_page_list(self): self.executable("pdftoppm", RENDER_STUB) for mode in ("failed", "timeout", "gap"): @@ -808,6 +923,19 @@ def test_partial_comments_are_unavailable_not_published(self): self.assertEqual(self.manifest()["comments"]["status"], "unavailable") self.assertFalse((self.root / "review/comments.json").exists()) + def test_serialized_comments_over_limit_remain_an_optional_failure(self): + result = self.run_bundle( + "--include-comments", "review", live=True, STUB_MODE="expanded-comments" + ) + self.assertEqual(result.returncode, 0, result.stderr) + manifest = self.manifest() + self.assertEqual(manifest["status"], "complete") + self.assertEqual(manifest["comments"]["status"], "unavailable") + self.assertEqual(manifest["comments"]["reason"], "comments-size-limit") + self.assertNotIn("comments.json", manifest["artifacts"]) + self.assertFalse((self.root / "review/comments.json").exists()) + self.assertTrue((self.root / "review/index.html").is_file()) + if __name__ == "__main__": unittest.main() From d0dbd2f01416bff7697f08f198dbcda53060be40 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:39:06 -0300 Subject: [PATCH 15/19] fix(docs): accept canonical export receipts --- .changeset/docs-review-bundle.md | 3 +- .github/workflows/ci.yml | 10 ++ examples/docs-review-bundle/README.md | 22 +++- .../docs-review-bundle/docs_review_bundle.py | 2 +- .../test_docs_review_bundle.py | 119 +++++++++++++++++- 5 files changed, 150 insertions(+), 6 deletions(-) diff --git a/.changeset/docs-review-bundle.md b/.changeset/docs-review-bundle.md index e8504e8ce..286e9e7f0 100644 --- a/.changeset/docs-review-bundle.md +++ b/.changeset/docs-review-bundle.md @@ -7,4 +7,5 @@ native exports, safe DOCX raster extraction, local HTML, optional PDF page previews with explicitly unverified page coverage, revision observations, and an offline fixture workflow. Preserve nested image occurrences and legitimate asset reuse, and keep oversized optional comments from failing -the required bundle. +the required bundle. Verify each export's exact canonical destination against +the real CLI receipt. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 797c27d85..6953d6464 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -53,8 +53,18 @@ jobs: os: [ubuntu-latest, macos-latest] steps: - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4 + - name: Install Rust + uses: dtolnay/rust-toolchain@631a55b12751854ce901bb631d5902ceb48146f7 # stable + - name: Cache cargo + uses: Swatinem/rust-cache@ad397744b0d591a723ab90405b7247fac0e6b8db # v2 + with: + key: docs-review-bundle-${{ matrix.os }} + - name: Build CLI for export contract regression + run: cargo build --locked - name: Test stdlib companion with synthetic fixtures run: python3 -B -m unittest discover -s examples/docs-review-bundle -p 'test_*.py' -v + env: + GWS_TEST_BINARY: ${{ github.workspace }}/target/debug/gws changes: name: Detect Changes diff --git a/examples/docs-review-bundle/README.md b/examples/docs-review-bundle/README.md index 1943ac148..7975497f3 100644 --- a/examples/docs-review-bundle/README.md +++ b/examples/docs-review-bundle/README.md @@ -55,7 +55,9 @@ The companion requests `docs documents get` with `includeTabsContent: true` and uses `drive files export` with `--format json` and fixed relative `--output` filenames. Every `gws` subprocess runs inside the new bundle. Export success, MIME type, destination and byte count are checked against the -written artifact. This requires the existing `gws` binary-export receipt format. +written artifact. The receipt must name the exact canonical absolute destination, +as returned by `gws`; a matching basename alone is insufficient. This requires +the existing `gws` binary-export receipt format. Open `index.html` locally. The index has no JavaScript or remote dependencies. It contains a sandboxed PDF frame, optional page images, a paragraph/table @@ -192,5 +194,19 @@ python3 -B -m unittest discover \ Tests use generated JSON/PDF/DOCX files and explicit `gws`/renderer executables as stubs. Their subprocess environment omits real authentication settings; -no installed `gws`, real document, or network request is needed. CI runs this -same command in a dedicated Linux/macOS job. +no installed `gws`, real document, or network request is needed for those tests. + +To include the real CLI export-contract regression: + +```sh +cargo build --locked +GWS_TEST_BINARY="$PWD/target/debug/gws" python3 -B -m unittest discover \ + -s examples/docs-review-bundle -p 'test_*.py' -v +``` + +This additional test uses cached synthetic Discovery, a dummy token, isolated +configuration and ADC paths, and a loopback HTTP server. It downloads all three +generated exports through the real CLI and checks that the bundle completes. +It never contacts Google or uses real credentials. The dedicated Linux/macOS CI +job builds `gws` and always enables this test; local runs without +`GWS_TEST_BINARY` explicitly skip it. diff --git a/examples/docs-review-bundle/docs_review_bundle.py b/examples/docs-review-bundle/docs_review_bundle.py index 841dbe2d3..2ba9ee7c0 100755 --- a/examples/docs-review-bundle/docs_review_bundle.py +++ b/examples/docs-review-bundle/docs_review_bundle.py @@ -649,7 +649,7 @@ def build_bundle(args, bundle, manifest): data = read_bytes(bundle / name) if ( receipt.get("status") != "success" - or receipt.get("saved_file") != name + or receipt.get("saved_file") != str((bundle / name).resolve()) or str(receipt.get("mimeType", "")).split(";")[0].strip().lower() != mime or type(receipt.get("bytes")) is not int or receipt["bytes"] != len(data) diff --git a/examples/docs-review-bundle/test_docs_review_bundle.py b/examples/docs-review-bundle/test_docs_review_bundle.py index 84e9a6e0e..299ab0614 100644 --- a/examples/docs-review-bundle/test_docs_review_bundle.py +++ b/examples/docs-review-bundle/test_docs_review_bundle.py @@ -14,6 +14,7 @@ import base64 import contextlib import hashlib +import http.server from html.parser import HTMLParser import importlib.util import io @@ -24,8 +25,10 @@ import subprocess import sys import tempfile +import threading import unittest from unittest import mock +import urllib.parse import zipfile @@ -287,7 +290,11 @@ def fixtures(path): shutil.copyfile(fixture / name, name) print(json.dumps({ "status": "error" if mode == "bad-export-status" else "success", - "saved_file": name, + "saved_file": ( + str(Path.cwd().parent / "other" / name) if mode == "wrong-export-path" + else name if mode == "relative-export-path" + else str(Path(name).resolve()) + ), "mimeType": params["mimeType"], "bytes": 1 if mode == "wrong-export-size" else (fixture / name).stat().st_size, })) @@ -399,6 +406,116 @@ def manifest(self, directory="review"): return json.loads((self.root / directory / "manifest.json").read_text()) +class ExportReceiptTests(BundleTestCase): + def test_canonical_receipts_complete_all_exports(self): + result = self.run_bundle("review", live=True) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertEqual(self.manifest()["status"], "complete") + + def test_wrong_destination_or_relative_receipt_is_refused(self): + for mode in ["wrong-export-path", "relative-export-path"]: + with self.subTest(mode=mode): + result = self.run_bundle(mode, live=True, STUB_MODE=mode) + self.assertEqual(result.returncode, 1, result.stderr) + self.assertEqual(self.manifest(mode)["error"], "invalid-export-receipt") + + +@unittest.skipUnless(os.environ.get("GWS_TEST_BINARY"), "Set GWS_TEST_BINARY for real CLI coverage") +class RealCliExportTests(BundleTestCase): + def test_real_cli_exports_complete_bundle_with_canonical_receipts(self): + binary = Path(os.environ["GWS_TEST_BINARY"]).resolve(strict=True) + fixture = self.fixture + requests = [] + exports = { + "application/pdf": "document.pdf", + "application/vnd.openxmlformats-officedocument.wordprocessingml.document": "document.docx", + "text/markdown": "document.md", + } + + class Handler(http.server.BaseHTTPRequestHandler): + def log_message(self, *_args): + pass + + def do_GET(self): + parsed = urllib.parse.urlsplit(self.path) + path = urllib.parse.unquote(parsed.path) + params = urllib.parse.parse_qs(parsed.query) + requests.append((path, params, self.headers.get("Authorization"))) + if path == "/documents/synthetic-doc": + data = (fixture / "source.json").read_bytes() + mime = "application/json" + elif path == "/files/synthetic-doc/export": + mime = params.get("mimeType", [""])[0] + if mime not in exports: + self.send_error(400) + return + data = (fixture / exports[mime]).read_bytes() + else: + self.send_error(404) + return + self.send_response(200) + self.send_header("Content-Type", mime) + self.send_header("Content-Length", str(len(data))) + self.end_headers() + self.wfile.write(data) + + server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + config = self.root / "real-cli-config" + cache = config / "cache" + cache.mkdir(parents=True) + (self.root / ".env").write_text("") + for service, version, resource, method, id_field, path in [ + ("docs", "v1", "documents", "get", "documentId", "documents/{documentId}"), + ("drive", "v3", "files", "export", "fileId", "files/{fileId}/export"), + ]: + discovery = { + "name": service, "version": version, + "rootUrl": f"http://127.0.0.1:{server.server_port}/", + "resources": {resource: {"methods": {method: { + "httpMethod": "GET", "path": path, + "parameters": {id_field: { + "type": "string", "location": "path", "required": True, + }}, + }}}}, + } + (cache / f"{service}_{version}.json").write_text(json.dumps(discovery)) + env = { + **self.env, + "GOOGLE_WORKSPACE_CLI_CONFIG_DIR": str(config), + "GOOGLE_WORKSPACE_CLI_TOKEN": "synthetic-loopback-token", + "GOOGLE_APPLICATION_CREDENTIALS": str(self.root / "absent-adc.json"), + "GOOGLE_WORKSPACE_CLI_KEYRING_BACKEND": "file", + "GOOGLE_WORKSPACE_PROJECT_ID": "synthetic-loopback-project", + "HTTP_PROXY": "http://127.0.0.1:1", + "HTTPS_PROXY": "http://127.0.0.1:1", + "ALL_PROXY": "http://127.0.0.1:1", + "NO_PROXY": "127.0.0.1,localhost", + } + result = subprocess.run( + [sys.executable, "-B", str(SCRIPT), "--document-id", "synthetic-doc", + "--gws", str(binary), "--timeout", "10", "review"], + cwd=self.root, env=env, text=True, capture_output=True, timeout=25, + ) + self.assertEqual(result.returncode, 0, result.stderr) + manifest = self.manifest() + self.assertEqual(manifest["status"], "complete") + self.assertEqual(manifest["revisions"]["status"], "unchanged") + for name in exports.values(): + data = (self.root / "review" / name).read_bytes() + self.assertEqual(data, (fixture / name).read_bytes()) + self.assertEqual(manifest["artifacts"][name]["bytes"], len(data)) + self.assertEqual(len(requests), 5) + self.assertTrue(all(auth == "Bearer synthetic-loopback-token" + for _, _, auth in requests)) + finally: + server.shutdown() + server.server_close() + thread.join(timeout=5) + + class ExtractionTests(BundleTestCase): def extract(self, **limits): return self.api().extract_docx( From d73bb60bd48f02e8a367ccb7e8e51f92a670252b Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:51:16 -0300 Subject: [PATCH 16/19] ci(docs): isolate companion checks with verified action pins --- .github/workflows/ci.yml | 15 ---------- .github/workflows/docs-review.yml | 46 +++++++++++++++++++++++++++++++ 2 files changed, 46 insertions(+), 15 deletions(-) create mode 100644 .github/workflows/docs-review.yml diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6953d6464..7918eb92d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -30,21 +30,6 @@ env: SCCACHE_IGNORE_SERVER_IO_ERROR: "true" jobs: - docs-review: - name: Docs Review Python Example - runs-on: ${{ matrix.os }} - strategy: - matrix: - os: [ubuntu-latest, macos-latest] - env: - PYTHONDONTWRITEBYTECODE: "1" - steps: - - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4 - - name: Test with synthetic fixtures and stub gws - run: | - python3 --version - python3 -m unittest discover -s examples/docs-review -p 'test_*.py' -v - docs-review-bundle: name: Docs Review Bundle (Python) runs-on: ${{ matrix.os }} diff --git a/.github/workflows/docs-review.yml b/.github/workflows/docs-review.yml new file mode 100644 index 000000000..a530b13f8 --- /dev/null +++ b/.github/workflows/docs-review.yml @@ -0,0 +1,46 @@ +# Copyright 2026 Google LLC +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +name: Docs Review Example + +on: + push: + branches: [main] + pull_request: + branches: [main] + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: ${{ github.ref != 'refs/heads/main' }} + +jobs: + docs-review: + name: Docs Review Python Example + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [ubuntu-latest, macos-latest] + env: + PYTHONDONTWRITEBYTECODE: "1" + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + persist-credentials: false + - name: Test with synthetic fixtures and stub gws + run: | + python3 --version + python3 -m unittest discover -s examples/docs-review -p 'test_*.py' -v From e3eb4cbba57a600fd7bee67df8059a2c27ceaeb9 Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 12:51:16 -0300 Subject: [PATCH 17/19] ci(docs): isolate companion checks with verified action pins --- .github/workflows/ci.yml | 21 --------- .github/workflows/docs-review-bundle.yml | 54 ++++++++++++++++++++++++ 2 files changed, 54 insertions(+), 21 deletions(-) create mode 100644 .github/workflows/docs-review-bundle.yml diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7918eb92d..1eedc9972 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -30,27 +30,6 @@ env: SCCACHE_IGNORE_SERVER_IO_ERROR: "true" jobs: - docs-review-bundle: - name: Docs Review Bundle (Python) - runs-on: ${{ matrix.os }} - strategy: - matrix: - os: [ubuntu-latest, macos-latest] - steps: - - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4 - - name: Install Rust - uses: dtolnay/rust-toolchain@631a55b12751854ce901bb631d5902ceb48146f7 # stable - - name: Cache cargo - uses: Swatinem/rust-cache@ad397744b0d591a723ab90405b7247fac0e6b8db # v2 - with: - key: docs-review-bundle-${{ matrix.os }} - - name: Build CLI for export contract regression - run: cargo build --locked - - name: Test stdlib companion with synthetic fixtures - run: python3 -B -m unittest discover -s examples/docs-review-bundle -p 'test_*.py' -v - env: - GWS_TEST_BINARY: ${{ github.workspace }}/target/debug/gws - changes: name: Detect Changes runs-on: ubuntu-latest diff --git a/.github/workflows/docs-review-bundle.yml b/.github/workflows/docs-review-bundle.yml new file mode 100644 index 000000000..a0f651c61 --- /dev/null +++ b/.github/workflows/docs-review-bundle.yml @@ -0,0 +1,54 @@ +# Copyright 2026 Google LLC +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +name: Docs Review Bundle + +on: + push: + branches: [main] + pull_request: + branches: [main] + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: ${{ github.ref != 'refs/heads/main' }} + +jobs: + docs-review-bundle: + name: Docs Review Bundle (Python) + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [ubuntu-latest, macos-latest] + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + persist-credentials: false + - name: Install Rust + uses: dtolnay/rust-toolchain@d1031067263f94b142dd6c0ce24c5eb9d02d52a0 # master + with: + toolchain: stable + - name: Cache cargo + uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 + with: + key: docs-review-bundle-${{ matrix.os }} + - name: Build CLI for export contract regression + run: cargo build --locked + - name: Test stdlib companion with synthetic fixtures + run: python3 -B -m unittest discover -s examples/docs-review-bundle -p 'test_*.py' -v + env: + GWS_TEST_BINARY: ${{ github.workspace }}/target/debug/gws From 8a1cfbdddf5e2b424edd504ccf4bd2cbbcfbc29e Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 14:17:32 -0300 Subject: [PATCH 18/19] test(cli): consume complete synthetic HTTP requests --- crates/google-workspace-cli/tests/dry_run.rs | 62 ++++++++++++++++++++ 1 file changed, 62 insertions(+) diff --git a/crates/google-workspace-cli/tests/dry_run.rs b/crates/google-workspace-cli/tests/dry_run.rs index bcd578e8e..c760dccd4 100644 --- a/crates/google-workspace-cli/tests/dry_run.rs +++ b/crates/google-workspace-cli/tests/dry_run.rs @@ -57,6 +57,9 @@ impl NetworkTrap { let worker = thread::spawn(move || loop { match listener.accept() { Ok((mut stream, _)) => { + // Accepted sockets may inherit the listener's nonblocking + // mode. Request reads need the timeout below on every OS. + stream.set_nonblocking(false).unwrap(); // Record the connection even if the client fails before // sending HTTP (for example, during TLS/proxy setup). let mut requests = captured.lock().unwrap(); @@ -72,6 +75,25 @@ impl NetworkTrap { Ok(size) => header.extend_from_slice(&buffer[..size]), } } + // TCP may deliver headers and body in separate reads. + // Consume the body before closing the connection, or unread + // request bytes can reset it and hide our synthetic 403. + if let Some(end) = header.windows(4).position(|w| w == b"\r\n\r\n") { + let body_len = String::from_utf8_lossy(&header[..end]) + .lines() + .filter_map(|line| line.split_once(':')) + .find(|(name, _)| name.eq_ignore_ascii_case("content-length")) + .map(|(_, value)| value.trim().parse::().unwrap()) + .unwrap_or(0); + let expected_len = end + 4 + body_len; + while header.len() < expected_len { + let remaining = (expected_len - header.len()).min(buffer.len()); + match stream.read(&mut buffer[..remaining]) { + Ok(0) | Err(_) => break, + Ok(size) => header.extend_from_slice(&buffer[..size]), + } + } + } *requests.last_mut().unwrap() = String::from_utf8_lossy(&header).into_owned(); let body = r#"{"error":{"code":403,"message":"Synthetic denial","errors":[{"reason":"forbidden"}]}}"#; let _ = write!( @@ -587,3 +609,43 @@ fn raw_real_request_uses_token_and_preserves_api_failure() { fn docs_write_real_request_uses_token_and_preserves_api_failure() { assert_authenticated_api_failure(WRITE); } + +#[test] +fn network_trap_waits_for_split_request_body_before_responding() { + use std::net::TcpStream; + + let trap = NetworkTrap::new(); + let address = trap + .url + .strip_prefix("http://") + .unwrap() + .trim_end_matches('/'); + let mut stream = TcpStream::connect(address).unwrap(); + stream + .set_read_timeout(Some(Duration::from_millis(100))) + .unwrap(); + stream + .write_all(b"POST / HTTP/1.1\r\nHost: localhost\r\nContent-Length: 5\r\n\r\n") + .unwrap(); + + let mut byte = [0; 1]; + let error = stream + .read(&mut byte) + .expect_err("must consume the body before responding"); + assert!(matches!( + error.kind(), + std::io::ErrorKind::WouldBlock | std::io::ErrorKind::TimedOut + )); + + stream.write_all(b"hello").unwrap(); + stream + .set_read_timeout(Some(Duration::from_secs(2))) + .unwrap(); + let mut response = String::new(); + stream.read_to_string(&mut response).unwrap(); + assert!( + response.starts_with("HTTP/1.1 403 Forbidden\r\n"), + "{response}" + ); + assert!(trap.requests.lock().unwrap()[0].ends_with("hello")); +} From 18410e92c55e74f793f1f9097bee145802a041ac Mon Sep 17 00:00:00 2001 From: ratovarius Date: Fri, 11 Sep 2026 14:18:08 -0300 Subject: [PATCH 19/19] chore(fork): publish maintained Docs workflow improvements --- .changeset/maintained-public-fork.md | 7 + .github/CODEOWNERS | 13 +- .github/upstream-workflows/README.md | 21 +++ .../audit.yml | 0 .../automation.yml | 0 .../{workflows => upstream-workflows}/ci.yml | 0 .../{workflows => upstream-workflows}/cla.yml | 0 .../coverage.yml | 0 .../generate-skills.yml | 0 .../policy.yml | 0 .../publish-skills.yml | 0 .../release-changesets.yml | 0 .../release.yml | 0 .../stale.yml | 0 .github/workflows/fork-ci.yml | 64 +++++++++ AGENTS.md | 8 +- CONTRIBUTING.md | 129 ++++++++++++++++++ FORK.md | 74 ++++++++++ README.md | 69 ++++++---- SECURITY.md | 14 +- crates/google-workspace-cli/Cargo.toml | 4 +- crates/google-workspace-cli/README.md | 20 +-- .../src/generate_skills.rs | 4 +- crates/google-workspace-cli/src/main.rs | 5 +- crates/google-workspace/Cargo.toml | 4 +- package.json | 8 +- skills/gws-shared/SKILL.md | 4 +- 27 files changed, 380 insertions(+), 68 deletions(-) create mode 100644 .changeset/maintained-public-fork.md create mode 100644 .github/upstream-workflows/README.md rename .github/{workflows => upstream-workflows}/audit.yml (100%) rename .github/{workflows => upstream-workflows}/automation.yml (100%) rename .github/{workflows => upstream-workflows}/ci.yml (100%) rename .github/{workflows => upstream-workflows}/cla.yml (100%) rename .github/{workflows => upstream-workflows}/coverage.yml (100%) rename .github/{workflows => upstream-workflows}/generate-skills.yml (100%) rename .github/{workflows => upstream-workflows}/policy.yml (100%) rename .github/{workflows => upstream-workflows}/publish-skills.yml (100%) rename .github/{workflows => upstream-workflows}/release-changesets.yml (100%) rename .github/{workflows => upstream-workflows}/release.yml (100%) rename .github/{workflows => upstream-workflows}/stale.yml (100%) create mode 100644 .github/workflows/fork-ci.yml create mode 100644 CONTRIBUTING.md create mode 100644 FORK.md diff --git a/.changeset/maintained-public-fork.md b/.changeset/maintained-public-fork.md new file mode 100644 index 000000000..874117e7b --- /dev/null +++ b/.changeset/maintained-public-fork.md @@ -0,0 +1,7 @@ +--- +"@googleworkspace/cli": patch +--- + +Publish the integrated Docs workflow improvements in the independent ratovarius +fork with upstream attribution, source-install instructions, contribution +tracking, and credential-independent CI. diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index 3c8b319f2..6252656b8 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -1,10 +1,3 @@ -# Codeowners - -# Core engine code strictly requires your review -# Isolates agents to `skills/` or `src/helpers/` unless absolutely necessary -/src/main.rs @jpoehnelt -/src/executor.rs @jpoehnelt -/src/discovery.rs @jpoehnelt -/src/commands.rs @jpoehnelt -/src/auth.rs @jpoehnelt -/src/schema.rs @jpoehnelt +# Review ownership for the independently maintained ratovarius fork. +# Original project authorship is preserved in FORK.md and the source history. +* @ratovarius diff --git a/.github/upstream-workflows/README.md b/.github/upstream-workflows/README.md new file mode 100644 index 000000000..8a0b025a7 --- /dev/null +++ b/.github/upstream-workflows/README.md @@ -0,0 +1,21 @@ +# Upstream automation reference + +These eleven workflow files are preserved unchanged from +`googleworkspace/cli` at `a3768d0e82ad83cca2da97724e46bea4ff0e6dbd`. +GitHub Actions only loads workflows from `.github/workflows`, so these archived +copies do not run in this fork. + +The upstream workflows include Google-specific CLA, bot, package publishing, +live-account smoke tests, and policy integrations. The fork instead runs +`fork-ci.yml`, `docs-review.yml`, and `docs-review-bundle.yml` with read-only +permissions and synthetic test data. They need no Google account credentials. + +This initial fork CI covers Rust tests/builds and the two Python companions on +Linux and macOS, plus Rust formatting and Clippy. It does not reproduce the +upstream release matrix, Nix checks, coverage reporting, dependency audit, or +scheduled skill regeneration. + +When syncing upstream, review changes here and any newly introduced active +workflows. Port useful checks deliberately; restore publishing only after +configuring a distinct fork release identity and destinations. See +[`CONTRIBUTING.md`](../../CONTRIBUTING.md). diff --git a/.github/workflows/audit.yml b/.github/upstream-workflows/audit.yml similarity index 100% rename from .github/workflows/audit.yml rename to .github/upstream-workflows/audit.yml diff --git a/.github/workflows/automation.yml b/.github/upstream-workflows/automation.yml similarity index 100% rename from .github/workflows/automation.yml rename to .github/upstream-workflows/automation.yml diff --git a/.github/workflows/ci.yml b/.github/upstream-workflows/ci.yml similarity index 100% rename from .github/workflows/ci.yml rename to .github/upstream-workflows/ci.yml diff --git a/.github/workflows/cla.yml b/.github/upstream-workflows/cla.yml similarity index 100% rename from .github/workflows/cla.yml rename to .github/upstream-workflows/cla.yml diff --git a/.github/workflows/coverage.yml b/.github/upstream-workflows/coverage.yml similarity index 100% rename from .github/workflows/coverage.yml rename to .github/upstream-workflows/coverage.yml diff --git a/.github/workflows/generate-skills.yml b/.github/upstream-workflows/generate-skills.yml similarity index 100% rename from .github/workflows/generate-skills.yml rename to .github/upstream-workflows/generate-skills.yml diff --git a/.github/workflows/policy.yml b/.github/upstream-workflows/policy.yml similarity index 100% rename from .github/workflows/policy.yml rename to .github/upstream-workflows/policy.yml diff --git a/.github/workflows/publish-skills.yml b/.github/upstream-workflows/publish-skills.yml similarity index 100% rename from .github/workflows/publish-skills.yml rename to .github/upstream-workflows/publish-skills.yml diff --git a/.github/workflows/release-changesets.yml b/.github/upstream-workflows/release-changesets.yml similarity index 100% rename from .github/workflows/release-changesets.yml rename to .github/upstream-workflows/release-changesets.yml diff --git a/.github/workflows/release.yml b/.github/upstream-workflows/release.yml similarity index 100% rename from .github/workflows/release.yml rename to .github/upstream-workflows/release.yml diff --git a/.github/workflows/stale.yml b/.github/upstream-workflows/stale.yml similarity index 100% rename from .github/workflows/stale.yml rename to .github/upstream-workflows/stale.yml diff --git a/.github/workflows/fork-ci.yml b/.github/workflows/fork-ci.yml new file mode 100644 index 000000000..a92b56149 --- /dev/null +++ b/.github/workflows/fork-ci.yml @@ -0,0 +1,64 @@ +name: Fork CI + +on: + push: + branches: [main] + pull_request: + branches: [main] + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: ${{ github.ref != 'refs/heads/main' }} + +env: + CARGO_TERM_COLOR: always + +jobs: + test: + name: Rust tests (${{ matrix.os }}) + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-latest] + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + persist-credentials: false + - name: Install Rust + uses: dtolnay/rust-toolchain@d1031067263f94b142dd6c0ce24c5eb9d02d52a0 # master + with: + toolchain: stable + - name: Cache cargo + uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 + with: + key: fork-test-${{ matrix.os }} + - name: Test workspace with synthetic fixtures + run: cargo test --workspace --locked + - name: Build workspace + run: cargo build --workspace --locked + + lint: + name: Rust formatting and Clippy + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4.3.1 + with: + persist-credentials: false + - name: Install Rust + uses: dtolnay/rust-toolchain@d1031067263f94b142dd6c0ce24c5eb9d02d52a0 # master + with: + toolchain: stable + components: rustfmt, clippy + - name: Cache cargo + uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 + with: + key: fork-lint + - name: Check formatting + run: cargo fmt --all -- --check + - name: Clippy + run: cargo clippy --workspace --locked -- -D warnings diff --git a/AGENTS.md b/AGENTS.md index 196fc4fda..8c2347d19 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,5 +1,9 @@ # AGENTS.md +This is the independently maintained `ratovarius/cli` fork. Read +[FORK.md](FORK.md) for upstream attribution and the feature ledger, and +[CONTRIBUTING.md](CONTRIBUTING.md) for fork PRs, upstream submissions, and CI. + ## Project Overview `gws` is a Rust CLI tool for interacting with Google Workspace APIs. It dynamically generates its command surface at runtime by parsing Google Discovery Service JSON documents. @@ -13,7 +17,7 @@ ## Build & Test > [!IMPORTANT] -> **Test Coverage**: The `codecov/patch` check requires that new or modified lines are covered by tests. When adding code, extract testable helper functions rather than embedding logic in `main`/`run` where it's hard to unit-test. Run `cargo test` locally and verify new branches are exercised. +> **Test Coverage**: Cover new or modified behavior with tests. When adding code, extract testable helper functions rather than embedding logic in `main`/`run` where it's hard to unit-test. Run `cargo test` locally and verify new branches are exercised. Upstream submissions may additionally require `codecov/patch`; this fork's active checks are documented in `CONTRIBUTING.md`. ```bash cargo build # Build in dev mode @@ -33,7 +37,7 @@ Every PR must include a changeset file. Create one at `.changeset/gws +

gws — ratovarius fork

+ +**An independently maintained public fork of +[googleworkspace/cli](https://github.com/googleworkspace/cli).** +Built on the original project's work, with additional Google Docs reading, +review, export, and credential-handling improvements. Original authorship, +history, and the Apache-2.0 license are preserved. + +See [what this fork adds and its upstream PRs](FORK.md), or +[contribute here](CONTRIBUTING.md). This fork can develop independently while +focused improvements are offered back to the original project. **One CLI for all of Google Workspace — built for humans and AI agents.**
Drive, Gmail, Calendar, and every Workspace API. Zero boilerplate. Structured JSON output. 40+ agent skills included. > [!NOTE] -> This is **not** an officially supported Google product. +> This fork is maintained by [ratovarius](https://github.com/ratovarius). +> It is **not** an officially supported Google product.

- npm version - license - CI status - install size + license + Fork CI status


-⬇️ **[Download the latest release for your OS](https://github.com/googleworkspace/cli/releases)** +**[Install this fork from source](#installation)** · +[Report an issue](https://github.com/ratovarius/cli/issues) `gws` doesn't ship a static list of commands. It reads Google's own [Discovery Service](https://developers.google.com/discovery) at runtime and builds its entire command surface dynamically. When Google Workspace adds an API endpoint or method, `gws` picks it up automatically. @@ -38,37 +48,38 @@ Drive, Gmail, Calendar, and every Workspace API. Zero boilerplate. Structured JS ## Prerequisites -- **Node.js 18+** — for `npm install` (or download a pre-built binary from [GitHub Releases](https://github.com/googleworkspace/cli/releases)) +- **Stable Rust and Cargo** — to build this fork from source +- **Python 3.10+** — for the optional Docs review and export companions (POSIX systems) - **A Google Cloud project** — required for OAuth credentials. You can create one via the [Google Cloud Console](https://console.cloud.google.com/) or with the [`gcloud` CLI](https://cloud.google.com/sdk/docs/install) or with the `gws auth setup` command. - **A Google account** with access to Google Workspace ## Installation -The recommended way to install `gws` is to download the pre-built binary for your OS and architecture from the **[GitHub Releases](https://github.com/googleworkspace/cli/releases)** page. Extract the archive and place the `gws` binary in your `$PATH`. - -For convenience, you can also use `npm` to automate downloading the appropriate binary from GitHub Releases: - -```bash -npm install -g @googleworkspace/cli -``` - -Or build from source: +Install the maintained fork's `main` from source: ```bash -cargo install --git https://github.com/googleworkspace/cli --locked +cargo install --git https://github.com/ratovarius/cli --branch main --locked google-workspace-cli ``` -A Nix flake is also available at `github:googleworkspace/cli` +To use the Docs review companions as well, keep a source checkout: ```bash -nix run github:googleworkspace/cli +git clone https://github.com/ratovarius/cli.git +cd cli +cargo build --workspace --locked +export PATH="$PWD/target/debug:$PATH" ``` -On macOS and Linux, you can also install via [Homebrew](https://brew.sh/): +See [Docs review](examples/docs-review/README.md) and +[visual review bundles](examples/docs-review-bundle/README.md) for their commands. +Check `command -v gws` to confirm which installed binary your shell will use. -```bash -brew install googleworkspace-cli -``` +This initial fork publication provides source on GitHub. The upstream +`@googleworkspace/cli` npm package, `google-workspace-cli` crates.io package, +Homebrew package, and [upstream releases](https://github.com/googleworkspace/cli/releases) +install the original distribution and do not include fork-only improvements. +For a reproducible source build, replace `--branch main` with `--rev COMMIT_SHA` +using the fork commit you have reviewed. ## Quick Start @@ -295,11 +306,11 @@ The repo ships 100+ Agent Skills (`SKILL.md` files) — one for every supported ```bash # Install all skills at once -npx skills add https://github.com/googleworkspace/cli +npx skills add https://github.com/ratovarius/cli # Or pick only what you need -npx skills add https://github.com/googleworkspace/cli/tree/main/skills/gws-drive -npx skills add https://github.com/googleworkspace/cli/tree/main/skills/gws-gmail +npx skills add https://github.com/ratovarius/cli/tree/main/skills/gws-drive +npx skills add https://github.com/ratovarius/cli/tree/main/skills/gws-gmail ```
@@ -313,7 +324,7 @@ ln -s $(pwd)/skills/gws-* ~/.openclaw/skills/ cp -r skills/gws-drive skills/gws-gmail ~/.openclaw/skills/ ``` -The `gws-shared` skill includes an `install` block so OpenClaw auto-installs the CLI via `npm` if `gws` isn't on PATH. +Install this fork using the source instructions above before using agent skills. An agent installer that uses the upstream npm package will install the original distribution.
@@ -327,7 +338,7 @@ The `gws-shared` skill includes an `install` block so OpenClaw auto-installs the 2. Install the extension into the Gemini CLI: ```bash - gemini extensions install https://github.com/googleworkspace/cli + gemini extensions install https://github.com/ratovarius/cli ``` Installing this extension gives your Gemini CLI agent direct access to all `gws` commands and Google Workspace agent skills. Because `gws` handles its own authentication securely, you simply need to authenticate your terminal once prior to using the agent, and the extension will automatically inherit your credentials. diff --git a/SECURITY.md b/SECURITY.md index 07bc436f3..13c40a975 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -1,6 +1,12 @@ # Report a security issue -To report a security issue, please use [https://g.co/vulnz](https://g.co/vulnz). We use -[https://g.co/vulnz](https://g.co/vulnz) for our intake, and do coordination and disclosure here on -GitHub (including using GitHub Security Advisory). The Google Security Team will -respond within 5 working days of your report on [https://g.co/vulnz](https://g.co/vulnz). +For vulnerabilities in this independently maintained fork, use +[GitHub's private vulnerability reporting](https://github.com/ratovarius/cli/security/advisories/new). +Include the affected commit, reproduction steps, and impact. Do not include +real OAuth tokens, document contents, or other credentials in a public issue. + +The fork maintainer coordinates reports for changes maintained here. This fork +does not promise a Google Security Team response or a fixed response time. + +For vulnerabilities also affecting the original project, follow the +[upstream security policy](https://github.com/googleworkspace/cli/security/policy). diff --git a/crates/google-workspace-cli/Cargo.toml b/crates/google-workspace-cli/Cargo.toml index 058b109e6..78929df0d 100644 --- a/crates/google-workspace-cli/Cargo.toml +++ b/crates/google-workspace-cli/Cargo.toml @@ -18,8 +18,8 @@ version = "0.22.5" edition = "2021" description = "Google Workspace CLI — dynamic command surface from Discovery Service" license = "Apache-2.0" -repository = "https://github.com/googleworkspace/cli" -homepage = "https://github.com/googleworkspace/cli" +repository = "https://github.com/ratovarius/cli" +homepage = "https://github.com/ratovarius/cli" readme = "README.md" authors = ["Justin Poehnelt"] keywords = ["cli", "google-workspace", "google", "drive", "gmail"] diff --git a/crates/google-workspace-cli/README.md b/crates/google-workspace-cli/README.md index 901fae46a..ae06d9d31 100644 --- a/crates/google-workspace-cli/README.md +++ b/crates/google-workspace-cli/README.md @@ -4,18 +4,20 @@ `gws` dynamically generates its command surface at runtime by reading Google's [Discovery Service](https://developers.google.com/discovery). Drive, Gmail, Calendar, and every Workspace API — zero boilerplate, structured JSON output, 40+ agent skills included. -## Install +## Install this fork -Download the pre-built binary for your OS and architecture from the **[GitHub Releases](https://github.com/googleworkspace/cli/releases)** page. - -Alternatively, you can use package managers as a convenience layer: +This is the independently maintained [ratovarius fork](https://github.com/ratovarius/cli) +of [googleworkspace/cli](https://github.com/googleworkspace/cli). +Original authorship and the Apache-2.0 license are retained. ```bash -npm install -g @googleworkspace/cli # npm (downloads GitHub release binary) -cargo install google-workspace-cli # crates.io -nix run github:googleworkspace/cli # nix +cargo install --git https://github.com/ratovarius/cli --branch main --locked google-workspace-cli ``` +Upstream registry packages and binary releases do not include fork-only changes. +See the [fork ledger](https://github.com/ratovarius/cli/blob/main/FORK.md) +for improvements and upstream PRs. + ## Quick Start ```bash @@ -26,8 +28,8 @@ gws gmail users.messages list --params '{"maxResults": 3}' ## Documentation -See the [full README](https://github.com/googleworkspace/cli#readme) for authentication setup, helper commands, agent skills, and more. +See the [full README](https://github.com/ratovarius/cli#readme) for authentication setup, helper commands, agent skills, and more. ## License -Apache-2.0 — see [LICENSE](https://github.com/googleworkspace/cli/blob/main/LICENSE). +Apache-2.0 — see [LICENSE](https://github.com/ratovarius/cli/blob/main/LICENSE). diff --git a/crates/google-workspace-cli/src/generate_skills.rs b/crates/google-workspace-cli/src/generate_skills.rs index ce2654077..e99591b4a 100644 --- a/crates/google-workspace-cli/src/generate_skills.rs +++ b/crates/google-workspace-cli/src/generate_skills.rs @@ -747,8 +747,8 @@ gws [sub-resource] [flags] ## Community & Feedback Etiquette -- Encourage agents/users to star the repository when they find the project useful: `https://github.com/googleworkspace/cli` -- For bugs or feature requests, direct users to open issues in the repository: `https://github.com/googleworkspace/cli/issues` +- This is the independently maintained `https://github.com/ratovarius/cli` fork of `https://github.com/googleworkspace/cli`; see `FORK.md` for attribution and upstream contributions. +- For bugs or feature requests, direct users to open issues in the fork: `https://github.com/ratovarius/cli/issues` - Before creating a new issue, **always** search existing issues and feature requests first - If a matching issue already exists, add context by commenting on the existing thread instead of creating a duplicate "# diff --git a/crates/google-workspace-cli/src/main.rs b/crates/google-workspace-cli/src/main.rs index ea53a93b8..63de1d1c0 100644 --- a/crates/google-workspace-cli/src/main.rs +++ b/crates/google-workspace-cli/src/main.rs @@ -541,8 +541,9 @@ fn print_usage() { } println!(); println!("COMMUNITY:"); - println!(" Star the repo: https://github.com/googleworkspace/cli"); - println!(" Report bugs / request features: https://github.com/googleworkspace/cli/issues"); + println!(" Fork: https://github.com/ratovarius/cli"); + println!(" Upstream: https://github.com/googleworkspace/cli"); + println!(" Report bugs / request features: https://github.com/ratovarius/cli/issues"); println!(" Please search existing issues first; if one already exists, comment there."); println!(); println!("DISCLAIMER:"); diff --git a/crates/google-workspace/Cargo.toml b/crates/google-workspace/Cargo.toml index 3dc9f0830..5351565ed 100644 --- a/crates/google-workspace/Cargo.toml +++ b/crates/google-workspace/Cargo.toml @@ -18,8 +18,8 @@ version = "0.22.5" edition = "2021" description = "Google Workspace API client — Discovery Document types, service registry, and HTTP utilities" license = "Apache-2.0" -repository = "https://github.com/googleworkspace/cli" -homepage = "https://github.com/googleworkspace/cli" +repository = "https://github.com/ratovarius/cli" +homepage = "https://github.com/ratovarius/cli" readme = "README.md" authors = ["Justin Poehnelt"] keywords = ["google-workspace", "google", "discovery", "api-client"] diff --git a/package.json b/package.json index 2a556aade..1084846b6 100644 --- a/package.json +++ b/package.json @@ -2,19 +2,19 @@ "name": "@googleworkspace/cli", "version": "0.22.5", "private": true, - "description": "Google Workspace CLI — dynamic command surface from Discovery Service", + "description": "Google Workspace CLI \u2014 dynamic command surface from Discovery Service", "license": "Apache-2.0", "repository": { "type": "git", - "url": "https://github.com/googleworkspace/cli.git" + "url": "https://github.com/ratovarius/cli.git" }, "author": { "name": "Justin Poehnelt", "email": "justin.poehnelt@gmail.com" }, - "homepage": "https://github.com/googleworkspace/cli", + "homepage": "https://github.com/ratovarius/cli", "bugs": { - "url": "https://github.com/googleworkspace/cli/issues" + "url": "https://github.com/ratovarius/cli/issues" }, "scripts": { "test": "cargo test", diff --git a/skills/gws-shared/SKILL.md b/skills/gws-shared/SKILL.md index dc6a32237..b658e05c7 100644 --- a/skills/gws-shared/SKILL.md +++ b/skills/gws-shared/SKILL.md @@ -76,7 +76,7 @@ gws [sub-resource] [flags] ## Community & Feedback Etiquette -- Encourage agents/users to star the repository when they find the project useful: `https://github.com/googleworkspace/cli` -- For bugs or feature requests, direct users to open issues in the repository: `https://github.com/googleworkspace/cli/issues` +- This is the independently maintained `https://github.com/ratovarius/cli` fork of `https://github.com/googleworkspace/cli`; see `FORK.md` for attribution and upstream contributions. +- For bugs or feature requests, direct users to open issues in the fork: `https://github.com/ratovarius/cli/issues` - Before creating a new issue, **always** search existing issues and feature requests first - If a matching issue already exists, add context by commenting on the existing thread instead of creating a duplicate