From 695d282c5b47771f95ec924502a2c0670f404ccb Mon Sep 17 00:00:00 2001 From: ivanpaghubasan Date: Mon, 28 Sep 2026 15:13:22 +0800 Subject: [PATCH] feat(rules): add side-effect bounds rules (batch 1) Adds side_effect_bounds.yaml to the openai_sdk, pydantic_ai, mcp, and langchain packs: OAI-030/031, PYD-014, MCP-030, LC-025. Flags tools where the model chooses the recipient or amount of a side effect with no visible bound (severity high, confidence 0.5). Mirrored byte-for-byte from the engine fixture. No manifest or schema change; existing predicates only. --- langchain/side_effect_bounds.yaml | 110 ++++++++++++++ mcp/side_effect_bounds.yaml | 110 ++++++++++++++ openai_sdk/side_effect_bounds.yaml | 213 ++++++++++++++++++++++++++++ pydantic_ai/side_effect_bounds.yaml | 115 +++++++++++++++ 4 files changed, 548 insertions(+) create mode 100644 langchain/side_effect_bounds.yaml create mode 100644 mcp/side_effect_bounds.yaml create mode 100644 openai_sdk/side_effect_bounds.yaml create mode 100644 pydantic_ai/side_effect_bounds.yaml diff --git a/langchain/side_effect_bounds.yaml b/langchain/side_effect_bounds.yaml new file mode 100644 index 0000000..7fe3d5c --- /dev/null +++ b/langchain/side_effect_bounds.yaml @@ -0,0 +1,110 @@ +policy: + id: langchain_side_effect_bounds + name: LangChain unbounded side-effect parameters + category: langchain + description: > + Rules that flag a side-effecting LangChain tool (send/notify/refund/ + charge/...) whose recipient or amount parameter is free-form model input + with no visible bound — no allow-list constraining who a message can go + to, no cap or enum constraining how much money moves. This is a different + failure mode from LC-017 (idempotency): a duplicate call repeats an + action within a bound the caller already accepted, while an unbounded + recipient or amount lets a single call move further than the caller + intended. + +rules: + - id: LC-025 + title: Side-effecting tool lets the model choose the recipient or amount with no visible bound + severity: high + confidence: 0.5 + language: python + applies_to: + - langchain_tool + scope: tool + match: + all: + - any: + - all: + - name_has_prefix: + - send_ + - notify_ + - param_name_matches: + contains: + - recipient + exact: + - to + - cc + - bcc + - email + - email_address + - phone + - phone_number + - to_email + - to_address + - to_number + - all: + - name_has_prefix: + - refund_ + - charge_ + - pay_ + - payout_ + - transfer_ + - issue_ + - param_name_matches: + contains: + - amount + exact: + - total + - price + suffixes: + - cents + - not: + has_body_text: + - "(le=" + - " le=" + - "(lt=" + - " lt=" + - "conint(" + - "condecimal(" + - "confloat(" + - "Literal[" + - "MAX_" + - "_MAX" + - "_LIMIT" + - "_limit" + - "ALLOWED" + - "allowed_" + - "ALLOWLIST" + - "allow_list" + - "WHITELIST" + - "whitelist" + - "endswith(\"@" + - "endswith('@" + - "interrupt(" + explanation: > + This LangChain tool's name signals a send/notify or a refund/charge/pay + side effect, and it takes a free-form recipient or amount parameter + with no visible bound: no allow-list constraining who the message + reaches, and no cap, range, or enum constraining how much money moves. + A ReAct agent's tool arguments come from the model's own reasoning, + which prompt injection, a misread task, or an upstream tool's poisoned + observation can steer — and that model can pass any address or any + amount into this call, with no checkpoint before it executes. This is a + distinct failure from LC-017 (idempotency): a duplicate call repeats an + action inside a bound the caller already accepted; this rule is about + a single call moving further — a wider blast radius, not a repeated + one. A tool body that calls LangGraph's interrupt(...) to pause and + wait for a human decision before acting is a gate and silences this + rule. + fix: > + Do not let the model supply the recipient or the amount unconstrained. + Derive the recipient from trusted context (a value bound to the + conversation or the authenticated user) rather than a model parameter, + or constrain it to an allow-list / domain allow-list validated before + the call executes. Cap the amount with a Pydantic field constraint on + the tool's args_schema (Annotated[int, Field(le=...)], conint(le=...)) + AND enforce the same cap server-side, since a static bound can be + bypassed by a differently-shaped call. For anything above a low, + pre-approved threshold, call interrupt(...) from inside the tool to + require a human to approve the specific recipient or amount before the + side effect runs. diff --git a/mcp/side_effect_bounds.yaml b/mcp/side_effect_bounds.yaml new file mode 100644 index 0000000..dd5d3e5 --- /dev/null +++ b/mcp/side_effect_bounds.yaml @@ -0,0 +1,110 @@ +policy: + id: mcp_side_effect_bounds + name: MCP unbounded side-effect parameters + category: mcp + description: > + Rules that flag a side-effecting MCP tool (send/notify/refund/charge/...) + whose recipient or amount parameter is free-form model input with no + visible bound — no allow-list constraining who a message can go to, no + cap or enum constraining how much money moves. This is a different + failure mode from MCP-007 (idempotency): a duplicate call repeats an + action within a bound the caller already accepted, while an unbounded + recipient or amount lets a single call move further than the caller + intended. + +rules: + - id: MCP-030 + title: Side-effecting tool lets the model choose the recipient or amount with no visible bound + severity: high + confidence: 0.5 + language: python + applies_to: + - mcp_tool + scope: tool + match: + all: + - any: + - all: + - name_has_prefix: + - send_ + - notify_ + - param_name_matches: + contains: + - recipient + exact: + - to + - cc + - bcc + - email + - email_address + - phone + - phone_number + - to_email + - to_address + - to_number + - all: + - name_has_prefix: + - refund_ + - charge_ + - pay_ + - payout_ + - transfer_ + - issue_ + - param_name_matches: + contains: + - amount + exact: + - total + - price + suffixes: + - cents + - not: + has_body_text: + - "(le=" + - " le=" + - "(lt=" + - " lt=" + - "conint(" + - "condecimal(" + - "confloat(" + - "Literal[" + - "MAX_" + - "_MAX" + - "_LIMIT" + - "_limit" + - "ALLOWED" + - "allowed_" + - "ALLOWLIST" + - "allow_list" + - "WHITELIST" + - "whitelist" + - "endswith(\"@" + - "endswith('@" + - "elicit(" + explanation: > + This MCP tool's name signals a send/notify or a refund/charge/pay side + effect, and it takes a free-form recipient or amount parameter with no + visible bound: no allow-list constraining who the message reaches, and + no cap, range, or enum constraining how much money moves. A client + talks to this server on behalf of a model whose reasoning can be + steered by prompt injection, by its own misreading of the task, or by + an upstream tool's poisoned output — and that model can pass any + address or any amount into this call, with the handler acting on it + directly. This is a distinct failure from MCP-007 (idempotency): a + duplicate call repeats an action inside a bound the caller already + accepted; this rule is about a single call moving further — a wider + blast radius, not a repeated one. A handler that calls ctx.elicit(...) + to ask the human operator for confirmation before acting is a gate and + silences this rule. + fix: > + Do not let the model supply the recipient or the amount unconstrained. + Derive the recipient from trusted context (a value bound to the + authenticated session, not a tool argument) rather than a model + parameter, or constrain it to an allow-list / domain allow-list + validated before the call executes. Cap the amount with a Pydantic + field constraint on the tool's input model (Annotated[int, + Field(le=...)], conint(le=...)) AND enforce the same cap server-side, + since a static bound can be bypassed by a differently-shaped call. For + anything above a low, pre-approved threshold, use the MCP elicitation + capability (ctx.elicit(...)) to require the human on the other end of + the client to confirm before the handler proceeds. diff --git a/openai_sdk/side_effect_bounds.yaml b/openai_sdk/side_effect_bounds.yaml new file mode 100644 index 0000000..4624452 --- /dev/null +++ b/openai_sdk/side_effect_bounds.yaml @@ -0,0 +1,213 @@ +policy: + id: openai_sdk_side_effect_bounds + name: OpenAI Agents SDK unbounded side-effect parameters + category: openai_sdk + description: > + Rules that flag a side-effecting OpenAI Agents SDK tool (send/notify/ + refund/charge/...) whose recipient or amount parameter is free-form model + input with no visible bound — no allow-list constraining who a message can + go to, no cap or enum constraining how much money moves. This is a + different failure mode from OAI-009/019 (idempotency): a duplicate call + repeats an action within a bound the caller already accepted, while an + unbounded recipient or amount lets a single call move further than the + caller intended. Retry/duplication is OAI-009/019's job; volume is + OAI-101/104's job (max_turns). + +rules: + - id: OAI-030 + title: Side-effecting tool lets the model choose the recipient or amount with no visible bound + severity: high + confidence: 0.5 + language: python + applies_to: + - openai_tool + scope: tool + match: + all: + - any: + - all: + - name_has_prefix: + - send_ + - notify_ + - param_name_matches: + contains: + - recipient + exact: + - to + - cc + - bcc + - email + - email_address + - phone + - phone_number + - to_email + - to_address + - to_number + - all: + - name_has_prefix: + - refund_ + - charge_ + - pay_ + - payout_ + - transfer_ + - issue_ + - param_name_matches: + contains: + - amount + exact: + - total + - price + suffixes: + - cents + - not: + has_body_text: + - "(le=" + - " le=" + - "(lt=" + - " lt=" + - "conint(" + - "condecimal(" + - "confloat(" + - "Literal[" + - "MAX_" + - "_MAX" + - "_LIMIT" + - "_limit" + - "ALLOWED" + - "allowed_" + - "ALLOWLIST" + - "allow_list" + - "WHITELIST" + - "whitelist" + - "endswith(\"@" + - "endswith('@" + - any: + - not: + tool_decorator_kwarg_present: + - needs_approval + - tool_decorator_kwarg_value: + kwarg: needs_approval + value: "False" + explanation: > + This @function_tool's name signals a send/notify or a refund/charge/pay + side effect, and it takes a free-form recipient or amount parameter with + no visible bound: no allow-list constraining who the message reaches, no + cap, range, or enum constraining how much money moves, and no + needs_approval gate. A model steered by prompt injection, by its own + misreading of the task, or by an upstream tool's poisoned output can + pass any address or any amount to this call, and the tool will act on + it. This is a distinct failure from OAI-009/019 (idempotency): a + duplicate call repeats an action inside a bound the caller already + accepted; this rule is about a single call moving further — a wider + blast radius, not a repeated one. Passing needs_approval=True or a + per-call approval callable is a gate and silences this rule, exactly as + it does for OAI-014. + fix: > + Do not let the model supply the recipient or the amount unconstrained. + Derive the recipient from trusted context (the authenticated user's + on-file address, a ticket's assignee) rather than a model parameter, or + constrain it to an allow-list / domain allow-list validated before the + call executes. Cap the amount with a schema bound (Annotated[int, + Field(le=...)], conint(le=...), or a plain range check) AND enforce the + same cap server-side, since a static bound can be bypassed by a + differently-shaped call. For anything above a low, pre-approved + threshold, pass needs_approval=True and route the approval through a + human. + + - id: OAI-031 + title: TypeScript side-effecting tool lets the model choose the recipient or amount with no visible bound + severity: high + confidence: 0.5 + language: typescript + applies_to: + - openai_tool + scope: tool + match: + all: + - any: + - all: + - name_has_prefix: + - send + - notify + - param_name_matches: + contains: + - recipient + exact: + - to + - cc + - bcc + - email + - emailAddress + - toEmail + - toAddress + - phone + - phoneNumber + - toNumber + - all: + - name_has_prefix: + - refund + - charge + - payout + - transfer + - issue + - param_name_matches: + contains: + - amount + exact: + - total + - price + suffixes: + - cents + - not: + has_body_text: + - ".max(" + - ".lte(" + - ".lt(" + - "maximum" + - "z.enum(" + - "nativeEnum(" + - "\"enum\"" + - "enum:" + - "MAX_" + - "_MAX" + - "maxAmount" + - "_LIMIT" + - "Limit" + - "ALLOWED" + - "allowedDomains" + - "AllowList" + - "allowList" + - "WHITELIST" + - "whitelist" + - "endsWith(\"@" + - "endsWith('@" + - any: + - not: + tool_decorator_kwarg_present: + - needsApproval + - tool_decorator_kwarg_value: + kwarg: needsApproval + value: "false" + explanation: > + This OpenAI Agents SDK tool authored in TypeScript has a name that + signals a send/notify or a refund/charge/pay side effect, and it takes a + free-form recipient or amount parameter with no visible bound: no + allow-list constraining who the message reaches, no cap, max, or enum + constraining how much money moves, and no needsApproval gate. A model + steered by prompt injection, by its own misreading of the task, or by an + upstream tool's poisoned output can pass any address or any amount to + this call. This is a distinct failure from OAI-019 (idempotency): a + duplicate call repeats an action inside a bound the caller already + accepted; this rule is about a single call moving further. Passing + needsApproval: true or a per-call approval function is a gate and + silences this rule, the same needsApproval option OAI-014 checks on the + Python side of this SDK. + fix: > + Do not let the model supply the recipient or the amount unconstrained. + Derive the recipient from trusted context rather than a model + parameter, or constrain it with a zod allow-list / domain check + validated before the call executes. Cap the amount with a zod bound + (z.number().max(...)) or an enum AND enforce the same cap server-side, + since a static bound in the schema can be bypassed by a differently- + shaped call. For anything above a low, pre-approved threshold, set + needsApproval: true and route the approval through a human. diff --git a/pydantic_ai/side_effect_bounds.yaml b/pydantic_ai/side_effect_bounds.yaml new file mode 100644 index 0000000..3965868 --- /dev/null +++ b/pydantic_ai/side_effect_bounds.yaml @@ -0,0 +1,115 @@ +policy: + id: pydantic_ai_side_effect_bounds + name: Pydantic AI unbounded side-effect parameters + category: pydantic_ai + description: > + Rules that flag a side-effecting Pydantic AI tool (send/notify/refund/ + charge/...) whose recipient or amount parameter is free-form model input + with no visible bound — no allow-list constraining who a message can go + to, no cap or enum constraining how much money moves. This is a different + failure mode from PYD-007 (idempotency): a duplicate call repeats an + action within a bound the caller already accepted, while an unbounded + recipient or amount lets a single call move further than the caller + intended. + +rules: + - id: PYD-014 + title: Side-effecting tool lets the model choose the recipient or amount with no visible bound + severity: high + confidence: 0.5 + language: python + applies_to: + - pydantic_ai_tool + scope: tool + match: + all: + - any: + - all: + - name_has_prefix: + - send_ + - notify_ + - param_name_matches: + contains: + - recipient + exact: + - to + - cc + - bcc + - email + - email_address + - phone + - phone_number + - to_email + - to_address + - to_number + - all: + - name_has_prefix: + - refund_ + - charge_ + - pay_ + - payout_ + - transfer_ + - issue_ + - param_name_matches: + contains: + - amount + exact: + - total + - price + suffixes: + - cents + - not: + has_body_text: + - "(le=" + - " le=" + - "(lt=" + - " lt=" + - "conint(" + - "condecimal(" + - "confloat(" + - "Literal[" + - "MAX_" + - "_MAX" + - "_LIMIT" + - "_limit" + - "ALLOWED" + - "allowed_" + - "ALLOWLIST" + - "allow_list" + - "WHITELIST" + - "whitelist" + - "endswith(\"@" + - "endswith('@" + - any: + - not: + tool_decorator_kwarg_present: + - requires_approval + - tool_decorator_kwarg_value: + kwarg: requires_approval + value: "False" + explanation: > + This Pydantic AI tool's name signals a send/notify or a refund/charge/ + pay side effect, and it takes a free-form recipient or amount parameter + with no visible bound: no allow-list constraining who the message + reaches, no cap, range, or enum constraining how much money moves, and + no requires_approval gate. A model steered by prompt injection, by its + own misreading of the task, or by an upstream tool's poisoned output can + pass any address or any amount to this call, and the tool will act on + it. This is a distinct failure from PYD-007 (idempotency): a duplicate + call repeats an action inside a bound the caller already accepted; this + rule is about a single call moving further — a wider blast radius, not + a repeated one. Setting requires_approval=True on @agent.tool / + @agent.tool_plain routes the call through Pydantic AI's deferred-tool + approval flow and is a gate that silences this rule. + fix: > + Do not let the model supply the recipient or the amount unconstrained. + Derive the recipient from trusted context (the authenticated user's + on-file address, the run's originating conversation) rather than a + model parameter, or constrain it to an allow-list / domain allow-list + validated before the call executes. Cap the amount with a Pydantic + field constraint (Annotated[int, Field(le=...)], conint(le=...)) AND + enforce the same cap server-side, since a static bound can be bypassed + by a differently-shaped call. For anything above a low, pre-approved + threshold, set requires_approval=True and handle the resulting + DeferredToolRequests in your run loop so a human signs off before the + call executes.