From 699112056aabeda3aa4526f0c8bf48738aff23df Mon Sep 17 00:00:00 2001 From: Karl Liu Date: Fri, 31 Jul 2026 15:14:49 +0200 Subject: [PATCH 1/9] Fix AWS benchmark matrix compatibility --- benchmarks/100.webapps/130.crud-api/README.md | 7 ++- .../411.image-recognition/README.md | 8 +-- .../411.image-recognition/config.json | 1 + .../500.scientific/503.graph-bfs/README.md | 6 ++- .../500.scientific/503.graph-bfs/input.py | 19 ++++++- .../504.dna-visualisation/README.md | 4 ++ .../python/requirements.txt.3.10 | 7 +++ docs/build.md | 20 +++++++- docs/platforms.md | 12 +++++ sebs/benchmark.py | 26 ++++++++++ tests/test_aws_matrix_fixes.py | 49 +++++++++++++++++++ 11 files changed, 152 insertions(+), 7 deletions(-) create mode 100644 benchmarks/500.scientific/504.dna-visualisation/python/requirements.txt.3.10 create mode 100644 tests/test_aws_matrix_fixes.py diff --git a/benchmarks/100.webapps/130.crud-api/README.md b/benchmarks/100.webapps/130.crud-api/README.md index 11e07a2ed..628114890 100644 --- a/benchmarks/100.webapps/130.crud-api/README.md +++ b/benchmarks/100.webapps/130.crud-api/README.md @@ -1,9 +1,14 @@ # 130.crud-api - CRUD API **Type:** Webapps -**Languages:** Python +**Languages:** Python, Node.js **Architecture:** x64, arm64 ## Description The benchmark implements a simple CRUD application simulating a webstore cart. It offers three basic methods: add new item (`PUT`), get an item (`GET`), and query all items in a cart. It uses the NoSQL storage, with each item stored using cart id as primary key and item id as secondary key. The Python implementation uses cloud-native libraries to access the database. + +On AWS, the identity running SeBS must be allowed to create, describe, seed, +and clean up the benchmark's DynamoDB table. The Lambda execution role must +allow item reads, writes, and queries. See the AWS section of the +[platform documentation](../../../docs/platforms.md) for the exact actions. diff --git a/benchmarks/400.inference/411.image-recognition/README.md b/benchmarks/400.inference/411.image-recognition/README.md index 3b3e5cb1d..4c2288459 100644 --- a/benchmarks/400.inference/411.image-recognition/README.md +++ b/benchmarks/400.inference/411.image-recognition/README.md @@ -1,7 +1,7 @@ # 411.image-recognition - Image Recognition **Type:** Inference -**Languages:** Python +**Languages:** Python, C++ **Architecture:** x64 ## Description @@ -16,8 +16,10 @@ The minimal memory amount is set to 768 MiB due to GCP requirements. It works wi > This benchmark contains PyTorch which is often too large to fit into a code package. Up to Python 3.7, we can directly ship the dependencies. For Python 3.8, we use an additional zipping step that requires additional setup during the first run, making cold invocations slower. Warm invocations are not affected. > [!WARNING] -> This benchmark does not work on AWS with Python 3.9 and code package due to excessive code size. While it is possible to ship the benchmark by zipping `torchvision` and `numpy` (see `benchmarks/400.inference/411.image-recognition/python/package.sh`), this significantly affects cold startup. On the lowest supported memory configuration of 512 MB, the cold startup can reach 30 seconds, making HTTP trigger unusable due to 30 second timeout of API gateway. Use Docker deployments for these configurations. +> This benchmark does not fit the AWS Lambda uncompressed code-package limit +> with any Python version or C++ runtime currently supported by SeBS. SeBS marks +> AWS package deployment as unsupported for this benchmark. Use +> `--system-variant container`; retrying the ZIP package cannot succeed. > [!WARNING] > This benchmark does not work on GCP functions gen1 with Python 3.8+ due to excessive code size. Use container deployments on Google Cloud Run for these configurations. - diff --git a/benchmarks/400.inference/411.image-recognition/config.json b/benchmarks/400.inference/411.image-recognition/config.json index 8c5010fc3..f15c28494 100644 --- a/benchmarks/400.inference/411.image-recognition/config.json +++ b/benchmarks/400.inference/411.image-recognition/config.json @@ -3,5 +3,6 @@ "memory": 768, "languages": ["python", "cpp"], "modules": ["storage"], + "system_variants": {"aws": ["container"]}, "cpp_dependencies": ["sdk", "torch", "opencv"] } diff --git a/benchmarks/500.scientific/503.graph-bfs/README.md b/benchmarks/500.scientific/503.graph-bfs/README.md index ee27ea620..0222894bc 100644 --- a/benchmarks/500.scientific/503.graph-bfs/README.md +++ b/benchmarks/500.scientific/503.graph-bfs/README.md @@ -1,9 +1,13 @@ # 503.graph-bfs - Graph BFS **Type:** Scientific -**Languages:** Python +**Languages:** Python, C++ **Architecture:** x64, arm64 ## Description The benchmark represents scientific computations offloaded to serverless functions. It uses the `python-igraph` library to generate an input graph and process it with the Breadth-First Search (BFS) algorithm. + +Python 3.9 uses `python-igraph` 0.9, which reports the BFS root as its own +parent. Newer igraph versions report `-1`. Output validation canonicalizes these +equivalent root sentinels before checking the deterministic result checksum. diff --git a/benchmarks/500.scientific/503.graph-bfs/input.py b/benchmarks/500.scientific/503.graph-bfs/input.py index 5ece44eca..aa190059a 100644 --- a/benchmarks/500.scientific/503.graph-bfs/input.py +++ b/benchmarks/500.scientific/503.graph-bfs/input.py @@ -45,7 +45,24 @@ def validate_output(data_dir: str | None, input_config: dict, output: dict, lang seed = input_config.get('seed') if seed == 42 and size in expected_checksums[language]: - serialized = json.dumps(result, separators=(',', ':')) + # igraph 0.9 uses the root itself as its parent, while igraph 0.11 + # uses -1. Both represent the same BFS tree, so canonicalize the root + # sentinel before comparing deterministic output checksums. + normalized_result = [ + list(value) if isinstance(value, (list, tuple)) else value + for value in result + ] + order, parents = normalized_result[0], normalized_result[2] + if isinstance(order, list) and order and isinstance(parents, list): + root = order[0] + if ( + isinstance(root, int) + and 0 <= root < len(parents) + and parents[root] in (-1, root) + ): + parents[root] = -1 + + serialized = json.dumps(normalized_result, separators=(',', ':')) actual_checksum = hashlib.md5(serialized.encode()).hexdigest() expected_checksum = expected_checksums[language][size] if actual_checksum != expected_checksum: diff --git a/benchmarks/500.scientific/504.dna-visualisation/README.md b/benchmarks/500.scientific/504.dna-visualisation/README.md index 2440a6bf9..38cd3206b 100644 --- a/benchmarks/500.scientific/504.dna-visualisation/README.md +++ b/benchmarks/500.scientific/504.dna-visualisation/README.md @@ -7,3 +7,7 @@ ## Description This benchmark is inspired by the [DNAVisualization](https://github.com/Benjamin-Lee/DNAvisualization.org) project and it implements processing the `.fasta` file with the `squiggle` Python library. + +AWS Python 3.10 and 3.11 use version-specific dependency files. NumPy, +ContourPy, and Pillow are pinned to versions with wheels compatible with the +Lambda build images so that package construction does not require a compiler. diff --git a/benchmarks/500.scientific/504.dna-visualisation/python/requirements.txt.3.10 b/benchmarks/500.scientific/504.dna-visualisation/python/requirements.txt.3.10 new file mode 100644 index 000000000..3602556e0 --- /dev/null +++ b/benchmarks/500.scientific/504.dna-visualisation/python/requirements.txt.3.10 @@ -0,0 +1,7 @@ +# Copyright 2020-2025 ETH Zurich and the SeBS authors. All rights reserved. +squiggle==0.3.1 +# Lambda images use glibc 2.26, for which there are no wheels on new packages +# we have to fix version to prevent compilation from source +numpy==2.2.6 +contourpy==1.3.2 +pillow==10.3.0 diff --git a/docs/build.md b/docs/build.md index c6c7f4bab..3d63698a7 100644 --- a/docs/build.md +++ b/docs/build.md @@ -38,6 +38,25 @@ and `nosql` for NoSQL databases. Each module corresponds to a set of packages th **Package Code** - we move files to create the directory structure expected on each cloud platform and create a final deployment package. An example of a customization is Azure Functions, where additional JSON configuration files are needed. +### Benchmark-specific system variants + +A benchmark can restrict deployment variants for a platform in `config.json`: + +```json +"system_variants": {"aws": ["container"]} +``` + +This restriction reuses the existing provider-neutral `--system-variant` CLI +option. The selected platform interprets its value: AWS supports `package` and +`container`, while GCP supports `function-gen1`, `function-gen2`, and +`container`. It is therefore not named `--aws-system-variant`. + +When this optional mapping is absent, the benchmark accepts every variant +supported by the platform. When present, SeBS rejects unsupported combinations +before building or allocating cloud resources and reports the allowed variant. +For example, `411.image-recognition` is container-only on AWS because its Python +and C++ dependencies exceed Lambda's uncompressed ZIP limit. + ## Docker Image Deployment ```mermaid @@ -235,4 +254,3 @@ the variant code invalidates the cached package automatically. | **Add Dockerfile** | Create `dockerfiles////Dockerfile.run`. | | **Register in systems.json** | Add the variant name to `variant_images` for the appropriate language and system. | | **Build image** | Run `python tools/build_docker_images.py --deployment --language --language-variant `. | - diff --git a/docs/platforms.md b/docs/platforms.md index 14aae20c2..4b4a3a02c 100644 --- a/docs/platforms.md +++ b/docs/platforms.md @@ -65,6 +65,18 @@ You can provide a [role](https://docs.aws.amazon.com/lambda/latest/dg/lambda-int with permissions to access AWS Lambda and S3; otherwise, one will be created automatically. To use a user-defined lambda role, set the name in config JSON - see an example in `configs/example.json`. +Benchmarks with the `nosql` module, such as `130.crud-api`, also use DynamoDB. +Prepare both permission layers before running these benchmarks; SeBS does not +grant DynamoDB permissions automatically: + +| Principal | Required actions | Resource | +| --- | --- | --- | +| Identity running SeBS (for example, the `sebs` IAM user) | `dynamodb:CreateTable`, `dynamodb:DescribeTable`, `dynamodb:PutItem`; add `dynamodb:DeleteTable` for cleanup | `arn:aws:dynamodb:::table/sebs-benchmarks-*` | +| Lambda execution role (the default is `sebs-lambda-role`) | `dynamodb:PutItem`, `dynamodb:GetItem`, `dynamodb:Query` | `arn:aws:dynamodb:::table/sebs-benchmarks-*` | + +The default role created by SeBS only receives S3 and CloudWatch Logs access, +so attach the second set of actions to that role explicitly. + You can pass the credentials either using the default AWS-specific environment variables: ``` diff --git a/sebs/benchmark.py b/sebs/benchmark.py index 32cf62b9f..df3ba42b1 100644 --- a/sebs/benchmark.py +++ b/sebs/benchmark.py @@ -118,6 +118,7 @@ class BenchmarkConfig: memory: Memory allocation in MB languages: List of supported programming languages modules: List of benchmark modules/features required + system_variants: Optional deployment-variant restrictions by platform """ @@ -128,6 +129,7 @@ def __init__( language_specs: List["LanguageSpec"], modules: List[BenchmarkModule], cpp_dependencies: Optional[List[CppDependencies]] = None, + system_variants: Optional[Dict[str, List[str]]] = None, ): """ Initialize a benchmark configuration. @@ -137,12 +139,14 @@ def __init__( memory: Memory allocation in MB languages: List of supported programming languages modules: List of benchmark modules/features required + system_variants: Optional deployment-variant restrictions by platform """ self._timeout = timeout self._memory = memory self._language_specs = language_specs self._modules = modules self._cpp_dependencies = cpp_dependencies or [] + self._system_variants = system_variants or {} @property def timeout(self) -> int: @@ -225,6 +229,15 @@ def supports(self, language: Language, variant: str) -> bool: """Return True when language + variant combination is declared in config.json.""" return variant in self.supported_variants(language) + def supported_system_variants(self, deployment: str) -> Optional[List[str]]: + """Return a platform restriction, or None when every variant is supported.""" + return self._system_variants.get(deployment) + + def supports_system_variant(self, deployment: str, variant: str) -> bool: + """Return whether a deployment variant is supported on the platform.""" + supported = self.supported_system_variants(deployment) + return supported is None or variant in supported + @staticmethod def deserialize(json_object: dict) -> BenchmarkConfig: """ @@ -244,6 +257,7 @@ def deserialize(json_object: dict) -> BenchmarkConfig: cpp_dependencies=[ CppDependencies.deserialize(x) for x in json_object.get("cpp_dependencies", []) ], + system_variants=json_object.get("system_variants"), ) @@ -621,6 +635,18 @@ def __init__( self.benchmark, self.language, self._language_variant ) ) + if not self.benchmark_config.supports_system_variant( + self._deployment_name, self._system_variant.value + ): + supported = self.benchmark_config.supported_system_variants(self._deployment_name) + raise RuntimeError( + "Benchmark {} does not support system variant {} on {}; use {}".format( + self.benchmark, + self._system_variant.value, + self._deployment_name, + ", ".join(supported or []), + ) + ) self._cache_client = cache_client self._docker_client = docker_client self._system_config = system_config diff --git a/tests/test_aws_matrix_fixes.py b/tests/test_aws_matrix_fixes.py new file mode 100644 index 000000000..49c46c2eb --- /dev/null +++ b/tests/test_aws_matrix_fixes.py @@ -0,0 +1,49 @@ +#!/usr/bin/env python3 +# Copyright 2020-2025 ETH Zurich and the SeBS authors. All rights reserved. + +import json +import runpy +import unittest +from pathlib import Path + +from sebs.benchmark import BenchmarkConfig + + +ROOT = Path(__file__).resolve().parents[1] + + +class AWSMatrixFixesTest(unittest.TestCase): + def test_411_is_container_only_on_aws(self): + with (ROOT / "benchmarks/400.inference/411.image-recognition/config.json").open() as f: + config = BenchmarkConfig.deserialize(json.load(f)) + + self.assertFalse(config.supports_system_variant("aws", "package")) + self.assertTrue(config.supports_system_variant("aws", "container")) + self.assertTrue(config.supports_system_variant("local", "package")) + + def test_igraph_root_parent_conventions_validate_equally(self): + module = runpy.run_path( + str(ROOT / "benchmarks/500.scientific/503.graph-bfs/input.py") + ) + validate_output = module["validate_output"] + result = [list(range(10)), [0, 1, 10], [0] * 10] + + self.assertIsNone( + validate_output(None, {"size": 10, "seed": 42}, {"result": result}, "python") + ) + result[2][0] = 7 + self.assertIn( + "checksum mismatch", + validate_output(None, {"size": 10, "seed": 42}, {"result": result}, "python"), + ) + + def test_python_310_uses_the_wheel_compatible_dna_pins(self): + requirements = ROOT / "benchmarks/500.scientific/504.dna-visualisation/python" + self.assertEqual( + (requirements / "requirements.txt.3.10").read_text(), + (requirements / "requirements.txt.3.11").read_text(), + ) + + +if __name__ == "__main__": + unittest.main() From f6e8c1401d0f71e3b86ab1415710fff8aa297128 Mon Sep 17 00:00:00 2001 From: Karl Liu Date: Sat, 1 Aug 2026 04:50:22 +0200 Subject: [PATCH 2/9] Match existing documentation wrapping style --- benchmarks/100.webapps/130.crud-api/README.md | 5 +---- .../400.inference/411.image-recognition/README.md | 5 +---- benchmarks/500.scientific/503.graph-bfs/README.md | 4 +--- .../500.scientific/504.dna-visualisation/README.md | 4 +--- docs/build.md | 13 +++---------- 5 files changed, 7 insertions(+), 24 deletions(-) diff --git a/benchmarks/100.webapps/130.crud-api/README.md b/benchmarks/100.webapps/130.crud-api/README.md index 628114890..34c25383e 100644 --- a/benchmarks/100.webapps/130.crud-api/README.md +++ b/benchmarks/100.webapps/130.crud-api/README.md @@ -8,7 +8,4 @@ The benchmark implements a simple CRUD application simulating a webstore cart. It offers three basic methods: add new item (`PUT`), get an item (`GET`), and query all items in a cart. It uses the NoSQL storage, with each item stored using cart id as primary key and item id as secondary key. The Python implementation uses cloud-native libraries to access the database. -On AWS, the identity running SeBS must be allowed to create, describe, seed, -and clean up the benchmark's DynamoDB table. The Lambda execution role must -allow item reads, writes, and queries. See the AWS section of the -[platform documentation](../../../docs/platforms.md) for the exact actions. +On AWS, the identity running SeBS must be allowed to create, describe, seed, and clean up the benchmark's DynamoDB table. The Lambda execution role must allow item reads, writes, and queries. See the AWS section of the [platform documentation](../../../docs/platforms.md) for the exact actions. diff --git a/benchmarks/400.inference/411.image-recognition/README.md b/benchmarks/400.inference/411.image-recognition/README.md index 4c2288459..f87f59cee 100644 --- a/benchmarks/400.inference/411.image-recognition/README.md +++ b/benchmarks/400.inference/411.image-recognition/README.md @@ -16,10 +16,7 @@ The minimal memory amount is set to 768 MiB due to GCP requirements. It works wi > This benchmark contains PyTorch which is often too large to fit into a code package. Up to Python 3.7, we can directly ship the dependencies. For Python 3.8, we use an additional zipping step that requires additional setup during the first run, making cold invocations slower. Warm invocations are not affected. > [!WARNING] -> This benchmark does not fit the AWS Lambda uncompressed code-package limit -> with any Python version or C++ runtime currently supported by SeBS. SeBS marks -> AWS package deployment as unsupported for this benchmark. Use -> `--system-variant container`; retrying the ZIP package cannot succeed. +> This benchmark does not fit the AWS Lambda uncompressed code-package limit with any Python version or C++ runtime currently supported by SeBS. SeBS marks AWS package deployment as unsupported for this benchmark. Use `--system-variant container`; retrying the ZIP package cannot succeed. > [!WARNING] > This benchmark does not work on GCP functions gen1 with Python 3.8+ due to excessive code size. Use container deployments on Google Cloud Run for these configurations. diff --git a/benchmarks/500.scientific/503.graph-bfs/README.md b/benchmarks/500.scientific/503.graph-bfs/README.md index 0222894bc..c710908e2 100644 --- a/benchmarks/500.scientific/503.graph-bfs/README.md +++ b/benchmarks/500.scientific/503.graph-bfs/README.md @@ -8,6 +8,4 @@ The benchmark represents scientific computations offloaded to serverless functions. It uses the `python-igraph` library to generate an input graph and process it with the Breadth-First Search (BFS) algorithm. -Python 3.9 uses `python-igraph` 0.9, which reports the BFS root as its own -parent. Newer igraph versions report `-1`. Output validation canonicalizes these -equivalent root sentinels before checking the deterministic result checksum. +Python 3.9 uses `python-igraph` 0.9, which reports the BFS root as its own parent. Newer igraph versions report `-1`. Output validation canonicalizes these equivalent root sentinels before checking the deterministic result checksum. diff --git a/benchmarks/500.scientific/504.dna-visualisation/README.md b/benchmarks/500.scientific/504.dna-visualisation/README.md index 38cd3206b..d3216846c 100644 --- a/benchmarks/500.scientific/504.dna-visualisation/README.md +++ b/benchmarks/500.scientific/504.dna-visualisation/README.md @@ -8,6 +8,4 @@ This benchmark is inspired by the [DNAVisualization](https://github.com/Benjamin-Lee/DNAvisualization.org) project and it implements processing the `.fasta` file with the `squiggle` Python library. -AWS Python 3.10 and 3.11 use version-specific dependency files. NumPy, -ContourPy, and Pillow are pinned to versions with wheels compatible with the -Lambda build images so that package construction does not require a compiler. +AWS Python 3.10 and 3.11 use version-specific dependency files. NumPy, ContourPy, and Pillow are pinned to versions with wheels compatible with the Lambda build images so that package construction does not require a compiler. diff --git a/docs/build.md b/docs/build.md index 3d63698a7..b45d28159 100644 --- a/docs/build.md +++ b/docs/build.md @@ -46,16 +46,9 @@ A benchmark can restrict deployment variants for a platform in `config.json`: "system_variants": {"aws": ["container"]} ``` -This restriction reuses the existing provider-neutral `--system-variant` CLI -option. The selected platform interprets its value: AWS supports `package` and -`container`, while GCP supports `function-gen1`, `function-gen2`, and -`container`. It is therefore not named `--aws-system-variant`. - -When this optional mapping is absent, the benchmark accepts every variant -supported by the platform. When present, SeBS rejects unsupported combinations -before building or allocating cloud resources and reports the allowed variant. -For example, `411.image-recognition` is container-only on AWS because its Python -and C++ dependencies exceed Lambda's uncompressed ZIP limit. +This restriction reuses the existing provider-neutral `--system-variant` CLI option. The selected platform interprets its value: AWS supports `package` and `container`, while GCP supports `function-gen1`, `function-gen2`, and `container`. It is therefore not named `--aws-system-variant`. + +When this optional mapping is absent, the benchmark accepts every variant supported by the platform. When present, SeBS rejects unsupported combinations before building or allocating cloud resources and reports the allowed variant. For example, `411.image-recognition` is container-only on AWS because its Python and C++ dependencies exceed Lambda's uncompressed ZIP limit. ## Docker Image Deployment From cab807811b7c0915989212b6201a72ac22354307 Mon Sep 17 00:00:00 2001 From: Karl Liu Date: Sun, 2 Aug 2026 04:31:53 +0200 Subject: [PATCH 3/9] Configure DynamoDB access for AWS Lambda --- benchmarks/100.webapps/130.crud-api/README.md | 2 +- docs/platforms.md | 41 ++++++++--- sebs/aws/config.py | 73 ++++++++++++++++--- tests/test_aws_matrix_fixes.py | 49 +++++++++++++ 4 files changed, 140 insertions(+), 25 deletions(-) diff --git a/benchmarks/100.webapps/130.crud-api/README.md b/benchmarks/100.webapps/130.crud-api/README.md index 34c25383e..940ab44ab 100644 --- a/benchmarks/100.webapps/130.crud-api/README.md +++ b/benchmarks/100.webapps/130.crud-api/README.md @@ -8,4 +8,4 @@ The benchmark implements a simple CRUD application simulating a webstore cart. It offers three basic methods: add new item (`PUT`), get an item (`GET`), and query all items in a cart. It uses the NoSQL storage, with each item stored using cart id as primary key and item id as secondary key. The Python implementation uses cloud-native libraries to access the database. -On AWS, the identity running SeBS must be allowed to create, describe, seed, and clean up the benchmark's DynamoDB table. The Lambda execution role must allow item reads, writes, and queries. See the AWS section of the [platform documentation](../../../docs/platforms.md) for the exact actions. +On AWS, the identity running SeBS must be allowed to create, describe, seed, and clean up the benchmark's DynamoDB table. SeBS grants the required item reads, writes, and queries to its default Lambda execution role; custom roles must provide them explicitly. See the AWS section of the [platform documentation](../../../docs/platforms.md) for the exact actions. diff --git a/docs/platforms.md b/docs/platforms.md index 4b4a3a02c..3fcc3f915 100644 --- a/docs/platforms.md +++ b/docs/platforms.md @@ -57,25 +57,42 @@ jq 'del(.config.deployment.credentials)' | sponge ## AWS Lambda AWS provides one year of free services, including a significant amount of computing time in AWS Lambda. -To work with AWS, you need to provide access and secret keys to a role with permissions -sufficient to manage functions and S3 resources. -Additionally, the account must have `AmazonAPIGatewayAdministrator` permission to set up -automatically AWS HTTP trigger. -You can provide a [role](https://docs.aws.amazon.com/lambda/latest/dg/lambda-intro-execution-role.html) -with permissions to access AWS Lambda and S3; otherwise, one will be created automatically. -To use a user-defined lambda role, set the name in config JSON - see an example in `configs/example.json`. +To work with AWS, provide access and secret keys for an IAM identity with permissions +sufficient to manage Lambda functions and S3 resources. Container deployments also +require ECR permissions. `AmazonAPIGatewayAdministrator` is needed only when using API +Gateway instead of the default Lambda Function URLs. +Collecting server-side metrics and invocation errors requires `logs:StartQuery` and +`logs:GetQueryResults`. Resource cleanup additionally uses `logs:DescribeLogGroups` +and `logs:DeleteLogGroup`. + +You can provide a [Lambda execution role](https://docs.aws.amazon.com/lambda/latest/dg/lambda-intro-execution-role.html) +or let SeBS create and manage the default `sebs-lambda-role`. To use a custom role, set +its ARN as `lambda-role` in the config JSON. SeBS does not modify a custom role, so it +must already provide S3, CloudWatch Logs, and benchmark-specific permissions. +The IAM identity running SeBS always needs `iam:PassRole` on the selected execution +role. + +When SeBS manages the default role, the IAM identity running SeBS must allow +`iam:GetRole`, `iam:CreateRole`, `iam:AttachRolePolicy`, and `iam:PutRolePolicy` on +`arn:aws:iam:::role/sebs-lambda-role`. SeBS attaches `AmazonS3FullAccess` +and `AWSLambdaBasicExecutionRole`, and adds an inline policy for the DynamoDB item +operations used by the bundled NoSQL benchmark. Benchmarks with the `nosql` module, such as `130.crud-api`, also use DynamoDB. -Prepare both permission layers before running these benchmarks; SeBS does not -grant DynamoDB permissions automatically: +The IAM identity running SeBS performs table management and input seeding itself, so +prepare these permissions manually: | Principal | Required actions | Resource | | --- | --- | --- | | Identity running SeBS (for example, the `sebs` IAM user) | `dynamodb:CreateTable`, `dynamodb:DescribeTable`, `dynamodb:PutItem`; add `dynamodb:DeleteTable` for cleanup | `arn:aws:dynamodb:::table/sebs-benchmarks-*` | -| Lambda execution role (the default is `sebs-lambda-role`) | `dynamodb:PutItem`, `dynamodb:GetItem`, `dynamodb:Query` | `arn:aws:dynamodb:::table/sebs-benchmarks-*` | -The default role created by SeBS only receives S3 and CloudWatch Logs access, -so attach the second set of actions to that role explicitly. +The AWS-managed [`AmazonDynamoDBFullAccess_v2`](https://docs.aws.amazon.com/aws-managed-policy/latest/reference/AmazonDynamoDBFullAccess_v2.html) +policy covers these caller-side DynamoDB actions but grants broader access than SeBS +requires. + +SeBS grants `dynamodb:PutItem`, `dynamodb:GetItem`, and `dynamodb:Query` on the same +table prefix to the default execution role. Add those actions yourself when using a +custom execution role. You can pass the credentials either using the default AWS-specific environment variables: diff --git a/sebs/aws/config.py b/sebs/aws/config.py index 83230641a..943882166 100644 --- a/sebs/aws/config.py +++ b/sebs/aws/config.py @@ -225,6 +225,8 @@ class AWSResources(Resources): _docker_password: Docker registry password _container_repository: ECR repository name _lambda_role: IAM role ARN for Lambda execution + _lambda_role_user_provided: Whether the role came from explicit configuration + _lambda_role_permissions_configured: Whether default role permissions were configured _http_apis: Dictionary of HTTP API configurations """ @@ -385,6 +387,8 @@ def __init__( self._docker_password: Optional[str] = password if password != "" else None self._container_repository: Optional[str] = None self._lambda_role = "" + self._lambda_role_user_provided: bool = False + self._lambda_role_permissions_configured: bool = False self._http_apis: Dict[str, AWSResources.HTTPApi] = {} self._function_urls: Dict[str, AWSResources.FunctionURL] = {} self._use_function_url: bool = True @@ -472,9 +476,10 @@ def function_url_auth_type(self, value: FunctionURLAuthType): def lambda_role(self, boto3_session: boto3.session.Session) -> str: """Get or create IAM role for Lambda execution. - Creates a Lambda execution role with S3 and basic execution permissions - if it doesn't already exist. The role allows Lambda functions to access - S3 and write CloudWatch logs. + Creates a Lambda execution role with S3, DynamoDB, and basic execution + permissions if it doesn't already exist. The role allows Lambda functions + to access SeBS resources and write CloudWatch logs. Explicitly configured + roles are returned unchanged. Args: boto3_session: Boto3 session for AWS API calls @@ -485,8 +490,16 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: Raises: ClientError: If IAM operations fail """ + role_name = "sebs-lambda-role" + if ( + self._lambda_role_user_provided + or self._lambda_role_permissions_configured + or (self._lambda_role and not self._lambda_role.endswith(f":role/{role_name}")) + ): + return self._lambda_role + + iam_client = boto3_session.client(service_name="iam") if not self._lambda_role: - iam_client = boto3_session.client(service_name="iam") trust_policy = { "Version": "2012-10-17", "Statement": [ @@ -497,11 +510,6 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: } ], } - role_name = "sebs-lambda-role" - attached_policies = [ - "arn:aws:iam::aws:policy/AmazonS3FullAccess", - "arn:aws:iam::aws:policy/service-role/AWSLambdaBasicExecutionRole", - ] try: out = iam_client.get_role(RoleName=role_name) self._lambda_role = out["Role"]["Arn"] @@ -517,9 +525,43 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: "Sleep 10 seconds to avoid problems when using role immediately." ) time.sleep(10) - # Attach basic AWS Lambda and S3 policies. - for policy in attached_policies: - iam_client.attach_role_policy(RoleName=role_name, PolicyArn=policy) + + arn_parts = self._lambda_role.split(":", 5) + if ( + len(arn_parts) != 6 + or arn_parts[0] != "arn" + or arn_parts[2] != "iam" + or not arn_parts[4] + ): + raise RuntimeError(f"Invalid Lambda execution role ARN: {self._lambda_role}") + partition = arn_parts[1] + account_id = arn_parts[4] + + for policy in ( + "arn:aws:iam::aws:policy/AmazonS3FullAccess", + "arn:aws:iam::aws:policy/service-role/AWSLambdaBasicExecutionRole", + ): + iam_client.attach_role_policy(RoleName=role_name, PolicyArn=policy) + + dynamodb_policy = { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": ["dynamodb:GetItem", "dynamodb:PutItem", "dynamodb:Query"], + "Resource": ( + f"arn:{partition}:dynamodb:{self.region}:{account_id}:" + "table/sebs-benchmarks-*" + ), + } + ], + } + iam_client.put_role_policy( + RoleName=role_name, + PolicyName="sebs-dynamodb-access", + PolicyDocument=json.dumps(dynamodb_policy), + ) + self._lambda_role_permissions_configured = True return self._lambda_role def http_api( @@ -1132,6 +1174,13 @@ def deserialize(config: dict, cache: Cache, handlers: LoggingHandlers) -> Resour ret.logging_handlers = handlers ret.logging.info("No resources for AWS found, initialize!") + configured_lambda_role = config.get("lambda-role") or config.get("resources", {}).get( + "lambda-role" + ) + if configured_lambda_role: + ret._lambda_role = configured_lambda_role + ret._lambda_role_user_provided = True + return ret diff --git a/tests/test_aws_matrix_fixes.py b/tests/test_aws_matrix_fixes.py index 49c46c2eb..29bdd0c05 100644 --- a/tests/test_aws_matrix_fixes.py +++ b/tests/test_aws_matrix_fixes.py @@ -5,14 +5,63 @@ import runpy import unittest from pathlib import Path +from unittest.mock import Mock +from sebs.aws.config import AWSResources from sebs.benchmark import BenchmarkConfig +from sebs.utils import LoggingHandlers ROOT = Path(__file__).resolve().parents[1] class AWSMatrixFixesTest(unittest.TestCase): + def test_cached_default_lambda_role_receives_dynamodb_access(self): + resources = AWSResources() + resources.region = "us-east-1" + AWSResources.initialize( + resources, + {"lambda-role": "arn:aws:iam::123456789012:role/sebs-lambda-role"}, + ) + iam_client = Mock() + session = Mock() + session.client.return_value = iam_client + + resources.lambda_role(session) + + self.assertEqual( + iam_client.put_role_policy.call_args.kwargs["RoleName"], "sebs-lambda-role" + ) + self.assertEqual( + iam_client.put_role_policy.call_args.kwargs["PolicyName"], + "sebs-dynamodb-access", + ) + policy = json.loads(iam_client.put_role_policy.call_args.kwargs["PolicyDocument"]) + self.assertEqual( + policy["Statement"][0]["Action"], + ["dynamodb:GetItem", "dynamodb:PutItem", "dynamodb:Query"], + ) + self.assertEqual( + policy["Statement"][0]["Resource"], + "arn:aws:dynamodb:us-east-1:123456789012:table/sebs-benchmarks-*", + ) + + resources.lambda_role(session) + self.assertEqual(iam_client.put_role_policy.call_count, 1) + + def test_explicit_custom_lambda_role_is_not_modified(self): + role_arn = "arn:aws:iam::123456789012:role/sebs-lambda-role" + cache = Mock() + cache.get_config.return_value = None + resources = AWSResources.deserialize( + {"lambda-role": role_arn, "resources": {}}, cache, LoggingHandlers() + ) + session = Mock() + + self.assertEqual(resources.lambda_role(session), role_arn) + + session.client.assert_not_called() + def test_411_is_container_only_on_aws(self): with (ROOT / "benchmarks/400.inference/411.image-recognition/config.json").open() as f: config = BenchmarkConfig.deserialize(json.load(f)) From 682f95f19d0ec925d3e91011709ae42a258b15ed Mon Sep 17 00:00:00 2001 From: Karl Liu Date: Sun, 2 Aug 2026 05:13:59 +0200 Subject: [PATCH 4/9] Document AWS ECR permissions --- docs/platforms.md | 42 ++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 40 insertions(+), 2 deletions(-) diff --git a/docs/platforms.md b/docs/platforms.md index 3fcc3f915..d38fa79ee 100644 --- a/docs/platforms.md +++ b/docs/platforms.md @@ -59,12 +59,50 @@ jq 'del(.config.deployment.credentials)' | sponge AWS provides one year of free services, including a significant amount of computing time in AWS Lambda. To work with AWS, provide access and secret keys for an IAM identity with permissions sufficient to manage Lambda functions and S3 resources. Container deployments also -require ECR permissions. `AmazonAPIGatewayAdministrator` is needed only when using API -Gateway instead of the default Lambda Function URLs. +require the ECR permissions listed below. `AmazonAPIGatewayAdministrator` is needed only +when using API Gateway instead of the default Lambda Function URLs. Collecting server-side metrics and invocation errors requires `logs:StartQuery` and `logs:GetQueryResults`. Resource cleanup additionally uses `logs:DescribeLogGroups` and `logs:DeleteLogGroup`. +For a `container` deployment, the IAM identity running SeBS creates and inspects the +`sebs-benchmarks-*` ECR repository, pushes images, and creates Lambda functions from +those images. The following policy covers the required repository and image operations: + +```json +{ + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": [ + "ecr:BatchCheckLayerAvailability", + "ecr:BatchGetImage", + "ecr:CompleteLayerUpload", + "ecr:CreateRepository", + "ecr:DescribeImages", + "ecr:DescribeRepositories", + "ecr:GetDownloadUrlForLayer", + "ecr:GetRepositoryPolicy", + "ecr:InitiateLayerUpload", + "ecr:PutImage", + "ecr:SetRepositoryPolicy", + "ecr:UploadLayerPart" + ], + "Resource": "arn:aws:ecr:::repository/sebs-benchmarks-*" + }, + { + "Effect": "Allow", + "Action": "ecr:GetAuthorizationToken", + "Resource": "*" + } + ] +} +``` + +Add `ecr:DeleteRepository` on the same repository ARN when using SeBS resource cleanup. +AWS documents the image-push actions in [IAM permissions for pushing an image](https://docs.aws.amazon.com/AmazonECR/latest/userguide/image-push-iam.html) and the additional image-access and repository-policy actions in [Create a Lambda function using a container image](https://docs.aws.amazon.com/lambda/latest/dg/images-create.html). The AWS-managed [`AmazonEC2ContainerRegistryFullAccess`](https://docs.aws.amazon.com/aws-managed-policy/latest/reference/AmazonEC2ContainerRegistryFullAccess.html) policy also covers these operations but grants broader access than SeBS requires. + You can provide a [Lambda execution role](https://docs.aws.amazon.com/lambda/latest/dg/lambda-intro-execution-role.html) or let SeBS create and manage the default `sebs-lambda-role`. To use a custom role, set its ARN as `lambda-role` in the config JSON. SeBS does not modify a custom role, so it From 6ba391b1178af196de42c386ba432fe124514e32 Mon Sep 17 00:00:00 2001 From: Karl Liu Date: Sun, 2 Aug 2026 05:55:52 +0200 Subject: [PATCH 5/9] Restore Lambda role reuse behavior --- benchmarks/100.webapps/130.crud-api/README.md | 2 +- docs/platforms.md | 19 ++-- sebs/aws/config.py | 90 ++++++++----------- tests/test_aws_matrix_fixes.py | 11 ++- 4 files changed, 55 insertions(+), 67 deletions(-) diff --git a/benchmarks/100.webapps/130.crud-api/README.md b/benchmarks/100.webapps/130.crud-api/README.md index 940ab44ab..4f09fc0fb 100644 --- a/benchmarks/100.webapps/130.crud-api/README.md +++ b/benchmarks/100.webapps/130.crud-api/README.md @@ -8,4 +8,4 @@ The benchmark implements a simple CRUD application simulating a webstore cart. It offers three basic methods: add new item (`PUT`), get an item (`GET`), and query all items in a cart. It uses the NoSQL storage, with each item stored using cart id as primary key and item id as secondary key. The Python implementation uses cloud-native libraries to access the database. -On AWS, the identity running SeBS must be allowed to create, describe, seed, and clean up the benchmark's DynamoDB table. SeBS grants the required item reads, writes, and queries to its default Lambda execution role; custom roles must provide them explicitly. See the AWS section of the [platform documentation](../../../docs/platforms.md) for the exact actions. +On AWS, the identity running SeBS must be allowed to create, describe, seed, and clean up the benchmark's DynamoDB table. When SeBS resolves its default Lambda execution role, it grants the required item reads, writes, and queries; configured or cached roles must provide them explicitly. See the AWS section of the [platform documentation](../../../docs/platforms.md) for the exact actions. diff --git a/docs/platforms.md b/docs/platforms.md index d38fa79ee..7a5572250 100644 --- a/docs/platforms.md +++ b/docs/platforms.md @@ -105,16 +105,19 @@ AWS documents the image-push actions in [IAM permissions for pushing an image](h You can provide a [Lambda execution role](https://docs.aws.amazon.com/lambda/latest/dg/lambda-intro-execution-role.html) or let SeBS create and manage the default `sebs-lambda-role`. To use a custom role, set -its ARN as `lambda-role` in the config JSON. SeBS does not modify a custom role, so it -must already provide S3, CloudWatch Logs, and benchmark-specific permissions. +its ARN as `lambda-role` in the config JSON. SeBS reuses a role ARN loaded from the +configuration or cache without modifying it, so that role must already provide S3, +CloudWatch Logs, and benchmark-specific permissions. The IAM identity running SeBS always needs `iam:PassRole` on the selected execution role. -When SeBS manages the default role, the IAM identity running SeBS must allow -`iam:GetRole`, `iam:CreateRole`, `iam:AttachRolePolicy`, and `iam:PutRolePolicy` on +When no role ARN is loaded, SeBS resolves the default role. The IAM identity running +SeBS must allow `iam:GetRole`, `iam:CreateRole`, `iam:AttachRolePolicy`, and +`iam:PutRolePolicy` on `arn:aws:iam:::role/sebs-lambda-role`. SeBS attaches `AmazonS3FullAccess` and `AWSLambdaBasicExecutionRole`, and adds an inline policy for the DynamoDB item -operations used by the bundled NoSQL benchmark. +operations used by the bundled NoSQL benchmark. A cached default role is reused without +updating its policies. Benchmarks with the `nosql` module, such as `130.crud-api`, also use DynamoDB. The IAM identity running SeBS performs table management and input seeding itself, so @@ -128,9 +131,9 @@ The AWS-managed [`AmazonDynamoDBFullAccess_v2`](https://docs.aws.amazon.com/aws- policy covers these caller-side DynamoDB actions but grants broader access than SeBS requires. -SeBS grants `dynamodb:PutItem`, `dynamodb:GetItem`, and `dynamodb:Query` on the same -table prefix to the default execution role. Add those actions yourself when using a -custom execution role. +When it resolves the default execution role, SeBS grants `dynamodb:PutItem`, +`dynamodb:GetItem`, and `dynamodb:Query` on the same table prefix. Add those actions +yourself when using a configured or cached execution role. You can pass the credentials either using the default AWS-specific environment variables: diff --git a/sebs/aws/config.py b/sebs/aws/config.py index 943882166..f2373cef5 100644 --- a/sebs/aws/config.py +++ b/sebs/aws/config.py @@ -225,8 +225,6 @@ class AWSResources(Resources): _docker_password: Docker registry password _container_repository: ECR repository name _lambda_role: IAM role ARN for Lambda execution - _lambda_role_user_provided: Whether the role came from explicit configuration - _lambda_role_permissions_configured: Whether default role permissions were configured _http_apis: Dictionary of HTTP API configurations """ @@ -387,8 +385,6 @@ def __init__( self._docker_password: Optional[str] = password if password != "" else None self._container_repository: Optional[str] = None self._lambda_role = "" - self._lambda_role_user_provided: bool = False - self._lambda_role_permissions_configured: bool = False self._http_apis: Dict[str, AWSResources.HTTPApi] = {} self._function_urls: Dict[str, AWSResources.FunctionURL] = {} self._use_function_url: bool = True @@ -478,8 +474,7 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: Creates a Lambda execution role with S3, DynamoDB, and basic execution permissions if it doesn't already exist. The role allows Lambda functions - to access SeBS resources and write CloudWatch logs. Explicitly configured - roles are returned unchanged. + to access SeBS resources and write CloudWatch logs. Args: boto3_session: Boto3 session for AWS API calls @@ -490,16 +485,8 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: Raises: ClientError: If IAM operations fail """ - role_name = "sebs-lambda-role" - if ( - self._lambda_role_user_provided - or self._lambda_role_permissions_configured - or (self._lambda_role and not self._lambda_role.endswith(f":role/{role_name}")) - ): - return self._lambda_role - - iam_client = boto3_session.client(service_name="iam") if not self._lambda_role: + iam_client = boto3_session.client(service_name="iam") trust_policy = { "Version": "2012-10-17", "Statement": [ @@ -510,6 +497,7 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: } ], } + role_name = "sebs-lambda-role" try: out = iam_client.get_role(RoleName=role_name) self._lambda_role = out["Role"]["Arn"] @@ -526,42 +514,41 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: ) time.sleep(10) - arn_parts = self._lambda_role.split(":", 5) - if ( - len(arn_parts) != 6 - or arn_parts[0] != "arn" - or arn_parts[2] != "iam" - or not arn_parts[4] - ): - raise RuntimeError(f"Invalid Lambda execution role ARN: {self._lambda_role}") - partition = arn_parts[1] - account_id = arn_parts[4] - - for policy in ( - "arn:aws:iam::aws:policy/AmazonS3FullAccess", - "arn:aws:iam::aws:policy/service-role/AWSLambdaBasicExecutionRole", - ): - iam_client.attach_role_policy(RoleName=role_name, PolicyArn=policy) - - dynamodb_policy = { - "Version": "2012-10-17", - "Statement": [ - { - "Effect": "Allow", - "Action": ["dynamodb:GetItem", "dynamodb:PutItem", "dynamodb:Query"], - "Resource": ( - f"arn:{partition}:dynamodb:{self.region}:{account_id}:" - "table/sebs-benchmarks-*" - ), - } - ], - } - iam_client.put_role_policy( - RoleName=role_name, - PolicyName="sebs-dynamodb-access", - PolicyDocument=json.dumps(dynamodb_policy), - ) - self._lambda_role_permissions_configured = True + arn_parts = self._lambda_role.split(":", 5) + if ( + len(arn_parts) != 6 + or arn_parts[0] != "arn" + or arn_parts[2] != "iam" + or not arn_parts[4] + ): + raise RuntimeError(f"Invalid Lambda execution role ARN: {self._lambda_role}") + partition = arn_parts[1] + account_id = arn_parts[4] + + for policy in ( + "arn:aws:iam::aws:policy/AmazonS3FullAccess", + "arn:aws:iam::aws:policy/service-role/AWSLambdaBasicExecutionRole", + ): + iam_client.attach_role_policy(RoleName=role_name, PolicyArn=policy) + + dynamodb_policy = { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": ["dynamodb:GetItem", "dynamodb:PutItem", "dynamodb:Query"], + "Resource": ( + f"arn:{partition}:dynamodb:{self.region}:{account_id}:" + "table/sebs-benchmarks-*" + ), + } + ], + } + iam_client.put_role_policy( + RoleName=role_name, + PolicyName="sebs-dynamodb-access", + PolicyDocument=json.dumps(dynamodb_policy), + ) return self._lambda_role def http_api( @@ -1179,7 +1166,6 @@ def deserialize(config: dict, cache: Cache, handlers: LoggingHandlers) -> Resour ) if configured_lambda_role: ret._lambda_role = configured_lambda_role - ret._lambda_role_user_provided = True return ret diff --git a/tests/test_aws_matrix_fixes.py b/tests/test_aws_matrix_fixes.py index 29bdd0c05..5e4769eab 100644 --- a/tests/test_aws_matrix_fixes.py +++ b/tests/test_aws_matrix_fixes.py @@ -16,14 +16,13 @@ class AWSMatrixFixesTest(unittest.TestCase): - def test_cached_default_lambda_role_receives_dynamodb_access(self): + def test_new_default_lambda_role_receives_dynamodb_access(self): resources = AWSResources() resources.region = "us-east-1" - AWSResources.initialize( - resources, - {"lambda-role": "arn:aws:iam::123456789012:role/sebs-lambda-role"}, - ) iam_client = Mock() + iam_client.get_role.return_value = { + "Role": {"Arn": "arn:aws:iam::123456789012:role/sebs-lambda-role"} + } session = Mock() session.client.return_value = iam_client @@ -49,7 +48,7 @@ def test_cached_default_lambda_role_receives_dynamodb_access(self): resources.lambda_role(session) self.assertEqual(iam_client.put_role_policy.call_count, 1) - def test_explicit_custom_lambda_role_is_not_modified(self): + def test_configured_lambda_role_is_not_modified(self): role_arn = "arn:aws:iam::123456789012:role/sebs-lambda-role" cache = Mock() cache.get_config.return_value = None From 31a698533dc64e41b25738c27af21bb9ad99dbee Mon Sep 17 00:00:00 2001 From: Karl Liu Date: Mon, 3 Aug 2026 01:09:54 +0200 Subject: [PATCH 6/9] Fix AWS report parsing with application logs --- docs/platforms.md | 2 ++ sebs/aws/aws.py | 10 ++++++---- tests/test_aws_matrix_fixes.py | 23 +++++++++++++++++++++++ 3 files changed, 31 insertions(+), 4 deletions(-) diff --git a/docs/platforms.md b/docs/platforms.md index 7a5572250..ed681c776 100644 --- a/docs/platforms.md +++ b/docs/platforms.md @@ -158,6 +158,8 @@ or in the JSON input configuration: } ``` +Direct library invocations obtain provider duration and billing metrics from AWS `START` and `REPORT` log records. Application log lines may be interleaved with these records; SeBS ignores unstructured fields while parsing the provider metrics. + ### Lambda Function URLs vs API Gateway SeBS supports two methods for HTTP-based function invocation on AWS Lambda: diff --git a/sebs/aws/aws.py b/sebs/aws/aws.py index dd7077d18..5751cf8eb 100644 --- a/sebs/aws/aws.py +++ b/sebs/aws/aws.py @@ -671,10 +671,12 @@ def parse_aws_report( "REPORT RequestId: abc123\tDuration: 100.00 ms\tBilled Duration: 100 ms\t..." """ aws_vals = {} - for line in log.split("\t"): - if not line.isspace(): - split = line.split(":") - aws_vals[split[0]] = split[1].split()[0] + for log_line in log.splitlines(): + for field in log_line.split("\t"): + key, separator, value = field.partition(":") + values = value.split() + if separator and values: + aws_vals[key] = values[0] if "START RequestId" in aws_vals: request_id = aws_vals["START RequestId"] elif "REPORT RequestId" in aws_vals: diff --git a/tests/test_aws_matrix_fixes.py b/tests/test_aws_matrix_fixes.py index 5e4769eab..abe23355f 100644 --- a/tests/test_aws_matrix_fixes.py +++ b/tests/test_aws_matrix_fixes.py @@ -7,8 +7,10 @@ from pathlib import Path from unittest.mock import Mock +from sebs.aws.aws import AWS from sebs.aws.config import AWSResources from sebs.benchmark import BenchmarkConfig +from sebs.faas.function import ExecutionResult from sebs.utils import LoggingHandlers @@ -16,6 +18,27 @@ class AWSMatrixFixesTest(unittest.TestCase): + def test_aws_report_parser_ignores_application_log_fields(self): + request_id = "1ebf703c-b814-4eb8-b26e-788f53d5e328" + log = ( + f"START RequestId: {request_id} Version: $LATEST\n" + f"2026-08-02T09:33:20.853Z\t{request_id}\tERROR\t" + "(node:8) [DEP0040] DeprecationWarning: The `punycode` module is deprecated.\n" + f"END RequestId: {request_id}\n" + f"REPORT RequestId: {request_id}\tDuration: 9467.75 ms\t" + "Billed Duration: 10014 ms\tMemory Size: 128 MB\t" + "Max Memory Used: 112 MB\tInit Duration: 546.19 ms\n" + ) + result = ExecutionResult() + + self.assertEqual(AWS.parse_aws_report(log, result), request_id) + self.assertEqual(result.request_id, request_id) + self.assertEqual(result.provider_times.execution, 9467750) + self.assertEqual(result.provider_times.initialization, 546190) + self.assertEqual(result.stats.memory_used, 112.0) + self.assertEqual(result.billing.billed_time, 10014) + self.assertEqual(result.billing.memory, 128) + def test_new_default_lambda_role_receives_dynamodb_access(self): resources = AWSResources() resources.region = "us-east-1" From 9668ae6093db76c8479fb80fede8e2aa3c94a51f Mon Sep 17 00:00:00 2001 From: Karl Liu Date: Mon, 3 Aug 2026 06:07:41 +0200 Subject: [PATCH 7/9] Tighten AWS fix documentation --- benchmarks/100.webapps/130.crud-api/README.md | 2 - .../411.image-recognition/README.md | 5 +- docs/build.md | 7 +- docs/platforms.md | 98 +++++-------------- tests/test_aws_matrix_fixes.py | 27 ++--- 5 files changed, 44 insertions(+), 95 deletions(-) diff --git a/benchmarks/100.webapps/130.crud-api/README.md b/benchmarks/100.webapps/130.crud-api/README.md index 4f09fc0fb..0cb335051 100644 --- a/benchmarks/100.webapps/130.crud-api/README.md +++ b/benchmarks/100.webapps/130.crud-api/README.md @@ -7,5 +7,3 @@ ## Description The benchmark implements a simple CRUD application simulating a webstore cart. It offers three basic methods: add new item (`PUT`), get an item (`GET`), and query all items in a cart. It uses the NoSQL storage, with each item stored using cart id as primary key and item id as secondary key. The Python implementation uses cloud-native libraries to access the database. - -On AWS, the identity running SeBS must be allowed to create, describe, seed, and clean up the benchmark's DynamoDB table. When SeBS resolves its default Lambda execution role, it grants the required item reads, writes, and queries; configured or cached roles must provide them explicitly. See the AWS section of the [platform documentation](../../../docs/platforms.md) for the exact actions. diff --git a/benchmarks/400.inference/411.image-recognition/README.md b/benchmarks/400.inference/411.image-recognition/README.md index f87f59cee..30bdd4cb7 100644 --- a/benchmarks/400.inference/411.image-recognition/README.md +++ b/benchmarks/400.inference/411.image-recognition/README.md @@ -1,7 +1,7 @@ # 411.image-recognition - Image Recognition **Type:** Inference -**Languages:** Python, C++ +**Languages:** Python **Architecture:** x64 ## Description @@ -16,7 +16,8 @@ The minimal memory amount is set to 768 MiB due to GCP requirements. It works wi > This benchmark contains PyTorch which is often too large to fit into a code package. Up to Python 3.7, we can directly ship the dependencies. For Python 3.8, we use an additional zipping step that requires additional setup during the first run, making cold invocations slower. Warm invocations are not affected. > [!WARNING] -> This benchmark does not fit the AWS Lambda uncompressed code-package limit with any Python version or C++ runtime currently supported by SeBS. SeBS marks AWS package deployment as unsupported for this benchmark. Use `--system-variant container`; retrying the ZIP package cannot succeed. +> This benchmark exceeds the AWS Lambda uncompressed code-package limit. Use `--system-variant container` on AWS. > [!WARNING] > This benchmark does not work on GCP functions gen1 with Python 3.8+ due to excessive code size. Use container deployments on Google Cloud Run for these configurations. + diff --git a/docs/build.md b/docs/build.md index b45d28159..99e627272 100644 --- a/docs/build.md +++ b/docs/build.md @@ -38,7 +38,7 @@ and `nosql` for NoSQL databases. Each module corresponds to a set of packages th **Package Code** - we move files to create the directory structure expected on each cloud platform and create a final deployment package. An example of a customization is Azure Functions, where additional JSON configuration files are needed. -### Benchmark-specific system variants +### Benchmark-Specific System Variants A benchmark can restrict deployment variants for a platform in `config.json`: @@ -46,9 +46,7 @@ A benchmark can restrict deployment variants for a platform in `config.json`: "system_variants": {"aws": ["container"]} ``` -This restriction reuses the existing provider-neutral `--system-variant` CLI option. The selected platform interprets its value: AWS supports `package` and `container`, while GCP supports `function-gen1`, `function-gen2`, and `container`. It is therefore not named `--aws-system-variant`. - -When this optional mapping is absent, the benchmark accepts every variant supported by the platform. When present, SeBS rejects unsupported combinations before building or allocating cloud resources and reports the allowed variant. For example, `411.image-recognition` is container-only on AWS because its Python and C++ dependencies exceed Lambda's uncompressed ZIP limit. +The values match the platform's `--system-variant` choices. Without this mapping, all platform variants are allowed. SeBS checks the restriction before building the benchmark or creating cloud resources. ## Docker Image Deployment @@ -247,3 +245,4 @@ the variant code invalidates the cached package automatically. | **Add Dockerfile** | Create `dockerfiles////Dockerfile.run`. | | **Register in systems.json** | Add the variant name to `variant_images` for the appropriate language and system. | | **Build image** | Run `python tools/build_docker_images.py --deployment --language --language-variant `. | + diff --git a/docs/platforms.md b/docs/platforms.md index ed681c776..8e0690b2a 100644 --- a/docs/platforms.md +++ b/docs/platforms.md @@ -58,82 +58,32 @@ jq 'del(.config.deployment.credentials)' | sponge AWS provides one year of free services, including a significant amount of computing time in AWS Lambda. To work with AWS, provide access and secret keys for an IAM identity with permissions -sufficient to manage Lambda functions and S3 resources. Container deployments also -require the ECR permissions listed below. `AmazonAPIGatewayAdministrator` is needed only -when using API Gateway instead of the default Lambda Function URLs. -Collecting server-side metrics and invocation errors requires `logs:StartQuery` and -`logs:GetQueryResults`. Resource cleanup additionally uses `logs:DescribeLogGroups` -and `logs:DeleteLogGroup`. +sufficient to manage functions and S3 resources. -For a `container` deployment, the IAM identity running SeBS creates and inspects the -`sebs-benchmarks-*` ECR repository, pushes images, and creates Lambda functions from -those images. The following policy covers the required repository and image operations: +### Quick Start Permissions -```json -{ - "Version": "2012-10-17", - "Statement": [ - { - "Effect": "Allow", - "Action": [ - "ecr:BatchCheckLayerAvailability", - "ecr:BatchGetImage", - "ecr:CompleteLayerUpload", - "ecr:CreateRepository", - "ecr:DescribeImages", - "ecr:DescribeRepositories", - "ecr:GetDownloadUrlForLayer", - "ecr:GetRepositoryPolicy", - "ecr:InitiateLayerUpload", - "ecr:PutImage", - "ecr:SetRepositoryPolicy", - "ecr:UploadLayerPart" - ], - "Resource": "arn:aws:ecr:::repository/sebs-benchmarks-*" - }, - { - "Effect": "Allow", - "Action": "ecr:GetAuthorizationToken", - "Resource": "*" - } - ] -} -``` +For package and container deployments, NoSQL benchmarks, log queries, and cleanup, +attach these AWS managed policies to the IAM identity running SeBS: + +* `AWSLambda_FullAccess` +* `AmazonS3FullAccess` +* `IAMFullAccess` +* `CloudWatchLogsFullAccess` +* `AmazonDynamoDBFullAccess_v2` +* `AmazonEC2ContainerRegistryFullAccess` + +This bundle is intentionally broad and should be used only in a dedicated experiment +account. Add `AmazonAPIGatewayAdministrator` only when using API Gateway instead of the +default Lambda Function URLs. + +The [Lambda execution role](https://docs.aws.amazon.com/lambda/latest/dg/lambda-intro-execution-role.html) +is separate from the IAM identity running SeBS. Set its ARN as `lambda-role` in the config +JSON, or let SeBS resolve `sebs-lambda-role`. -Add `ecr:DeleteRepository` on the same repository ARN when using SeBS resource cleanup. -AWS documents the image-push actions in [IAM permissions for pushing an image](https://docs.aws.amazon.com/AmazonECR/latest/userguide/image-push-iam.html) and the additional image-access and repository-policy actions in [Create a Lambda function using a container image](https://docs.aws.amazon.com/lambda/latest/dg/images-create.html). The AWS-managed [`AmazonEC2ContainerRegistryFullAccess`](https://docs.aws.amazon.com/aws-managed-policy/latest/reference/AmazonEC2ContainerRegistryFullAccess.html) policy also covers these operations but grants broader access than SeBS requires. - -You can provide a [Lambda execution role](https://docs.aws.amazon.com/lambda/latest/dg/lambda-intro-execution-role.html) -or let SeBS create and manage the default `sebs-lambda-role`. To use a custom role, set -its ARN as `lambda-role` in the config JSON. SeBS reuses a role ARN loaded from the -configuration or cache without modifying it, so that role must already provide S3, -CloudWatch Logs, and benchmark-specific permissions. -The IAM identity running SeBS always needs `iam:PassRole` on the selected execution -role. - -When no role ARN is loaded, SeBS resolves the default role. The IAM identity running -SeBS must allow `iam:GetRole`, `iam:CreateRole`, `iam:AttachRolePolicy`, and -`iam:PutRolePolicy` on -`arn:aws:iam:::role/sebs-lambda-role`. SeBS attaches `AmazonS3FullAccess` -and `AWSLambdaBasicExecutionRole`, and adds an inline policy for the DynamoDB item -operations used by the bundled NoSQL benchmark. A cached default role is reused without -updating its policies. - -Benchmarks with the `nosql` module, such as `130.crud-api`, also use DynamoDB. -The IAM identity running SeBS performs table management and input seeding itself, so -prepare these permissions manually: - -| Principal | Required actions | Resource | -| --- | --- | --- | -| Identity running SeBS (for example, the `sebs` IAM user) | `dynamodb:CreateTable`, `dynamodb:DescribeTable`, `dynamodb:PutItem`; add `dynamodb:DeleteTable` for cleanup | `arn:aws:dynamodb:::table/sebs-benchmarks-*` | - -The AWS-managed [`AmazonDynamoDBFullAccess_v2`](https://docs.aws.amazon.com/aws-managed-policy/latest/reference/AmazonDynamoDBFullAccess_v2.html) -policy covers these caller-side DynamoDB actions but grants broader access than SeBS -requires. - -When it resolves the default execution role, SeBS grants `dynamodb:PutItem`, -`dynamodb:GetItem`, and `dynamodb:Query` on the same table prefix. Add those actions -yourself when using a configured or cached execution role. +SeBS does not modify an execution role loaded from the configuration or cache. When SeBS +resolves the default role itself, it attaches `AmazonS3FullAccess` and +`AWSLambdaBasicExecutionRole`, then adds the DynamoDB item permissions used by NoSQL +benchmarks. You can pass the credentials either using the default AWS-specific environment variables: @@ -158,7 +108,7 @@ or in the JSON input configuration: } ``` -Direct library invocations obtain provider duration and billing metrics from AWS `START` and `REPORT` log records. Application log lines may be interleaved with these records; SeBS ignores unstructured fields while parsing the provider metrics. +Direct library invocations parse AWS `START` and `REPORT` records for provider metrics and tolerate interleaved application log fields. ### Lambda Function URLs vs API Gateway diff --git a/tests/test_aws_matrix_fixes.py b/tests/test_aws_matrix_fixes.py index abe23355f..b952e9fd4 100644 --- a/tests/test_aws_matrix_fixes.py +++ b/tests/test_aws_matrix_fixes.py @@ -18,7 +18,7 @@ class AWSMatrixFixesTest(unittest.TestCase): - def test_aws_report_parser_ignores_application_log_fields(self): + def test_aws_report_parser_tolerates_application_log_fields(self): request_id = "1ebf703c-b814-4eb8-b26e-788f53d5e328" log = ( f"START RequestId: {request_id} Version: $LATEST\n" @@ -71,18 +71,21 @@ def test_new_default_lambda_role_receives_dynamodb_access(self): resources.lambda_role(session) self.assertEqual(iam_client.put_role_policy.call_count, 1) - def test_configured_lambda_role_is_not_modified(self): + def test_configured_or_cached_lambda_role_is_not_modified(self): role_arn = "arn:aws:iam::123456789012:role/sebs-lambda-role" - cache = Mock() - cache.get_config.return_value = None - resources = AWSResources.deserialize( - {"lambda-role": role_arn, "resources": {}}, cache, LoggingHandlers() + cases = ( + ({"lambda-role": role_arn, "resources": {}}, None), + ({}, {"resources": {"lambda-role": role_arn}}), ) - session = Mock() - - self.assertEqual(resources.lambda_role(session), role_arn) + for config, cached_config in cases: + with self.subTest(config=config, cached_config=cached_config): + cache = Mock() + cache.get_config.return_value = cached_config + resources = AWSResources.deserialize(config, cache, LoggingHandlers()) + session = Mock() - session.client.assert_not_called() + self.assertEqual(resources.lambda_role(session), role_arn) + session.client.assert_not_called() def test_411_is_container_only_on_aws(self): with (ROOT / "benchmarks/400.inference/411.image-recognition/config.json").open() as f: @@ -93,9 +96,7 @@ def test_411_is_container_only_on_aws(self): self.assertTrue(config.supports_system_variant("local", "package")) def test_igraph_root_parent_conventions_validate_equally(self): - module = runpy.run_path( - str(ROOT / "benchmarks/500.scientific/503.graph-bfs/input.py") - ) + module = runpy.run_path(str(ROOT / "benchmarks/500.scientific/503.graph-bfs/input.py")) validate_output = module["validate_output"] result = [list(range(10)), [0, 1, 10], [0] * 10] From 40871e6a82e482fe40c1f1e221effa87465241b8 Mon Sep 17 00:00:00 2001 From: Karl Liu Date: Sun, 13 Sep 2026 21:33:28 +0200 Subject: [PATCH 8/9] Simplify AWS execution role ARN validation --- sebs/aws/config.py | 13 ++++--------- 1 file changed, 4 insertions(+), 9 deletions(-) diff --git a/sebs/aws/config.py b/sebs/aws/config.py index f2373cef5..794b0a563 100644 --- a/sebs/aws/config.py +++ b/sebs/aws/config.py @@ -16,6 +16,7 @@ import base64 import json import os +import re import time from enum import Enum from typing import TYPE_CHECKING, cast, Dict, List, Optional, Tuple @@ -514,16 +515,10 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: ) time.sleep(10) - arn_parts = self._lambda_role.split(":", 5) - if ( - len(arn_parts) != 6 - or arn_parts[0] != "arn" - or arn_parts[2] != "iam" - or not arn_parts[4] - ): + arn_match = re.fullmatch(r"arn:([^:]+):iam::([^:]+):role/.+", self._lambda_role) + if arn_match is None: raise RuntimeError(f"Invalid Lambda execution role ARN: {self._lambda_role}") - partition = arn_parts[1] - account_id = arn_parts[4] + partition, account_id = arn_match.groups() for policy in ( "arn:aws:iam::aws:policy/AmazonS3FullAccess", From dd8581b78125d37b339a52ef44afde85455a6ea5 Mon Sep 17 00:00:00 2001 From: Karl Liu Date: Sun, 13 Sep 2026 22:20:27 +0200 Subject: [PATCH 9/9] Address AWS review documentation and regression test feedback --- .../411.image-recognition/README.md | 3 +- docs/contributing.md | 15 ++++ tests/test_aws_matrix_fixes.py | 85 +++++++++++++++---- 3 files changed, 83 insertions(+), 20 deletions(-) diff --git a/benchmarks/400.inference/411.image-recognition/README.md b/benchmarks/400.inference/411.image-recognition/README.md index 30bdd4cb7..9a32e473d 100644 --- a/benchmarks/400.inference/411.image-recognition/README.md +++ b/benchmarks/400.inference/411.image-recognition/README.md @@ -16,8 +16,7 @@ The minimal memory amount is set to 768 MiB due to GCP requirements. It works wi > This benchmark contains PyTorch which is often too large to fit into a code package. Up to Python 3.7, we can directly ship the dependencies. For Python 3.8, we use an additional zipping step that requires additional setup during the first run, making cold invocations slower. Warm invocations are not affected. > [!WARNING] -> This benchmark exceeds the AWS Lambda uncompressed code-package limit. Use `--system-variant container` on AWS. +> Use `--system-variant container` on AWS. This benchmark does not work on AWS with Python 3.9 and code package due to excessive code size. While it is possible to ship the benchmark by zipping `torchvision` and `numpy` (see `benchmarks/400.inference/411.image-recognition/python/package.sh`), this significantly affects cold startup. On the lowest supported memory configuration of 512 MB, the cold startup can reach 30 seconds, making HTTP trigger unusable due to 30 second timeout of API gateway. Use Docker deployments for these configurations. > [!WARNING] > This benchmark does not work on GCP functions gen1 with Python 3.8+ due to excessive code size. Use container deployments on Google Cloud Run for these configurations. - diff --git a/docs/contributing.md b/docs/contributing.md index deed86bae..5f38fe207 100644 --- a/docs/contributing.md +++ b/docs/contributing.md @@ -28,6 +28,21 @@ pip install .[dev] uv sync --extra dev ``` +### AWS Regression Checks + +Run the focused offline checks from the repository root: + +```bash +python -m unittest discover -s tests -p test_aws_matrix_fixes.py -v +``` + +These tests cover report parsing, missing default-role creation and DynamoDB policy +attachment, preservation of configured/cached roles without IAM modification, +benchmark initialization with AWS deployment variants, and both valid BFS root-parent +sentinels. AWS calls and benchmark-data setup are mocked. The tests do not validate +cloud deployment or dependency-wheel compatibility; those require separate builds +and benchmark runs. Requirements files for different runtimes need not be identical. + ## Docker Images SeBS uses Docker images for building functions and running benchmarks: diff --git a/tests/test_aws_matrix_fixes.py b/tests/test_aws_matrix_fixes.py index b952e9fd4..da4c2738d 100644 --- a/tests/test_aws_matrix_fixes.py +++ b/tests/test_aws_matrix_fixes.py @@ -5,11 +5,12 @@ import runpy import unittest from pathlib import Path -from unittest.mock import Mock +from unittest.mock import Mock, patch from sebs.aws.aws import AWS from sebs.aws.config import AWSResources -from sebs.benchmark import BenchmarkConfig +from sebs.benchmark import Benchmark +from sebs.experiments.config import Config as ExperimentConfig from sebs.faas.function import ExecutionResult from sebs.utils import LoggingHandlers @@ -18,7 +19,10 @@ class AWSMatrixFixesTest(unittest.TestCase): + """Cover AWS matrix failures without building images or contacting AWS.""" + def test_aws_report_parser_tolerates_application_log_fields(self): + """Extract provider metrics despite interleaved application warnings.""" request_id = "1ebf703c-b814-4eb8-b26e-788f53d5e328" log = ( f"START RequestId: {request_id} Version: $LATEST\n" @@ -40,17 +44,28 @@ def test_aws_report_parser_tolerates_application_log_fields(self): self.assertEqual(result.billing.memory, 128) def test_new_default_lambda_role_receives_dynamodb_access(self): + """Create a missing default role and attach its scoped DynamoDB policy once.""" + + class NoSuchEntityException(Exception): + """Stand in for the missing-role exception from the IAM client.""" + + pass + resources = AWSResources() resources.region = "us-east-1" iam_client = Mock() - iam_client.get_role.return_value = { + iam_client.exceptions.NoSuchEntityException = NoSuchEntityException + iam_client.get_role.side_effect = NoSuchEntityException() + iam_client.create_role.return_value = { "Role": {"Arn": "arn:aws:iam::123456789012:role/sebs-lambda-role"} } session = Mock() session.client.return_value = iam_client - resources.lambda_role(session) + with patch("sebs.aws.config.time.sleep"): + resources.lambda_role(session) + iam_client.create_role.assert_called_once() self.assertEqual( iam_client.put_role_policy.call_args.kwargs["RoleName"], "sebs-lambda-role" ) @@ -72,6 +87,7 @@ def test_new_default_lambda_role_receives_dynamodb_access(self): self.assertEqual(iam_client.put_role_policy.call_count, 1) def test_configured_or_cached_lambda_role_is_not_modified(self): + """Respect supplied roles without requiring IAM policy-management permissions.""" role_arn = "arn:aws:iam::123456789012:role/sebs-lambda-role" cases = ( ({"lambda-role": role_arn, "resources": {}}, None), @@ -87,19 +103,59 @@ def test_configured_or_cached_lambda_role_is_not_modified(self): self.assertEqual(resources.lambda_role(session), role_arn) session.client.assert_not_called() - def test_411_is_container_only_on_aws(self): - with (ROOT / "benchmarks/400.inference/411.image-recognition/config.json").open() as f: - config = BenchmarkConfig.deserialize(json.load(f)) - - self.assertFalse(config.supports_system_variant("aws", "package")) - self.assertTrue(config.supports_system_variant("aws", "container")) - self.assertTrue(config.supports_system_variant("local", "package")) + def test_411_system_variant_is_validated_during_initialization(self): + """Reject AWS package deployment and accept container deployment at initialization.""" + experiment = { + "update_code": False, + "update_storage": False, + "download_results": False, + "runtime": {"language": "python", "version": "3.10"}, + "architecture": "x64", + } + cache = Mock() + cache.get_container.return_value = None + cache.get_functions.return_value = {} + + with self.assertRaisesRegex( + RuntimeError, + "does not support system variant package on aws; use container", + ): + Benchmark( + "411.image-recognition", + "aws", + ExperimentConfig.deserialize({**experiment, "system_variant": "package"}), + Mock(), + str(ROOT), + cache, + Mock(), + ) + + with ( + patch("sebs.benchmark.ensure_benchmarks_data"), + patch("sebs.benchmark.load_benchmark_input"), + ): + benchmark = Benchmark( + "411.image-recognition", + "aws", + ExperimentConfig.deserialize({**experiment, "system_variant": "container"}), + Mock(), + str(ROOT), + cache, + Mock(), + ) + + self.assertEqual(benchmark.system_variant.value, "container") def test_igraph_root_parent_conventions_validate_equally(self): + """Accept both valid BFS root-parent sentinels while rejecting invalid output.""" module = runpy.run_path(str(ROOT / "benchmarks/500.scientific/503.graph-bfs/input.py")) validate_output = module["validate_output"] result = [list(range(10)), [0, 1, 10], [0] * 10] + self.assertIsNone( + validate_output(None, {"size": 10, "seed": 42}, {"result": result}, "python") + ) + result[2][0] = -1 self.assertIsNone( validate_output(None, {"size": 10, "seed": 42}, {"result": result}, "python") ) @@ -109,13 +165,6 @@ def test_igraph_root_parent_conventions_validate_equally(self): validate_output(None, {"size": 10, "seed": 42}, {"result": result}, "python"), ) - def test_python_310_uses_the_wheel_compatible_dna_pins(self): - requirements = ROOT / "benchmarks/500.scientific/504.dna-visualisation/python" - self.assertEqual( - (requirements / "requirements.txt.3.10").read_text(), - (requirements / "requirements.txt.3.11").read_text(), - ) - if __name__ == "__main__": unittest.main()