diff --git a/benchmarks/100.webapps/130.crud-api/README.md b/benchmarks/100.webapps/130.crud-api/README.md index 11e07a2ed..0cb335051 100644 --- a/benchmarks/100.webapps/130.crud-api/README.md +++ b/benchmarks/100.webapps/130.crud-api/README.md @@ -1,7 +1,7 @@ # 130.crud-api - CRUD API **Type:** Webapps -**Languages:** Python +**Languages:** Python, Node.js **Architecture:** x64, arm64 ## Description diff --git a/benchmarks/400.inference/411.image-recognition/README.md b/benchmarks/400.inference/411.image-recognition/README.md index 3b3e5cb1d..9a32e473d 100644 --- a/benchmarks/400.inference/411.image-recognition/README.md +++ b/benchmarks/400.inference/411.image-recognition/README.md @@ -16,8 +16,7 @@ The minimal memory amount is set to 768 MiB due to GCP requirements. It works wi > This benchmark contains PyTorch which is often too large to fit into a code package. Up to Python 3.7, we can directly ship the dependencies. For Python 3.8, we use an additional zipping step that requires additional setup during the first run, making cold invocations slower. Warm invocations are not affected. > [!WARNING] -> This benchmark does not work on AWS with Python 3.9 and code package due to excessive code size. While it is possible to ship the benchmark by zipping `torchvision` and `numpy` (see `benchmarks/400.inference/411.image-recognition/python/package.sh`), this significantly affects cold startup. On the lowest supported memory configuration of 512 MB, the cold startup can reach 30 seconds, making HTTP trigger unusable due to 30 second timeout of API gateway. Use Docker deployments for these configurations. +> Use `--system-variant container` on AWS. This benchmark does not work on AWS with Python 3.9 and code package due to excessive code size. While it is possible to ship the benchmark by zipping `torchvision` and `numpy` (see `benchmarks/400.inference/411.image-recognition/python/package.sh`), this significantly affects cold startup. On the lowest supported memory configuration of 512 MB, the cold startup can reach 30 seconds, making HTTP trigger unusable due to 30 second timeout of API gateway. Use Docker deployments for these configurations. > [!WARNING] > This benchmark does not work on GCP functions gen1 with Python 3.8+ due to excessive code size. Use container deployments on Google Cloud Run for these configurations. - diff --git a/benchmarks/400.inference/411.image-recognition/config.json b/benchmarks/400.inference/411.image-recognition/config.json index 8c5010fc3..f15c28494 100644 --- a/benchmarks/400.inference/411.image-recognition/config.json +++ b/benchmarks/400.inference/411.image-recognition/config.json @@ -3,5 +3,6 @@ "memory": 768, "languages": ["python", "cpp"], "modules": ["storage"], + "system_variants": {"aws": ["container"]}, "cpp_dependencies": ["sdk", "torch", "opencv"] } diff --git a/benchmarks/500.scientific/503.graph-bfs/README.md b/benchmarks/500.scientific/503.graph-bfs/README.md index ee27ea620..c710908e2 100644 --- a/benchmarks/500.scientific/503.graph-bfs/README.md +++ b/benchmarks/500.scientific/503.graph-bfs/README.md @@ -1,9 +1,11 @@ # 503.graph-bfs - Graph BFS **Type:** Scientific -**Languages:** Python +**Languages:** Python, C++ **Architecture:** x64, arm64 ## Description The benchmark represents scientific computations offloaded to serverless functions. It uses the `python-igraph` library to generate an input graph and process it with the Breadth-First Search (BFS) algorithm. + +Python 3.9 uses `python-igraph` 0.9, which reports the BFS root as its own parent. Newer igraph versions report `-1`. Output validation canonicalizes these equivalent root sentinels before checking the deterministic result checksum. diff --git a/benchmarks/500.scientific/503.graph-bfs/input.py b/benchmarks/500.scientific/503.graph-bfs/input.py index 5ece44eca..aa190059a 100644 --- a/benchmarks/500.scientific/503.graph-bfs/input.py +++ b/benchmarks/500.scientific/503.graph-bfs/input.py @@ -45,7 +45,24 @@ def validate_output(data_dir: str | None, input_config: dict, output: dict, lang seed = input_config.get('seed') if seed == 42 and size in expected_checksums[language]: - serialized = json.dumps(result, separators=(',', ':')) + # igraph 0.9 uses the root itself as its parent, while igraph 0.11 + # uses -1. Both represent the same BFS tree, so canonicalize the root + # sentinel before comparing deterministic output checksums. + normalized_result = [ + list(value) if isinstance(value, (list, tuple)) else value + for value in result + ] + order, parents = normalized_result[0], normalized_result[2] + if isinstance(order, list) and order and isinstance(parents, list): + root = order[0] + if ( + isinstance(root, int) + and 0 <= root < len(parents) + and parents[root] in (-1, root) + ): + parents[root] = -1 + + serialized = json.dumps(normalized_result, separators=(',', ':')) actual_checksum = hashlib.md5(serialized.encode()).hexdigest() expected_checksum = expected_checksums[language][size] if actual_checksum != expected_checksum: diff --git a/benchmarks/500.scientific/504.dna-visualisation/README.md b/benchmarks/500.scientific/504.dna-visualisation/README.md index 2440a6bf9..d3216846c 100644 --- a/benchmarks/500.scientific/504.dna-visualisation/README.md +++ b/benchmarks/500.scientific/504.dna-visualisation/README.md @@ -7,3 +7,5 @@ ## Description This benchmark is inspired by the [DNAVisualization](https://github.com/Benjamin-Lee/DNAvisualization.org) project and it implements processing the `.fasta` file with the `squiggle` Python library. + +AWS Python 3.10 and 3.11 use version-specific dependency files. NumPy, ContourPy, and Pillow are pinned to versions with wheels compatible with the Lambda build images so that package construction does not require a compiler. diff --git a/benchmarks/500.scientific/504.dna-visualisation/python/requirements.txt.3.10 b/benchmarks/500.scientific/504.dna-visualisation/python/requirements.txt.3.10 new file mode 100644 index 000000000..3602556e0 --- /dev/null +++ b/benchmarks/500.scientific/504.dna-visualisation/python/requirements.txt.3.10 @@ -0,0 +1,7 @@ +# Copyright 2020-2025 ETH Zurich and the SeBS authors. All rights reserved. +squiggle==0.3.1 +# Lambda images use glibc 2.26, for which there are no wheels on new packages +# we have to fix version to prevent compilation from source +numpy==2.2.6 +contourpy==1.3.2 +pillow==10.3.0 diff --git a/docs/build.md b/docs/build.md index c6c7f4bab..99e627272 100644 --- a/docs/build.md +++ b/docs/build.md @@ -38,6 +38,16 @@ and `nosql` for NoSQL databases. Each module corresponds to a set of packages th **Package Code** - we move files to create the directory structure expected on each cloud platform and create a final deployment package. An example of a customization is Azure Functions, where additional JSON configuration files are needed. +### Benchmark-Specific System Variants + +A benchmark can restrict deployment variants for a platform in `config.json`: + +```json +"system_variants": {"aws": ["container"]} +``` + +The values match the platform's `--system-variant` choices. Without this mapping, all platform variants are allowed. SeBS checks the restriction before building the benchmark or creating cloud resources. + ## Docker Image Deployment ```mermaid diff --git a/docs/contributing.md b/docs/contributing.md index deed86bae..5f38fe207 100644 --- a/docs/contributing.md +++ b/docs/contributing.md @@ -28,6 +28,21 @@ pip install .[dev] uv sync --extra dev ``` +### AWS Regression Checks + +Run the focused offline checks from the repository root: + +```bash +python -m unittest discover -s tests -p test_aws_matrix_fixes.py -v +``` + +These tests cover report parsing, missing default-role creation and DynamoDB policy +attachment, preservation of configured/cached roles without IAM modification, +benchmark initialization with AWS deployment variants, and both valid BFS root-parent +sentinels. AWS calls and benchmark-data setup are mocked. The tests do not validate +cloud deployment or dependency-wheel compatibility; those require separate builds +and benchmark runs. Requirements files for different runtimes need not be identical. + ## Docker Images SeBS uses Docker images for building functions and running benchmarks: diff --git a/docs/platforms.md b/docs/platforms.md index b01937c40..e91bb080c 100644 --- a/docs/platforms.md +++ b/docs/platforms.md @@ -57,16 +57,36 @@ If you have JSON result files, such as `experiment.json` from a benchmark run or jq 'del(.config.deployment.credentials)' | sponge ``` -# AWS Lambda +## AWS Lambda AWS provides one year of free services, including a significant amount of computing time in AWS Lambda. -To work with AWS, you need to provide access and secret keys to a role with permissions +To work with AWS, provide access and secret keys for an IAM identity with permissions sufficient to manage functions and S3 resources. -Additionally, the account must have `AmazonAPIGatewayAdministrator` permission to set up -automatically an AWS HTTP trigger. -You can provide a [role](https://docs.aws.amazon.com/lambda/latest/dg/lambda-intro-execution-role.html) -with permissions to access AWS Lambda and S3; otherwise, one will be created automatically. -To use a user-defined Lambda role, set the name in the config JSON - see an example in `configs/example.json`. + +### Quick Start Permissions + +For package and container deployments, NoSQL benchmarks, log queries, and cleanup, +attach these AWS managed policies to the IAM identity running SeBS: + +* `AWSLambda_FullAccess` +* `AmazonS3FullAccess` +* `IAMFullAccess` +* `CloudWatchLogsFullAccess` +* `AmazonDynamoDBFullAccess_v2` +* `AmazonEC2ContainerRegistryFullAccess` + +This bundle is intentionally broad and should be used only in a dedicated experiment +account. Add `AmazonAPIGatewayAdministrator` only when using API Gateway instead of the +default Lambda Function URLs. + +The [Lambda execution role](https://docs.aws.amazon.com/lambda/latest/dg/lambda-intro-execution-role.html) +is separate from the IAM identity running SeBS. Set its ARN as `lambda-role` in the config +JSON (see `configs/example.json`), or let SeBS resolve `sebs-lambda-role`. + +SeBS does not modify an execution role loaded from the configuration or cache. When SeBS +resolves the default role itself, it attaches `AmazonS3FullAccess` and +`AWSLambdaBasicExecutionRole`, then adds the DynamoDB item permissions used by NoSQL +benchmarks. You can pass the credentials either using the default AWS-specific environment variables: @@ -91,6 +111,8 @@ or in the JSON input configuration: } ``` +Direct library invocations parse AWS `START` and `REPORT` records for provider metrics and tolerate interleaved application log fields. + ### Lambda Function URLs vs API Gateway SeBS supports two methods for HTTP-based function invocation on AWS Lambda: diff --git a/sebs/aws/aws.py b/sebs/aws/aws.py index dd7077d18..5751cf8eb 100644 --- a/sebs/aws/aws.py +++ b/sebs/aws/aws.py @@ -671,10 +671,12 @@ def parse_aws_report( "REPORT RequestId: abc123\tDuration: 100.00 ms\tBilled Duration: 100 ms\t..." """ aws_vals = {} - for line in log.split("\t"): - if not line.isspace(): - split = line.split(":") - aws_vals[split[0]] = split[1].split()[0] + for log_line in log.splitlines(): + for field in log_line.split("\t"): + key, separator, value = field.partition(":") + values = value.split() + if separator and values: + aws_vals[key] = values[0] if "START RequestId" in aws_vals: request_id = aws_vals["START RequestId"] elif "REPORT RequestId" in aws_vals: diff --git a/sebs/aws/config.py b/sebs/aws/config.py index 83230641a..794b0a563 100644 --- a/sebs/aws/config.py +++ b/sebs/aws/config.py @@ -16,6 +16,7 @@ import base64 import json import os +import re import time from enum import Enum from typing import TYPE_CHECKING, cast, Dict, List, Optional, Tuple @@ -472,9 +473,9 @@ def function_url_auth_type(self, value: FunctionURLAuthType): def lambda_role(self, boto3_session: boto3.session.Session) -> str: """Get or create IAM role for Lambda execution. - Creates a Lambda execution role with S3 and basic execution permissions - if it doesn't already exist. The role allows Lambda functions to access - S3 and write CloudWatch logs. + Creates a Lambda execution role with S3, DynamoDB, and basic execution + permissions if it doesn't already exist. The role allows Lambda functions + to access SeBS resources and write CloudWatch logs. Args: boto3_session: Boto3 session for AWS API calls @@ -498,10 +499,6 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: ], } role_name = "sebs-lambda-role" - attached_policies = [ - "arn:aws:iam::aws:policy/AmazonS3FullAccess", - "arn:aws:iam::aws:policy/service-role/AWSLambdaBasicExecutionRole", - ] try: out = iam_client.get_role(RoleName=role_name) self._lambda_role = out["Role"]["Arn"] @@ -517,9 +514,36 @@ def lambda_role(self, boto3_session: boto3.session.Session) -> str: "Sleep 10 seconds to avoid problems when using role immediately." ) time.sleep(10) - # Attach basic AWS Lambda and S3 policies. - for policy in attached_policies: + + arn_match = re.fullmatch(r"arn:([^:]+):iam::([^:]+):role/.+", self._lambda_role) + if arn_match is None: + raise RuntimeError(f"Invalid Lambda execution role ARN: {self._lambda_role}") + partition, account_id = arn_match.groups() + + for policy in ( + "arn:aws:iam::aws:policy/AmazonS3FullAccess", + "arn:aws:iam::aws:policy/service-role/AWSLambdaBasicExecutionRole", + ): iam_client.attach_role_policy(RoleName=role_name, PolicyArn=policy) + + dynamodb_policy = { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": ["dynamodb:GetItem", "dynamodb:PutItem", "dynamodb:Query"], + "Resource": ( + f"arn:{partition}:dynamodb:{self.region}:{account_id}:" + "table/sebs-benchmarks-*" + ), + } + ], + } + iam_client.put_role_policy( + RoleName=role_name, + PolicyName="sebs-dynamodb-access", + PolicyDocument=json.dumps(dynamodb_policy), + ) return self._lambda_role def http_api( @@ -1132,6 +1156,12 @@ def deserialize(config: dict, cache: Cache, handlers: LoggingHandlers) -> Resour ret.logging_handlers = handlers ret.logging.info("No resources for AWS found, initialize!") + configured_lambda_role = config.get("lambda-role") or config.get("resources", {}).get( + "lambda-role" + ) + if configured_lambda_role: + ret._lambda_role = configured_lambda_role + return ret diff --git a/sebs/benchmark.py b/sebs/benchmark.py index 32cf62b9f..df3ba42b1 100644 --- a/sebs/benchmark.py +++ b/sebs/benchmark.py @@ -118,6 +118,7 @@ class BenchmarkConfig: memory: Memory allocation in MB languages: List of supported programming languages modules: List of benchmark modules/features required + system_variants: Optional deployment-variant restrictions by platform """ @@ -128,6 +129,7 @@ def __init__( language_specs: List["LanguageSpec"], modules: List[BenchmarkModule], cpp_dependencies: Optional[List[CppDependencies]] = None, + system_variants: Optional[Dict[str, List[str]]] = None, ): """ Initialize a benchmark configuration. @@ -137,12 +139,14 @@ def __init__( memory: Memory allocation in MB languages: List of supported programming languages modules: List of benchmark modules/features required + system_variants: Optional deployment-variant restrictions by platform """ self._timeout = timeout self._memory = memory self._language_specs = language_specs self._modules = modules self._cpp_dependencies = cpp_dependencies or [] + self._system_variants = system_variants or {} @property def timeout(self) -> int: @@ -225,6 +229,15 @@ def supports(self, language: Language, variant: str) -> bool: """Return True when language + variant combination is declared in config.json.""" return variant in self.supported_variants(language) + def supported_system_variants(self, deployment: str) -> Optional[List[str]]: + """Return a platform restriction, or None when every variant is supported.""" + return self._system_variants.get(deployment) + + def supports_system_variant(self, deployment: str, variant: str) -> bool: + """Return whether a deployment variant is supported on the platform.""" + supported = self.supported_system_variants(deployment) + return supported is None or variant in supported + @staticmethod def deserialize(json_object: dict) -> BenchmarkConfig: """ @@ -244,6 +257,7 @@ def deserialize(json_object: dict) -> BenchmarkConfig: cpp_dependencies=[ CppDependencies.deserialize(x) for x in json_object.get("cpp_dependencies", []) ], + system_variants=json_object.get("system_variants"), ) @@ -621,6 +635,18 @@ def __init__( self.benchmark, self.language, self._language_variant ) ) + if not self.benchmark_config.supports_system_variant( + self._deployment_name, self._system_variant.value + ): + supported = self.benchmark_config.supported_system_variants(self._deployment_name) + raise RuntimeError( + "Benchmark {} does not support system variant {} on {}; use {}".format( + self.benchmark, + self._system_variant.value, + self._deployment_name, + ", ".join(supported or []), + ) + ) self._cache_client = cache_client self._docker_client = docker_client self._system_config = system_config diff --git a/tests/test_aws_matrix_fixes.py b/tests/test_aws_matrix_fixes.py new file mode 100644 index 000000000..da4c2738d --- /dev/null +++ b/tests/test_aws_matrix_fixes.py @@ -0,0 +1,170 @@ +#!/usr/bin/env python3 +# Copyright 2020-2025 ETH Zurich and the SeBS authors. All rights reserved. + +import json +import runpy +import unittest +from pathlib import Path +from unittest.mock import Mock, patch + +from sebs.aws.aws import AWS +from sebs.aws.config import AWSResources +from sebs.benchmark import Benchmark +from sebs.experiments.config import Config as ExperimentConfig +from sebs.faas.function import ExecutionResult +from sebs.utils import LoggingHandlers + + +ROOT = Path(__file__).resolve().parents[1] + + +class AWSMatrixFixesTest(unittest.TestCase): + """Cover AWS matrix failures without building images or contacting AWS.""" + + def test_aws_report_parser_tolerates_application_log_fields(self): + """Extract provider metrics despite interleaved application warnings.""" + request_id = "1ebf703c-b814-4eb8-b26e-788f53d5e328" + log = ( + f"START RequestId: {request_id} Version: $LATEST\n" + f"2026-08-02T09:33:20.853Z\t{request_id}\tERROR\t" + "(node:8) [DEP0040] DeprecationWarning: The `punycode` module is deprecated.\n" + f"END RequestId: {request_id}\n" + f"REPORT RequestId: {request_id}\tDuration: 9467.75 ms\t" + "Billed Duration: 10014 ms\tMemory Size: 128 MB\t" + "Max Memory Used: 112 MB\tInit Duration: 546.19 ms\n" + ) + result = ExecutionResult() + + self.assertEqual(AWS.parse_aws_report(log, result), request_id) + self.assertEqual(result.request_id, request_id) + self.assertEqual(result.provider_times.execution, 9467750) + self.assertEqual(result.provider_times.initialization, 546190) + self.assertEqual(result.stats.memory_used, 112.0) + self.assertEqual(result.billing.billed_time, 10014) + self.assertEqual(result.billing.memory, 128) + + def test_new_default_lambda_role_receives_dynamodb_access(self): + """Create a missing default role and attach its scoped DynamoDB policy once.""" + + class NoSuchEntityException(Exception): + """Stand in for the missing-role exception from the IAM client.""" + + pass + + resources = AWSResources() + resources.region = "us-east-1" + iam_client = Mock() + iam_client.exceptions.NoSuchEntityException = NoSuchEntityException + iam_client.get_role.side_effect = NoSuchEntityException() + iam_client.create_role.return_value = { + "Role": {"Arn": "arn:aws:iam::123456789012:role/sebs-lambda-role"} + } + session = Mock() + session.client.return_value = iam_client + + with patch("sebs.aws.config.time.sleep"): + resources.lambda_role(session) + + iam_client.create_role.assert_called_once() + self.assertEqual( + iam_client.put_role_policy.call_args.kwargs["RoleName"], "sebs-lambda-role" + ) + self.assertEqual( + iam_client.put_role_policy.call_args.kwargs["PolicyName"], + "sebs-dynamodb-access", + ) + policy = json.loads(iam_client.put_role_policy.call_args.kwargs["PolicyDocument"]) + self.assertEqual( + policy["Statement"][0]["Action"], + ["dynamodb:GetItem", "dynamodb:PutItem", "dynamodb:Query"], + ) + self.assertEqual( + policy["Statement"][0]["Resource"], + "arn:aws:dynamodb:us-east-1:123456789012:table/sebs-benchmarks-*", + ) + + resources.lambda_role(session) + self.assertEqual(iam_client.put_role_policy.call_count, 1) + + def test_configured_or_cached_lambda_role_is_not_modified(self): + """Respect supplied roles without requiring IAM policy-management permissions.""" + role_arn = "arn:aws:iam::123456789012:role/sebs-lambda-role" + cases = ( + ({"lambda-role": role_arn, "resources": {}}, None), + ({}, {"resources": {"lambda-role": role_arn}}), + ) + for config, cached_config in cases: + with self.subTest(config=config, cached_config=cached_config): + cache = Mock() + cache.get_config.return_value = cached_config + resources = AWSResources.deserialize(config, cache, LoggingHandlers()) + session = Mock() + + self.assertEqual(resources.lambda_role(session), role_arn) + session.client.assert_not_called() + + def test_411_system_variant_is_validated_during_initialization(self): + """Reject AWS package deployment and accept container deployment at initialization.""" + experiment = { + "update_code": False, + "update_storage": False, + "download_results": False, + "runtime": {"language": "python", "version": "3.10"}, + "architecture": "x64", + } + cache = Mock() + cache.get_container.return_value = None + cache.get_functions.return_value = {} + + with self.assertRaisesRegex( + RuntimeError, + "does not support system variant package on aws; use container", + ): + Benchmark( + "411.image-recognition", + "aws", + ExperimentConfig.deserialize({**experiment, "system_variant": "package"}), + Mock(), + str(ROOT), + cache, + Mock(), + ) + + with ( + patch("sebs.benchmark.ensure_benchmarks_data"), + patch("sebs.benchmark.load_benchmark_input"), + ): + benchmark = Benchmark( + "411.image-recognition", + "aws", + ExperimentConfig.deserialize({**experiment, "system_variant": "container"}), + Mock(), + str(ROOT), + cache, + Mock(), + ) + + self.assertEqual(benchmark.system_variant.value, "container") + + def test_igraph_root_parent_conventions_validate_equally(self): + """Accept both valid BFS root-parent sentinels while rejecting invalid output.""" + module = runpy.run_path(str(ROOT / "benchmarks/500.scientific/503.graph-bfs/input.py")) + validate_output = module["validate_output"] + result = [list(range(10)), [0, 1, 10], [0] * 10] + + self.assertIsNone( + validate_output(None, {"size": 10, "seed": 42}, {"result": result}, "python") + ) + result[2][0] = -1 + self.assertIsNone( + validate_output(None, {"size": 10, "seed": 42}, {"result": result}, "python") + ) + result[2][0] = 7 + self.assertIn( + "checksum mismatch", + validate_output(None, {"size": 10, "seed": 42}, {"result": result}, "python"), + ) + + +if __name__ == "__main__": + unittest.main()