From 3110fee8d8fd6bf50547965d301df8ee84c166fa Mon Sep 17 00:00:00 2001 From: Willian Cesar Cincerre Da Silva <3329-willian.cincerre@users.noreply.code.ifoodcorp.com.br> Date: Tue, 15 Sep 2026 06:13:26 -0300 Subject: [PATCH 1/9] feat(amazon-vpc-cni): add new integration for Amazon VPC CNI --- amazon_vpc_cni/CHANGELOG.md | 7 + amazon_vpc_cni/README.md | 66 ++ amazon_vpc_cni/assets/configuration/spec.yaml | 14 + amazon_vpc_cni/assets/service_checks.json | 11 + amazon_vpc_cni/datadog_checks/__init__.py | 1 + .../amazon_vpc_cni/__about__.py | 1 + .../datadog_checks/amazon_vpc_cni/__init__.py | 4 + .../datadog_checks/amazon_vpc_cni/check.py | 21 + .../amazon_vpc_cni/config_models/__init__.py | 20 + .../amazon_vpc_cni/config_models/defaults.py | 124 ++++ .../amazon_vpc_cni/config_models/instance.py | 180 ++++++ .../amazon_vpc_cni/config_models/shared.py | 41 ++ .../config_models/validators.py | 9 + .../amazon_vpc_cni/data/conf.yaml.example | 610 ++++++++++++++++++ .../datadog_checks/amazon_vpc_cni/metrics.py | 33 + amazon_vpc_cni/hatch.toml | 4 + amazon_vpc_cni/manifest.json | 58 ++ amazon_vpc_cni/metadata.csv | 33 + amazon_vpc_cni/pyproject.toml | 60 ++ amazon_vpc_cni/tests/__init__.py | 0 amazon_vpc_cni/tests/common.py | 46 ++ amazon_vpc_cni/tests/conftest.py | 57 ++ .../tests/docker/docker-compose.yaml | 8 + amazon_vpc_cni/tests/fixtures/metrics.txt | 97 +++ amazon_vpc_cni/tests/test_e2e.py | 23 + amazon_vpc_cni/tests/test_integration.py | 12 + amazon_vpc_cni/tests/test_unit.py | 35 + 27 files changed, 1575 insertions(+) create mode 100644 amazon_vpc_cni/CHANGELOG.md create mode 100644 amazon_vpc_cni/README.md create mode 100644 amazon_vpc_cni/assets/configuration/spec.yaml create mode 100644 amazon_vpc_cni/assets/service_checks.json create mode 100644 amazon_vpc_cni/datadog_checks/__init__.py create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/__about__.py create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/__init__.py create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/check.py create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/__init__.py create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/defaults.py create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/instance.py create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/shared.py create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/validators.py create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/data/conf.yaml.example create mode 100644 amazon_vpc_cni/datadog_checks/amazon_vpc_cni/metrics.py create mode 100644 amazon_vpc_cni/hatch.toml create mode 100644 amazon_vpc_cni/manifest.json create mode 100644 amazon_vpc_cni/metadata.csv create mode 100644 amazon_vpc_cni/pyproject.toml create mode 100644 amazon_vpc_cni/tests/__init__.py create mode 100644 amazon_vpc_cni/tests/common.py create mode 100644 amazon_vpc_cni/tests/conftest.py create mode 100644 amazon_vpc_cni/tests/docker/docker-compose.yaml create mode 100644 amazon_vpc_cni/tests/fixtures/metrics.txt create mode 100644 amazon_vpc_cni/tests/test_e2e.py create mode 100644 amazon_vpc_cni/tests/test_integration.py create mode 100644 amazon_vpc_cni/tests/test_unit.py diff --git a/amazon_vpc_cni/CHANGELOG.md b/amazon_vpc_cni/CHANGELOG.md new file mode 100644 index 000000000..eb9034136 --- /dev/null +++ b/amazon_vpc_cni/CHANGELOG.md @@ -0,0 +1,7 @@ +# CHANGELOG - Amazon VPC CNI + +## 0.1.0 / 2026-09-15 + +***Added***: + +* Initial release of the Amazon VPC CNI integration. diff --git a/amazon_vpc_cni/README.md b/amazon_vpc_cni/README.md new file mode 100644 index 000000000..1068cd315 --- /dev/null +++ b/amazon_vpc_cni/README.md @@ -0,0 +1,66 @@ +# Agent Check: Amazon VPC CNI + +## Overview + +The [Amazon VPC CNI plugin for Kubernetes][13] is the networking plugin deployed on each Amazon EC2 node in an Amazon EKS cluster. It manages elastic network interfaces (ENIs) and assigns private IPv4 or IPv6 addresses from the VPC to each pod. + +This integration scrapes the Prometheus metrics exposed by the VPC CNI plugin (`awscni_*`) and provides visibility into IP address allocation, ENI usage, AWS API activity, and IPAMD health. It lets you replace ad-hoc `awscni_*` custom metrics with an official, curated integration. + +## Setup + +### Installation + +If you are using Agent v6.8+ follow the instructions below to install the Amazon VPC CNI check on your host. See the dedicated Agent guide for [installing community integrations][1] to install checks with the [Agent Manager][2] or in a [Docker environment][4]. + +1. [Download and launch the Datadog Agent][3]. +2. Run the following command to install the Agent integration: + + ```shell + datadog-agent integration install -t datadog-amazon-vpc-cni== + ``` + +3. Configure your integration in the same way as core [integrations][5]. + +### Configuration + +Enable Prometheus metrics on the `aws-node` DaemonSet by setting the environment variable `ENABLE_PROMETHEUS_METRICS` to `true`. The plugin then exposes metrics at `http://localhost:61678/metrics` on each node. + +1. Edit the `amazon_vpc_cni.d/conf.yaml` file in the `conf.d/` folder at the root of your [Agent's configuration directory][6] to start collecting your Amazon VPC CNI metrics. See the [sample amazon_vpc_cni.d/conf.yaml][7] for all available configuration options. + +2. [Restart the Agent][8]. + +### Validation + +[Run the Agent's status subcommand][9] and look for `amazon_vpc_cni` under the Checks section. + +## Data Collected + +### Metrics + +See [metadata.csv][10] for a list of metrics provided by this integration. + +### Service Checks + +See [service_checks.json][11] for a list of service checks provided by this integration. + +### Events + +The Amazon VPC CNI integration does not include any events. + +## Troubleshooting + +Need help? Contact [Datadog support][12]. + +[1]: https://docs.datadoghq.com/agent/guide/use-community-integrations/ +[2]: https://docs.datadoghq.com/agent/guide/agent-commands/?tab=agentv6v7#start-stop-and-restart-the-agent +[3]: https://app.datadoghq.com/account/settings/agent/latest +[4]: https://docs.datadoghq.com/agent/guide/use-community-integrations/?tab=docker +[5]: https://docs.datadoghq.com/getting_started/integrations/ +[6]: https://docs.datadoghq.com/agent/guide/agent-configuration-files/#agent-configuration-directory +[7]: https://github.com/DataDog/integrations-extras/blob/master/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/data/conf.yaml.example +[8]: https://docs.datadoghq.com/agent/guide/agent-commands/#start-stop-and-restart-the-agent +[9]: https://docs.datadoghq.com/agent/guide/agent-commands/#agent-status-and-information +[10]: https://github.com/DataDog/integrations-extras/blob/master/amazon_vpc_cni/metadata.csv +[11]: https://github.com/DataDog/integrations-extras/blob/master/amazon_vpc_cni/assets/service_checks.json +[12]: https://docs.datadoghq.com/help/ +[13]: https://github.com/aws/amazon-vpc-cni-k8s diff --git a/amazon_vpc_cni/assets/configuration/spec.yaml b/amazon_vpc_cni/assets/configuration/spec.yaml new file mode 100644 index 000000000..419ba880e --- /dev/null +++ b/amazon_vpc_cni/assets/configuration/spec.yaml @@ -0,0 +1,14 @@ +name: Amazon VPC CNI +files: +- name: amazon_vpc_cni.yaml + options: + - template: init_config + options: + - template: init_config/default + - template: instances + options: + - template: instances/openmetrics + overrides: + openmetrics_endpoint.value.example: http://localhost:61678/metrics + openmetrics_endpoint.description: | + Endpoint exposing the Amazon VPC CNI Prometheus metrics. diff --git a/amazon_vpc_cni/assets/service_checks.json b/amazon_vpc_cni/assets/service_checks.json new file mode 100644 index 000000000..9dd7afed7 --- /dev/null +++ b/amazon_vpc_cni/assets/service_checks.json @@ -0,0 +1,11 @@ +[ + { + "agent_version": "7.59.0", + "integration": "Amazon VPC CNI", + "check": "amazon_vpc_cni.openmetrics.health", + "statuses": ["ok", "critical"], + "groups": ["host", "endpoint"], + "name": "Amazon VPC CNI OpenMetrics endpoint health", + "description": "Returns `CRITICAL` if the Agent is unable to connect to the Amazon VPC CNI OpenMetrics endpoint, otherwise returns `OK`." + } +] diff --git a/amazon_vpc_cni/datadog_checks/__init__.py b/amazon_vpc_cni/datadog_checks/__init__.py new file mode 100644 index 000000000..69e3be50d --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/__init__.py @@ -0,0 +1 @@ +__path__ = __import__('pkgutil').extend_path(__path__, __name__) diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/__about__.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/__about__.py new file mode 100644 index 000000000..b794fd409 --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/__about__.py @@ -0,0 +1 @@ +__version__ = '0.1.0' diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/__init__.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/__init__.py new file mode 100644 index 000000000..963df70d7 --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/__init__.py @@ -0,0 +1,4 @@ +from .__about__ import __version__ +from .check import AmazonVpcCniCheck + +__all__ = ['__version__', 'AmazonVpcCniCheck'] diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/check.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/check.py new file mode 100644 index 000000000..943c736de --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/check.py @@ -0,0 +1,21 @@ +from datadog_checks.base import OpenMetricsBaseCheckV2 + +from .metrics import METRIC_MAP + + +class AmazonVpcCniCheck(OpenMetricsBaseCheckV2): + __NAMESPACE__ = 'amazon_vpc_cni' + DEFAULT_METRIC_LIMIT = 0 + + def __init__(self, name, init_config, instances): + super(AmazonVpcCniCheck, self).__init__(name, init_config, instances) + + def get_default_config(self): + return { + 'metrics': [METRIC_MAP], + 'send_distribution_sums_as_monotonic': 'true', + 'send_distribution_counts_as_monotonic': 'true', + } + + def check(self, _): + super().check(_) diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/__init__.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/__init__.py new file mode 100644 index 000000000..5c2bf5c9f --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/__init__.py @@ -0,0 +1,20 @@ +# This file is autogenerated. +# To change this file you should edit assets/configuration/spec.yaml and then run the following commands: +# ddev -x validate config -s +# ddev -x validate models -s + +from .instance import InstanceConfig +from .shared import SharedConfig + + +class ConfigMixin: + _config_model_instance: InstanceConfig + _config_model_shared: SharedConfig + + @property + def config(self) -> InstanceConfig: + return self._config_model_instance + + @property + def shared_config(self) -> SharedConfig: + return self._config_model_shared diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/defaults.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/defaults.py new file mode 100644 index 000000000..a14932fbf --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/defaults.py @@ -0,0 +1,124 @@ +# This file is autogenerated. +# To change this file you should edit assets/configuration/spec.yaml and then run the following commands: +# ddev -x validate config -s +# ddev -x validate models -s + + +def instance_allow_redirects(): + return True + + +def instance_auth_type(): + return 'basic' + + +def instance_cache_metric_wildcards(): + return True + + +def instance_cache_shared_labels(): + return True + + +def instance_collect_counters_with_distributions(): + return False + + +def instance_collect_histogram_buckets(): + return True + + +def instance_disable_generic_tags(): + return False + + +def instance_empty_default_hostname(): + return False + + +def instance_enable_health_service_check(): + return True + + +def instance_enable_legacy_tags_normalization(): + return True + + +def instance_histogram_buckets_as_distributions(): + return False + + +def instance_ignore_connection_errors(): + return False + + +def instance_kerberos_auth(): + return 'disabled' + + +def instance_kerberos_delegate(): + return False + + +def instance_kerberos_force_initiate(): + return False + + +def instance_log_requests(): + return False + + +def instance_min_collection_interval(): + return 15 + + +def instance_non_cumulative_histogram_buckets(): + return False + + +def instance_persist_connections(): + return False + + +def instance_request_size(): + return 16 + + +def instance_skip_proxy(): + return False + + +def instance_tag_by_endpoint(): + return True + + +def instance_telemetry(): + return False + + +def instance_timeout(): + return 10 + + +def instance_tls_ignore_warning(): + return False + + +def instance_tls_use_host_header(): + return False + + +def instance_tls_verify(): + return True + + +def instance_use_latest_spec(): + return False + + +def instance_use_legacy_auth_encoding(): + return True + + +def instance_use_process_start_time(): + return False diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/instance.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/instance.py new file mode 100644 index 000000000..465c3c599 --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/instance.py @@ -0,0 +1,180 @@ +# This file is autogenerated. +# To change this file you should edit assets/configuration/spec.yaml and then run the following commands: +# ddev -x validate config -s +# ddev -x validate models -s + +from __future__ import annotations + +from types import MappingProxyType +from typing import Any, Optional, Union + +from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator +from typing_extensions import Literal + +from datadog_checks.base.utils.functions import identity +from datadog_checks.base.utils.models import validation + +from . import defaults, validators + + +SECURE_FIELD_NAMES = frozenset( + ['auth_token', 'kerberos_cache', 'kerberos_keytab', 'tls_ca_cert', 'tls_cert', 'tls_private_key'] +) + + +class AuthToken(BaseModel): + model_config = ConfigDict( + arbitrary_types_allowed=True, + frozen=True, + ) + reader: Optional[MappingProxyType[str, Any]] = None + writer: Optional[MappingProxyType[str, Any]] = None + + +class ExtraMetrics(BaseModel): + model_config = ConfigDict( + arbitrary_types_allowed=True, + extra='allow', + frozen=True, + ) + name: Optional[str] = None + type: Optional[str] = None + + +class MetricPatterns(BaseModel): + model_config = ConfigDict( + arbitrary_types_allowed=True, + frozen=True, + ) + exclude: Optional[tuple[str, ...]] = None + include: Optional[tuple[str, ...]] = None + + +class Metrics(BaseModel): + model_config = ConfigDict( + arbitrary_types_allowed=True, + extra='allow', + frozen=True, + ) + name: Optional[str] = None + type: Optional[str] = None + + +class Proxy(BaseModel): + model_config = ConfigDict( + arbitrary_types_allowed=True, + frozen=True, + ) + http: Optional[str] = None + https: Optional[str] = None + no_proxy: Optional[tuple[str, ...]] = None + + +class ShareLabels(BaseModel): + model_config = ConfigDict( + arbitrary_types_allowed=True, + frozen=True, + ) + labels: Optional[tuple[str, ...]] = None + match: Optional[tuple[str, ...]] = None + + +class InstanceConfig(BaseModel): + model_config = ConfigDict( + validate_default=True, + arbitrary_types_allowed=True, + frozen=True, + ) + allow_redirects: Optional[bool] = None + auth_token: Optional[AuthToken] = None + auth_type: Optional[str] = None + aws_host: Optional[str] = None + aws_region: Optional[str] = None + aws_service: Optional[str] = None + cache_metric_wildcards: Optional[bool] = None + cache_shared_labels: Optional[bool] = None + collect_counters_with_distributions: Optional[bool] = None + collect_histogram_buckets: Optional[bool] = None + connect_timeout: Optional[float] = None + disable_generic_tags: Optional[bool] = None + empty_default_hostname: Optional[bool] = None + enable_health_service_check: Optional[bool] = None + enable_legacy_tags_normalization: Optional[bool] = None + exclude_labels: Optional[tuple[str, ...]] = None + exclude_metrics: Optional[tuple[str, ...]] = None + exclude_metrics_by_labels: Optional[MappingProxyType[str, Union[bool, tuple[str, ...]]]] = None + extra_headers: Optional[MappingProxyType[str, Any]] = None + extra_metrics: Optional[tuple[Union[str, MappingProxyType[str, Union[str, ExtraMetrics]]], ...]] = None + headers: Optional[MappingProxyType[str, Any]] = None + histogram_buckets_as_distributions: Optional[bool] = None + hostname_format: Optional[str] = None + hostname_label: Optional[str] = None + ignore_connection_errors: Optional[bool] = None + ignore_tags: Optional[tuple[str, ...]] = None + include_labels: Optional[tuple[str, ...]] = None + kerberos_auth: Optional[Literal['required', 'optional', 'disabled']] = None + kerberos_cache: Optional[str] = None + kerberos_delegate: Optional[bool] = None + kerberos_force_initiate: Optional[bool] = None + kerberos_hostname: Optional[str] = None + kerberos_keytab: Optional[str] = None + kerberos_principal: Optional[str] = None + log_requests: Optional[bool] = None + metric_patterns: Optional[MetricPatterns] = None + metrics: Optional[tuple[Union[str, MappingProxyType[str, Union[str, Metrics]]], ...]] = None + min_collection_interval: Optional[float] = None + namespace: Optional[str] = Field(None, pattern='\\w*') + non_cumulative_histogram_buckets: Optional[bool] = None + ntlm_domain: Optional[str] = None + openmetrics_endpoint: str + password: Optional[str] = None + persist_connections: Optional[bool] = None + proxy: Optional[Proxy] = None + raw_line_filters: Optional[tuple[str, ...]] = None + raw_metric_prefix: Optional[str] = None + read_timeout: Optional[float] = None + rename_labels: Optional[MappingProxyType[str, Any]] = None + request_size: Optional[float] = None + service: Optional[str] = None + share_labels: Optional[MappingProxyType[str, Union[bool, ShareLabels]]] = None + skip_proxy: Optional[bool] = None + tag_by_endpoint: Optional[bool] = None + tags: Optional[tuple[str, ...]] = None + telemetry: Optional[bool] = None + timeout: Optional[float] = None + tls_ca_cert: Optional[str] = None + tls_cert: Optional[str] = None + tls_ciphers: Optional[tuple[str, ...]] = None + tls_ignore_warning: Optional[bool] = None + tls_private_key: Optional[str] = None + tls_protocols_allowed: Optional[tuple[str, ...]] = None + tls_use_host_header: Optional[bool] = None + tls_verify: Optional[bool] = None + use_latest_spec: Optional[bool] = None + use_legacy_auth_encoding: Optional[bool] = None + use_process_start_time: Optional[bool] = None + username: Optional[str] = None + + @model_validator(mode='before') + def _initial_validation(cls, values): + return validation.core.initialize_config(getattr(validators, 'initialize_instance', identity)(values)) + + @field_validator('*', mode='before') + def _validate(cls, value, info): + field = cls.model_fields[info.field_name] + field_name = field.alias or info.field_name + if field_name in info.context['configured_fields']: + value = getattr(validators, f'instance_{info.field_name}', identity)(value, field=field) + + if info.field_name in SECURE_FIELD_NAMES: + validation.security.check_field_trusted_provider( + info.field_name, value, info.context.get('security_config') + ) + else: + value = getattr(defaults, f'instance_{info.field_name}', lambda: value)() + + return validation.utils.make_immutable(value) + + @model_validator(mode='after') + def _final_validation(cls, model): + return validation.core.check_model(getattr(validators, 'check_instance', identity)(model)) diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/shared.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/shared.py new file mode 100644 index 000000000..4017c43e2 --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/shared.py @@ -0,0 +1,41 @@ +# This file is autogenerated. +# To change this file you should edit assets/configuration/spec.yaml and then run the following commands: +# ddev -x validate config -s +# ddev -x validate models -s + +from __future__ import annotations + +from typing import Optional + +from pydantic import BaseModel, ConfigDict, field_validator, model_validator + +from datadog_checks.base.utils.functions import identity +from datadog_checks.base.utils.models import validation + +from . import validators + + +class SharedConfig(BaseModel): + model_config = ConfigDict( + validate_default=True, + arbitrary_types_allowed=True, + frozen=True, + ) + service: Optional[str] = None + + @model_validator(mode='before') + def _initial_validation(cls, values): + return validation.core.initialize_config(getattr(validators, 'initialize_shared', identity)(values)) + + @field_validator('*', mode='before') + def _validate(cls, value, info): + field = cls.model_fields[info.field_name] + field_name = field.alias or info.field_name + if field_name in info.context['configured_fields']: + value = getattr(validators, f'shared_{info.field_name}', identity)(value, field=field) + + return validation.utils.make_immutable(value) + + @model_validator(mode='after') + def _final_validation(cls, model): + return validation.core.check_model(getattr(validators, 'check_shared', identity)(model)) diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/validators.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/validators.py new file mode 100644 index 000000000..39523e4f9 --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/config_models/validators.py @@ -0,0 +1,9 @@ +# Here you can include additional config validators or transformers +# +# def initialize_instance(values, **kwargs): +# if 'my_option' not in values and 'my_legacy_option' in values: +# values['my_option'] = values['my_legacy_option'] +# if values.get('my_number') > 10: +# raise ValueError('my_number max value is 10, got %s' % str(values.get('my_number'))) +# +# return values diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/data/conf.yaml.example b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/data/conf.yaml.example new file mode 100644 index 000000000..42e1e323d --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/data/conf.yaml.example @@ -0,0 +1,610 @@ +## All options defined here are available to all instances. +# +init_config: + + ## @param service - string - optional + ## Attach the tag `service:` to every metric, event, and service check emitted by this integration. + ## + ## Additionally, this sets the default `service` for every log source. + # + # service: + +## Every instance is scheduled independently of the others. +# +instances: + + ## @param openmetrics_endpoint - string - required + ## Endpoint exposing the Amazon VPC CNI Prometheus metrics. + # + - openmetrics_endpoint: http://localhost:61678/metrics + + ## @param raw_metric_prefix - string - optional + ## A prefix that is removed from all exposed metric names, if present. + ## All configuration options will use the prefix-less name. + # + # raw_metric_prefix: _ + + ## @param extra_metrics - (list of string or mapping) - optional + ## This list defines metrics to collect from the `openmetrics_endpoint`, in addition to + ## what the check collects by default. If the check already collects a metric, then + ## metric definitions here take precedence. Metrics may be defined in 3 ways: + ## + ## 1. If the item is a string, then it represents the exposed metric name, and + ## the sent metric name will be identical. For example: + ## ``` + ## extra_metrics: + ## - + ## - + ## ``` + ## 2. If the item is a mapping, then the keys represent the exposed metric names. + ## + ## 1. If a value is a string, then it represents the sent metric name. For example: + ## ``` + ## extra_metrics: + ## - : + ## - : + ## ``` + ## 2. If a value is a mapping, then it must have a `name` and/or `type` key. + ## The `name` represents the sent metric name, and the `type` represents how + ## the metric should be handled, overriding any type information the endpoint + ## may provide. For example: + ## ``` + ## extra_metrics: + ## - : + ## name: + ## type: + ## - : + ## name: + ## type: + ## ``` + ## The supported native types are `gauge`, `counter`, `histogram`, and `summary`. + ## + ## Note: To collect counter metrics with names ending in `_total`, specify the metric name without the `_total` + ## suffix. For example, to collect the counter metric `promhttp_metric_handler_requests_total`, specify + ## `promhttp_metric_handler_requests`. This submits to Datadog the metric name appended with `.count`. + ## For more information, see: + ## https://github.com/OpenObservability/OpenMetrics/blob/main/specification/OpenMetrics.md#suffixes + ## + ## Regular expressions may be used to match the exposed metric names, for example: + ## ``` + ## extra_metrics: + ## - ^network_(ingress|egress)_.+ + ## - .+: + ## type: gauge + ## ``` + # + # extra_metrics: + # - + # - : + # - : + # name: + # type: + + ## @param exclude_metrics - list of strings - optional + ## A list of metrics to exclude, with each entry being either + ## the exact metric name or a regular expression. + ## + ## In order to exclude all metrics but the ones matching a specific filter, + ## you can use a negative lookahead regex like: + ## - ^(?!foo).*$ + # + # exclude_metrics: [] + + ## @param exclude_metrics_by_labels - mapping - optional + ## A mapping of labels to exclude metrics with matching label name and their corresponding metric values. To match + ## all values of a label, set it to `true`. + ## + ## Note: Label filtering happens before `rename_labels`. + ## + ## For example, the following configuration instructs the check to exclude all metrics with + ## a label `worker` or a label `pid` with the value of either `23` or `42`. + ## + ## exclude_metrics_by_labels: + ## worker: true + ## pid: + ## - '23' + ## - '42' + # + # exclude_metrics_by_labels: {} + + ## @param exclude_labels - list of strings - optional + ## A list of labels to exclude, useful for high cardinality values like timestamps or UUIDs. + ## May be used in conjunction with `include_labels`. + ## Labels defined in `exclude_labels` will take precedence in case of overlap. + ## + ## Note: Label filtering happens before `rename_labels`. + # + # exclude_labels: [] + + ## @param include_labels - list of strings - optional + ## A list of labels to include. May be used in conjunction with `exclude_labels`. + ## Labels defined in `exclude_labels` will take precedence in case of overlap. + ## + ## Note: Label filtering happens before `rename_labels`. + # + # include_labels: [] + + ## @param rename_labels - mapping - optional + ## A mapping of label names to their new names. + # + # rename_labels: + # : + # : + + ## @param enable_health_service_check - boolean - optional - default: true + ## Whether or not to send a service check named `.openmetrics.health` which reports + ## the health of the `openmetrics_endpoint`. + # + # enable_health_service_check: true + + ## @param ignore_connection_errors - boolean - optional - default: false + ## Whether or not to ignore connection errors when scraping `openmetrics_endpoint`. + # + # ignore_connection_errors: false + + ## @param hostname_label - string - optional + ## Override the hostname for every metric submission with the value of one of its labels. + # + # hostname_label: + + ## @param hostname_format - string - optional + ## When `hostname_label` is set, this instructs the check how to format the values. The string + ## `` is replaced by the value of the label defined by `hostname_label`. + # + # hostname_format: + + ## @param collect_histogram_buckets - boolean - optional - default: true + ## Whether or not to send histogram buckets. + # + # collect_histogram_buckets: true + + ## @param non_cumulative_histogram_buckets - boolean - optional - default: false + ## Whether or not histogram buckets are non-cumulative and to come with a `lower_bound` tag. + # + # non_cumulative_histogram_buckets: false + + ## @param histogram_buckets_as_distributions - boolean - optional - default: false + ## Whether or not to send histogram buckets as Datadog distribution metrics. This implicitly + ## enables the `collect_histogram_buckets` and `non_cumulative_histogram_buckets` options. + ## + ## Learn more about distribution metrics: + ## https://docs.datadoghq.com/developers/metrics/types/?tab=distribution#metric-types + # + # histogram_buckets_as_distributions: false + + ## @param collect_counters_with_distributions - boolean - optional - default: false + ## Whether or not to also collect the observation counter metrics ending in `.sum` and `.count` + ## when sending histogram buckets as Datadog distribution metrics. This implicitly enables the + ## `histogram_buckets_as_distributions` option. + # + # collect_counters_with_distributions: false + + ## @param use_process_start_time - boolean - optional - default: false + ## Whether to enable a heuristic for reporting counter values on the first scrape. When true, + ## the first time an endpoint is scraped, check `process_start_time_seconds` to decide whether zero + ## initial value can be assumed for counters. This requires keeping metrics in memory until the entire + ## response is received. + # + # use_process_start_time: false + + ## @param share_labels - mapping - optional + ## This mapping allows for the sharing of labels across multiple metrics. The keys represent the + ## exposed metrics from which to share labels, and the values are mappings that configure the + ## sharing behavior. Each mapping must have at least one of the following keys: + ## + ## - labels - This is a list of labels to share. All labels are shared if this is not set. + ## - match - This is a list of labels to match on other metrics as a condition for sharing. + ## - values - This is a list of allowed values as a condition for sharing. + ## + ## To unconditionally share all labels of a metric, set it to `true`. + ## + ## For example, the following configuration instructs the check to apply all labels from `metric_a` + ## to all other metrics, the `node` label from `metric_b` to only those metrics that have a `pod` + ## label value that matches the `pod` label value of `metric_b`, and all labels from `metric_c` + ## to all other metrics if their value is equal to `23` or `42`. + # + # share_labels: + # metric_a: true + # metric_b: + # labels: + # - node + # match: + # - pod + # metric_c: + # values: + # - 23 + # - 42 + + ## @param cache_shared_labels - boolean - optional - default: true + ## When `share_labels` is set, it instructs the check to cache labels collected from the first payload + ## for improved performance. + ## + ## Set this to `false` to compute label sharing for every payload at the risk of potentially increased memory usage. + # + # cache_shared_labels: true + + ## @param raw_line_filters - list of strings - optional + ## A list of regular expressions used to exclude lines read from the `openmetrics_endpoint` + ## from being parsed. + # + # raw_line_filters: [] + + ## @param cache_metric_wildcards - boolean - optional - default: true + ## Whether or not to cache data from metrics that are defined by regular expressions rather + ## than the full metric name. + # + # cache_metric_wildcards: true + + ## @param telemetry - boolean - optional - default: false + ## Whether or not to submit metrics prefixed by `.telemetry.` for debugging purposes. + # + # telemetry: false + + ## @param ignore_tags - list of strings - optional + ## A list of regular expressions used to ignore tags added by Autodiscovery and entries in the `tags` option. + # + # ignore_tags: + # - + # - + # - + + ## @param proxy - mapping - optional + ## This overrides the `proxy` setting in `init_config`. + ## + ## Set HTTP or HTTPS proxies for this instance. Use the `no_proxy` list + ## to specify hosts that must bypass proxies. + ## + ## The SOCKS protocol is also supported, for example: + ## + ## socks5://user:pass@host:port + ## + ## Using the scheme `socks5` causes the DNS resolution to happen on the + ## client, rather than on the proxy server. This is in line with `curl`, + ## which uses the scheme to decide whether to do the DNS resolution on + ## the client or proxy. If you want to resolve the domains on the proxy + ## server, use `socks5h` as the scheme. + # + # proxy: + # http: http://: + # https: https://: + # no_proxy: + # - + # - + + ## @param skip_proxy - boolean - optional - default: false + ## This overrides the `skip_proxy` setting in `init_config`. + ## + ## If set to `true`, this makes the check bypass any proxy + ## settings enabled and attempt to reach services directly. + # + # skip_proxy: false + + ## @param auth_type - string - optional - default: basic + ## The type of authentication to use. The available types (and related options) are: + ## ``` + ## - basic + ## |__ username + ## |__ password + ## |__ use_legacy_auth_encoding + ## - digest + ## |__ username + ## |__ password + ## - ntlm + ## |__ ntlm_domain + ## |__ password + ## - kerberos + ## |__ kerberos_auth + ## |__ kerberos_cache + ## |__ kerberos_delegate + ## |__ kerberos_force_initiate + ## |__ kerberos_hostname + ## |__ kerberos_keytab + ## |__ kerberos_principal + ## - aws + ## |__ aws_region + ## |__ aws_host + ## |__ aws_service + ## ``` + ## The `aws` auth type relies on boto3 to automatically gather AWS credentials, for example: from `.aws/credentials`. + ## Details: https://boto3.amazonaws.com/v1/documentation/api/latest/guide/configuration.html#configuring-credentials + # + # auth_type: basic + + ## @param use_legacy_auth_encoding - boolean - optional - default: true + ## When `auth_type` is set to `basic`, this determines whether to encode as `latin1` rather than `utf-8`. + # + # use_legacy_auth_encoding: true + + ## @param username - string - optional + ## The username to use if services are behind basic or digest auth. + # + # username: + + ## @param password - string - optional + ## The password to use if services are behind basic or NTLM auth. + # + # password: + + ## @param ntlm_domain - string - optional + ## If your services use NTLM authentication, specify + ## the domain used in the check. For NTLM Auth, append + ## the username to domain, not as the `username` parameter. + # + # ntlm_domain: \ + + ## @param kerberos_auth - string - optional - default: disabled + ## If your services use Kerberos authentication, you can specify the Kerberos + ## strategy to use between: + ## + ## - required + ## - optional + ## - disabled + ## + ## See https://github.com/requests/requests-kerberos#mutual-authentication + # + # kerberos_auth: disabled + + ## @param kerberos_cache - string - optional + ## Sets the KRB5CCNAME environment variable. + ## It should point to a credential cache with a valid TGT. + # + # kerberos_cache: + + ## @param kerberos_delegate - boolean - optional - default: false + ## Set to `true` to enable Kerberos delegation of credentials to a server that requests delegation. + ## + ## See https://github.com/requests/requests-kerberos#delegation + # + # kerberos_delegate: false + + ## @param kerberos_force_initiate - boolean - optional - default: false + ## Set to `true` to preemptively initiate the Kerberos GSS exchange and + ## present a Kerberos ticket on the initial request (and all subsequent). + ## + ## See https://github.com/requests/requests-kerberos#preemptive-authentication + # + # kerberos_force_initiate: false + + ## @param kerberos_hostname - string - optional + ## Override the hostname used for the Kerberos GSS exchange if its DNS name doesn't + ## match its Kerberos hostname, for example: behind a content switch or load balancer. + ## + ## See https://github.com/requests/requests-kerberos#hostname-override + # + # kerberos_hostname: + + ## @param kerberos_principal - string - optional + ## Set an explicit principal, to force Kerberos to look for a + ## matching credential cache for the named user. + ## + ## See https://github.com/requests/requests-kerberos#explicit-principal + # + # kerberos_principal: + + ## @param kerberos_keytab - string - optional + ## Set the path to your Kerberos key tab file. + # + # kerberos_keytab: + + ## @param auth_token - mapping - optional + ## This allows for the use of authentication information from dynamic sources. + ## Both a reader and writer must be configured. + ## + ## The available readers are: + ## + ## - type: file + ## path (required): The absolute path for the file to read from. + ## pattern: A regular expression pattern with a single capture group used to find the + ## token rather than using the entire file, for example: Your secret is (.+) + ## - type: oauth + ## url (required): The token endpoint. + ## client_id (required): The client identifier. + ## client_secret (required): The client secret. + ## basic_auth: Whether the provider expects credentials to be transmitted in + ## an HTTP Basic Auth header. The default is: false + ## options: Mapping of additional options to pass to the provider, such as the audience + ## or the scope. For example: + ## options: + ## audience: https://example.com + ## scope: read:example + ## + ## The available writers are: + ## + ## - type: header + ## name (required): The name of the field, for example: Authorization + ## value: The template value, for example `Bearer `. The default is: + ## placeholder: The substring in `value` to replace with the token, defaults to: + # + # auth_token: + # reader: + # type: + # : + # : + # writer: + # type: + # : + # : + + ## @param aws_region - string - optional + ## If your services require AWS Signature Version 4 signing, set the region. + ## + ## See https://docs.aws.amazon.com/general/latest/gr/signature-version-4.html + # + # aws_region: + + ## @param aws_host - string - optional + ## If your services require AWS Signature Version 4 signing, set the host. + ## This only needs the hostname and does not require the protocol (HTTP, HTTPS, and more). + ## For example, if connecting to https://us-east-1.amazonaws.com/, set `aws_host` to `us-east-1.amazonaws.com`. + ## + ## Note: This setting is not necessary for official integrations. + ## + ## See https://docs.aws.amazon.com/general/latest/gr/signature-version-4.html + # + # aws_host: + + ## @param aws_service - string - optional + ## If your services require AWS Signature Version 4 signing, set the service code. For a list + ## of available service codes, see https://docs.aws.amazon.com/general/latest/gr/rande.html + ## + ## Note: This setting is not necessary for official integrations. + ## + ## See https://docs.aws.amazon.com/general/latest/gr/signature-version-4.html + # + # aws_service: + + ## @param tls_verify - boolean - optional - default: true + ## Instructs the check to validate the TLS certificate of services. + # + # tls_verify: true + + ## @param tls_use_host_header - boolean - optional - default: false + ## If a `Host` header is set, this enables its use for SNI (matching against the TLS certificate CN or SAN). + # + # tls_use_host_header: false + + ## @param tls_ignore_warning - boolean - optional - default: false + ## If `tls_verify` is disabled, security warnings are logged by the check. + ## Disable those by setting `tls_ignore_warning` to true. + # + # tls_ignore_warning: false + + ## @param tls_cert - string - optional + ## The path to a single file in PEM format containing a certificate as well as any + ## number of CA certificates needed to establish the certificate's authenticity for + ## use when connecting to services. It may also contain an unencrypted private key to use. + # + # tls_cert: + + ## @param tls_private_key - string - optional + ## The unencrypted private key to use for `tls_cert` when connecting to services. This is + ## required if `tls_cert` is set and it does not already contain a private key. + # + # tls_private_key: + + ## @param tls_ca_cert - string - optional + ## The path to a file of concatenated CA certificates in PEM format or a directory + ## containing several CA certificates in PEM format. If a directory, the directory + ## must have been processed using the `openssl rehash` command. See: + ## https://www.openssl.org/docs/man3.2/man1/c_rehash.html + # + # tls_ca_cert: + + ## @param tls_protocols_allowed - list of strings - optional + ## The expected versions of TLS/SSL when fetching intermediate certificates. + ## Only `SSLv3`, `TLSv1.2`, `TLSv1.3` are allowed by default. The possible values are: + ## SSLv3 + ## TLSv1 + ## TLSv1.1 + ## TLSv1.2 + ## TLSv1.3 + # + # tls_protocols_allowed: + # - SSLv3 + # - TLSv1.2 + # - TLSv1.3 + + ## @param tls_ciphers - list of strings - optional + ## The list of ciphers suites to use when connecting to an endpoint. If not specified, + ## `ALL` ciphers are used. For list of ciphers see: + ## https://www.openssl.org/docs/man1.0.2/man1/ciphers.html + # + # tls_ciphers: + # - TLS_AES_256_GCM_SHA384 + # - TLS_CHACHA20_POLY1305_SHA256 + # - TLS_AES_128_GCM_SHA256 + + ## @param headers - mapping - optional + ## The headers parameter allows you to send specific headers with every request. + ## You can use it for explicitly specifying the host header or adding headers for + ## authorization purposes. + ## + ## This overrides any default headers. + # + # headers: + # Host: + # X-Auth-Token: + + ## @param extra_headers - mapping - optional + ## Additional headers to send with every request. + # + # extra_headers: + # Host: + # X-Auth-Token: + + ## @param timeout - number - optional - default: 10 + ## The timeout for accessing services. + ## + ## This overrides the `timeout` setting in `init_config`. + # + # timeout: 10 + + ## @param connect_timeout - number - optional + ## The connect timeout for accessing services. Defaults to `timeout`. + # + # connect_timeout: + + ## @param read_timeout - number - optional + ## The read timeout for accessing services. Defaults to `timeout`. + # + # read_timeout: + + ## @param request_size - number - optional - default: 16 + ## The number of kibibytes (KiB) to read from streaming HTTP responses at a time. + # + # request_size: 16 + + ## @param log_requests - boolean - optional - default: false + ## Whether or not to debug log the HTTP(S) requests made, including the method and URL. + # + # log_requests: false + + ## @param persist_connections - boolean - optional - default: false + ## Whether or not to persist cookies and use connection pooling for improved performance. + # + # persist_connections: false + + ## @param allow_redirects - boolean - optional - default: true + ## Whether or not to allow URL redirection. + # + # allow_redirects: true + + ## @param tags - list of strings - optional + ## A list of tags to attach to every metric and service check emitted by this instance. + ## + ## Learn more about tagging at https://docs.datadoghq.com/tagging + # + # tags: + # - : + # - : + + ## @param service - string - optional + ## Attach the tag `service:` to every metric, event, and service check emitted by this integration. + ## + ## Overrides any `service` defined in the `init_config` section. + # + # service: + + ## @param min_collection_interval - number - optional - default: 15 + ## This changes the collection interval of the check. For more information, see: + ## https://docs.datadoghq.com/developers/write_agent_check/#collection-interval + # + # min_collection_interval: 15 + + ## @param empty_default_hostname - boolean - optional - default: false + ## This forces the check to send metrics with no hostname. + ## + ## This is useful for cluster-level checks. + # + # empty_default_hostname: false + + ## @param metric_patterns - mapping - optional + ## A mapping of metrics to include or exclude, with each entry being a regular expression. + ## + ## Metrics defined in `exclude` will take precedence in case of overlap. + # + # metric_patterns: + # include: + # - + # exclude: + # - diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/metrics.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/metrics.py new file mode 100644 index 000000000..704c17568 --- /dev/null +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/metrics.py @@ -0,0 +1,33 @@ +METRIC_MAP = { + ## IPAMD + 'awscni_ipamd_error_count': 'ipamd_error_count', + 'awscni_ipamd_action_inprogress': 'ipamd_action_inprogress', + 'awscni_reconcile_count': 'reconcile_count', + 'awscni_add_ip_req_count': 'add_ip_req_count', + 'awscni_del_ip_req_count': 'del_ip_req_count', + 'awscni_pod_eni_error_count': 'pod_eni_error_count', + ## AWS API + 'awscni_aws_api_latency_ms': 'aws_api_latency_ms', + 'awscni_aws_api_error_count': 'aws_api_error_count', + 'awscni_aws_utils_error_count': 'aws_utils_error_count', + 'awscni_ec2api_req_count': 'ec2api_req_count', + 'awscni_ec2api_error_count': 'ec2api_error_count', + ## ENI / IP allocation + 'awscni_eni_max': 'eni_max', + 'awscni_ip_max': 'ip_max', + 'awscni_eni_allocated': 'eni_allocated', + 'awscni_total_ip_addresses': 'total_ip_addresses', + 'awscni_assigned_ip_addresses': 'assigned_ip_addresses', + 'awscni_force_removed_enis': 'force_removed_enis', + 'awscni_force_removed_ips': 'force_removed_ips', + 'awscni_total_ipv4_prefixes': 'total_ipv4_prefixes', + 'awscni_assigned_ip_per_cidr': 'assigned_ip_per_cidr', + 'awscni_no_available_ip_addresses': 'no_available_ip_addresses', + 'awscni_assigned_ip_per_eni': 'assigned_ip_per_eni', + ## Startup duration + 'awscni_ipamd_startup_duration_seconds': 'ipamd_startup_duration_seconds', + 'awscni_ipamd_node_initialization_duration_seconds': 'ipamd_node_initialization_duration_seconds', + ## Connmark + 'awscni_connmark_backend': 'connmark_backend', + 'awscni_connmark_reconcile_total': 'connmark_reconcile_total', +} diff --git a/amazon_vpc_cni/hatch.toml b/amazon_vpc_cni/hatch.toml new file mode 100644 index 000000000..fee245500 --- /dev/null +++ b/amazon_vpc_cni/hatch.toml @@ -0,0 +1,4 @@ +[env.collectors.datadog-checks] + +[[envs.default.matrix]] +python = ["3.13"] diff --git a/amazon_vpc_cni/manifest.json b/amazon_vpc_cni/manifest.json new file mode 100644 index 000000000..6bb2437d6 --- /dev/null +++ b/amazon_vpc_cni/manifest.json @@ -0,0 +1,58 @@ +{ + "manifest_version": "2.0.0", + "app_uuid": "b566f4c9-aea1-4fcd-a19b-7a34f8d049dd", + "app_id": "amazon-vpc-cni", + "owner": "agent-integrations", + "display_on_public_website": true, + "tile": { + "overview": "README.md#Overview", + "configuration": "README.md#Setup", + "support": "README.md#Support", + "changelog": "CHANGELOG.md", + "description": "Monitor the Amazon VPC CNI plugin for Kubernetes, tracking IP address allocation, ENI usage, and AWS API activity.", + "title": "Amazon VPC CNI", + "media": [], + "classifier_tags": [ + "Category::Metrics", + "Offering::Integration", + "Supported OS::Linux", + "Supported OS::macOS", + "Supported OS::Windows" + ], + "resources": [ + { + "resource_type": "other", + "url": "https://github.com/aws/amazon-vpc-cni-k8s" + } + ] + }, + "assets": { + "integration": { + "auto_install": true, + "source_type_id": 89444319, + "source_type_name": "Amazon VPC CNI", + "configuration": { + "spec": "assets/configuration/spec.yaml" + }, + "events": { + "creates_events": false + }, + "metrics": { + "prefix": "amazon_vpc_cni.", + "check": "amazon_vpc_cni.assigned_ip_addresses", + "metadata_path": "metadata.csv" + }, + "service_checks": { + "metadata_path": "assets/service_checks.json" + } + }, + "dashboards": {}, + "monitors": {} + }, + "author": { + "support_email": "willianccs@gmail.com", + "name": "Willian Cesar Cincerre da Silva", + "homepage": "https://github.com/DataDog/integrations-extras", + "sales_email": "willian@appoena.io" + } +} diff --git a/amazon_vpc_cni/metadata.csv b/amazon_vpc_cni/metadata.csv new file mode 100644 index 000000000..d9428d0cc --- /dev/null +++ b/amazon_vpc_cni/metadata.csv @@ -0,0 +1,33 @@ +metric_name,metric_type,interval,unit_name,per_unit_name,description,orientation,integration,short_name,curated_metric +amazon_vpc_cni.ipamd_action_inprogress,gauge,,,,The number of ipamd actions in progress.,0,amazon_vpc_cni,ipamd actions in progress, +amazon_vpc_cni.eni_max,gauge,,,,The maximum number of ENIs that can be attached to the instance (accounting for unmanaged ENIs).,0,amazon_vpc_cni,eni max, +amazon_vpc_cni.ip_max,gauge,,,,The maximum number of IP addresses that can be allocated to the instance.,0,amazon_vpc_cni,ip max, +amazon_vpc_cni.eni_allocated,gauge,,,,The number of ENIs allocated.,0,amazon_vpc_cni,eni allocated, +amazon_vpc_cni.total_ip_addresses,gauge,,,,The total number of IP addresses.,0,amazon_vpc_cni,total ip addresses, +amazon_vpc_cni.assigned_ip_addresses,gauge,,,,The number of IP addresses assigned to pods.,0,amazon_vpc_cni,assigned ip addresses, +amazon_vpc_cni.total_ipv4_prefixes,gauge,,,,The total number of IPv4 prefixes.,0,amazon_vpc_cni,total ipv4 prefixes, +amazon_vpc_cni.assigned_ip_per_cidr,gauge,,,,The total number of IP addresses assigned per CIDR.,0,amazon_vpc_cni,assigned ip per cidr, +amazon_vpc_cni.assigned_ip_per_eni,gauge,,,,The number of allocated IPs partitioned by ENI.,0,amazon_vpc_cni,assigned ip per eni, +amazon_vpc_cni.connmark_backend,gauge,,,,The connmark backend in use; the active backend series is set to 1.,0,amazon_vpc_cni,connmark backend, +amazon_vpc_cni.ipamd_error_count.count,count,,,,The number of errors encountered in ipamd.,-1,amazon_vpc_cni,ipamd errors, +amazon_vpc_cni.reconcile_count.count,count,,,,The number of times ipamd reconciles on ENIs and IP/Prefix addresses.,0,amazon_vpc_cni,reconcile count, +amazon_vpc_cni.add_ip_req_count.count,count,,,,The number of add IP address requests.,0,amazon_vpc_cni,add ip requests, +amazon_vpc_cni.del_ip_req_count.count,count,,,,The number of delete IP address requests.,0,amazon_vpc_cni,delete ip requests, +amazon_vpc_cni.pod_eni_error_count.count,count,,,,The number of errors encountered for pod ENIs.,-1,amazon_vpc_cni,pod eni errors, +amazon_vpc_cni.aws_api_error_count.count,count,,,,The number of times AWS API returns an error.,-1,amazon_vpc_cni,aws api errors, +amazon_vpc_cni.aws_utils_error_count.count,count,,,,The number of errors not handled in the awsutils library.,-1,amazon_vpc_cni,aws utils errors, +amazon_vpc_cni.ec2api_req_count.count,count,,,,The number of requests made to EC2 APIs by the CNI.,0,amazon_vpc_cni,ec2 api requests, +amazon_vpc_cni.ec2api_error_count.count,count,,,,The number of failed EC2 API requests.,-1,amazon_vpc_cni,ec2 api errors, +amazon_vpc_cni.force_removed_enis.count,count,,,,The number of ENIs force removed while they had assigned pods.,0,amazon_vpc_cni,force removed enis, +amazon_vpc_cni.force_removed_ips.count,count,,,,The number of IPs force removed while they had assigned pods.,0,amazon_vpc_cni,force removed ips, +amazon_vpc_cni.no_available_ip_addresses.count,count,,,,The number of pod IP assignments that fail due to no available IP addresses.,-1,amazon_vpc_cni,no available ip addresses, +amazon_vpc_cni.connmark_reconcile_total.count,count,,,,The number of connmark reconciliations by result.,0,amazon_vpc_cni,connmark reconciles, +amazon_vpc_cni.aws_api_latency_ms.count,count,,,,Number of AWS API latency observations.,0,amazon_vpc_cni,aws api latency count, +amazon_vpc_cni.aws_api_latency_ms.sum,count,,,,Sum of AWS API call latencies.,0,amazon_vpc_cni,aws api latency sum, +amazon_vpc_cni.aws_api_latency_ms.quantile,gauge,,,,AWS API call latency quantiles.,0,amazon_vpc_cni,aws api latency quantile, +amazon_vpc_cni.ipamd_startup_duration_seconds.count,count,,second,,Number of IPAMD startup duration observations.,0,amazon_vpc_cni,ipamd startup duration count, +amazon_vpc_cni.ipamd_startup_duration_seconds.sum,count,,second,,Sum of IPAMD startup durations.,0,amazon_vpc_cni,ipamd startup duration sum, +amazon_vpc_cni.ipamd_startup_duration_seconds.bucket,count,,second,,Histogram buckets of IPAMD startup durations.,0,amazon_vpc_cni,ipamd startup duration bucket, +amazon_vpc_cni.ipamd_node_initialization_duration_seconds.count,count,,second,,Number of node initialization duration observations.,0,amazon_vpc_cni,node init duration count, +amazon_vpc_cni.ipamd_node_initialization_duration_seconds.sum,count,,second,,Sum of node initialization durations.,0,amazon_vpc_cni,node init duration sum, +amazon_vpc_cni.ipamd_node_initialization_duration_seconds.bucket,count,,second,,Histogram buckets of node initialization durations.,0,amazon_vpc_cni,node init duration bucket, diff --git a/amazon_vpc_cni/pyproject.toml b/amazon_vpc_cni/pyproject.toml new file mode 100644 index 000000000..2901b40e5 --- /dev/null +++ b/amazon_vpc_cni/pyproject.toml @@ -0,0 +1,60 @@ +[build-system] +requires = [ + "hatchling>=0.13.0", +] +build-backend = "hatchling.build" + +[project] +name = "datadog-amazon-vpc-cni" +description = "The Amazon VPC CNI check" +readme = "README.md" +license = "BSD-3-Clause" +requires-python = ">=3.9" +keywords = [ + "datadog", + "datadog agent", + "datadog check", + "amazon_vpc_cni", +] +authors = [ + { name = "Willian Cesar Cincerre da Silva", email = "willianccs@gmail.com" }, +] +classifiers = [ + "Development Status :: 5 - Production/Stable", + "Intended Audience :: Developers", + "Intended Audience :: System Administrators", + "License :: OSI Approved :: BSD License", + "Private :: Do Not Upload", + "Programming Language :: Python :: 3.13", + "Topic :: System :: Monitoring", +] +dependencies = [ + "datadog-checks-base>=37.42.0", +] +dynamic = [ + "version", +] + +[project.optional-dependencies] +deps = [] + +[project.urls] +Source = "https://github.com/DataDog/integrations-extras" + +[tool.hatch.version] +path = "datadog_checks/amazon_vpc_cni/__about__.py" + +[tool.hatch.build.targets.sdist] +include = [ + "/datadog_checks", + "/tests", + "/manifest.json", +] + +[tool.hatch.build.targets.wheel] +include = [ + "/datadog_checks/amazon_vpc_cni", +] +dev-mode-dirs = [ + ".", +] diff --git a/amazon_vpc_cni/tests/__init__.py b/amazon_vpc_cni/tests/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/amazon_vpc_cni/tests/common.py b/amazon_vpc_cni/tests/common.py new file mode 100644 index 000000000..b88a0de7e --- /dev/null +++ b/amazon_vpc_cni/tests/common.py @@ -0,0 +1,46 @@ +import os + +HERE = os.path.dirname(os.path.abspath(__file__)) + +EXPECTED_PROMETHEUS_METRICS = [ + 'amazon_vpc_cni.ipamd_action_inprogress', + 'amazon_vpc_cni.eni_max', + 'amazon_vpc_cni.ip_max', + 'amazon_vpc_cni.eni_allocated', + 'amazon_vpc_cni.total_ip_addresses', + 'amazon_vpc_cni.assigned_ip_addresses', + 'amazon_vpc_cni.total_ipv4_prefixes', + 'amazon_vpc_cni.assigned_ip_per_cidr', + 'amazon_vpc_cni.assigned_ip_per_eni', + 'amazon_vpc_cni.connmark_backend', + 'amazon_vpc_cni.ipamd_error_count.count', + 'amazon_vpc_cni.reconcile_count.count', + 'amazon_vpc_cni.add_ip_req_count.count', + 'amazon_vpc_cni.del_ip_req_count.count', + 'amazon_vpc_cni.pod_eni_error_count.count', + 'amazon_vpc_cni.aws_api_error_count.count', + 'amazon_vpc_cni.aws_utils_error_count.count', + 'amazon_vpc_cni.ec2api_req_count.count', + 'amazon_vpc_cni.ec2api_error_count.count', + 'amazon_vpc_cni.force_removed_enis.count', + 'amazon_vpc_cni.force_removed_ips.count', + 'amazon_vpc_cni.no_available_ip_addresses.count', + 'amazon_vpc_cni.connmark_reconcile_total.count', + 'amazon_vpc_cni.aws_api_latency_ms.count', + 'amazon_vpc_cni.aws_api_latency_ms.sum', + 'amazon_vpc_cni.aws_api_latency_ms.quantile', + 'amazon_vpc_cni.ipamd_startup_duration_seconds.count', + 'amazon_vpc_cni.ipamd_startup_duration_seconds.sum', + 'amazon_vpc_cni.ipamd_startup_duration_seconds.bucket', + 'amazon_vpc_cni.ipamd_node_initialization_duration_seconds.count', + 'amazon_vpc_cni.ipamd_node_initialization_duration_seconds.sum', + 'amazon_vpc_cni.ipamd_node_initialization_duration_seconds.bucket', +] + +MOCKED_INSTANCE = {'openmetrics_endpoint': 'http://localhost:61678/metrics'} + +BAD_HOSTNAME_INSTANCE = {'openmetrics_endpoint': 'http://invalid-hostname:61678/metrics'} + + +def get_fixture_path(filename): + return os.path.join(HERE, 'fixtures', filename) diff --git a/amazon_vpc_cni/tests/conftest.py b/amazon_vpc_cni/tests/conftest.py new file mode 100644 index 000000000..347a2029c --- /dev/null +++ b/amazon_vpc_cni/tests/conftest.py @@ -0,0 +1,57 @@ +import os +from unittest import mock + +import pytest + +from datadog_checks.amazon_vpc_cni import AmazonVpcCniCheck +from datadog_checks.dev import docker_run, get_docker_hostname, get_here +from datadog_checks.dev.conditions import CheckEndpoints + +HERE = get_here() +INSTANCE_URL = f"http://{get_docker_hostname()}:61678/metrics" + + +@pytest.fixture(scope='session') +def dd_environment(): + compose_file = os.path.join(HERE, 'docker', 'docker-compose.yaml') + conditions = [ + CheckEndpoints(INSTANCE_URL, attempts=120, wait=2), + ] + with docker_run(compose_file, conditions=conditions): + instances = {'instances': [{'openmetrics_endpoint': INSTANCE_URL}]} + yield instances + + +@pytest.fixture +def instance(): + return { + 'openmetrics_endpoint': INSTANCE_URL, + } + + +@pytest.fixture +def check(instance): + return AmazonVpcCniCheck('amazon_vpc_cni', {}, [instance]) + + +@pytest.fixture() +def mock_prometheus_metrics(): + fixture_file = os.path.join(HERE, 'fixtures', 'metrics.txt') + try: + with open(fixture_file, 'r') as f: + content = f.read() + except FileNotFoundError: + pytest.fail(f"Metrics fixture file not found: {fixture_file}") + except Exception as e: + pytest.fail(f"Error reading metrics fixture file: {fixture_file}. Error: {e}") + + with mock.patch( + 'requests.Session.get', + return_value=mock.MagicMock( + status_code=200, + iter_lines=lambda **kwargs: content.split('\n'), + headers={'Content-Type': 'text/plain'}, + close=lambda: None, + ), + ): + yield diff --git a/amazon_vpc_cni/tests/docker/docker-compose.yaml b/amazon_vpc_cni/tests/docker/docker-compose.yaml new file mode 100644 index 000000000..b053e7d17 --- /dev/null +++ b/amazon_vpc_cni/tests/docker/docker-compose.yaml @@ -0,0 +1,8 @@ +services: + amazon-vpc-cni-mock: + image: python:3.13-alpine + command: python -m http.server 61678 --directory /metrics + ports: + - "61678:61678" + volumes: + - ../fixtures:/metrics diff --git a/amazon_vpc_cni/tests/fixtures/metrics.txt b/amazon_vpc_cni/tests/fixtures/metrics.txt new file mode 100644 index 000000000..051b8018b --- /dev/null +++ b/amazon_vpc_cni/tests/fixtures/metrics.txt @@ -0,0 +1,97 @@ +# HELP awscni_ipamd_error_count The number of errors encountered in ipamd +# TYPE awscni_ipamd_error_count counter +awscni_ipamd_error_count{fn="increaseDatastorePool"} 2 +awscni_ipamd_error_count{fn="decreaseDatastorePool"} 1 +# HELP awscni_ipamd_action_inprogress The number of ipamd actions in progress +# TYPE awscni_ipamd_action_inprogress gauge +awscni_ipamd_action_inprogress{fn="nodeInit"} 1 +awscni_ipamd_action_inprogress{fn="increaseDatastorePool"} 0 +# HELP awscni_eni_max The maximum number of ENIs that can be attached to the instance, accounting for unmanaged ENIs +# TYPE awscni_eni_max gauge +awscni_eni_max 4 +# HELP awscni_ip_max The maximum number of IP addresses that can be allocated to the instance +# TYPE awscni_ip_max gauge +awscni_ip_max 30 +# HELP awscni_reconcile_count The number of times ipamd reconciles on ENIs and IP/Prefix addresses +# TYPE awscni_reconcile_count counter +awscni_reconcile_count{fn="eniReconcileAdd"} 120 +awscni_reconcile_count{fn="eniReconcileDel"} 3 +# HELP awscni_add_ip_req_count The number of add IP address requests +# TYPE awscni_add_ip_req_count counter +awscni_add_ip_req_count 450 +# HELP awscni_del_ip_req_count The number of delete IP address requests +# TYPE awscni_del_ip_req_count counter +awscni_del_ip_req_count{reason="pod_deleted"} 200 +# HELP awscni_pod_eni_error_count The number of errors encountered for pod ENIs +# TYPE awscni_pod_eni_error_count counter +awscni_pod_eni_error_count{fn="podENIHandler"} 4 +# HELP awscni_aws_api_latency_ms AWS API call latency in ms +# TYPE awscni_aws_api_latency_ms summary +awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.5"} 15 +awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.9"} 45 +awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.99"} 80 +awscni_aws_api_latency_ms_sum{api="DescribeNetworkInterfaces",error="false",status="200"} 1200 +awscni_aws_api_latency_ms_count{api="DescribeNetworkInterfaces",error="false",status="200"} 60 +# HELP awscni_aws_api_error_count The number of times AWS API returns an error +# TYPE awscni_aws_api_error_count counter +awscni_aws_api_error_count{api="DescribeNetworkInterfaces",error="RequestLimitExceeded"} 3 +# HELP awscni_aws_utils_error_count The number of errors not handled in awsutils library +# TYPE awscni_aws_utils_error_count counter +awscni_aws_utils_error_count{fn="GetENILimit",error="NoENILimit"} 1 +# HELP awscni_ec2api_req_count The number of requests made to EC2 APIs by CNI +# TYPE awscni_ec2api_req_count counter +awscni_ec2api_req_count{fn="DescribeInstances"} 90 +# HELP awscni_ec2api_error_count The number of failed EC2 APIs requests +# TYPE awscni_ec2api_error_count counter +awscni_ec2api_error_count{fn="DescribeInstances"} 2 +# HELP awscni_eni_allocated The number of ENIs allocated +# TYPE awscni_eni_allocated gauge +awscni_eni_allocated 3 +# HELP awscni_total_ip_addresses The total number of IP addresses +# TYPE awscni_total_ip_addresses gauge +awscni_total_ip_addresses 24 +# HELP awscni_assigned_ip_addresses The number of IP addresses assigned to pods +# TYPE awscni_assigned_ip_addresses gauge +awscni_assigned_ip_addresses 18 +# HELP awscni_force_removed_enis The number of ENIs force removed while they had assigned pods +# TYPE awscni_force_removed_enis counter +awscni_force_removed_enis 0 +# HELP awscni_force_removed_ips The number of IPs force removed while they had assigned pods +# TYPE awscni_force_removed_ips counter +awscni_force_removed_ips 0 +# HELP awscni_total_ipv4_prefixes The total number of IPv4 prefixes +# TYPE awscni_total_ipv4_prefixes gauge +awscni_total_ipv4_prefixes 8 +# HELP awscni_assigned_ip_per_cidr The total number of IP addresses assigned per cidr +# TYPE awscni_assigned_ip_per_cidr gauge +awscni_assigned_ip_per_cidr{cidr="192.168.0.0/24"} 18 +# HELP awscni_no_available_ip_addresses The number of pod IP assignments that fail due to no available IP addresses +# TYPE awscni_no_available_ip_addresses counter +awscni_no_available_ip_addresses 5 +# HELP awscni_assigned_ip_per_eni The number of allocated ips partitioned by eni +# TYPE awscni_assigned_ip_per_eni gauge +awscni_assigned_ip_per_eni{eni="eni-0a1b2c3d"} 8 +awscni_assigned_ip_per_eni{eni="eni-0d4e5f6a"} 10 +# HELP awscni_ipamd_startup_duration_seconds The duration of IPAMD startup from process start to ready to serve CNI requests +# TYPE awscni_ipamd_startup_duration_seconds histogram +awscni_ipamd_startup_duration_seconds_bucket{success="true",with_api_server="true",failure_reason="",le="0.1"} 0 +awscni_ipamd_startup_duration_seconds_bucket{success="true",with_api_server="true",failure_reason="",le="0.5"} 0 +awscni_ipamd_startup_duration_seconds_bucket{success="true",with_api_server="true",failure_reason="",le="1"} 1 +awscni_ipamd_startup_duration_seconds_bucket{success="true",with_api_server="true",failure_reason="",le="+Inf"} 1 +awscni_ipamd_startup_duration_seconds_sum{success="true",with_api_server="true",failure_reason=""} 0.75 +awscni_ipamd_startup_duration_seconds_count{success="true",with_api_server="true",failure_reason=""} 1 +# HELP awscni_ipamd_node_initialization_duration_seconds The duration of node initialization during IPAMD startup +# TYPE awscni_ipamd_node_initialization_duration_seconds histogram +awscni_ipamd_node_initialization_duration_seconds_bucket{success="true",le="0.5"} 1 +awscni_ipamd_node_initialization_duration_seconds_bucket{success="true",le="1"} 1 +awscni_ipamd_node_initialization_duration_seconds_bucket{success="true",le="+Inf"} 1 +awscni_ipamd_node_initialization_duration_seconds_sum{success="true"} 0.4 +awscni_ipamd_node_initialization_duration_seconds_count{success="true"} 1 +# HELP awscni_connmark_backend The connmark backend in use; the active backend's series is set to 1 +# TYPE awscni_connmark_backend gauge +awscni_connmark_backend{backend="iptables"} 1 +awscni_connmark_backend{backend="ebpf"} 0 +# HELP awscni_connmark_reconcile_total The number of connmark Setup/Cleanup reconciliations by result +# TYPE awscni_connmark_reconcile_total counter +awscni_connmark_reconcile_total{result="success"} 40 +awscni_connmark_reconcile_total{result="error"} 1 diff --git a/amazon_vpc_cni/tests/test_e2e.py b/amazon_vpc_cni/tests/test_e2e.py new file mode 100644 index 000000000..c6f27f9a2 --- /dev/null +++ b/amazon_vpc_cni/tests/test_e2e.py @@ -0,0 +1,23 @@ +import pytest + +from datadog_checks.base.constants import ServiceCheck +from datadog_checks.dev.utils import get_metadata_metrics + +from .common import EXPECTED_PROMETHEUS_METRICS + + +@pytest.mark.e2e +def test_e2e_service_check_ok(dd_agent_check, aggregator, instance, mock_prometheus_metrics): + dd_agent_check(instance) + aggregator.assert_service_check('amazon_vpc_cni.openmetrics.health', ServiceCheck.OK) + + +@pytest.mark.e2e +def test_e2e_assert_metrics(dd_agent_check, aggregator, instance, mock_prometheus_metrics): + dd_agent_check(instance) + + for metric in EXPECTED_PROMETHEUS_METRICS: + aggregator.assert_metric(metric, at_least=0) + + aggregator.assert_all_metrics_covered() + aggregator.assert_metrics_using_metadata(get_metadata_metrics()) diff --git a/amazon_vpc_cni/tests/test_integration.py b/amazon_vpc_cni/tests/test_integration.py new file mode 100644 index 000000000..faa1a2372 --- /dev/null +++ b/amazon_vpc_cni/tests/test_integration.py @@ -0,0 +1,12 @@ +import pytest + + +@pytest.mark.integration +@pytest.mark.usefixtures('dd_environment') +def test_connect_ok(dd_run_check, aggregator, instance): + from datadog_checks.amazon_vpc_cni import AmazonVpcCniCheck + from datadog_checks.base.constants import ServiceCheck + + check = AmazonVpcCniCheck('amazon_vpc_cni', {}, [instance]) + dd_run_check(check) + aggregator.assert_service_check('amazon_vpc_cni.openmetrics.health', ServiceCheck.OK) diff --git a/amazon_vpc_cni/tests/test_unit.py b/amazon_vpc_cni/tests/test_unit.py new file mode 100644 index 000000000..456976045 --- /dev/null +++ b/amazon_vpc_cni/tests/test_unit.py @@ -0,0 +1,35 @@ +import pytest + +from datadog_checks.amazon_vpc_cni import AmazonVpcCniCheck +from datadog_checks.base.constants import ServiceCheck +from datadog_checks.dev.utils import get_metadata_metrics + +from .common import BAD_HOSTNAME_INSTANCE, EXPECTED_PROMETHEUS_METRICS + +pytestmark = [pytest.mark.unit] + + +def test_connect_exception(dd_run_check, aggregator, caplog): + with pytest.raises(Exception, match="Failed to resolve 'invalid-hostname'|Max retries exceeded"): + check = AmazonVpcCniCheck('amazon_vpc_cni', {}, [BAD_HOSTNAME_INSTANCE]) + dd_run_check(check) + + aggregator.assert_service_check('amazon_vpc_cni.openmetrics.health', ServiceCheck.CRITICAL) + + +def test_check_mock_amazon_vpc_cni_metrics(dd_run_check, aggregator, check, mock_prometheus_metrics): + dd_run_check(check) + for metric_name in EXPECTED_PROMETHEUS_METRICS: + aggregator.assert_metric(metric_name, at_least=0) + aggregator.assert_metrics_using_metadata(get_metadata_metrics()) + + aggregator.assert_service_check('amazon_vpc_cni.openmetrics.health', ServiceCheck.OK) + + +def test_empty_instance(dd_run_check): + with pytest.raises( + Exception, + match='\nopenmetrics_endpoint\n Field required', + ): + check = AmazonVpcCniCheck('amazon_vpc_cni', {}, [{}]) + dd_run_check(check) From 2862d175cdc7807a909786c3c3dad13b47e629ad Mon Sep 17 00:00:00 2001 From: Willian Cesar Cincerre Da Silva <3329-willian.cincerre@users.noreply.code.ifoodcorp.com.br> Date: Tue, 15 Sep 2026 14:08:17 -0300 Subject: [PATCH 2/9] fix(amazon-vpc-cni): serve mock metrics at /metrics for integration tests --- amazon_vpc_cni/tests/fixtures/metrics | 97 +++++++++++++++++++++++++++ 1 file changed, 97 insertions(+) create mode 100644 amazon_vpc_cni/tests/fixtures/metrics diff --git a/amazon_vpc_cni/tests/fixtures/metrics b/amazon_vpc_cni/tests/fixtures/metrics new file mode 100644 index 000000000..051b8018b --- /dev/null +++ b/amazon_vpc_cni/tests/fixtures/metrics @@ -0,0 +1,97 @@ +# HELP awscni_ipamd_error_count The number of errors encountered in ipamd +# TYPE awscni_ipamd_error_count counter +awscni_ipamd_error_count{fn="increaseDatastorePool"} 2 +awscni_ipamd_error_count{fn="decreaseDatastorePool"} 1 +# HELP awscni_ipamd_action_inprogress The number of ipamd actions in progress +# TYPE awscni_ipamd_action_inprogress gauge +awscni_ipamd_action_inprogress{fn="nodeInit"} 1 +awscni_ipamd_action_inprogress{fn="increaseDatastorePool"} 0 +# HELP awscni_eni_max The maximum number of ENIs that can be attached to the instance, accounting for unmanaged ENIs +# TYPE awscni_eni_max gauge +awscni_eni_max 4 +# HELP awscni_ip_max The maximum number of IP addresses that can be allocated to the instance +# TYPE awscni_ip_max gauge +awscni_ip_max 30 +# HELP awscni_reconcile_count The number of times ipamd reconciles on ENIs and IP/Prefix addresses +# TYPE awscni_reconcile_count counter +awscni_reconcile_count{fn="eniReconcileAdd"} 120 +awscni_reconcile_count{fn="eniReconcileDel"} 3 +# HELP awscni_add_ip_req_count The number of add IP address requests +# TYPE awscni_add_ip_req_count counter +awscni_add_ip_req_count 450 +# HELP awscni_del_ip_req_count The number of delete IP address requests +# TYPE awscni_del_ip_req_count counter +awscni_del_ip_req_count{reason="pod_deleted"} 200 +# HELP awscni_pod_eni_error_count The number of errors encountered for pod ENIs +# TYPE awscni_pod_eni_error_count counter +awscni_pod_eni_error_count{fn="podENIHandler"} 4 +# HELP awscni_aws_api_latency_ms AWS API call latency in ms +# TYPE awscni_aws_api_latency_ms summary +awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.5"} 15 +awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.9"} 45 +awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.99"} 80 +awscni_aws_api_latency_ms_sum{api="DescribeNetworkInterfaces",error="false",status="200"} 1200 +awscni_aws_api_latency_ms_count{api="DescribeNetworkInterfaces",error="false",status="200"} 60 +# HELP awscni_aws_api_error_count The number of times AWS API returns an error +# TYPE awscni_aws_api_error_count counter +awscni_aws_api_error_count{api="DescribeNetworkInterfaces",error="RequestLimitExceeded"} 3 +# HELP awscni_aws_utils_error_count The number of errors not handled in awsutils library +# TYPE awscni_aws_utils_error_count counter +awscni_aws_utils_error_count{fn="GetENILimit",error="NoENILimit"} 1 +# HELP awscni_ec2api_req_count The number of requests made to EC2 APIs by CNI +# TYPE awscni_ec2api_req_count counter +awscni_ec2api_req_count{fn="DescribeInstances"} 90 +# HELP awscni_ec2api_error_count The number of failed EC2 APIs requests +# TYPE awscni_ec2api_error_count counter +awscni_ec2api_error_count{fn="DescribeInstances"} 2 +# HELP awscni_eni_allocated The number of ENIs allocated +# TYPE awscni_eni_allocated gauge +awscni_eni_allocated 3 +# HELP awscni_total_ip_addresses The total number of IP addresses +# TYPE awscni_total_ip_addresses gauge +awscni_total_ip_addresses 24 +# HELP awscni_assigned_ip_addresses The number of IP addresses assigned to pods +# TYPE awscni_assigned_ip_addresses gauge +awscni_assigned_ip_addresses 18 +# HELP awscni_force_removed_enis The number of ENIs force removed while they had assigned pods +# TYPE awscni_force_removed_enis counter +awscni_force_removed_enis 0 +# HELP awscni_force_removed_ips The number of IPs force removed while they had assigned pods +# TYPE awscni_force_removed_ips counter +awscni_force_removed_ips 0 +# HELP awscni_total_ipv4_prefixes The total number of IPv4 prefixes +# TYPE awscni_total_ipv4_prefixes gauge +awscni_total_ipv4_prefixes 8 +# HELP awscni_assigned_ip_per_cidr The total number of IP addresses assigned per cidr +# TYPE awscni_assigned_ip_per_cidr gauge +awscni_assigned_ip_per_cidr{cidr="192.168.0.0/24"} 18 +# HELP awscni_no_available_ip_addresses The number of pod IP assignments that fail due to no available IP addresses +# TYPE awscni_no_available_ip_addresses counter +awscni_no_available_ip_addresses 5 +# HELP awscni_assigned_ip_per_eni The number of allocated ips partitioned by eni +# TYPE awscni_assigned_ip_per_eni gauge +awscni_assigned_ip_per_eni{eni="eni-0a1b2c3d"} 8 +awscni_assigned_ip_per_eni{eni="eni-0d4e5f6a"} 10 +# HELP awscni_ipamd_startup_duration_seconds The duration of IPAMD startup from process start to ready to serve CNI requests +# TYPE awscni_ipamd_startup_duration_seconds histogram +awscni_ipamd_startup_duration_seconds_bucket{success="true",with_api_server="true",failure_reason="",le="0.1"} 0 +awscni_ipamd_startup_duration_seconds_bucket{success="true",with_api_server="true",failure_reason="",le="0.5"} 0 +awscni_ipamd_startup_duration_seconds_bucket{success="true",with_api_server="true",failure_reason="",le="1"} 1 +awscni_ipamd_startup_duration_seconds_bucket{success="true",with_api_server="true",failure_reason="",le="+Inf"} 1 +awscni_ipamd_startup_duration_seconds_sum{success="true",with_api_server="true",failure_reason=""} 0.75 +awscni_ipamd_startup_duration_seconds_count{success="true",with_api_server="true",failure_reason=""} 1 +# HELP awscni_ipamd_node_initialization_duration_seconds The duration of node initialization during IPAMD startup +# TYPE awscni_ipamd_node_initialization_duration_seconds histogram +awscni_ipamd_node_initialization_duration_seconds_bucket{success="true",le="0.5"} 1 +awscni_ipamd_node_initialization_duration_seconds_bucket{success="true",le="1"} 1 +awscni_ipamd_node_initialization_duration_seconds_bucket{success="true",le="+Inf"} 1 +awscni_ipamd_node_initialization_duration_seconds_sum{success="true"} 0.4 +awscni_ipamd_node_initialization_duration_seconds_count{success="true"} 1 +# HELP awscni_connmark_backend The connmark backend in use; the active backend's series is set to 1 +# TYPE awscni_connmark_backend gauge +awscni_connmark_backend{backend="iptables"} 1 +awscni_connmark_backend{backend="ebpf"} 0 +# HELP awscni_connmark_reconcile_total The number of connmark Setup/Cleanup reconciliations by result +# TYPE awscni_connmark_reconcile_total counter +awscni_connmark_reconcile_total{result="success"} 40 +awscni_connmark_reconcile_total{result="error"} 1 From 1a95ba23b2e74310018999096752ebe9a5ab2791 Mon Sep 17 00:00:00 2001 From: Willian Cesar Cincerre Da Silva <3329-willian.cincerre@users.noreply.code.ifoodcorp.com.br> Date: Tue, 15 Sep 2026 14:15:47 -0300 Subject: [PATCH 3/9] Add CODEOWNERS entry for Amazon VPC CNI --- .github/CODEOWNERS | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index e1a4e0a4f..be984e8cb 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -20,6 +20,7 @@ code-coverage.datadog.yml @DataDog/agent-integr /alertnow/ @Uijong-Kim uijong.kim@bespinglobal.com /algorithmia/ @koverholt support@algorithmia.io @DataDog/ecosystems-review /altostra/ @yevk @altostra/engineering @DataDog/ecosystems-review +/amazon_vpc_cni/ @willianccs willianccs@gmail.com /ambassador/ @datawire hello@datawire.io /amixr/ @iskhakov ildar@amixr.io /apache-apisix/ @bisakhmondal bisakhmondal00@gmail.com From 63720d8a631ee34cc5424490de1903de203c7af7 Mon Sep 17 00:00:00 2001 From: Willian Cesar Cincerre Da Silva <3329-willian.cincerre@users.noreply.code.ifoodcorp.com.br> Date: Tue, 15 Sep 2026 14:20:39 -0300 Subject: [PATCH 4/9] Add Amazon VPC CNI to test-all matrix --- .github/workflows/test-all.yml | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/.github/workflows/test-all.yml b/.github/workflows/test-all.yml index 0660e3479..7f5c38b6b 100644 --- a/.github/workflows/test-all.yml +++ b/.github/workflows/test-all.yml @@ -63,6 +63,25 @@ jobs: test-py3: ${{ inputs.test-py3 }} setup-env-vars: "${{ inputs.setup-env-vars }}" secrets: inherit + ja9c433d: + uses: DataDog/integrations-core/.github/workflows/test-target.yml@5f6b58dfb609ea0afb37762612dbb6e2c37d2933 + with: + job-name: Amazon VPC CNI + target: amazon_vpc_cni + platform: linux + runner: '["ubuntu-22.04"]' + repo: "${{ inputs.repo }}" + context: ${{ inputs.context }} + python-version: "${{ inputs.python-version }}" + latest: ${{ inputs.latest }} + agent-image: "${{ inputs.agent-image }}" + agent-image-py2: "${{ inputs.agent-image-py2 }}" + agent-image-windows: "${{ inputs.agent-image-windows }}" + agent-image-windows-py2: "${{ inputs.agent-image-windows-py2 }}" + test-py2: ${{ inputs.test-py2 }} + test-py3: ${{ inputs.test-py3 }} + setup-env-vars: "${{ inputs.setup-env-vars }}" + secrets: inherit j7a4721a: uses: DataDog/integrations-core/.github/workflows/test-target.yml@5f6b58dfb609ea0afb37762612dbb6e2c37d2933 with: From fa6472d90c49388a4dd3013e211822c379467c4f Mon Sep 17 00:00:00 2001 From: Willian Cesar Cincerre Da Silva <3329-willian.cincerre@users.noreply.code.ifoodcorp.com.br> Date: Sat, 19 Sep 2026 00:03:19 -0300 Subject: [PATCH 5/9] fix(amazon-vpc-cni): address code review feedback - Fix connmark_reconcile mapping (prometheus_client strips the _total suffix) - Strengthen unit test assertions (at_least=1, check_symmetric_inclusion) - Restore @DataDog/ecosystems-review on CODEOWNERS entries and add it to amazon_vpc_cni - Correct duration sub-metric units in metadata.csv - Simplify check.py (drop unused V1-only config and pass-through overrides) - Add missing sagemaker metrics for full parity - Remove unused mock fixture from e2e tests --- .github/CODEOWNERS | 12 ++++++------ .../datadog_checks/amazon_vpc_cni/check.py | 8 -------- .../datadog_checks/amazon_vpc_cni/metrics.py | 4 +++- amazon_vpc_cni/metadata.csv | 14 ++++++++------ amazon_vpc_cni/tests/common.py | 2 ++ amazon_vpc_cni/tests/fixtures/metrics | 6 ++++++ amazon_vpc_cni/tests/fixtures/metrics.txt | 6 ++++++ amazon_vpc_cni/tests/test_e2e.py | 4 ++-- amazon_vpc_cni/tests/test_unit.py | 4 ++-- 9 files changed, 35 insertions(+), 25 deletions(-) diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index 5b54ae3ac..e6bdcc59b 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -20,12 +20,12 @@ code-coverage.datadog.yml @DataDog/agent-integr /alertnow/ @Uijong-Kim uijong.kim@bespinglobal.com @DataDog/ecosystems-review /algorithmia/ @koverholt support@algorithmia.io @DataDog/ecosystems-review /altostra/ @yevk @altostra/engineering @DataDog/ecosystems-review -/amazon_vpc_cni/ @willianccs willianccs@gmail.com -/ambassador/ @datawire hello@datawire.io -/amixr/ @iskhakov ildar@amixr.io -/apache-apisix/ @bisakhmondal bisakhmondal00@gmail.com -/apollo/ @sachindshinde sachin@apollographql.com -/appkeeper/ @numno1 +/amazon_vpc_cni/ @willianccs willianccs@gmail.com @DataDog/ecosystems-review +/ambassador/ @datawire hello@datawire.io @DataDog/ecosystems-review +/amixr/ @iskhakov ildar@amixr.io @DataDog/ecosystems-review +/apache-apisix/ @bisakhmondal bisakhmondal00@gmail.com @DataDog/ecosystems-review +/apollo/ @sachindshinde sachin@apollographql.com @DataDog/ecosystems-review +/appkeeper/ @numno1 @DataDog/ecosystems-review /apptrail/ @Samrose-Ahmed samrose@apptrail.com @DataDog/ecosystems-review /aqua/ @DataDog/container-integrations /artie/ @danafallon @DataDog/ecosystems-review diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/check.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/check.py index 943c736de..2730a42b6 100644 --- a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/check.py +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/check.py @@ -7,15 +7,7 @@ class AmazonVpcCniCheck(OpenMetricsBaseCheckV2): __NAMESPACE__ = 'amazon_vpc_cni' DEFAULT_METRIC_LIMIT = 0 - def __init__(self, name, init_config, instances): - super(AmazonVpcCniCheck, self).__init__(name, init_config, instances) - def get_default_config(self): return { 'metrics': [METRIC_MAP], - 'send_distribution_sums_as_monotonic': 'true', - 'send_distribution_counts_as_monotonic': 'true', } - - def check(self, _): - super().check(_) diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/metrics.py b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/metrics.py index 704c17568..24f6cc1e4 100644 --- a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/metrics.py +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/metrics.py @@ -12,6 +12,8 @@ 'awscni_aws_utils_error_count': 'aws_utils_error_count', 'awscni_ec2api_req_count': 'ec2api_req_count', 'awscni_ec2api_error_count': 'ec2api_error_count', + 'awscni_sagemakerapi_req_count': 'sagemakerapi_req_count', + 'awscni_sagemakerapi_error_count': 'sagemakerapi_error_count', ## ENI / IP allocation 'awscni_eni_max': 'eni_max', 'awscni_ip_max': 'ip_max', @@ -29,5 +31,5 @@ 'awscni_ipamd_node_initialization_duration_seconds': 'ipamd_node_initialization_duration_seconds', ## Connmark 'awscni_connmark_backend': 'connmark_backend', - 'awscni_connmark_reconcile_total': 'connmark_reconcile_total', + 'awscni_connmark_reconcile': 'connmark_reconcile_total', } diff --git a/amazon_vpc_cni/metadata.csv b/amazon_vpc_cni/metadata.csv index d9428d0cc..a06e7d86a 100644 --- a/amazon_vpc_cni/metadata.csv +++ b/amazon_vpc_cni/metadata.csv @@ -18,16 +18,18 @@ amazon_vpc_cni.aws_api_error_count.count,count,,,,The number of times AWS API re amazon_vpc_cni.aws_utils_error_count.count,count,,,,The number of errors not handled in the awsutils library.,-1,amazon_vpc_cni,aws utils errors, amazon_vpc_cni.ec2api_req_count.count,count,,,,The number of requests made to EC2 APIs by the CNI.,0,amazon_vpc_cni,ec2 api requests, amazon_vpc_cni.ec2api_error_count.count,count,,,,The number of failed EC2 API requests.,-1,amazon_vpc_cni,ec2 api errors, +amazon_vpc_cni.sagemakerapi_req_count.count,count,,,,The number of requests made to SageMaker APIs by the CNI.,0,amazon_vpc_cni,sagemaker api requests, +amazon_vpc_cni.sagemakerapi_error_count.count,count,,,,The number of failed SageMaker API requests.,-1,amazon_vpc_cni,sagemaker api errors, amazon_vpc_cni.force_removed_enis.count,count,,,,The number of ENIs force removed while they had assigned pods.,0,amazon_vpc_cni,force removed enis, amazon_vpc_cni.force_removed_ips.count,count,,,,The number of IPs force removed while they had assigned pods.,0,amazon_vpc_cni,force removed ips, amazon_vpc_cni.no_available_ip_addresses.count,count,,,,The number of pod IP assignments that fail due to no available IP addresses.,-1,amazon_vpc_cni,no available ip addresses, amazon_vpc_cni.connmark_reconcile_total.count,count,,,,The number of connmark reconciliations by result.,0,amazon_vpc_cni,connmark reconciles, amazon_vpc_cni.aws_api_latency_ms.count,count,,,,Number of AWS API latency observations.,0,amazon_vpc_cni,aws api latency count, -amazon_vpc_cni.aws_api_latency_ms.sum,count,,,,Sum of AWS API call latencies.,0,amazon_vpc_cni,aws api latency sum, -amazon_vpc_cni.aws_api_latency_ms.quantile,gauge,,,,AWS API call latency quantiles.,0,amazon_vpc_cni,aws api latency quantile, -amazon_vpc_cni.ipamd_startup_duration_seconds.count,count,,second,,Number of IPAMD startup duration observations.,0,amazon_vpc_cni,ipamd startup duration count, +amazon_vpc_cni.aws_api_latency_ms.sum,count,,millisecond,,Sum of AWS API call latencies.,0,amazon_vpc_cni,aws api latency sum, +amazon_vpc_cni.aws_api_latency_ms.quantile,gauge,,millisecond,,AWS API call latency quantiles.,0,amazon_vpc_cni,aws api latency quantile, +amazon_vpc_cni.ipamd_startup_duration_seconds.count,count,,,,Number of IPAMD startup duration observations.,0,amazon_vpc_cni,ipamd startup duration count, amazon_vpc_cni.ipamd_startup_duration_seconds.sum,count,,second,,Sum of IPAMD startup durations.,0,amazon_vpc_cni,ipamd startup duration sum, -amazon_vpc_cni.ipamd_startup_duration_seconds.bucket,count,,second,,Histogram buckets of IPAMD startup durations.,0,amazon_vpc_cni,ipamd startup duration bucket, -amazon_vpc_cni.ipamd_node_initialization_duration_seconds.count,count,,second,,Number of node initialization duration observations.,0,amazon_vpc_cni,node init duration count, +amazon_vpc_cni.ipamd_startup_duration_seconds.bucket,count,,,,Histogram buckets of IPAMD startup durations.,0,amazon_vpc_cni,ipamd startup duration bucket, +amazon_vpc_cni.ipamd_node_initialization_duration_seconds.count,count,,,,Number of node initialization duration observations.,0,amazon_vpc_cni,node init duration count, amazon_vpc_cni.ipamd_node_initialization_duration_seconds.sum,count,,second,,Sum of node initialization durations.,0,amazon_vpc_cni,node init duration sum, -amazon_vpc_cni.ipamd_node_initialization_duration_seconds.bucket,count,,second,,Histogram buckets of node initialization durations.,0,amazon_vpc_cni,node init duration bucket, +amazon_vpc_cni.ipamd_node_initialization_duration_seconds.bucket,count,,,,Histogram buckets of node initialization durations.,0,amazon_vpc_cni,node init duration bucket, diff --git a/amazon_vpc_cni/tests/common.py b/amazon_vpc_cni/tests/common.py index b88a0de7e..97002cf82 100644 --- a/amazon_vpc_cni/tests/common.py +++ b/amazon_vpc_cni/tests/common.py @@ -22,6 +22,8 @@ 'amazon_vpc_cni.aws_utils_error_count.count', 'amazon_vpc_cni.ec2api_req_count.count', 'amazon_vpc_cni.ec2api_error_count.count', + 'amazon_vpc_cni.sagemakerapi_req_count.count', + 'amazon_vpc_cni.sagemakerapi_error_count.count', 'amazon_vpc_cni.force_removed_enis.count', 'amazon_vpc_cni.force_removed_ips.count', 'amazon_vpc_cni.no_available_ip_addresses.count', diff --git a/amazon_vpc_cni/tests/fixtures/metrics b/amazon_vpc_cni/tests/fixtures/metrics index 051b8018b..a4f90ecff 100644 --- a/amazon_vpc_cni/tests/fixtures/metrics +++ b/amazon_vpc_cni/tests/fixtures/metrics @@ -44,6 +44,12 @@ awscni_ec2api_req_count{fn="DescribeInstances"} 90 # HELP awscni_ec2api_error_count The number of failed EC2 APIs requests # TYPE awscni_ec2api_error_count counter awscni_ec2api_error_count{fn="DescribeInstances"} 2 +# HELP awscni_sagemakerapi_req_count The number of requests made to SageMaker APIs by CNI +# TYPE awscni_sagemakerapi_req_count counter +awscni_sagemakerapi_req_count{fn="InvokeEndpoint"} 12 +# HELP awscni_sagemakerapi_error_count The number of failed SageMaker API requests +# TYPE awscni_sagemakerapi_error_count counter +awscni_sagemakerapi_error_count{fn="InvokeEndpoint"} 1 # HELP awscni_eni_allocated The number of ENIs allocated # TYPE awscni_eni_allocated gauge awscni_eni_allocated 3 diff --git a/amazon_vpc_cni/tests/fixtures/metrics.txt b/amazon_vpc_cni/tests/fixtures/metrics.txt index 051b8018b..a4f90ecff 100644 --- a/amazon_vpc_cni/tests/fixtures/metrics.txt +++ b/amazon_vpc_cni/tests/fixtures/metrics.txt @@ -44,6 +44,12 @@ awscni_ec2api_req_count{fn="DescribeInstances"} 90 # HELP awscni_ec2api_error_count The number of failed EC2 APIs requests # TYPE awscni_ec2api_error_count counter awscni_ec2api_error_count{fn="DescribeInstances"} 2 +# HELP awscni_sagemakerapi_req_count The number of requests made to SageMaker APIs by CNI +# TYPE awscni_sagemakerapi_req_count counter +awscni_sagemakerapi_req_count{fn="InvokeEndpoint"} 12 +# HELP awscni_sagemakerapi_error_count The number of failed SageMaker API requests +# TYPE awscni_sagemakerapi_error_count counter +awscni_sagemakerapi_error_count{fn="InvokeEndpoint"} 1 # HELP awscni_eni_allocated The number of ENIs allocated # TYPE awscni_eni_allocated gauge awscni_eni_allocated 3 diff --git a/amazon_vpc_cni/tests/test_e2e.py b/amazon_vpc_cni/tests/test_e2e.py index c6f27f9a2..48db44b43 100644 --- a/amazon_vpc_cni/tests/test_e2e.py +++ b/amazon_vpc_cni/tests/test_e2e.py @@ -7,13 +7,13 @@ @pytest.mark.e2e -def test_e2e_service_check_ok(dd_agent_check, aggregator, instance, mock_prometheus_metrics): +def test_e2e_service_check_ok(dd_agent_check, aggregator, instance): dd_agent_check(instance) aggregator.assert_service_check('amazon_vpc_cni.openmetrics.health', ServiceCheck.OK) @pytest.mark.e2e -def test_e2e_assert_metrics(dd_agent_check, aggregator, instance, mock_prometheus_metrics): +def test_e2e_assert_metrics(dd_agent_check, aggregator, instance): dd_agent_check(instance) for metric in EXPECTED_PROMETHEUS_METRICS: diff --git a/amazon_vpc_cni/tests/test_unit.py b/amazon_vpc_cni/tests/test_unit.py index 456976045..62a3880d6 100644 --- a/amazon_vpc_cni/tests/test_unit.py +++ b/amazon_vpc_cni/tests/test_unit.py @@ -20,8 +20,8 @@ def test_connect_exception(dd_run_check, aggregator, caplog): def test_check_mock_amazon_vpc_cni_metrics(dd_run_check, aggregator, check, mock_prometheus_metrics): dd_run_check(check) for metric_name in EXPECTED_PROMETHEUS_METRICS: - aggregator.assert_metric(metric_name, at_least=0) - aggregator.assert_metrics_using_metadata(get_metadata_metrics()) + aggregator.assert_metric(metric_name, at_least=1) + aggregator.assert_metrics_using_metadata(get_metadata_metrics(), check_symmetric_inclusion=True) aggregator.assert_service_check('amazon_vpc_cni.openmetrics.health', ServiceCheck.OK) From 47ed8b1c35f057bad4097466ce5cef59d5d89bbe Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Willian=20C=C3=A9sar?= <7793624+willianccs@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:10:35 -0300 Subject: [PATCH 6/9] Update amazon_vpc_cni/README.md Co-authored-by: Joe Peeples --- amazon_vpc_cni/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/amazon_vpc_cni/README.md b/amazon_vpc_cni/README.md index 1068cd315..a80662872 100644 --- a/amazon_vpc_cni/README.md +++ b/amazon_vpc_cni/README.md @@ -10,7 +10,7 @@ This integration scrapes the Prometheus metrics exposed by the VPC CNI plugin (` ### Installation -If you are using Agent v6.8+ follow the instructions below to install the Amazon VPC CNI check on your host. See the dedicated Agent guide for [installing community integrations][1] to install checks with the [Agent Manager][2] or in a [Docker environment][4]. +If you are using Datadog Agent v6.8+, follow the instructions below to install the Amazon VPC CNI check on your host. See the dedicated Agent guide for [installing community integrations][1] to install checks with the [Agent Manager][2] or in a [Docker environment][4]. 1. [Download and launch the Datadog Agent][3]. 2. Run the following command to install the Agent integration: From ce0680b424a328324c7421be3774b52308c34838 Mon Sep 17 00:00:00 2001 From: Willian Cesar Cincerre Da Silva <3329-willian.cincerre@users.noreply.code.ifoodcorp.com.br> Date: Mon, 28 Sep 2026 14:45:58 -0300 Subject: [PATCH 7/9] fix(amazon-vpc-cni): address Agent review feedback - Document node-reachability caveat for openmetrics_endpoint and use DISABLE_METRICS=false toggle - Remove unsupported aws_api_latency_ms.quantile from fixtures and metadata - Strengthen e2e metric assertions (at_least=1, symmetric inclusion) --- amazon_vpc_cni/README.md | 4 +++- amazon_vpc_cni/assets/configuration/spec.yaml | 7 +++++++ .../datadog_checks/amazon_vpc_cni/data/conf.yaml.example | 7 +++++++ amazon_vpc_cni/metadata.csv | 1 - amazon_vpc_cni/tests/common.py | 1 - amazon_vpc_cni/tests/fixtures/metrics | 3 --- amazon_vpc_cni/tests/fixtures/metrics.txt | 3 --- amazon_vpc_cni/tests/test_e2e.py | 4 ++-- 8 files changed, 19 insertions(+), 11 deletions(-) diff --git a/amazon_vpc_cni/README.md b/amazon_vpc_cni/README.md index a80662872..46114a8e9 100644 --- a/amazon_vpc_cni/README.md +++ b/amazon_vpc_cni/README.md @@ -23,7 +23,9 @@ If you are using Datadog Agent v6.8+, follow the instructions below to install t ### Configuration -Enable Prometheus metrics on the `aws-node` DaemonSet by setting the environment variable `ENABLE_PROMETHEUS_METRICS` to `true`. The plugin then exposes metrics at `http://localhost:61678/metrics` on each node. +Metrics are enabled by default on the `aws-node` DaemonSet. To control them, set the environment variable `DISABLE_METRICS` to `false`. The plugin exposes metrics at `http://localhost:61678/metrics` on each node. + +Because the `aws-node` DaemonSet runs with host networking, `localhost` only resolves to the IPAMD endpoint when the Agent also runs with host networking (or directly on the node). If the Agent runs as a pod without host networking, set `openmetrics_endpoint` to the node's IP address, for example `http://:61678/metrics`, or use Autodiscovery to resolve the node address. 1. Edit the `amazon_vpc_cni.d/conf.yaml` file in the `conf.d/` folder at the root of your [Agent's configuration directory][6] to start collecting your Amazon VPC CNI metrics. See the [sample amazon_vpc_cni.d/conf.yaml][7] for all available configuration options. diff --git a/amazon_vpc_cni/assets/configuration/spec.yaml b/amazon_vpc_cni/assets/configuration/spec.yaml index 419ba880e..e24025258 100644 --- a/amazon_vpc_cni/assets/configuration/spec.yaml +++ b/amazon_vpc_cni/assets/configuration/spec.yaml @@ -12,3 +12,10 @@ files: openmetrics_endpoint.value.example: http://localhost:61678/metrics openmetrics_endpoint.description: | Endpoint exposing the Amazon VPC CNI Prometheus metrics. + + The `aws-node` DaemonSet runs with host networking, so `localhost` only + resolves to the IPAMD endpoint when the Agent also runs with host + networking (or directly on the node host). If the Agent runs as a pod + without host networking, set this to the node's IP address, for example + `http://:61678/metrics`, or use Autodiscovery with a template + that resolves the node address. diff --git a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/data/conf.yaml.example b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/data/conf.yaml.example index 42e1e323d..d8a5acede 100644 --- a/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/data/conf.yaml.example +++ b/amazon_vpc_cni/datadog_checks/amazon_vpc_cni/data/conf.yaml.example @@ -15,6 +15,13 @@ instances: ## @param openmetrics_endpoint - string - required ## Endpoint exposing the Amazon VPC CNI Prometheus metrics. + ## + ## The `aws-node` DaemonSet runs with host networking, so `localhost` only + ## resolves to the IPAMD endpoint when the Agent also runs with host + ## networking (or directly on the node host). If the Agent runs as a pod + ## without host networking, set this to the node's IP address, for example + ## `http://:61678/metrics`, or use Autodiscovery with a template + ## that resolves the node address. # - openmetrics_endpoint: http://localhost:61678/metrics diff --git a/amazon_vpc_cni/metadata.csv b/amazon_vpc_cni/metadata.csv index a06e7d86a..0a0d8641e 100644 --- a/amazon_vpc_cni/metadata.csv +++ b/amazon_vpc_cni/metadata.csv @@ -26,7 +26,6 @@ amazon_vpc_cni.no_available_ip_addresses.count,count,,,,The number of pod IP ass amazon_vpc_cni.connmark_reconcile_total.count,count,,,,The number of connmark reconciliations by result.,0,amazon_vpc_cni,connmark reconciles, amazon_vpc_cni.aws_api_latency_ms.count,count,,,,Number of AWS API latency observations.,0,amazon_vpc_cni,aws api latency count, amazon_vpc_cni.aws_api_latency_ms.sum,count,,millisecond,,Sum of AWS API call latencies.,0,amazon_vpc_cni,aws api latency sum, -amazon_vpc_cni.aws_api_latency_ms.quantile,gauge,,millisecond,,AWS API call latency quantiles.,0,amazon_vpc_cni,aws api latency quantile, amazon_vpc_cni.ipamd_startup_duration_seconds.count,count,,,,Number of IPAMD startup duration observations.,0,amazon_vpc_cni,ipamd startup duration count, amazon_vpc_cni.ipamd_startup_duration_seconds.sum,count,,second,,Sum of IPAMD startup durations.,0,amazon_vpc_cni,ipamd startup duration sum, amazon_vpc_cni.ipamd_startup_duration_seconds.bucket,count,,,,Histogram buckets of IPAMD startup durations.,0,amazon_vpc_cni,ipamd startup duration bucket, diff --git a/amazon_vpc_cni/tests/common.py b/amazon_vpc_cni/tests/common.py index 97002cf82..2952a73b0 100644 --- a/amazon_vpc_cni/tests/common.py +++ b/amazon_vpc_cni/tests/common.py @@ -30,7 +30,6 @@ 'amazon_vpc_cni.connmark_reconcile_total.count', 'amazon_vpc_cni.aws_api_latency_ms.count', 'amazon_vpc_cni.aws_api_latency_ms.sum', - 'amazon_vpc_cni.aws_api_latency_ms.quantile', 'amazon_vpc_cni.ipamd_startup_duration_seconds.count', 'amazon_vpc_cni.ipamd_startup_duration_seconds.sum', 'amazon_vpc_cni.ipamd_startup_duration_seconds.bucket', diff --git a/amazon_vpc_cni/tests/fixtures/metrics b/amazon_vpc_cni/tests/fixtures/metrics index a4f90ecff..dc19ef1b0 100644 --- a/amazon_vpc_cni/tests/fixtures/metrics +++ b/amazon_vpc_cni/tests/fixtures/metrics @@ -27,9 +27,6 @@ awscni_del_ip_req_count{reason="pod_deleted"} 200 awscni_pod_eni_error_count{fn="podENIHandler"} 4 # HELP awscni_aws_api_latency_ms AWS API call latency in ms # TYPE awscni_aws_api_latency_ms summary -awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.5"} 15 -awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.9"} 45 -awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.99"} 80 awscni_aws_api_latency_ms_sum{api="DescribeNetworkInterfaces",error="false",status="200"} 1200 awscni_aws_api_latency_ms_count{api="DescribeNetworkInterfaces",error="false",status="200"} 60 # HELP awscni_aws_api_error_count The number of times AWS API returns an error diff --git a/amazon_vpc_cni/tests/fixtures/metrics.txt b/amazon_vpc_cni/tests/fixtures/metrics.txt index a4f90ecff..dc19ef1b0 100644 --- a/amazon_vpc_cni/tests/fixtures/metrics.txt +++ b/amazon_vpc_cni/tests/fixtures/metrics.txt @@ -27,9 +27,6 @@ awscni_del_ip_req_count{reason="pod_deleted"} 200 awscni_pod_eni_error_count{fn="podENIHandler"} 4 # HELP awscni_aws_api_latency_ms AWS API call latency in ms # TYPE awscni_aws_api_latency_ms summary -awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.5"} 15 -awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.9"} 45 -awscni_aws_api_latency_ms{api="DescribeNetworkInterfaces",error="false",status="200",quantile="0.99"} 80 awscni_aws_api_latency_ms_sum{api="DescribeNetworkInterfaces",error="false",status="200"} 1200 awscni_aws_api_latency_ms_count{api="DescribeNetworkInterfaces",error="false",status="200"} 60 # HELP awscni_aws_api_error_count The number of times AWS API returns an error diff --git a/amazon_vpc_cni/tests/test_e2e.py b/amazon_vpc_cni/tests/test_e2e.py index 48db44b43..202e0a03f 100644 --- a/amazon_vpc_cni/tests/test_e2e.py +++ b/amazon_vpc_cni/tests/test_e2e.py @@ -17,7 +17,7 @@ def test_e2e_assert_metrics(dd_agent_check, aggregator, instance): dd_agent_check(instance) for metric in EXPECTED_PROMETHEUS_METRICS: - aggregator.assert_metric(metric, at_least=0) + aggregator.assert_metric(metric, at_least=1) aggregator.assert_all_metrics_covered() - aggregator.assert_metrics_using_metadata(get_metadata_metrics()) + aggregator.assert_metrics_using_metadata(get_metadata_metrics(), check_symmetric_inclusion=True) From 42462bdc27d8e1c3f9897fff3a3e23760cc1c036 Mon Sep 17 00:00:00 2001 From: Willian Cesar Cincerre Da Silva <3329-willian.cincerre@users.noreply.code.ifoodcorp.com.br> Date: Mon, 28 Sep 2026 15:03:33 -0300 Subject: [PATCH 8/9] fix(amazon-vpc-cni): flush monotonic counters in e2e test OpenMetrics V2 counters are not emitted on the first scrape (flush_first_value), so a single dd_agent_check run produced no *.count metrics. Run with rate=True, matching the gatekeeper/aerospike_enterprise pattern, so counters flush. --- amazon_vpc_cni/tests/test_e2e.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/amazon_vpc_cni/tests/test_e2e.py b/amazon_vpc_cni/tests/test_e2e.py index 202e0a03f..0b5477c8d 100644 --- a/amazon_vpc_cni/tests/test_e2e.py +++ b/amazon_vpc_cni/tests/test_e2e.py @@ -14,7 +14,7 @@ def test_e2e_service_check_ok(dd_agent_check, aggregator, instance): @pytest.mark.e2e def test_e2e_assert_metrics(dd_agent_check, aggregator, instance): - dd_agent_check(instance) + dd_agent_check(instance, rate=True) for metric in EXPECTED_PROMETHEUS_METRICS: aggregator.assert_metric(metric, at_least=1) From 579a99f939023f16dc02d19db66491a5b4c14d6a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Willian=20C=C3=A9sar?= <7793624+willianccs@users.noreply.github.com> Date: Thu, 1 Oct 2026 13:30:54 -0300 Subject: [PATCH 9/9] Update amazon_vpc_cni/README.md Co-authored-by: bgoldberg122 --- amazon_vpc_cni/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/amazon_vpc_cni/README.md b/amazon_vpc_cni/README.md index 46114a8e9..3a2ab56b3 100644 --- a/amazon_vpc_cni/README.md +++ b/amazon_vpc_cni/README.md @@ -55,7 +55,7 @@ Need help? Contact [Datadog support][12]. [1]: https://docs.datadoghq.com/agent/guide/use-community-integrations/ [2]: https://docs.datadoghq.com/agent/guide/agent-commands/?tab=agentv6v7#start-stop-and-restart-the-agent -[3]: https://app.datadoghq.com/account/settings/agent/latest +[3]: /account/settings/agent/latest [4]: https://docs.datadoghq.com/agent/guide/use-community-integrations/?tab=docker [5]: https://docs.datadoghq.com/getting_started/integrations/ [6]: https://docs.datadoghq.com/agent/guide/agent-configuration-files/#agent-configuration-directory