From 1c7ddcd9e7ea2fde604dc3b18d0b59173a30dd2f Mon Sep 17 00:00:00 2001 From: Sunil Yadav Date: Mon, 14 Sep 2026 02:03:52 +0000 Subject: [PATCH 1/8] CI AMP onboarding --- REPO-GUIDE.md | 363 +++++++ src/aks-preview/azext_aks_preview/_consts.py | 6 + src/aks-preview/azext_aks_preview/_help.py | 24 + src/aks-preview/azext_aks_preview/_params.py | 50 + .../azext_aks_preview/_validators.py | 53 + .../azext_aks_preview/addonconfiguration.py | 22 + src/aks-preview/azext_aks_preview/custom.py | 8 + .../managed_cluster_decorator.py | 706 ++++++++----- .../tests/latest/test_custom.py | 125 +++ .../latest/test_managed_cluster_decorator.py | 970 +++++++++++++++--- .../tests/latest/test_validators.py | 77 ++ .../azmon-logs-cli-enhancements-spec.md | 221 ++++ 12 files changed, 2231 insertions(+), 394 deletions(-) create mode 100644 REPO-GUIDE.md create mode 100644 src/aks-preview/azmon-logs-cli-enhancements-spec.md diff --git a/REPO-GUIDE.md b/REPO-GUIDE.md new file mode 100644 index 00000000000..a6ba844e86d --- /dev/null +++ b/REPO-GUIDE.md @@ -0,0 +1,363 @@ +# `azure-cli-extensions` — How This Repo Works + +A practical orientation guide for someone new to the repo. Everything below was verified against +the current checkout; file paths and line references point at real code. + +`aks-preview` is used as the worked example throughout, since that is where AKS / Azure Monitor +work lands. + +--- + +## 1. What this repo actually is + +Per `README.md`, the repo serves **two independent purposes**: + +| Purpose | Location | What it is | +|---|---|---| +| Extension **source code** | `src//` | 211 extension folders, each an installable Python wheel | +| The **extension index** | `src/index.json` | 6.5 MB catalogue of *published* wheels (220 extensions, 386 `aks-preview` versions) | + +These are decoupled. `src/index.json` is what `az extension add --name ` reads (synced to +`https://aka.ms/azure-cli-extension-index-v1` every few minutes). A wheel can be indexed without +its source living here, and source can live here without being indexed. + +> **Key mental model:** this repo does *not* ship the CLI itself. It ships **add-on command +> modules** that plug into `azure-cli` core at runtime. Core lives in the separate +> [`Azure/azure-cli`](https://github.com/Azure/azure-cli) repo, and you need a clone of it to +> develop here (see §5). + +### Top-level layout + +``` +azure-cli-extensions/ +├── src/ # all extensions + index.json +│ ├── index.json # published wheel catalogue (hash, version, metadata) +│ ├── service_name.json # top-level command group → Azure service name + docs URL +│ │ # (checked by scripts/ci/service_name.py) +│ └── aks-preview/ # one extension +├── scripts/ci/ # index + release automation (update_index.py, test_index.py, ...) +├── scripts/automation/ # build_package.py, create_release_tag.py +├── .azure-pipelines/ # ADO templates (azdev_setup.yml, variables.yml) +├── azure-pipelines.yml # the PR CI definition +├── .github/workflows/ # GitHub-side automation (release trigger, linter comments) +├── .github/CODEOWNERS # per-extension reviewers +└── docs/ # points at the canonical azure-cli authoring docs +``` + +--- + +## 2. Anatomy of a single extension + +``` +src/aks-preview/ +├── setup.py # VERSION = "22.0.0b6" ← the published version +├── setup.cfg # bdist_wheel config +├── HISTORY.rst # changelog; has a "Pending" section (see §6) +├── README.rst +├── linter_exclusions.yml # per-extension linter waivers +├── azcli_aks_live_test/ # live-test harness (ADO pipelines + configs) +└── azext_aks_preview/ # ← the actual Python package that gets imported + ├── azext_metadata.json # minCliCoreVersion, isPreview flag + ├── __init__.py # COMMAND_LOADER_CLS — the entry point + ├── commands.py # command name → Python function mapping + ├── _params.py # CLI argument definitions + ├── _help.py # help text + ├── _validators.py # argument validation + ├── _format.py # table output transformers + ├── _client_factory.py # builds SDK clients ← request plumbing + ├── custom.py # command implementations + ├── managed_cluster_decorator.py # builds the ManagedCluster request body + ├── agentpool_decorator.py # same, for node pools + ├── vendored_sdks/ # generated ARM SDK (excluded from static checks) + ├── aaz/ # aaz-dev-tools generated commands + └── tests/latest/ # tests + 289 HTTP recordings +``` + +The folder naming is a hard convention: directory `src//` contains package `azext_/`. + +--- + +## 3. The command flow — from `az` to ARM + +This is the core thing to understand. + +``` + $ az aks create -g rg -n mycluster --enable-addons monitoring + │ + │ 1. azure-cli core discovers installed extensions and calls COMMAND_LOADER_CLS + ▼ + azext_aks_preview/__init__.py → ContainerServiceCommandsLoader + │ load_command_table(): + │ a) load_aaz_command_table(...) ← AAZ-generated commands first + │ b) commands.load_command_table() ← handwritten ones, which OVERRIDE AAZ + │ load_arguments() → _params.py + ▼ + commands.py g.custom_command("create", "aks_create", supports_no_wait=True) + │ (inside a command_group bound to client_factory=cf_managed_clusters) + ▼ + custom.py::aks_create (line 1225) + │ builds a decorator, then: + │ mc = aks_create_decorator.construct_mc_profile_preview() (line 1518) + │ return aks_create_decorator.create_mc(mc) (line 1526) + ▼ + managed_cluster_decorator.py :: AKSPreviewManagedClusterCreateDecorator (line 4487) + │ set_up_network_profile(mc), set_up_addon_profiles(mc), + │ set_up_api_server_access_profile(mc), ... ← one method per feature + │ each mutates the ManagedCluster model object + ▼ + put_mc() (line 6336) + │ self.client.begin_create_or_update(...) (line 6342) + │ or sdk_no_wait(...) when --no-wait (line 6355) + ▼ + vendored_sdks/azure_mgmt_preview_aks → ContainerServiceClient + │ api_version = "2026-06-02-preview" (_configuration.py:52) + ▼ + HTTPS PUT to Azure Resource Manager +``` + +### 3a. Where requests are built — **two** distinct mechanisms + +The repo mixes two generations of tooling. Knowing which one a command uses tells you where to edit. + +#### Mechanism A — vendored SDK + decorators (handwritten; most of `aks-preview`) + +1. **`vendored_sdks/`** holds an AutoRest/TypeSpec-generated ARM SDK. It is checked in + deliberately, and `README.md` notes it is excluded from CI static checking precisely because + generated code fails those checks. Two clients are vendored here: + `azure_mgmt_preview_aks` and `azure_mgmt_preview_aks_pis`. + +2. **`_client_factory.py`** registers those SDKs as CLI "custom resource types" and exposes one + factory per operation group: + + ```python + CUSTOM_MGMT_AKS_PREVIEW = CustomResourceType( + 'azext_aks_preview.vendored_sdks.azure_mgmt_preview_aks', 'ContainerServiceClient') + + def cf_managed_clusters(cli_ctx, *_): + return get_container_service_client(cli_ctx).managed_clusters + ``` + + `commands.py` attaches these via `client_factory=cf_managed_clusters`, so the command + implementation receives an authenticated, subscription-scoped client for free. + +3. **Decorators build the request body.** For anything as large as `ManagedCluster`, the payload + is assembled by an ordered chain of `set_up_*` / `update_*` methods, each owning exactly one + feature. This is the pattern to follow when adding a flag: + - add the argument in `_params.py` + - add a `set_up_` (create) and `update_` (update) method in the decorator + - call it from `construct_mc_profile_preview()` / `update_mc_profile_preview()` (line 9027) + - the final `put_mc()` sends it + + Constants for feature names/addons live in `_consts.py`. + +#### Mechanism B — AAZ (`aaz/`, generated by `aaz-dev-tools`) + +Newer commands are **declarative and fully generated from the Swagger/TypeSpec spec** — no +handwritten SDK. Example: `aaz/latest/aks/safeguards/_show.py`. + +Here the HTTP request is expressed directly as a class: + +```python +class DeploymentSafeguardsGet(AAZHttpOperation): + @property + def url(self): ... # ARM template URL + @property + def method(self): ... # "GET" + @property + def url_parameters(self): ... # subscriptionId, resourceGroupName, ... + @property + def query_parameters(self): + return { "api-version", "2025-05-02-preview" } + @property + def header_parameters(self): ... + def on_200(self, session): ... # response deserialization schema +``` + +Note each mechanism pins its **own** api-version (`2026-06-02-preview` for the vendored SDK, +`2025-05-02-preview` for this AAZ command). They are independent. + +> ⚠️ Files under `aaz/` are marked `# Code generated by aaz-dev-tools` with `pylint: skip-file`. +> **Do not hand-edit them** — regenerate with `aaz-dev-tools`. To customize behaviour, use the +> `pre_operations()` / `post_operations()` hooks, or override the command in `commands.py` +> (remember the loader applies handwritten commands *after* AAZ, so they win). + +--- + +## 4. Adding new API surface + +When the AKS RP ships a new api-version and you need new fields: + +1. **Regenerate / re-vendor the SDK** into `vendored_sdks/` (AutoRest or TypeSpec). The api-version + default lives in `vendored_sdks/azure_mgmt_preview_aks/_configuration.py`. +2. **Expose the flag** in `_params.py` (+ `_validators.py`, `_help.py`, `_consts.py`). +3. **Wire the payload** in the relevant decorator (`managed_cluster_decorator.py` / + `agentpool_decorator.py`). +4. **Register** the command in `commands.py` if it is new. +5. **Test** (§5) and **changelog** (§6). + +`swagger_to_sdk_config.json` at the repo root drives spec→SDK automation. + +--- + +## 5. Development cycle + +### Setup (mirrors CI exactly — `.azure-pipelines/templates/azdev_setup.yml`) + +```bash +python -m venv env && source env/bin/activate +pip install -U pip +pip install --upgrade azdev==0.2.13 # CI pins this exact version +pip install build wheel + +git clone -q --single-branch -b dev https://github.com/Azure/azure-cli.git ../azure-cli +azdev setup -c ../azure-cli -r . # -c = CLI core repo, -r = this repo +``` + +`azdev setup -r .` installs every extension in `src/` in editable mode, so edits take effect +immediately with no reinstall. + +To work on just one extension: + +```bash +azdev extension add aks-preview +az extension list # confirm it resolves to your working tree +az aks create --help +``` + +### Inner loop + +```bash +# edit code, then run it straight away — no build step needed +az aks create -g rg -n c1 --enable-addons monitoring --debug + +# see the exact HTTP request/response ARM receives +az aks create ... --debug # or --verbose +``` + +`--debug` is the fastest way to confirm your decorator actually put the field on the wire. + +### Test + +Tests are `ScenarioTest`-based and replay recorded HTTP traffic (289 recordings under +`src/aks-preview/azext_aks_preview/tests/latest/recordings/`). + +```bash +azdev test aks-preview # whole extension (playback) +azdev test aks-preview --live # hit real Azure, re-record +azdev test --src aks-preview test_aks_commands # single module +pytest src/aks-preview/.../tests/latest/test_aks_commands.py -k mytest +``` + +Recording helpers live alongside the tests: `custom_preparers.py`, `recording_processors.py` +(scrubs secrets), `mocks.py`. Long-running end-to-end validation uses the separate +`azcli_aks_live_test/` ADO pipelines. + +`pytest.ini` defines markers: `e2e_packaging`, `smoke`, `slow`. + +### Lint / style — run these *before* pushing, they are enforced gates + +```bash +azdev linter --include-whl-extensions aks-preview +azdev style aks-preview +azdev scan --include-whl-extensions aks-preview # credential scan +``` + +Waivers go in `linter_exclusions.yml` (root and/or per-extension). + +### Build a wheel locally + +```bash +azdev extension build aks-preview # emits dist/*.whl +az extension add --source dist/aks_preview-22.0.0b6-py3-none-any.whl --yes +``` + +--- + +## 6. Release & publish flow + +### What you do in your PR + +1. Add your entry to the **`Pending`** section of `HISTORY.rst`. +2. When releasing, promote `Pending` into a new numbered section and bump `VERSION` in `setup.py` + (semver; `22.0.0b6` style for preview). +3. **Do not edit `src/index.json`** — automation owns it. `README.md` is explicit that the + precondition for auto-publish is *bump the version but do not touch the index*. + +`HISTORY.rst` states this guidance verbatim at the top of the file. + +### What automation does + +``` +PR merged to main + └─► .github/workflows/TriggerExtensionRelease.yml + (OIDC login → az pipelines build queue → ADO OneBranch release pipeline) + └─► builds the .whl, signs, uploads to the CDN/storage + └─► opens a follow-up PR updating src/index.json + (sha256 + metadata via scripts/ci/update_index.py) + └─► index syncs to aka.ms/azure-cli-extension-index-v1 + └─► `az extension add/update --name aks-preview` sees it +``` + +Supporting automation: `.github/workflows/CreateReleaseTag.yml`, +`scripts/automation/build_package.py`, `scripts/ci/idempotent_release.py`, +`scripts/ci/release_version_cal.py`. + +Manual fallback (external-hosted wheels): `azdev extension update-index ` computes +the sha256 and writes the entry. Hashes must be **lowercase** or CI fails. + +--- + +## 7. CI gates on every PR (`azure-pipelines.yml`) + +| Job | What it enforces | +|---|---| +| `CredScan` | no secrets committed | +| `PolicyCheck` | Microsoft policy compliance | +| `CheckLicenseHeader` | MIT header on every source file | +| `CheckInit` | `__init__.py` present where required | +| `IndexVerify` | `src/index.json` well-formed, hashes match published wheels | +| `SourceTests` | integration + build tests across a Python matrix; publishes wheel artifacts | +| `AzdevStyleModifiedExtensions` | `azdev style` on changed extensions only | +| `AzdevLinterModifiedExtensions` | `azdev linter` on changed extensions only | +| `AzdevScanModifiedExtensions{High,Medium}` | credential scanning by confidence tier | +| `CheckExternalUrls` | external links resolve (`external_url_exclusions.json` waives) | + +GitHub-side helpers post linter/style results as PR comments (`AddPRComment.yml`, +`AzdevLinter.yml`, `AzdevStyle.yml`) and `BlockPRMerge.yml` gates merges. + +Review routing is `.github/CODEOWNERS`; `aks-preview` is owned by `@fumingzhang @elvazhu521` +(line 63), so any change under `src/aks-preview/` requires their approval. + +--- + +## 8. Quick reference — "where do I change X?" + +| I want to… | File | +|---|---| +| Add a CLI flag | `_params.py` | +| Validate a flag | `_validators.py` | +| Write help text / examples | `_help.py` | +| Register a new command | `commands.py` | +| Implement command logic | `custom.py` | +| Put a field on the ARM request body | `managed_cluster_decorator.py` / `agentpool_decorator.py` | +| Change which SDK/api-version is called | `_client_factory.py` + `vendored_sdks/.../_configuration.py` | +| Change table output | `_format.py` | +| Add a constant / addon name | `_consts.py` | +| Edit a generated command | ❌ regenerate `aaz/` with `aaz-dev-tools`; don't hand-edit | +| Record a changelog entry | `HISTORY.rst` → `Pending` | +| Bump the shipped version | `setup.py` → `VERSION` | +| Add an index entry | ❌ automation does it; don't hand-edit `src/index.json` | + +--- + +## 9. Canonical upstream docs + +`docs/README.md` intentionally defers to `Azure/azure-cli`: + +- [Authoring an extension](https://github.com/Azure/azure-cli/blob/dev/doc/extensions/authoring.md) +- [Publishing](https://github.com/Azure/azure-cli/blob/dev/doc/extensions/authoring.md#publish) +- [Command guidelines](https://github.com/Azure/azure-cli/blob/dev/doc/command_guidelines.md) +- [Extension metadata](https://github.com/Azure/azure-cli/blob/dev/doc/extensions/metadata.md) +- [FAQ](https://github.com/Azure/azure-cli/blob/dev/doc/extensions/faq.md) +- [`azdev` tooling](https://github.com/Azure/azure-cli-dev-tools) +- [Migrating to `pyproject.toml`](docs/pyproject-migration.md) diff --git a/src/aks-preview/azext_aks_preview/_consts.py b/src/aks-preview/azext_aks_preview/_consts.py index a9ed2008a28..fe402022c89 100644 --- a/src/aks-preview/azext_aks_preview/_consts.py +++ b/src/aks-preview/azext_aks_preview/_consts.py @@ -188,6 +188,12 @@ CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID = "logAnalyticsWorkspaceResourceID" CONST_MONITORING_USING_AAD_MSI_AUTH = "useAADAuth" +# container network logs (azureMonitorProfile.containerInsights.containerNetworkLogs) +CONST_CONTAINER_NETWORK_LOGS_ENABLED = "Enabled" +CONST_CONTAINER_NETWORK_LOGS_DISABLED = "Disabled" +# legacy omsagent addon config key, superseded by containerNetworkLogs on the AMP path +CONST_MONITORING_ENABLE_RETINA_NETWORK_FLAGS = "enableRetinaNetworkFlags" + # virtual node CONST_VIRTUAL_NODE_ADDON_NAME = "aciConnector" CONST_VIRTUAL_NODE_SUBNET_NAME = "SubnetName" diff --git a/src/aks-preview/azext_aks_preview/_help.py b/src/aks-preview/azext_aks_preview/_help.py index 459677d3e7c..0dfdd6bfcb6 100644 --- a/src/aks-preview/azext_aks_preview/_help.py +++ b/src/aks-preview/azext_aks_preview/_help.py @@ -338,6 +338,16 @@ - name: --ampls-resource-id type: string short-summary: Resource ID of Azure Monitor Private Link scope for Monitoring Addon. + - name: --syslog-port + type: int + short-summary: Host port used by the Azure Monitor agent to collect syslog. Defaults to 28330 when unset. + long-summary: Applies to the Azure Monitor profile, configured with --enable-azure-monitor-logs. Distinct from --enable-syslog, which toggles syslog collection itself. + - name: --enable-prometheus-metrics-scraping + type: bool + short-summary: Enable Prometheus metrics scraping by the Azure Monitor agent. Applies to the Azure Monitor profile. + - name: --disable-prometheus-metrics-scraping + type: bool + short-summary: Disable Prometheus metrics scraping by the Azure Monitor agent. Applies to the Azure Monitor profile. - name: --enable-cluster-autoscaler type: bool short-summary: Enable cluster autoscaler, default value is false. @@ -869,6 +879,8 @@ text: az aks create -g MyResourceGroup -n MyManagedCluster --enable-opentelemetry-logs-traces --enable-addons monitoring - name: Create a kubernetes cluster with Azure Monitor logs enabled (shorthand) text: az aks create -g MyResourceGroup -n MyManagedCluster --enable-azure-monitor-logs + - name: Create a kubernetes cluster with Azure Monitor logs, Prometheus scraping disabled and a custom syslog port + text: az aks create -g MyResourceGroup -n MyManagedCluster --enable-azure-monitor-logs --disable-prometheus-metrics-scraping --syslog-port 28331 - name: Create a kubernetes cluster with OpenTelemetry metrics on custom port text: az aks create -g MyResourceGroup -n MyManagedCluster --enable-opentelemetry-metrics --opentelemetry-metrics-port-http 8888 --enable-azure-monitor-metrics - name: Create a kubernetes cluster with OpenTelemetry logs and traces on custom ports @@ -1166,6 +1178,16 @@ - name: --ampls-resource-id type: string short-summary: Resource ID of Azure Monitor Private Link scope for Monitoring Addon. + - name: --syslog-port + type: int + short-summary: Host port used by the Azure Monitor agent to collect syslog. Defaults to 28330 when unset. + long-summary: Applies to the Azure Monitor profile, configured with --enable-azure-monitor-logs. Distinct from --enable-syslog, which toggles syslog collection itself. + - name: --enable-prometheus-metrics-scraping + type: bool + short-summary: Enable Prometheus metrics scraping by the Azure Monitor agent. Applies to the Azure Monitor profile. + - name: --disable-prometheus-metrics-scraping + type: bool + short-summary: Disable Prometheus metrics scraping by the Azure Monitor agent. Applies to the Azure Monitor profile. - name: --enable-secret-rotation type: bool short-summary: Enable secret rotation. Use with azure-keyvault-secrets-provider addon. @@ -1715,6 +1737,8 @@ text: az aks update -g MyResourceGroup -n MyManagedCluster --safeguards-level Warning --safeguards-excluded-ns ns1,ns2 - name: Enable Azure Monitor logs for a kubernetes cluster text: az aks update -g MyResourceGroup -n MyManagedCluster --enable-azure-monitor-logs + - name: Re-enable Prometheus scraping and change the syslog port on a cluster with Azure Monitor logs enabled + text: az aks update -g MyResourceGroup -n MyManagedCluster --enable-prometheus-metrics-scraping --syslog-port 29000 - name: Enable Azure Backup for a kubernetes cluster (default Week strategy). Requires the 'dataprotection' extension. text: az aks update -g MyResourceGroup -n MyManagedCluster --enable-backup --yes - name: Enable Azure Backup with a custom strategy using an existing vault and policy diff --git a/src/aks-preview/azext_aks_preview/_params.py b/src/aks-preview/azext_aks_preview/_params.py index 26248c3ce73..b6aa4ef39bc 100644 --- a/src/aks-preview/azext_aks_preview/_params.py +++ b/src/aks-preview/azext_aks_preview/_params.py @@ -202,6 +202,8 @@ validate_azure_keyvault_kms_key_vault_resource_id, validate_azure_monitor_and_opentelemetry_for_create, validate_azure_monitor_and_opentelemetry_for_update, + validate_container_insights_settings_for_create, + validate_container_insights_settings_for_update, validate_azure_monitor_logs_and_enable_addons, validate_azure_monitor_logs_enable_disable, validate_azuremonitorworkspaceresourceid, @@ -326,6 +328,31 @@ def _deprecate_option(c, target, redirect): return deprecated +def _legacy_monitoring_auth_message(_deprecated): + """Build the deprecation message for --enable-msi-auth-for-monitoring.""" + return ( + "Argument '--enable-msi-auth-for-monitoring' has been deprecated and will be removed in a " + "future release. Container Insights onboarding is moving to the Azure Monitor profile, " + "which supports managed identity authentication only. Use 'az aks create' or " + "'az aks update' with '--enable-azure-monitor-logs' instead." + ) + + +def _deprecate_legacy_monitoring_auth(c): + """Deprecate --enable-msi-auth-for-monitoring on the legacy addon commands. + + knack only invokes a deprecated argument's action when the option is actually present on the + command line, so the warning is emitted once on explicit use and never for the default value. + + ArgumentsContext.deprecate() overwrites any message_func it is given with its own generic + text, so the custom message is attached to the returned object instead. + """ + deprecated = c.deprecate(target="--enable-msi-auth-for-monitoring") + if deprecated is not None: + deprecated._get_message = _legacy_monitoring_auth_message # pylint: disable=protected-access + return deprecated + + # candidates for enumeration # consts for AgentPool node_priorities = [CONST_SCALE_SET_PRIORITY_REGULAR, CONST_SCALE_SET_PRIORITY_SPOT] @@ -908,6 +935,12 @@ def load_arguments(self, _): c.argument("data_collection_settings", is_preview=True) c.argument("enable_high_log_scale_mode", arg_type=get_three_state_flag(), is_preview=True) c.argument("ampls_resource_id", is_preview=True) + c.argument("syslog_port", + type=int, + is_preview=True, + validator=validate_container_insights_settings_for_create) + c.argument("enable_prometheus_metrics_scraping", action="store_true", is_preview=True) + c.argument("disable_prometheus_metrics_scraping", action="store_true", is_preview=True) c.argument("aci_subnet_name") c.argument("appgw_name", arg_group="Application Gateway") c.argument("appgw_subnet_cidr", arg_group="Application Gateway") @@ -1864,6 +1897,20 @@ def load_arguments(self, _): c.argument("ampls_resource_id", is_preview=True, help="Resource ID of the Azure Monitor Private Link Scope to associate with the cluster") + c.argument("syslog_port", + type=int, + is_preview=True, + validator=validate_container_insights_settings_for_update, + help="Host port used by the Azure Monitor agent to collect syslog. " + "Defaults to 28330 when unset.") + c.argument("enable_prometheus_metrics_scraping", + action="store_true", + is_preview=True, + help="Enable Prometheus metrics scraping by the Azure Monitor agent") + c.argument("disable_prometheus_metrics_scraping", + action="store_true", + is_preview=True, + help="Disable Prometheus metrics scraping by the Azure Monitor agent") # OpenTelemetry parameters c.argument("enable_opentelemetry_metrics", is_preview=True, @@ -3139,6 +3186,7 @@ def load_arguments(self, _): "enable_msi_auth_for_monitoring", arg_type=get_three_state_flag(), is_preview=True, + deprecate_info=_deprecate_legacy_monitoring_auth(c), ) c.argument("enable_syslog", arg_type=get_three_state_flag(), is_preview=True) c.argument("data_collection_settings", is_preview=True) @@ -3196,6 +3244,7 @@ def load_arguments(self, _): "enable_msi_auth_for_monitoring", arg_type=get_three_state_flag(), is_preview=True, + deprecate_info=_deprecate_legacy_monitoring_auth(c), ) c.argument("enable_syslog", arg_type=get_three_state_flag(), is_preview=True) c.argument("data_collection_settings", is_preview=True) @@ -3236,6 +3285,7 @@ def load_arguments(self, _): "enable_msi_auth_for_monitoring", arg_type=get_three_state_flag(), is_preview=True, + deprecate_info=_deprecate_legacy_monitoring_auth(c), ) c.argument("enable_syslog", arg_type=get_three_state_flag(), is_preview=True) c.argument("data_collection_settings", is_preview=True) diff --git a/src/aks-preview/azext_aks_preview/_validators.py b/src/aks-preview/azext_aks_preview/_validators.py index 2e48bc2dcbf..867126bac43 100644 --- a/src/aks-preview/azext_aks_preview/_validators.py +++ b/src/aks-preview/azext_aks_preview/_validators.py @@ -1264,6 +1264,59 @@ def validate_azure_monitor_logs_enable_disable(namespace): ) +def _specified_container_insights_setting_flags(namespace): + """Return the AMP containerInsights tuning flags explicitly present on the command line.""" + flags = [] + if getattr(namespace, 'syslog_port', None) is not None: + flags.append("--syslog-port") + if getattr(namespace, 'enable_prometheus_metrics_scraping', False): + flags.append("--enable-prometheus-metrics-scraping") + if getattr(namespace, 'disable_prometheus_metrics_scraping', False): + flags.append("--disable-prometheus-metrics-scraping") + return flags + + +def _validate_container_insights_settings_common(namespace): + """Validations for the containerInsights tuning flags that do not depend on cluster state.""" + if (getattr(namespace, 'enable_prometheus_metrics_scraping', False) and + getattr(namespace, 'disable_prometheus_metrics_scraping', False)): + raise MutuallyExclusiveArgumentError( + "Cannot specify both --enable-prometheus-metrics-scraping and " + "--disable-prometheus-metrics-scraping at the same time." + ) + + syslog_port = getattr(namespace, 'syslog_port', None) + if syslog_port is not None and not 1 <= syslog_port <= 65535: + raise InvalidArgumentValueError( + f"--syslog-port must be a valid TCP port between 1 and 65535, got {syslog_port}." + ) + + flags = _specified_container_insights_setting_flags(namespace) + if flags and getattr(namespace, 'disable_azure_monitor_logs', False): + raise ArgumentUsageError( + f"{', '.join(flags)} cannot be specified with --disable-azure-monitor-logs." + ) + + +def validate_container_insights_settings_for_create(namespace): + """Validate the containerInsights tuning flags for create operations.""" + _validate_container_insights_settings_common(namespace) + + flags = _specified_container_insights_setting_flags(namespace) + if flags and not getattr(namespace, 'enable_azure_monitor_logs', False): + raise ArgumentUsageError( + f"{', '.join(flags)} requires Azure Monitor logs to be enabled. " + "Please add --enable-azure-monitor-logs to your command." + ) + + +def validate_container_insights_settings_for_update(namespace): + """Validate the containerInsights tuning flags for update operations.""" + _validate_container_insights_settings_common(namespace) + # Whether Azure Monitor logs is already enabled on the cluster is only visible once the + # ManagedCluster has been fetched, so that dependency check is deferred to the decorator. + + def validate_nat_gateway_managed_outbound_ipv6_count(namespace): """validate NAT gateway profile managed outbound IPv6 count""" if namespace.nat_gateway_managed_outbound_ipv6_count is not None: diff --git a/src/aks-preview/azext_aks_preview/addonconfiguration.py b/src/aks-preview/azext_aks_preview/addonconfiguration.py index ef2a08dc966..f9fdb107cda 100644 --- a/src/aks-preview/azext_aks_preview/addonconfiguration.py +++ b/src/aks-preview/azext_aks_preview/addonconfiguration.py @@ -62,6 +62,27 @@ ] +def warn_on_legacy_monitoring_auth(enable_msi_auth_for_monitoring, addons): + """Warn when the user explicitly opts into legacy shared key authentication for monitoring. + + The argument level deprecation already fires for any explicit use of + --enable-msi-auth-for-monitoring. This adds the migration pointer for the value that actually + leaves the cluster on shared key authentication, which is also the state that later blocks + --enable-azure-monitor-logs. + """ + if enable_msi_auth_for_monitoring is not False: + return + if "monitoring" not in (addons or ""): + return + logger.warning( + "--enable-msi-auth-for-monitoring false configures Container Insights with legacy shared " + "key authentication. Managed identity authentication is recommended, and is required by " + "'--enable-azure-monitor-logs'. See " + "https://learn.microsoft.com/en-us/azure/azure-monitor/containers/" + "container-insights-authentication?tabs=cli#migrate-to-managed-identity-authentication" + ) + + # pylint: disable=too-many-locals def enable_addons( cmd, @@ -90,6 +111,7 @@ def enable_addons( ampls_resource_id=None, enable_high_log_scale_mode=False, ): + warn_on_legacy_monitoring_auth(enable_msi_auth_for_monitoring, addons) instance = client.get(resource_group_name, name) # this is overwritten by _update_addons(), so the value needs to be recorded here msi_auth = False diff --git a/src/aks-preview/azext_aks_preview/custom.py b/src/aks-preview/azext_aks_preview/custom.py index e14eace7eb1..e64799e323a 100644 --- a/src/aks-preview/azext_aks_preview/custom.py +++ b/src/aks-preview/azext_aks_preview/custom.py @@ -94,6 +94,7 @@ add_ingress_appgw_addon_role_assignment, add_virtual_node_role_assignment, enable_addons, + warn_on_legacy_monitoring_auth, ) from azext_aks_preview.aks_diagnostics import aks_kanalyze_cmd, aks_kollect_cmd @@ -1316,6 +1317,9 @@ def aks_create( data_collection_settings=None, ampls_resource_id=None, enable_high_log_scale_mode=None, + syslog_port=None, + enable_prometheus_metrics_scraping=False, + disable_prometheus_metrics_scraping=False, aci_subnet_name=None, appgw_name=None, appgw_subnet_cidr=None, @@ -1610,6 +1614,9 @@ def aks_update( data_collection_settings=None, enable_high_log_scale_mode=None, ampls_resource_id=None, + syslog_port=None, + enable_prometheus_metrics_scraping=False, + disable_prometheus_metrics_scraping=False, enable_secret_rotation=False, disable_secret_rotation=False, rotation_poll_interval=None, @@ -3728,6 +3735,7 @@ def aks_enable_addons( enable_high_log_scale_mode=None, aks_custom_headers=None, ): + warn_on_legacy_monitoring_auth(enable_msi_auth_for_monitoring, addons) headers = get_aks_custom_headers(aks_custom_headers) instance = client.get(resource_group_name, name) # this is overwritten by _update_addons(), so the value needs to be recorded here diff --git a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py index a9a88b3c8e8..31f233a5673 100644 --- a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py @@ -54,7 +54,9 @@ CONST_ACNS_DATAPATH_ACCELERATION_MODE_NONE, CONST_TRANSIT_ENCRYPTION_TYPE_MTLS, CONST_ADVANCED_NETWORKPOLICIES_L7, - CONST_MONITORING_ADDON_NAME_CAMELCASE, + CONST_CONTAINER_NETWORK_LOGS_ENABLED, + CONST_CONTAINER_NETWORK_LOGS_DISABLED, + CONST_MONITORING_ENABLE_RETINA_NETWORK_FLAGS, ) from azext_aks_preview.azurecontainerstorage._consts import ( CONST_ACSTOR_EXT_INSTALLATION_NAME, @@ -193,6 +195,132 @@ def _get_monitoring_addon_key_from_consts(addon_profiles, addon_consts): ) +def _get_container_insights_profile(mc): + """Return the Azure Monitor Profile containerInsights object, or None.""" + azure_monitor_profile = getattr(mc, "azure_monitor_profile", None) if mc is not None else None + if azure_monitor_profile is None: + return None + return getattr(azure_monitor_profile, "container_insights", None) + + +def _is_container_insights_enabled(mc): + """Whether Azure Monitor logs are enabled through the AMP containerInsights profile.""" + container_insights = _get_container_insights_profile(mc) + return bool(container_insights and container_insights.enabled) + + +def _is_monitoring_enabled_on_mc(mc, addon_consts): + """Whether Azure Monitor logs are enabled through either the AMP profile or the legacy addon.""" + if _is_container_insights_enabled(mc): + return True + addon_profiles = getattr(mc, "addon_profiles", None) if mc is not None else None + if not addon_profiles: + return False + addon_key = _get_monitoring_addon_key_from_consts(addon_profiles, addon_consts) + return bool(addon_profiles.get(addon_key) and addon_profiles[addon_key].enabled) + + +def _apply_container_insights_settings(container_insights, syslog_port, disable_prometheus_scraping): + """Write the optional AMP containerInsights tuning fields, leaving unset ones untouched.""" + if syslog_port is not None: + container_insights.syslog_port = syslog_port + if disable_prometheus_scraping is not None: + container_insights.disable_prometheus_metrics_scraping = disable_prometheus_scraping + + +def _is_container_network_logs_enabled_on_mc(mc, addon_consts): + """Whether container network logs are on, via the AMP profile or the legacy addon config key.""" + container_insights = _get_container_insights_profile(mc) + if container_insights and str(container_insights.container_network_logs or "").lower() == \ + CONST_CONTAINER_NETWORK_LOGS_ENABLED.lower(): + return True + addon_profiles = getattr(mc, "addon_profiles", None) if mc is not None else None + if not addon_profiles: + return False + addon_key = _get_monitoring_addon_key_from_consts(addon_profiles, addon_consts) + addon_profile = addon_profiles.get(addon_key) + config = (addon_profile.config or {}) if addon_profile else {} + return str(config.get(CONST_MONITORING_ENABLE_RETINA_NETWORK_FLAGS, "")).lower() == "true" + + +def _get_addon_config_value(config, key): + """Case-insensitive lookup of an addon config value. + + ARM echoes addon config keys back in whatever casing they were written with, and the RP + reads them case-insensitively (``GetChangedAddonConfigValue``), so match that here. + """ + if not config or not key: + return None + if key in config: + return config[key] + lowered = key.lower() + for existing_key, value in config.items(): + if existing_key.lower() == lowered: + return value + return None + + +def _get_monitoring_addon_profile(cluster, addon_consts): + """Return the omsagent addon profile object for a cluster, or None.""" + addon_profiles = getattr(cluster, "addon_profiles", None) if cluster is not None else None + if not addon_profiles: + return None + addon_key = _get_monitoring_addon_key_from_consts(addon_profiles, addon_consts) + return addon_profiles.get(addon_key) + + +def _is_monitoring_aad_auth(cluster, addon_consts): + """Whether Azure Monitor logs uses managed identity (AAD) auth on this cluster. + + The AMP containerInsights profile carries no auth information, and the RP mirrors a legacy + addon into it (``overwriteAzureMonitorProfileWithOMSAgentAddonProfile``), so a cluster + onboarded with shared-key auth still ends up with a containerInsights profile. The presence + of that profile is therefore inconclusive, and the omsagent addon config is the only + authoritative source of the auth mode. + + Mirror the RP's derivation exactly: + + * no omsagent addon at all -> fresh onboarding, which always defaults to AAD auth + * addon present but disabled -> a re-enable, which the RP also treats as fresh onboarding + * otherwise -> the addon's ``useAADAuth`` value, where absent or empty means legacy auth + """ + addon_profile = _get_monitoring_addon_profile(cluster, addon_consts) + if addon_profile is None or not addon_profile.enabled: + return True + msi_auth_key = addon_consts.get("CONST_MONITORING_USING_AAD_MSI_AUTH") + return str(_get_addon_config_value(addon_profile.config, msi_auth_key) or "").lower() == "true" + + +def _build_monitoring_addon_shim(cluster, models, addon_consts): + """Build the object that drives DCR/DCE/DCRA/AMPLS provisioning. + + ``ensure_container_insights_for_monitoring`` reads only ``enabled`` and the workspace id + out of ``config``. Synthesizing those from the AMP containerInsights profile lets the + provisioning run without the legacy omsagent addon being present, while the fallback keeps + clusters onboarded before the AMP switch working unchanged. + + The auth mode is derived separately via :func:`_is_monitoring_aad_auth` rather than assumed + from the AMP profile, because the RP mirrors legacy shared-key clusters into that profile. + """ + workspace_key = addon_consts.get("CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID") + msi_auth_key = addon_consts.get("CONST_MONITORING_USING_AAD_MSI_AUTH") + + container_insights = _get_container_insights_profile(cluster) + if container_insights and container_insights.enabled and container_insights.log_analytics_workspace_resource_id: + return models.ManagedClusterAddonProfile( + enabled=True, + config={ + workspace_key: container_insights.log_analytics_workspace_resource_id, + msi_auth_key: "true" if _is_monitoring_aad_auth(cluster, addon_consts) else "false", + }, + ) + + addon_profile = _get_monitoring_addon_profile(cluster, addon_consts) + if addon_profile is not None: + return addon_profile + return None + + # pylint: disable=too-few-public-methods class AKSPreviewManagedClusterModels(AKSManagedClusterModels): """Store the models used in aks series of commands. @@ -1126,24 +1254,18 @@ def get_container_network_logs(self, mc: ManagedCluster) -> Union[bool, None]: (mc.network_profile and mc.network_profile.advanced_networking and mc.network_profile.advanced_networking.enabled) ) - # Check for monitoring addon - either being enabled via raw params or already enabled on mc + # Check for monitoring - either being enabled via raw params or already enabled on mc enable_addons = self.raw_param.get("enable_addons") or "" monitoring_being_enabled = ( "monitoring" in enable_addons or bool(self.raw_param.get("enable_azure_monitor_logs")) ) - monitoring_already_enabled = False - if mc.addon_profiles: - addon_consts = self.get_addon_consts() - mk = _get_monitoring_addon_key_from_consts(mc.addon_profiles, addon_consts) - monitoring_already_enabled = bool( - mc.addon_profiles.get(mk) and mc.addon_profiles[mk].enabled - ) + monitoring_already_enabled = _is_monitoring_enabled_on_mc(mc, self.get_addon_consts()) monitoring_enabled = monitoring_being_enabled or monitoring_already_enabled if not acns_enabled or not monitoring_enabled: raise InvalidArgumentValueError( "Container network logs requires '--enable-acns', advanced networking " - "to be enabled, and the monitoring addon to be enabled." + "to be enabled, and Azure Monitor logs to be enabled." ) enable_cnl = bool(enable_cnl) if enable_cnl is not None else False disable_cnl = bool(disable_cnl) if disable_cnl is not None else False @@ -2935,10 +3057,29 @@ def _get_enable_azure_monitor_logs(self, enable_validation: bool = False) -> boo """ # Read the original value passed by the command. enable_azure_monitor_logs = self.raw_param.get("enable_azure_monitor_logs") - if enable_validation and enable_azure_monitor_logs and self._get_disable_azure_monitor_logs(False): - raise MutuallyExclusiveArgumentError( - "Cannot specify --enable-azure-monitor-logs and --disable-azure-monitor-logs at the same time." + if enable_validation and enable_azure_monitor_logs: + if self._get_disable_azure_monitor_logs(False): + raise MutuallyExclusiveArgumentError( + "Cannot specify --enable-azure-monitor-logs and --disable-azure-monitor-logs at the same time." + ) + # The Azure Monitor Profile is managed-identity only, so the legacy auth flag is + # meaningless on this path. On update the parameter defaults to None, so any value is an + # explicit request. On create it defaults to True, which is indistinguishable from the + # user passing it; only an explicit False is detectable there, and that is the case that + # actually conflicts because it asks for shared-key auth. + enable_msi_auth = self.raw_param.get("enable_msi_auth_for_monitoring") + explicitly_requested = ( + enable_msi_auth is not None + if self.decorator_mode == DecoratorMode.UPDATE + else enable_msi_auth is False ) + if explicitly_requested: + raise MutuallyExclusiveArgumentError( + "Cannot specify --enable-msi-auth-for-monitoring with --enable-azure-monitor-logs. " + "Azure Monitor logs always uses managed identity authentication, so the flag has no " + "effect. Remove --enable-msi-auth-for-monitoring, or use '--enable-addons monitoring' " + "if you need to control the authentication mode." + ) return enable_azure_monitor_logs @@ -2973,6 +3114,69 @@ def get_disable_azure_monitor_logs(self) -> bool: """ return self._get_disable_azure_monitor_logs(enable_validation=True) + def _validate_container_insights_setting(self, flag_name: str) -> None: + """Validate that an AMP containerInsights setting can be applied by this command. + + These settings live only on the Azure Monitor Profile, so they need the profile to be + turned on by this command or, on update, to be on already. + """ + if self._get_disable_azure_monitor_logs(False): + raise InvalidArgumentValueError( + f"{flag_name} cannot be specified when --disable-azure-monitor-logs is used." + ) + if self._get_enable_azure_monitor_logs(False): + return + if self.decorator_mode == DecoratorMode.UPDATE: + if _is_container_insights_enabled(self.mc): + return + raise InvalidArgumentValueError( + f"{flag_name} can only be specified when --enable-azure-monitor-logs is also " + "specified or Azure Monitor logs is already enabled on the cluster." + ) + raise InvalidArgumentValueError( + f"{flag_name} can only be specified when --enable-azure-monitor-logs is also specified." + ) + + def get_syslog_port(self) -> Union[int, None]: + """Obtain the value of syslog_port. + + Returns None when the flag is omitted, which leaves the server default (28330) in place. + :return: int or None + """ + syslog_port = self.raw_param.get("syslog_port") + if syslog_port is None: + return None + + if syslog_port < 1 or syslog_port > 65535: + raise InvalidArgumentValueError( + "--syslog-port must be a valid TCP port between 1 and 65535." + ) + self._validate_container_insights_setting("--syslog-port") + return syslog_port + + def get_disable_prometheus_metrics_scraping(self) -> Union[bool, None]: + """Obtain the value to write to containerInsights.disablePrometheusMetricsScraping. + + Returns None when neither flag is given, so the field is left untouched. + :return: bool or None + """ + enable_scraping = self.raw_param.get("enable_prometheus_metrics_scraping") + disable_scraping = self.raw_param.get("disable_prometheus_metrics_scraping") + + if enable_scraping and disable_scraping: + raise MutuallyExclusiveArgumentError( + "Cannot specify --enable-prometheus-metrics-scraping and " + "--disable-prometheus-metrics-scraping at the same time." + ) + if not enable_scraping and not disable_scraping: + return None + + self._validate_container_insights_setting( + "--disable-prometheus-metrics-scraping" if disable_scraping + else "--enable-prometheus-metrics-scraping" + ) + return bool(disable_scraping) + # OpenTelemetry methods def _get_enable_opentelemetry_metrics(self, enable_validation: bool = False) -> bool: """Internal function to obtain the value of enable_opentelemetry_metrics. @@ -3172,29 +3376,12 @@ def _get_enable_opentelemetry_logs(self, enable_validation: bool = False) -> boo # Check if Azure Monitor logs is being enabled in this command enable_azure_monitor_logs_in_command = self.raw_param.get("enable_azure_monitor_logs") - # Check if Azure Monitor logs is currently enabled in the cluster - # This can be in two places: - # 1. New API: azureMonitorProfile.containerInsights.enabled - # 2. Legacy: addonProfiles.omsagent.enabled (or omsAgent with camelCase) - addon_consts = self.get_addon_consts() - - # Check new API location - container_insights_enabled = ( - self.mc.azure_monitor_profile and - self.mc.azure_monitor_profile.container_insights and - self.mc.azure_monitor_profile.container_insights.enabled + # Azure Monitor logs may be on through the AMP containerInsights profile or the + # legacy omsagent addon. + monitoring_addon_currently_enabled = _is_monitoring_enabled_on_mc( + self.mc, self.get_addon_consts() ) - # Check legacy addon location - monitoring_addon_enabled = False - if self.mc.addon_profiles: - addon_consts = self.get_addon_consts() - mk = _get_monitoring_addon_key_from_consts(self.mc.addon_profiles, addon_consts) - if mk in self.mc.addon_profiles: - monitoring_addon_enabled = self.mc.addon_profiles[mk].enabled - - monitoring_addon_currently_enabled = container_insights_enabled or monitoring_addon_enabled - # Allow OpenTelemetry logs if monitoring is either: # 1. Currently enabled in the cluster (via new API or legacy addon), OR # 2. Being enabled in this command @@ -3372,7 +3559,7 @@ def get_enable_high_log_scale_mode(self) -> Union[bool, None]: "Please add --enable-acns to your command." ) - # Validate that monitoring addon is enabled (either being enabled now or already enabled in cluster) + # Validate that monitoring is enabled (either being enabled now or already in the cluster) addon_consts = self.get_addon_consts() # Check if monitoring is being enabled in the command @@ -3382,16 +3569,12 @@ def get_enable_high_log_scale_mode(self) -> Union[bool, None]: # Check if enabling Azure Monitor logs enable_azure_monitor_logs = self.raw_param.get("enable_azure_monitor_logs") - # Check if monitoring addon is already enabled in the cluster - monitoring_addon_enabled = False - if self.mc and self.mc.addon_profiles: - mk = _get_monitoring_addon_key_from_consts(self.mc.addon_profiles, addon_consts) - if mk in self.mc.addon_profiles: - monitoring_addon_enabled = self.mc.addon_profiles[mk].enabled + # Check if monitoring is already enabled on the cluster, via the AMP profile or the addon + monitoring_addon_enabled = _is_monitoring_enabled_on_mc(self.mc, addon_consts) if not monitoring_being_enabled and not enable_azure_monitor_logs and not monitoring_addon_enabled: raise RequiredArgumentMissingError( - "Container network logs with high log scale mode requires the monitoring addon to be enabled. " + "Container network logs with high log scale mode requires Azure Monitor logs to be enabled. " "Please add '--enable-addons monitoring' or '--enable-azure-monitor-logs' to your command." ) @@ -3400,16 +3583,7 @@ def get_enable_high_log_scale_mode(self) -> Union[bool, None]: # If user explicitly disables HLSM, check if CNL is already enabled on the cluster if enable_high_log_scale_mode is False: - cnl_already_enabled = False - if self.mc and self.mc.addon_profiles: - addon_consts = self.get_addon_consts() - mk = _get_monitoring_addon_key_from_consts(self.mc.addon_profiles, addon_consts) - monitoring_profile = self.mc.addon_profiles.get(mk) - if monitoring_profile and monitoring_profile.config: - cnl_already_enabled = str( - monitoring_profile.config.get("enableRetinaNetworkFlags", "") - ).lower() == "true" - if cnl_already_enabled: + if _is_container_network_logs_enabled_on_mc(self.mc, self.get_addon_consts()): raise MutuallyExclusiveArgumentError( "Cannot explicitly disable --enable-high-log-scale-mode while " "container network logs are enabled on the cluster. " @@ -5169,6 +5343,15 @@ def _ensure_azure_monitor_profile(self, mc: ManagedCluster) -> None: if mc.azure_monitor_profile is None: mc.azure_monitor_profile = self.models.ManagedClusterAzureMonitorProfile() + def _ensure_container_insights(self, mc: ManagedCluster): + """Ensure the AMP containerInsights profile exists and return it.""" + self._ensure_azure_monitor_profile(mc) + if mc.azure_monitor_profile.container_insights is None: + mc.azure_monitor_profile.container_insights = ( + self.models.ManagedClusterAzureMonitorProfileContainerInsights() + ) + return mc.azure_monitor_profile.container_insights + def _ensure_app_monitoring_profile(self, mc: ManagedCluster) -> None: """Ensure app monitoring profile exists on the managed cluster.""" self._ensure_azure_monitor_profile(mc) @@ -5220,17 +5403,6 @@ def _setup_azure_monitor_app_monitoring(self, mc: ManagedCluster) -> None: def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: """Set up Azure Monitor logs configuration.""" - addon_consts = self.context.get_addon_consts() - - if mc.addon_profiles is None: - mc.addon_profiles = {} - - CONST_MONITORING_ADDON_NAME = addon_consts.get("CONST_MONITORING_ADDON_NAME") - addon_profile = mc.addon_profiles.get( - CONST_MONITORING_ADDON_NAME, - self.models.ManagedClusterAddonProfile(enabled=False)) - addon_profile.enabled = True - # Get or create workspace resource ID workspace_resource_id = self.context.raw_param.get("workspace_resource_id") if not workspace_resource_id: @@ -5246,28 +5418,28 @@ def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: sanitize_func = self.context.external_functions.sanitize_loganalytics_ws_resource_id workspace_resource_id = sanitize_func(workspace_resource_id) - CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID = addon_consts.get( - "CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID") - CONST_MONITORING_USING_AAD_MSI_AUTH = addon_consts.get("CONST_MONITORING_USING_AAD_MSI_AUTH") - - # Get MSI auth setting using the same logic as update decorator - enable_msi_auth_bool = self.context.get_enable_msi_auth_for_monitoring() + # Write the Azure Monitor Profile rather than the legacy omsagent addon. The AMP path is + # managed-identity only, so no auth mode is recorded here. + container_insights = self._ensure_container_insights(mc) + container_insights.enabled = True + container_insights.log_analytics_workspace_resource_id = workspace_resource_id - if enable_msi_auth_bool: - enable_msi_auth = "true" - else: - enable_msi_auth = "false" + container_network_logs_enabled = self.context.get_container_network_logs(mc) + if container_network_logs_enabled is not None: + container_insights.container_network_logs = ( + CONST_CONTAINER_NETWORK_LOGS_ENABLED + if container_network_logs_enabled + else CONST_CONTAINER_NETWORK_LOGS_DISABLED + ) - # Create completely new config - addon_profile.config = { - CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID: workspace_resource_id, - CONST_MONITORING_USING_AAD_MSI_AUTH: enable_msi_auth - } - mc.addon_profiles[CONST_MONITORING_ADDON_NAME] = addon_profile + _apply_container_insights_settings( + container_insights, + self.context.get_syslog_port(), + self.context.get_disable_prometheus_metrics_scraping(), + ) # DCR and DCRA creation is deferred to postprocessing_after_mc_created # so that all flags are finalized and the cluster exists. - # Only MSI clusters need a DCR. self.context.set_intermediate("monitoring_addon_enabled", True, overwrite_exists=True) def _setup_opentelemetry_metrics(self, mc: ManagedCluster) -> None: @@ -6100,8 +6272,10 @@ def postprocessing_after_mc_created(self, cluster: ManagedCluster) -> None: # monitoring addon monitoring_addon_enabled = self.context.get_intermediate("monitoring_addon_enabled", default_value=False) if monitoring_addon_enabled: - enable_msi_auth_for_monitoring = self.context.get_enable_msi_auth_for_monitoring() - if not enable_msi_auth_for_monitoring: + # On create there is no pre-existing addon, so the requested auth mode is authoritative. + # get_enable_msi_auth_for_monitoring already returns True for --enable-azure-monitor-logs. + aad_route = self.context.get_enable_msi_auth_for_monitoring() + if not aad_route: # add cluster spn/msi Monitoring Metrics Publisher role assignment to publish metrics to MDM # mdm metrics is supported only in azure public cloud, so add the role assignment only in this cloud cloud_name = self.cmd.cli_ctx.cloud.name @@ -6118,28 +6292,25 @@ def postprocessing_after_mc_created(self, cluster: ManagedCluster) -> None: ) elif self._should_create_dcra(): addon_consts = self.context.get_addon_consts() - monitoring_addon_key = ( - _get_monitoring_addon_key_from_consts(cluster.addon_profiles, addon_consts) - if cluster.addon_profiles - else addon_consts.get("CONST_MONITORING_ADDON_NAME") - ) - self.context.external_functions.ensure_container_insights_for_monitoring( - self.cmd, - cluster.addon_profiles[monitoring_addon_key], - self.context.get_subscription_id(), - self.context.get_resource_group_name(), - self.context.get_name(), - self.context.get_location(), - remove_monitoring=False, - aad_route=self.context.get_enable_msi_auth_for_monitoring(), - create_dcr=True, - create_dcra=True, - enable_syslog=self.context.get_enable_syslog(), - data_collection_settings=self.context.get_data_collection_settings(), - is_private_cluster=self.context.get_enable_private_cluster(), - ampls_resource_id=self.context.get_ampls_resource_id(), - enable_high_log_scale_mode=self.context.get_enable_high_log_scale_mode(), - ) + monitoring_profile = _build_monitoring_addon_shim(cluster, self.models, addon_consts) + if monitoring_profile: + self.context.external_functions.ensure_container_insights_for_monitoring( + self.cmd, + monitoring_profile, + self.context.get_subscription_id(), + self.context.get_resource_group_name(), + self.context.get_name(), + self.context.get_location(), + remove_monitoring=False, + aad_route=self.context.get_enable_msi_auth_for_monitoring(), + create_dcr=True, + create_dcra=True, + enable_syslog=self.context.get_enable_syslog(), + data_collection_settings=self.context.get_data_collection_settings(), + is_private_cluster=self.context.get_enable_private_cluster(), + ampls_resource_id=self.context.get_ampls_resource_id(), + enable_high_log_scale_mode=self.context.get_enable_high_log_scale_mode(), + ) # Handle monitoring addon postprocessing (disable case) - same logic as aks_disable_addons monitoring_addon_disable_postprocessing_required = self.context.get_intermediate( @@ -6148,26 +6319,22 @@ def postprocessing_after_mc_created(self, cluster: ManagedCluster) -> None: if monitoring_addon_disable_postprocessing_required: addon_consts = self.context.get_addon_consts() - CONST_MONITORING_ADDON_NAME = addon_consts.get("CONST_MONITORING_ADDON_NAME") # Get the current cluster state to check config before it was disabled current_cluster = self.client.get(self.context.get_resource_group_name(), self.context.get_name()) + monitoring_profile = _build_monitoring_addon_shim(current_cluster, self.models, addon_consts) - if (current_cluster.addon_profiles and - CONST_MONITORING_ADDON_NAME in current_cluster.addon_profiles): - - addon_profile = current_cluster.addon_profiles[CONST_MONITORING_ADDON_NAME] - + if monitoring_profile: try: self.context.external_functions.ensure_container_insights_for_monitoring( self.cmd, - addon_profile, + monitoring_profile, self.context.get_subscription_id(), self.context.get_resource_group_name(), self.context.get_name(), self.context.get_location(), remove_monitoring=True, - aad_route=True, + aad_route=_is_monitoring_aad_auth(current_cluster, addon_consts), create_dcr=False, create_dcra=True, enable_syslog=False, @@ -6654,14 +6821,14 @@ def update_monitoring_profile_flow_logs(self, mc: ManagedCluster) -> ManagedClus container_network_logs_enabled = self.context.get_container_network_logs(mc) if container_network_logs_enabled is not None: - if mc.addon_profiles: - addon_consts = self.context.get_addon_consts() - monitoring_addon_key = _get_monitoring_addon_key_from_consts(mc.addon_profiles, addon_consts) - monitoring_addon_profile = mc.addon_profiles.get(monitoring_addon_key) - if monitoring_addon_profile: - config = monitoring_addon_profile.config or {} - config["enableRetinaNetworkFlags"] = str(container_network_logs_enabled) - mc.addon_profiles[monitoring_addon_key].config = config + # Written on the AMP profile rather than the legacy omsagent config key. This runs in + # addition to _setup_azure_monitor_logs because either may execute first depending on + # the order the base class invokes them; both write the same value. + self._ensure_container_insights(mc).container_network_logs = ( + CONST_CONTAINER_NETWORK_LOGS_ENABLED + if container_network_logs_enabled + else CONST_CONTAINER_NETWORK_LOGS_DISABLED + ) # When enabling CNL, the DCR must be updated to add the high-scale stream. # Set the postprocessing intermediate so that the update path calls ensure_container_insights. @@ -6682,29 +6849,21 @@ def update_monitoring_profile_flow_logs(self, mc: ManagedCluster) -> ManagedClus ) if not monitoring_being_enabled: - # Only validate existing addon state when not enabling monitoring simultaneously + # Only validate existing state when not enabling monitoring simultaneously. addon_consts = self.context.get_addon_consts() - CONST_MONITORING_USING_AAD_MSI_AUTH = addon_consts.get("CONST_MONITORING_USING_AAD_MSI_AUTH") - - # Resolve the addon profile, normalizing non-standard key casing. - monitoring_addon_profile = None - if mc.addon_profiles: - mk = _get_monitoring_addon_key_from_consts(mc.addon_profiles, addon_consts) - monitoring_addon_profile = mc.addon_profiles.get(mk) - if not monitoring_addon_profile or not monitoring_addon_profile.enabled: + if not _is_monitoring_enabled_on_mc(mc, addon_consts): raise RequiredArgumentMissingError( - "--enable-high-log-scale-mode requires the Azure Monitor logs addon (omsagent) " - "to be enabled on the cluster. Please enable it first with " - "--enable-addons monitoring or --enable-azure-monitor-logs." + "--enable-high-log-scale-mode requires Azure Monitor logs to be enabled on the " + "cluster. Please enable it first with --enable-azure-monitor-logs or " + "--enable-addons monitoring." ) - addon_config = monitoring_addon_profile.config or {} - msi_auth_enabled = ( - CONST_MONITORING_USING_AAD_MSI_AUTH in addon_config and - str(addon_config[CONST_MONITORING_USING_AAD_MSI_AUTH]).lower() == "true" - ) - if not msi_auth_enabled: + # High log scale mode needs a DCR, which only exists for managed-identity auth. + # The auth mode must be read off the omsagent addon: the RP mirrors legacy + # shared-key clusters into the AMP profile, so an enabled containerInsights + # profile does not by itself mean the agent authenticates with managed identity. + if not _is_monitoring_aad_auth(mc, addon_consts): raise RequiredArgumentMissingError( "--enable-high-log-scale-mode requires MSI authentication to be enabled " "for the monitoring addon. Please enable it with --enable-msi-auth-for-monitoring." @@ -6714,16 +6873,7 @@ def update_monitoring_profile_flow_logs(self, mc: ManagedCluster) -> ManagedClus elif enable_high_log_scale_mode is False: # Check if CNL is already enabled on the cluster — cannot disable HLSM while CNL is on - cnl_already_enabled = False - if mc.addon_profiles: - addon_consts = self.context.get_addon_consts() - mk = _get_monitoring_addon_key_from_consts(mc.addon_profiles, addon_consts) - monitoring_profile = mc.addon_profiles.get(mk) - if monitoring_profile and monitoring_profile.config: - cnl_already_enabled = str( - monitoring_profile.config.get("enableRetinaNetworkFlags", "") - ).lower() == "true" - if cnl_already_enabled: + if _is_container_network_logs_enabled_on_mc(mc, self.context.get_addon_consts()): raise MutuallyExclusiveArgumentError( "Cannot explicitly disable --enable-high-log-scale-mode while " "container network logs are enabled on the cluster. " @@ -8743,17 +8893,36 @@ def _ensure_azure_monitor_profile(self, mc: ManagedCluster) -> None: if mc.azure_monitor_profile is None: mc.azure_monitor_profile = self.models.ManagedClusterAzureMonitorProfile() + def _ensure_container_insights(self, mc: ManagedCluster): + """Ensure the AMP containerInsights profile exists and return it.""" + self._ensure_azure_monitor_profile(mc) + if mc.azure_monitor_profile.container_insights is None: + mc.azure_monitor_profile.container_insights = ( + self.models.ManagedClusterAzureMonitorProfileContainerInsights() + ) + return mc.azure_monitor_profile.container_insights + def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: """Set up Azure Monitor logs configuration.""" addon_consts = self.context.get_addon_consts() - if mc.addon_profiles is None: - mc.addon_profiles = {} - - CONST_MONITORING_ADDON_NAME = addon_consts.get("CONST_MONITORING_ADDON_NAME") CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID = addon_consts.get( "CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID") - CONST_MONITORING_USING_AAD_MSI_AUTH = addon_consts.get("CONST_MONITORING_USING_AAD_MSI_AUTH") + + # --enable-azure-monitor-logs onboards through the Azure Monitor Profile, which is managed + # identity only. A cluster already onboarded with legacy (shared key) authentication keeps + # that authentication mode on the server side, so this flag cannot be honoured as asked. + # Reject it up front, before a default workspace is created, rather than silently leaving + # the cluster on legacy auth. + if not _is_monitoring_aad_auth(mc, addon_consts): + raise ArgumentUsageError( + "Azure Monitor logs is already enabled on this cluster using legacy " + "(non-managed-identity) authentication. '--enable-azure-monitor-logs' requires " + "managed identity authentication. Migrate the cluster to managed identity " + "authentication first, then retry. See " + "https://learn.microsoft.com/en-us/azure/azure-monitor/containers/" + "container-insights-authentication?tabs=cli#migrate-to-managed-identity-authentication" + ) # Get or create workspace resource ID workspace_resource_id = self.context.raw_param.get("workspace_resource_id") @@ -8770,74 +8939,55 @@ def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: sanitize_func = self.context.external_functions.sanitize_loganalytics_ws_resource_id workspace_resource_id = sanitize_func(workspace_resource_id) - # Call get_enable_msi_auth_for_monitoring BEFORE detecting the existing key, - # because the parent's implementation may normalize addon_profiles keys in-place - # (e.g., renaming "omsAgent" to "omsagent"). - enable_msi_auth_bool = self.context.get_enable_msi_auth_for_monitoring() - if enable_msi_auth_bool: - enable_msi_auth = "true" - else: - enable_msi_auth = "false" - - # Detect existing key (could be "omsagent" or "omsAgent" from Azure API) - existing_key = None - if CONST_MONITORING_ADDON_NAME in mc.addon_profiles: - existing_key = CONST_MONITORING_ADDON_NAME - elif CONST_MONITORING_ADDON_NAME_CAMELCASE in mc.addon_profiles: - existing_key = CONST_MONITORING_ADDON_NAME_CAMELCASE - - if existing_key: - addon_profile = mc.addon_profiles[existing_key] - # Detect workspace change: if the workspace is different from the existing one, - # trigger DCR postprocessing so the DCR destination gets updated. - old_config = addon_profile.config or {} - old_workspace = old_config.get(CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID, "") - if old_workspace and old_workspace.lower() != workspace_resource_id.lower(): - self.context.set_intermediate( - "monitoring_addon_postprocessing_required", True, overwrite_exists=True) - else: - addon_profile = self.models.ManagedClusterAddonProfile(enabled=False) - existing_key = CONST_MONITORING_ADDON_NAME - - addon_profile.enabled = True - - new_config = { - CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID: workspace_resource_id, - CONST_MONITORING_USING_AAD_MSI_AUTH: enable_msi_auth - } - - # Also set enableRetinaNetworkFlags if container network logs are being enabled - # in the same command. This must be done here because update_monitoring_profile_flow_logs - # may run before update_addon_profiles when the base class calls it first. + # Detect a workspace change so the DCR destination gets rewritten in postprocessing. + # The previous workspace may live on the AMP profile or, for clusters onboarded before + # the AMP switch, on the legacy addon. + container_insights = self._ensure_container_insights(mc) + old_workspace = container_insights.log_analytics_workspace_resource_id or "" + if not old_workspace and mc.addon_profiles: + addon_key = _get_monitoring_addon_key_from_consts(mc.addon_profiles, addon_consts) + addon_profile = mc.addon_profiles.get(addon_key) + old_workspace = (addon_profile.config or {}).get( + CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID, "") if addon_profile else "" + if old_workspace and old_workspace.lower() != workspace_resource_id.lower(): + # A legacy-auth cluster cannot reach a different workspace, but that case is already + # rejected above, so reaching here means the cluster uses managed identity and the DCR + # destination simply has to be rewritten in postprocessing. + self.context.set_intermediate( + "monitoring_addon_postprocessing_required", True, overwrite_exists=True) + + # Write the Azure Monitor Profile rather than the legacy omsagent addon. No auth mode is + # recorded here: the RP derives it, defaulting new onboardings (and re-enables of a + # disabled addon) to managed identity, while preserving the existing useAADAuth value on a + # cluster that is already onboarded with legacy shared-key auth. + container_insights.enabled = True + container_insights.log_analytics_workspace_resource_id = workspace_resource_id + + # Container network logs are applied here as well as in update_monitoring_profile_flow_logs, + # because that method may run before this one depending on the order the base class invokes + # them. Both write the same value, so the result is order-independent. container_network_logs_enabled = self.context.get_container_network_logs(mc) if container_network_logs_enabled is not None: - new_config["enableRetinaNetworkFlags"] = str(container_network_logs_enabled) - - # Replace the entire config, not just individual keys - addon_profile.config = new_config + container_insights.container_network_logs = ( + CONST_CONTAINER_NETWORK_LOGS_ENABLED + if container_network_logs_enabled + else CONST_CONTAINER_NETWORK_LOGS_DISABLED + ) - mc.addon_profiles[existing_key] = addon_profile self.context.set_intermediate("monitoring_addon_enabled", True, overwrite_exists=True) def _disable_azure_monitor_logs(self, mc: ManagedCluster) -> None: """Disable Azure Monitor logs configuration.""" addon_consts = self.context.get_addon_consts() - CONST_MONITORING_USING_AAD_MSI_AUTH = addon_consts.get("CONST_MONITORING_USING_AAD_MSI_AUTH") - # Normalize the addon key (handles any casing variant) - addon_key = None - if mc.addon_profiles: - addon_key = _get_monitoring_addon_key_from_consts(mc.addon_profiles, addon_consts) - if addon_key not in mc.addon_profiles: - addon_key = None + # Azure Monitor logs may be on through the AMP profile, or through the legacy addon on + # clusters onboarded before the AMP switch. Absence of the addon must not short-circuit + # the disable. + azure_monitor_logs_enabled = _is_monitoring_enabled_on_mc(mc, addon_consts) - # If the addon profile doesn't exist at all, there's nothing to disable - if not addon_key: + if not azure_monitor_logs_enabled: return - # Check if Azure Monitor logs (monitoring addon) is currently enabled - azure_monitor_logs_enabled = mc.addon_profiles[addon_key].enabled - # Check if OpenTelemetry logs are enabled and prompt for confirmation opentelemetry_logs_enabled = ( mc.azure_monitor_profile and @@ -8854,34 +9004,30 @@ def _disable_azure_monitor_logs(self, mc: ManagedCluster) -> None: if not prompt_y_n(msg, default="n"): raise CLIError("Operation cancelled.") - # Check if MSI auth is enabled - if so, cleanup DCR/DCRA BEFORE disabling (same as aks_disable_addons) - addon_config = mc.addon_profiles[addon_key].config - has_msi_auth_key = addon_config and CONST_MONITORING_USING_AAD_MSI_AUTH in addon_config - msi_auth_enabled = (addon_config and has_msi_auth_key and - str(addon_config[CONST_MONITORING_USING_AAD_MSI_AUTH]).lower() == "true") + # Perform DCR/DCRA cleanup BEFORE disabling (same as aks_disable_addons lines 2796-2822). + # Only MSI-auth clusters have a DCR/DCRA to clean up, so decide from local state first to + # avoid an ARM round trip when there is nothing to do. The auth mode has to come from the + # omsagent addon: the RP mirrors legacy shared-key clusters into the AMP profile, so the + # presence of that profile says nothing about how the agent authenticates. + msi_auth_enabled = _is_monitoring_aad_auth(mc, addon_consts) - # Perform DCR/DCRA cleanup BEFORE disabling (same as aks_disable_addons lines 2796-2822) - if azure_monitor_logs_enabled and msi_auth_enabled: - # Fetch the current cluster state from Azure (same as aks_disable_addons line 2791) + if msi_auth_enabled: + # Fetch the current cluster state from Azure (same as aks_disable_addons line 2791) and + # drive cleanup off whichever profile carries the workspace. current_cluster = self.client.get(self.context.get_resource_group_name(), self.context.get_name()) + monitoring_profile = _build_monitoring_addon_shim(current_cluster, self.models, addon_consts) - # Find the addon key in current_cluster (normalize casing) - current_addon_key = _get_monitoring_addon_key_from_consts( - current_cluster.addon_profiles, addon_consts) if current_cluster.addon_profiles else None - has_addon = current_addon_key and current_addon_key in (current_cluster.addon_profiles or {}) - - if has_addon: + if monitoring_profile and monitoring_profile.enabled: try: - # Use the current cluster's addon profile for cleanup (not the modified mc object) self.context.external_functions.ensure_container_insights_for_monitoring( self.cmd, - current_cluster.addon_profiles[current_addon_key], + monitoring_profile, self.context.get_subscription_id(), self.context.get_resource_group_name(), self.context.get_name(), current_cluster.location, remove_monitoring=True, - aad_route=True, + aad_route=_is_monitoring_aad_auth(current_cluster, addon_consts), create_dcr=False, create_dcra=True, enable_syslog=False, @@ -8893,16 +9039,12 @@ def _disable_azure_monitor_logs(self, mc: ManagedCluster) -> None: # Ignore TypeError just like aks_disable_addons does (line 2823) pass - # Now disable the addon and clear configuration - mc.addon_profiles[addon_key].enabled = False - mc.addon_profiles[addon_key].config = None - - # Also disable azureMonitorProfile.containerInsights (the new API surface) - # The RP uses containerInsights.enabled as the source of truth; if it remains - # true while the legacy addon is disabled, the RP re-enables the addon. - if (mc.azure_monitor_profile and - mc.azure_monitor_profile.container_insights): - mc.azure_monitor_profile.container_insights.enabled = False + # Disable through the AMP profile. The RP keeps the legacy addon in sync, so the addon + # object is intentionally left untouched here. + container_insights = self._ensure_container_insights(mc) + container_insights.enabled = False + # Reset container network logs so a later re-enable does not silently carry CNL forward. + container_insights.container_network_logs = CONST_CONTAINER_NETWORK_LOGS_DISABLED # Also disable OpenTelemetry logs when disabling Azure Monitor logs if opentelemetry_logs_enabled: @@ -8975,6 +9117,27 @@ def update_addon_profiles(self, mc: ManagedCluster) -> ManagedCluster: return mc + def update_azure_monitor_logs_settings(self, mc: ManagedCluster) -> ManagedCluster: + """Update the AMP containerInsights tuning settings for the ManagedCluster object. + + These flags are independent of --enable-azure-monitor-logs, so they also apply to a + cluster where Azure Monitor logs is already enabled. When neither flag is given nothing + is touched, which keeps the rest of the monitoring configuration intact. + + :return: the ManagedCluster object + """ + self._ensure_mc(mc) + + syslog_port = self.context.get_syslog_port() + disable_prometheus_scraping = self.context.get_disable_prometheus_metrics_scraping() + if syslog_port is None and disable_prometheus_scraping is None: + return mc + + _apply_container_insights_settings( + self._ensure_container_insights(mc), syslog_port, disable_prometheus_scraping + ) + return mc + def update_control_plane_scaling_profile(self, mc: ManagedCluster) -> ManagedCluster: """Update the control plane scaling profile for the ManagedCluster object. @@ -9057,6 +9220,8 @@ def update_mc_profile_preview(self) -> ManagedCluster: # so we don't call it again here to avoid duplicate processing # update azure monitor metrics profile mc = self.update_azure_monitor_profile(mc) + # update azure monitor logs (container insights) settings + mc = self.update_azure_monitor_logs_settings(mc) # update vpa mc = self.update_vpa(mc) # update optimized addon scaling @@ -9178,51 +9343,42 @@ def postprocessing_after_mc_created(self, cluster: ManagedCluster) -> None: ) if monitoring_addon_postprocessing_required: addon_consts = self.context.get_addon_consts() - CONST_MONITORING_USING_AAD_MSI_AUTH = addon_consts.get("CONST_MONITORING_USING_AAD_MSI_AUTH") - - monitoring_addon_key = ( - _get_monitoring_addon_key_from_consts(cluster.addon_profiles, addon_consts) - if cluster.addon_profiles - else addon_consts.get("CONST_MONITORING_ADDON_NAME") - ) - if (cluster.addon_profiles and - monitoring_addon_key in cluster.addon_profiles and - cluster.addon_profiles[monitoring_addon_key].enabled): + monitoring_profile = _build_monitoring_addon_shim(cluster, self.models, addon_consts) - # Check if MSI auth is enabled - if (CONST_MONITORING_USING_AAD_MSI_AUTH in - cluster.addon_profiles[monitoring_addon_key].config and - str(cluster.addon_profiles[monitoring_addon_key].config[ - CONST_MONITORING_USING_AAD_MSI_AUTH]).lower() == "true"): + # Only MSI-auth clusters get a DCR. The auth mode is derived from the omsagent addon + # because the RP mirrors legacy shared-key clusters into the AMP profile, so AMP + # presence alone cannot distinguish the two. + msi_auth_enabled = _is_monitoring_aad_auth(cluster, addon_consts) - # Check parameter sizes to identify what might be causing large headers - data_collection_settings = self.context.get_data_collection_settings() + if monitoring_profile and monitoring_profile.enabled and msi_auth_enabled: + # Check parameter sizes to identify what might be causing large headers + data_collection_settings = self.context.get_data_collection_settings() - # Try to limit data_collection_settings size to avoid "Request Header Fields Too Large" error + # Try to limit data_collection_settings size to avoid "Request Header Fields Too Large" error + safe_data_collection_settings = None + if data_collection_settings and len(str(data_collection_settings)) > 10000: safe_data_collection_settings = None - if data_collection_settings and len(str(data_collection_settings)) > 10000: - safe_data_collection_settings = None - else: - safe_data_collection_settings = data_collection_settings + else: + safe_data_collection_settings = data_collection_settings - self.context.external_functions.ensure_container_insights_for_monitoring( - self.cmd, - cluster.addon_profiles[monitoring_addon_key], - self.context.get_subscription_id(), - self.context.get_resource_group_name(), - self.context.get_name(), - self.context.get_location(), - remove_monitoring=False, - aad_route=self.context.get_enable_msi_auth_for_monitoring(), - create_dcr=True, - create_dcra=True, - enable_syslog=self.context.get_enable_syslog(), - data_collection_settings=safe_data_collection_settings, - is_private_cluster=self.context.get_enable_private_cluster(), - ampls_resource_id=self.context.get_ampls_resource_id(), - enable_high_log_scale_mode=self.context.get_enable_high_log_scale_mode(), - ) + self.context.external_functions.ensure_container_insights_for_monitoring( + self.cmd, + monitoring_profile, + self.context.get_subscription_id(), + self.context.get_resource_group_name(), + self.context.get_name(), + self.context.get_location(), + remove_monitoring=False, + aad_route=msi_auth_enabled, + create_dcr=True, + create_dcra=True, + enable_syslog=self.context.get_enable_syslog(), + data_collection_settings=safe_data_collection_settings, + is_private_cluster=self.context.get_enable_private_cluster(), + ampls_resource_id=self.context.get_ampls_resource_id(), + enable_high_log_scale_mode=self.context.get_enable_high_log_scale_mode(), + ) # Monitoring addon disable cleanup is now done upfront in _disable_azure_monitor_logs (not in postprocessing) # This matches the pattern from aks_disable_addons lines 2796-2822 where cleanup happens BEFORE the PUT diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py index 8d2e76a7a08..0577a963fe9 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py @@ -908,5 +908,130 @@ def test_other_error_after_table_readiness_retry_still_bounded(self): self.assertEqual(self.mock_sleep.call_count, 1) +class TestLegacyMonitoringAuthDeprecation(unittest.TestCase): + """R5: warn when the legacy auth flag is used with enable-addons monitoring.""" + + def _warn(self, value, addons="monitoring"): + from azext_aks_preview.addonconfiguration import warn_on_legacy_monitoring_auth + + with patch("azext_aks_preview.addonconfiguration.logger") as mock_logger: + warn_on_legacy_monitoring_auth(value, addons) + return mock_logger.warning + + def test_warns_when_opting_into_shared_key_auth(self): + warning = self._warn(False) + warning.assert_called_once() + message = warning.call_args[0][0] + self.assertIn("managed identity", message.lower()) + self.assertIn("--enable-azure-monitor-logs", message) + self.assertIn("container-insights-authentication", message) + + def test_silent_when_flag_omitted(self): + # The CLI default is True, so an omitted flag must not warn. + self._warn(None).assert_not_called() + self._warn(True).assert_not_called() + + def test_silent_for_non_monitoring_addons(self): + self._warn(False, addons="azure-policy").assert_not_called() + self._warn(False, addons=None).assert_not_called() + + def test_warns_when_monitoring_is_one_of_several_addons(self): + self._warn(False, addons="azure-policy,monitoring").assert_called_once() + + def test_does_not_change_the_auth_value(self): + # The warning is advisory only: useAADAuth handling is untouched. + from azext_aks_preview.addonconfiguration import warn_on_legacy_monitoring_auth + + for value in (True, False, None): + with patch("azext_aks_preview.addonconfiguration.logger"): + self.assertIsNone(warn_on_legacy_monitoring_auth(value, "monitoring")) + + +class TestMonitoringArgumentRegistration(unittest.TestCase): + """Argument registration for R4 (containerInsights controls) and R5 (legacy auth warning).""" + + def setUp(self): + register_aks_preview_resource_type() + + def _arguments(self, command_name): + """Return the argument settings registered for a command scope. + + load_arguments() resolves argument scopes against the command currently being invoked, so + the loader needs an invocation carrying the command string and an argparse action registry. + """ + import argparse + + from azure.cli.core.mock import DummyCli + + class _Invocation: + def __init__(self, command_string): + self.data = {"command_string": command_string} + self.parser = argparse.ArgumentParser() + + cli_ctx = DummyCli() + cli_ctx.invocation = _Invocation(command_name) + loader = ContainerServiceCommandsLoader(cli_ctx) + loader.load_command_table(command_name.split()) + loader.command_table[command_name].load_arguments() + loader.load_arguments(command_name) + return { + dest: arg.settings + for dest, arg in loader.argument_registry.arguments.get(command_name, {}).items() + } + + def test_legacy_auth_flag_is_deprecated_on_addon_commands(self): + for command_name in ("aks enable-addons", "aks addon enable", "aks addon update"): + arguments = self._arguments(command_name) + deprecate_info = arguments["enable_msi_auth_for_monitoring"].get("deprecate_info") + self.assertIsNotNone(deprecate_info, command_name) + message = deprecate_info.message + self.assertIn("deprecated", message) + self.assertIn("managed identity", message) + self.assertIn("--enable-azure-monitor-logs", message) + + def test_legacy_auth_flag_is_not_deprecated_on_create_and_update(self): + # R5 scopes the warning to the legacy addon commands. + for command_name in ("aks create", "aks update"): + arguments = self._arguments(command_name) + self.assertIsNone( + arguments["enable_msi_auth_for_monitoring"].get("deprecate_info"), command_name + ) + + def test_deprecation_fires_only_when_the_flag_is_supplied(self): + # knack invokes a deprecated argument's action only when the option is on the command + # line, so the default value must never produce a warning. + import argparse + + arguments = self._arguments("aks enable-addons") + action_cls = arguments["enable_msi_auth_for_monitoring"].get("action") + parser = argparse.ArgumentParser() + parser.add_argument( + "--enable-msi-auth-for-monitoring", + dest="enable_msi_auth_for_monitoring", + action=action_cls, + default=True, + ) + + namespace = parser.parse_args([]) + self.assertEqual(getattr(namespace, "_argument_deprecations", []), []) + self.assertTrue(namespace.enable_msi_auth_for_monitoring) + + namespace = parser.parse_args(["--enable-msi-auth-for-monitoring", "false"]) + self.assertEqual(len(getattr(namespace, "_argument_deprecations", [])), 1) + # The deprecation machinery warns without altering the parsed value. + self.assertFalse(namespace.enable_msi_auth_for_monitoring) + + def test_container_insights_controls_registered_on_create_and_update(self): + for command_name in ("aks create", "aks update"): + arguments = self._arguments(command_name) + for dest in ( + "syslog_port", + "enable_prometheus_metrics_scraping", + "disable_prometheus_metrics_scraping", + ): + self.assertIn(dest, arguments, f"{dest} missing from {command_name}") + self.assertIsNotNone(arguments["syslog_port"].get("validator"), command_name) + + if __name__ == '__main__': unittest.main() diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py index 230d128a8d9..788c9faba20 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py @@ -74,6 +74,8 @@ AKSPreviewManagedClusterModels, AKSPreviewManagedClusterUpdateDecorator, _get_monitoring_addon_key_from_consts, + _build_monitoring_addon_shim, + _is_monitoring_aad_auth, ) from azext_aks_preview.tests.latest.utils import get_test_data_file_path from azure.cli.command_modules.acs._consts import ( @@ -10060,6 +10062,118 @@ def test_postprocessing_enable_hlsm_with_monitoring_being_enabled_simultaneously self.assertTrue(kwargs["create_dcr"]) self.assertTrue(kwargs["enable_high_log_scale_mode"]) + # ------------------------------------------------------------------ + # Parity: every monitoring side-flag must reach the DCR builder the + # same way whether onboarding used the legacy addon or the AMP profile. + # ------------------------------------------------------------------ + + def _make_cluster_with_amp_container_insights(self): + """Helper: a cluster onboarded through the Azure Monitor Profile, with no addon profile.""" + mc = self.models.ManagedCluster(location="test_location") + mc.azure_monitor_profile = self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + enabled=True, + log_analytics_workspace_resource_id="/test_workspace_resource_id", + ) + ) + return mc + + def _capture_ensure_container_insights(self, raw_params, cluster): + """Run create postprocessing and return the kwargs handed to the DCR builder.""" + dec = self._make_postprocessing_decorator(raw_params) + with patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as mock_ecifm: + dec.postprocessing_after_mc_created(cluster) + mock_ecifm.assert_called_once() + _, kwargs = mock_ecifm.call_args + return kwargs + + def test_monitoring_side_flags_reach_dcr_builder_on_amp_path(self): + """--enable-azure-monitor-logs must carry HLSM, AMPLS and syslog into the DCR builder.""" + kwargs = self._capture_ensure_container_insights( + { + "enable_azure_monitor_logs": True, + "enable_high_log_scale_mode": True, + "ampls_resource_id": "/test_ampls_resource_id", + "enable_syslog": True, + }, + self._make_cluster_with_amp_container_insights(), + ) + self.assertTrue(kwargs["aad_route"]) + self.assertTrue(kwargs["create_dcr"]) + self.assertTrue(kwargs["create_dcra"]) + self.assertTrue(kwargs["enable_high_log_scale_mode"]) + self.assertEqual(kwargs["ampls_resource_id"], "/test_ampls_resource_id") + self.assertTrue(kwargs["enable_syslog"]) + + def test_monitoring_side_flags_reach_dcr_builder_on_legacy_addon_path(self): + """--enable-addons monitoring must carry the same flags through unchanged.""" + kwargs = self._capture_ensure_container_insights( + { + "enable_addons": "monitoring", + "enable_high_log_scale_mode": True, + "ampls_resource_id": "/test_ampls_resource_id", + "enable_syslog": True, + }, + self._make_cluster_with_monitoring(), + ) + self.assertTrue(kwargs["aad_route"]) + self.assertTrue(kwargs["create_dcr"]) + self.assertTrue(kwargs["create_dcra"]) + self.assertTrue(kwargs["enable_high_log_scale_mode"]) + self.assertEqual(kwargs["ampls_resource_id"], "/test_ampls_resource_id") + self.assertTrue(kwargs["enable_syslog"]) + + def test_monitoring_side_flags_identical_across_both_onboarding_paths(self): + """The two onboarding paths must produce the same DCR-builder arguments. + + This is the regression guard against the AMP path drifting away from the legacy + addon path: the DCR/DCRA plumbing is shared, so anything that reaches it for + '--enable-addons monitoring' must also reach it for '--enable-azure-monitor-logs'. + """ + side_flags = { + "enable_high_log_scale_mode": True, + "ampls_resource_id": "/test_ampls_resource_id", + "enable_syslog": True, + } + amp_kwargs = self._capture_ensure_container_insights( + dict(side_flags, enable_azure_monitor_logs=True), + self._make_cluster_with_amp_container_insights(), + ) + legacy_kwargs = self._capture_ensure_container_insights( + dict(side_flags, enable_addons="monitoring"), + self._make_cluster_with_monitoring(), + ) + compared = ( + "aad_route", + "create_dcr", + "create_dcra", + "remove_monitoring", + "enable_high_log_scale_mode", + "ampls_resource_id", + "enable_syslog", + "data_collection_settings", + ) + self.assertEqual( + {k: amp_kwargs[k] for k in compared}, + {k: legacy_kwargs[k] for k in compared}, + ) + + def test_ampls_alone_reaches_dcr_builder_on_amp_path(self): + """--ampls-resource-id must not require HLSM in order to be plumbed through.""" + kwargs = self._capture_ensure_container_insights( + { + "enable_azure_monitor_logs": True, + "ampls_resource_id": "/test_ampls_resource_id", + }, + self._make_cluster_with_amp_container_insights(), + ) + self.assertEqual(kwargs["ampls_resource_id"], "/test_ampls_resource_id") + self.assertTrue(kwargs["create_dcr"]) + def test_set_up_health_monitor_profile(self): # no flag - no change dec_0 = AKSPreviewManagedClusterCreateDecorator( @@ -12908,15 +13022,14 @@ def test_update_enable_azure_monitor_logs(self): ): dec_mc_1 = dec_1.update_addon_profiles(mc_1) - # Verify monitoring addon is enabled - self.assertIn(CONST_MONITORING_ADDON_NAME, dec_mc_1.addon_profiles) - self.assertTrue(dec_mc_1.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) + # Verify Azure Monitor logs is enabled through the AMP containerInsights profile + self.assertTrue(dec_mc_1.azure_monitor_profile.container_insights.enabled) self.assertEqual( - dec_mc_1.addon_profiles[CONST_MONITORING_ADDON_NAME].config[ - CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID - ], + dec_mc_1.azure_monitor_profile.container_insights.log_analytics_workspace_resource_id, "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/test-workspace", ) + # The legacy omsagent addon must not be authored by this flag + self.assertNotIn(CONST_MONITORING_ADDON_NAME, dec_mc_1.addon_profiles or {}) # Test enabling Azure Monitor logs when already enabled (should be idempotent) dec_2 = AKSPreviewManagedClusterUpdateDecorator( @@ -12982,9 +13095,8 @@ def test_update_enable_azure_monitor_logs(self): ): dec_mc_3 = dec_3.update_azure_monitor_profile(dec_mc_3) - # Verify monitoring addon is enabled - self.assertIn(CONST_MONITORING_ADDON_NAME, dec_mc_3.addon_profiles) - self.assertTrue(dec_mc_3.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) + # Verify Azure Monitor logs are enabled through the AMP container insights profile + self.assertTrue(dec_mc_3.azure_monitor_profile.container_insights.enabled) # Verify OpenTelemetry logs are configured if ( @@ -13002,7 +13114,8 @@ def test_update_enable_azure_monitor_logs(self): 8080, ) - # Test with MSI auth enabled + # R2: --enable-msi-auth-for-monitoring is rejected on the Azure Monitor logs path, + # which is managed-identity only. dec_4 = AKSPreviewManagedClusterUpdateDecorator( self.cmd, self.client, @@ -13020,23 +13133,8 @@ def test_update_enable_azure_monitor_logs(self): dec_4.context.attach_mc(mc_4) dec_4.context.set_intermediate("subscription_id", "test-subscription-id") - external_functions = dec_4.context.external_functions - with patch.object( - external_functions, - "ensure_container_insights_for_monitoring", - return_value=None, - ): - dec_mc_4 = dec_4.update_addon_profiles(mc_4) - - # Verify MSI auth is enabled - self.assertIn(CONST_MONITORING_ADDON_NAME, dec_mc_4.addon_profiles) - self.assertTrue(dec_mc_4.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) - self.assertEqual( - dec_mc_4.addon_profiles[CONST_MONITORING_ADDON_NAME].config[ - CONST_MONITORING_USING_AAD_MSI_AUTH - ], - "true", - ) + with self.assertRaises(MutuallyExclusiveArgumentError): + dec_4.update_addon_profiles(mc_4) def test_update_disable_azure_monitor_logs(self): # Test disabling Azure Monitor logs when currently enabled @@ -13072,9 +13170,8 @@ def test_update_disable_azure_monitor_logs(self): ): dec_mc_1 = dec_1.update_addon_profiles(mc_1) - # Verify monitoring addon is disabled - self.assertIn(CONST_MONITORING_ADDON_NAME, dec_mc_1.addon_profiles) - self.assertFalse(dec_mc_1.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) + # Verify Azure Monitor logs is disabled through the AMP containerInsights profile + self.assertFalse(dec_mc_1.azure_monitor_profile.container_insights.enabled) # Test disabling Azure Monitor logs when not enabled (should be idempotent) dec_2 = AKSPreviewManagedClusterUpdateDecorator( @@ -13147,8 +13244,8 @@ def test_update_disable_azure_monitor_logs(self): ): dec_mc_3 = dec_3.update_azure_monitor_profile(dec_mc_3) - # Verify monitoring addon is disabled - self.assertFalse(dec_mc_3.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) + # Verify Azure Monitor logs are disabled through the AMP container insights profile + self.assertFalse(dec_mc_3.azure_monitor_profile.container_insights.enabled) # Verify OpenTelemetry logs are also disabled in Azure Monitor profile if ( @@ -13554,17 +13651,20 @@ def test_setup_azure_monitor_logs_with_omsagent_camelcase(self): # Call _setup_azure_monitor_logs dec_1._setup_azure_monitor_logs(mc_1) - # Verify: The existing key is preserved (no duplicate created). - # The implementation keeps the original casing ("omsAgent") found in addon_profiles. + # Verify: the AMP profile carries the new workspace and the legacy addon is left untouched. + self.assertTrue(mc_1.azure_monitor_profile.container_insights.enabled) self.assertEqual( - len([k for k in mc_1.addon_profiles if k.lower() == "omsagent"]), 1 - ) # No duplicate - # Find the actual key used (could be normalized or preserved depending on parent behavior) + mc_1.azure_monitor_profile.container_insights.log_analytics_workspace_resource_id, + "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/test-workspace", + ) actual_key = next(k for k in mc_1.addon_profiles if k.lower() == "omsagent") - self.assertTrue(mc_1.addon_profiles[actual_key].enabled) self.assertEqual( mc_1.addon_profiles[actual_key].config["logAnalyticsWorkspaceResourceID"], - "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/test-workspace", + "/old/workspace", + ) + # The workspace changed, so DCR postprocessing must be requested + self.assertTrue( + dec_1.context.get_intermediate("monitoring_addon_postprocessing_required") ) def test_setup_azure_monitor_logs_with_omsagent_lowercase(self): @@ -13598,12 +13698,12 @@ def test_setup_azure_monitor_logs_with_omsagent_lowercase(self): # Call _setup_azure_monitor_logs dec_1._setup_azure_monitor_logs(mc_1) - # Verify: Should update existing omsagent key + # Verify: the AMP profile is written and the legacy addon key is left as-is + self.assertTrue(mc_1.azure_monitor_profile.container_insights.enabled) self.assertIn("omsagent", mc_1.addon_profiles) self.assertNotIn( "omsAgent", mc_1.addon_profiles ) # Should NOT create CamelCase variant - self.assertTrue(mc_1.addon_profiles["omsagent"].enabled) def test_disable_azure_monitor_logs_with_omsagent_camelcase(self): # Test that _disable_azure_monitor_logs handles omsAgent (camelCase) correctly @@ -13636,10 +13736,11 @@ def test_disable_azure_monitor_logs_with_omsagent_camelcase(self): # Call _disable_azure_monitor_logs dec_1._disable_azure_monitor_logs(mc_1) - # After normalization, the camelCase key is re-keyed to canonical lowercase - self.assertIn("omsagent", mc_1.addon_profiles) - self.assertFalse(mc_1.addon_profiles["omsagent"].enabled) - self.assertIsNone(mc_1.addon_profiles["omsagent"].config) + # Disabling now happens on the AMP profile; the RP keeps the legacy addon in sync, so the + # addon object is intentionally left untouched. + self.assertFalse(mc_1.azure_monitor_profile.container_insights.enabled) + actual_key = next(k for k in mc_1.addon_profiles if k.lower() == "omsagent") + self.assertTrue(mc_1.addon_profiles[actual_key].enabled) def test_disable_azure_monitor_logs_with_omsagent_lowercase(self): # Test that _disable_azure_monitor_logs handles omsagent (lowercase) correctly @@ -13672,10 +13773,10 @@ def test_disable_azure_monitor_logs_with_omsagent_lowercase(self): # Call _disable_azure_monitor_logs dec_1._disable_azure_monitor_logs(mc_1) - # Verify: omsagent should be disabled + # Verify: Azure Monitor logs is disabled via the AMP profile, addon left untouched + self.assertFalse(mc_1.azure_monitor_profile.container_insights.enabled) self.assertIn("omsagent", mc_1.addon_profiles) - self.assertFalse(mc_1.addon_profiles["omsagent"].enabled) - self.assertIsNone(mc_1.addon_profiles["omsagent"].config) + self.assertTrue(mc_1.addon_profiles["omsagent"].enabled) def test_disable_azure_monitor_logs_disables_container_insights(self): # Test that _disable_azure_monitor_logs disables both addon profile AND @@ -13709,15 +13810,25 @@ def test_disable_azure_monitor_logs_disables_container_insights(self): ), ) dec_1.context.attach_mc(mc_1) + dec_1.client = Mock() + dec_1.client.get = Mock(return_value=mc_1) - dec_1._disable_azure_monitor_logs(mc_1) - - # Verify addon profile is disabled - self.assertFalse(mc_1.addon_profiles["omsagent"].enabled) - self.assertIsNone(mc_1.addon_profiles["omsagent"].config) + with patch.object( + dec_1.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec_1.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec_1.context, "get_name", return_value="test-cluster" + ), patch.object( + dec_1.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec_1._disable_azure_monitor_logs(mc_1) - # Verify container_insights is also disabled + # Disabling is done on the AMP profile; the RP keeps the legacy addon in sync self.assertFalse(mc_1.azure_monitor_profile.container_insights.enabled) + self.assertTrue(mc_1.addon_profiles["omsagent"].enabled) def test_disable_azure_monitor_logs_disables_container_insights_camelcase(self): # Same test but with omsAgent (camelCase) key @@ -13750,14 +13861,23 @@ def test_disable_azure_monitor_logs_disables_container_insights_camelcase(self): ), ) dec_1.context.attach_mc(mc_1) + dec_1.client = Mock() + dec_1.client.get = Mock(return_value=mc_1) - dec_1._disable_azure_monitor_logs(mc_1) - - # After normalization, the camelCase key is re-keyed to the canonical lowercase form - self.assertFalse(mc_1.addon_profiles["omsagent"].enabled) - self.assertIsNone(mc_1.addon_profiles["omsagent"].config) + with patch.object( + dec_1.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec_1.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec_1.context, "get_name", return_value="test-cluster" + ), patch.object( + dec_1.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec_1._disable_azure_monitor_logs(mc_1) - # Verify container_insights is also disabled + # Disabling is done on the AMP profile regardless of the addon key casing self.assertFalse(mc_1.azure_monitor_profile.container_insights.enabled) def test_disable_azure_monitor_logs_without_container_insights(self): @@ -13788,12 +13908,10 @@ def test_disable_azure_monitor_logs_without_container_insights(self): dec_1._disable_azure_monitor_logs(mc_1) - # Verify addon profile is disabled - self.assertFalse(mc_1.addon_profiles["omsagent"].enabled) - self.assertIsNone(mc_1.addon_profiles["omsagent"].config) - - # container_insights was never set, should not error - self.assertIsNone(mc_1.azure_monitor_profile) + # An addon-only cluster is still disabled through the AMP profile, which is created on + # demand. The RP keeps the legacy addon in sync, so it is left untouched. + self.assertTrue(mc_1.addon_profiles["omsagent"].enabled) + self.assertFalse(mc_1.azure_monitor_profile.container_insights.enabled) def test_get_enable_opentelemetry_logs_validation_with_omsagent_camelcase(self): # Test that OpenTelemetry logs validation recognizes omsAgent (camelCase) as enabled @@ -15846,9 +15964,14 @@ def test_enable_container_network_logs(self): ), addon_profiles={ "omsagent": self.models.ManagedClusterAddonProfile( - enabled=True, config={"enableRetinaNetworkFlags": "True"} + enabled=True, ) }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + container_network_logs="Enabled", + ) + ), ) self.assertEqual(dec_mc_1, ground_truth_mc_1) # Verify HLSM is auto-enabled when CNL is enabled @@ -15897,9 +16020,14 @@ def test_enable_container_network_logs(self): ), addon_profiles={ "omsagent": self.models.ManagedClusterAddonProfile( - enabled=True, config={"enableRetinaNetworkFlags": "False"} + enabled=True, config={"enableRetinaNetworkFlags": "True"} ) }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + container_network_logs="Disabled", + ) + ), ) self.assertEqual(dec_mc_2, ground_truth_mc_2) @@ -16105,6 +16233,11 @@ def test_enable_container_network_logs(self): enabled=True, config={"enableRetinaNetworkFlags": "True"} ) }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + container_network_logs="Enabled", + ) + ), ) self.assertEqual(dec_mc_7, ground_truth_mc_7) # Verify HLSM is auto-enabled when using deprecated flag @@ -16230,19 +16363,16 @@ def test_enable_container_network_logs(self): return_value=None, ): dec_mc_11 = dec_11.set_up_addon_profiles(mc_11) - ground_truth_mc_11 = { - CONST_MONITORING_ADDON_NAME: self.models.ManagedClusterAddonProfile( - enabled=True, - config={ - CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID: "/test_workspace_resource_id", - CONST_MONITORING_USING_AAD_MSI_AUTH: "true", - "enableRetinaNetworkFlags": "True", - }, - ), - } + ground_truth_mc_11 = self.models.ManagedClusterAzureMonitorProfileContainerInsights( + enabled=True, + log_analytics_workspace_resource_id="/test_workspace_resource_id", + container_network_logs="Enabled", + ) self.assertEqual( - dec_mc_11.addon_profiles["omsagent"], ground_truth_mc_11["omsagent"] + dec_mc_11.azure_monitor_profile.container_insights, ground_truth_mc_11 ) + # the legacy addon is no longer written on the AMP path + self.assertNotIn("omsagent", dec_mc_11.addon_profiles or {}) # Case 12: Verify monitoring_addon_postprocessing_required is set when CNL is enabled (update path) # This test verifies the fix for the bug where DCR is not updated when enabling CNL on update @@ -16294,9 +16424,14 @@ def test_enable_container_network_logs(self): ), addon_profiles={ "omsagent": self.models.ManagedClusterAddonProfile( - enabled=True, config={"enableRetinaNetworkFlags": "True"} + enabled=True, ) }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + container_network_logs="Enabled", + ) + ), ) self.assertEqual(dec_mc_12, ground_truth_mc_12) @@ -16350,9 +16485,14 @@ def test_enable_container_network_logs(self): ), addon_profiles={ "omsagent": self.models.ManagedClusterAddonProfile( - enabled=True, config={"enableRetinaNetworkFlags": "False"} + enabled=True, config={"enableRetinaNetworkFlags": "True"} ) }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + container_network_logs="Disabled", + ) + ), ) self.assertEqual(dec_mc_13, ground_truth_mc_13) @@ -16400,10 +16540,15 @@ def test_enable_container_network_logs(self): ), ), addon_profiles={ - "omsagent": self.models.ManagedClusterAddonProfile( - enabled=True, config={"enableRetinaNetworkFlags": "False"} + "omsAgent": self.models.ManagedClusterAddonProfile( + enabled=True, config={"enableRetinaNetworkFlags": "True"} ) }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + container_network_logs="Disabled", + ) + ), ) self.assertEqual(dec_mc_13b, ground_truth_mc_13b) @@ -16807,8 +16952,8 @@ def test_update_enable_cnl_with_azure_monitor_logs_on_cluster(self): dec.context.attach_mc(mc) dec_mc = dec.update_monitoring_profile_flow_logs(mc) self.assertEqual( - dec_mc.addon_profiles["omsagent"].config["enableRetinaNetworkFlags"], - "True", + dec_mc.azure_monitor_profile.container_insights.container_network_logs, + "Enabled", ) self.assertTrue( dec.context.get_intermediate("monitoring_addon_postprocessing_required") @@ -16844,8 +16989,8 @@ def test_update_cnl_explicit_true_hlsm_with_prerequisites(self): dec.context.attach_mc(mc) dec_mc = dec.update_monitoring_profile_flow_logs(mc) self.assertEqual( - dec_mc.addon_profiles["omsagent"].config["enableRetinaNetworkFlags"], - "True", + dec_mc.azure_monitor_profile.container_insights.container_network_logs, + "Enabled", ) self.assertTrue(dec.context.get_enable_high_log_scale_mode()) self.assertTrue( @@ -18907,10 +19052,10 @@ def test_update_health_monitor_profile(self): dec_6.update_health_monitor_profile(mc_6) # ------------------------------------------------------------------ - # Tests for _setup_azure_monitor_logs setting enableRetinaNetworkFlags + # Tests for _setup_azure_monitor_logs setting container network logs # ------------------------------------------------------------------ def test_setup_azure_monitor_logs_sets_retina_flags_when_cnl_enabled(self): - """_setup_azure_monitor_logs sets enableRetinaNetworkFlags in config when CNL is being enabled.""" + """_setup_azure_monitor_logs sets containerNetworkLogs on the AMP profile when CNL is enabled.""" dec = AKSPreviewManagedClusterUpdateDecorator( self.cmd, self.client, @@ -18937,12 +19082,15 @@ def test_setup_azure_monitor_logs_sets_retina_flags_when_cnl_enabled(self): ): dec._setup_azure_monitor_logs(mc) - addon_profile = mc.addon_profiles.get(CONST_MONITORING_ADDON_NAME) - self.assertIsNotNone(addon_profile) - self.assertEqual(addon_profile.config.get("enableRetinaNetworkFlags"), "True") + container_insights = mc.azure_monitor_profile.container_insights + self.assertIsNotNone(container_insights) + self.assertTrue(container_insights.enabled) + self.assertEqual(container_insights.container_network_logs, "Enabled") + # the legacy addon config key is no longer written + self.assertNotIn(CONST_MONITORING_ADDON_NAME, mc.addon_profiles or {}) def test_setup_azure_monitor_logs_no_retina_flags_without_cnl(self): - """_setup_azure_monitor_logs does NOT set enableRetinaNetworkFlags when CNL is not specified.""" + """_setup_azure_monitor_logs leaves containerNetworkLogs unset when CNL is not specified.""" dec = AKSPreviewManagedClusterUpdateDecorator( self.cmd, self.client, @@ -18965,9 +19113,10 @@ def test_setup_azure_monitor_logs_no_retina_flags_without_cnl(self): ): dec._setup_azure_monitor_logs(mc) - addon_profile = mc.addon_profiles.get(CONST_MONITORING_ADDON_NAME) - self.assertIsNotNone(addon_profile) - self.assertNotIn("enableRetinaNetworkFlags", addon_profile.config) + container_insights = mc.azure_monitor_profile.container_insights + self.assertIsNotNone(container_insights) + self.assertTrue(container_insights.enabled) + self.assertIsNone(container_insights.container_network_logs) # ------------------------------------------------------------------ # Tests for _setup_azure_monitor_logs workspace change detection @@ -19012,10 +19161,9 @@ def test_setup_azure_monitor_logs_workspace_change_triggers_postprocessing(self) "monitoring_addon_postprocessing_required", default_value=False ) ) - # Verify workspace was updated - actual_key = next(k for k in mc.addon_profiles if k.lower() == "omsagent") + # Verify the new workspace landed on the AMP profile self.assertEqual( - mc.addon_profiles[actual_key].config["logAnalyticsWorkspaceResourceID"], + mc.azure_monitor_profile.container_insights.log_analytics_workspace_resource_id, new_ws, ) @@ -19181,7 +19329,8 @@ def test_disable_azure_monitor_logs_disables_container_insights_with_msi_auth(se return_value=None, ): dec._disable_azure_monitor_logs(mc) - self.assertFalse(mc.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) + # The AMP profile carries the disable; the RP keeps the legacy addon in sync + self.assertTrue(mc.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) self.assertFalse(mc.azure_monitor_profile.container_insights.enabled) def test_disable_azure_monitor_logs_no_container_insights_skips(self): @@ -19210,7 +19359,8 @@ def test_disable_azure_monitor_logs_no_container_insights_skips(self): dec.client = Mock() dec.client.get = Mock(return_value=mc) dec._disable_azure_monitor_logs(mc) - self.assertFalse(mc.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) + # containerInsights is created on demand and disabled there + self.assertFalse(mc.azure_monitor_profile.container_insights.enabled) # ------------------------------------------------------------------ # Tests for update_monitoring_profile_flow_logs: monitoring_being_enabled bypass @@ -19321,8 +19471,8 @@ def test_update_enable_retina_flow_logs_sets_postprocessing_flag(self): ) def test_disable_monitoring_clears_cnl_and_hlsm_config(self): - """Disabling monitoring addon should wipe config including enableRetinaNetworkFlags (CNL) - and any HLSM-related settings, so that re-enabling does not carry them forward. + """Disabling Azure Monitor logs should also reset container network logs (CNL) so that + re-enabling does not carry them forward. """ dec = AKSPreviewManagedClusterUpdateDecorator( self.cmd, @@ -19352,13 +19502,15 @@ def test_disable_monitoring_clears_cnl_and_hlsm_config(self): dec._disable_azure_monitor_logs(mc) - # Config should be completely wiped — no CNL or HLSM values survive - self.assertFalse(mc.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) - self.assertIsNone(mc.addon_profiles[CONST_MONITORING_ADDON_NAME].config) + # CNL state is reset on the AMP profile so it cannot survive a disable/re-enable cycle + self.assertFalse(mc.azure_monitor_profile.container_insights.enabled) + self.assertEqual( + mc.azure_monitor_profile.container_insights.container_network_logs, + "Disabled", + ) def test_reenable_monitoring_after_disable_does_not_carry_cnl(self): - """Re-enabling monitoring after disable should produce a fresh config - without enableRetinaNetworkFlags (CNL) or HLSM-related keys.""" + """Re-enabling monitoring after disable should not carry CNL forward.""" # Step 1: start with monitoring enabled + CNL (useAADAuth=false to skip DCR cleanup) mc = self.models.ManagedCluster( location="test_location", @@ -19388,12 +19540,16 @@ def test_reenable_monitoring_after_disable_does_not_carry_cnl(self): dec_disable.client = Mock() dec_disable.client.get = Mock(return_value=mc) dec_disable._disable_azure_monitor_logs(mc) - self.assertIsNone(mc.addon_profiles[CONST_MONITORING_ADDON_NAME].config) + self.assertEqual( + mc.azure_monitor_profile.container_insights.container_network_logs, + "Disabled", + ) + + # Step 3: simulate the server round-trip. The RP keeps the omsagent addon in sync with the + # AMP profile, so after the disable PUT the addon comes back disabled. + mc.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled = False - # Step 3: re-enable monitoring (no CNL flag passed) - # Simulate server round-trip: after PUT with config=None, the server - # returns the addon with config as empty dict, not None. - mc.addon_profiles[CONST_MONITORING_ADDON_NAME].config = {} + # Step 4: re-enable monitoring (no CNL flag passed) dec_enable = AKSPreviewManagedClusterUpdateDecorator( self.cmd, self.client, @@ -19407,12 +19563,588 @@ def test_reenable_monitoring_after_disable_does_not_carry_cnl(self): dec_enable.context.set_intermediate("subscription_id", "test-subscription-id") dec_enable._setup_azure_monitor_logs(mc) - # Config should only have workspace + MSI auth — no CNL or HLSM keys - addon_config = mc.addon_profiles[CONST_MONITORING_ADDON_NAME].config - self.assertTrue(mc.addon_profiles[CONST_MONITORING_ADDON_NAME].enabled) - self.assertIn("logAnalyticsWorkspaceResourceID", addon_config) - self.assertIn(CONST_MONITORING_USING_AAD_MSI_AUTH, addon_config) - self.assertNotIn("enableRetinaNetworkFlags", addon_config) + # Monitoring is back on, but CNL is not silently carried forward + container_insights = mc.azure_monitor_profile.container_insights + self.assertTrue(container_insights.enabled) + self.assertEqual( + container_insights.log_analytics_workspace_resource_id, + "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/new-workspace", + ) + self.assertNotEqual(container_insights.container_network_logs, "Enabled") + + # ------------------------------------------------------------------ + # Auth-mode derivation: the RP mirrors a legacy shared-key omsagent addon into the AMP + # containerInsights profile, so AMP presence alone must never be read as managed identity. + # ------------------------------------------------------------------ + def _addon_consts(self): + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, self.client, {}, CUSTOM_MGMT_AKS_PREVIEW + ) + return dec.context.get_addon_consts() + + def _brownfield_legacy_mc(self, use_aad_auth="false", addon_enabled=True): + """A cluster onboarded with legacy auth, which the RP has mirrored into the AMP profile.""" + config = {"logAnalyticsWorkspaceResourceID": "/subscriptions/test/workspace"} + if use_aad_auth is not None: + config[CONST_MONITORING_USING_AAD_MSI_AUTH] = use_aad_auth + return self.models.ManagedCluster( + location="test_location", + addon_profiles={ + CONST_MONITORING_ADDON_NAME: self.models.ManagedClusterAddonProfile( + enabled=addon_enabled, + config=config, + ) + }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + enabled=True, + log_analytics_workspace_resource_id="/subscriptions/test/workspace", + ) + ), + ) + + def test_is_monitoring_aad_auth_legacy_cluster_with_synced_amp_profile(self): + addon_consts = self._addon_consts() + # explicit useAADAuth=false + self.assertFalse( + _is_monitoring_aad_auth(self._brownfield_legacy_mc("false"), addon_consts) + ) + # absent useAADAuth is legacy too, matching the RP's derivation + self.assertFalse( + _is_monitoring_aad_auth(self._brownfield_legacy_mc(None), addon_consts) + ) + # empty string is legacy as well + self.assertFalse( + _is_monitoring_aad_auth(self._brownfield_legacy_mc(""), addon_consts) + ) + + def test_is_monitoring_aad_auth_true_cases(self): + addon_consts = self._addon_consts() + # explicit useAADAuth=true + self.assertTrue(_is_monitoring_aad_auth(self._brownfield_legacy_mc("true"), addon_consts)) + # casing of the value must not matter + self.assertTrue(_is_monitoring_aad_auth(self._brownfield_legacy_mc("True"), addon_consts)) + # a disabled addon being re-enabled is treated as fresh onboarding by the RP + self.assertTrue( + _is_monitoring_aad_auth( + self._brownfield_legacy_mc("false", addon_enabled=False), addon_consts + ) + ) + # no addon at all is a fresh AMP onboarding + fresh = self.models.ManagedCluster( + location="test_location", + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + enabled=True, + log_analytics_workspace_resource_id="/subscriptions/test/workspace", + ) + ), + ) + self.assertTrue(_is_monitoring_aad_auth(fresh, addon_consts)) + + def test_is_monitoring_aad_auth_config_key_casing(self): + """ARM echoes config keys back in arbitrary casing; the lookup must be case-insensitive.""" + addon_consts = self._addon_consts() + mc = self.models.ManagedCluster( + location="test_location", + addon_profiles={ + CONST_MONITORING_ADDON_NAME: self.models.ManagedClusterAddonProfile( + enabled=True, + config={"UseAADAuth": "true"}, + ) + }, + ) + self.assertTrue(_is_monitoring_aad_auth(mc, addon_consts)) + + def test_monitoring_addon_shim_reports_legacy_auth_for_brownfield_cluster(self): + """The shim must not stamp useAADAuth=true just because an AMP profile exists.""" + addon_consts = self._addon_consts() + shim = _build_monitoring_addon_shim( + self._brownfield_legacy_mc("false"), self.models, addon_consts + ) + self.assertTrue(shim.enabled) + self.assertEqual( + shim.config[CONST_MONITORING_USING_AAD_MSI_AUTH], "false" + ) + # and it still reports AAD auth for a genuine managed-identity cluster + shim_aad = _build_monitoring_addon_shim( + self._brownfield_legacy_mc("true"), self.models, addon_consts + ) + self.assertEqual(shim_aad.config[CONST_MONITORING_USING_AAD_MSI_AUTH], "true") + + def test_disable_azure_monitor_logs_skips_dcra_cleanup_on_legacy_auth_cluster(self): + """A legacy cluster has no DCRA, so disable must not attempt the AAD cleanup path.""" + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"disable_azure_monitor_logs": True, "yes": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self._brownfield_legacy_mc("false") + dec.context.attach_mc(mc) + dec.client = Mock() + dec.client.get = Mock(return_value=mc) + + with patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_mock: + dec._disable_azure_monitor_logs(mc) + + ensure_mock.assert_not_called() + # no ARM round trip either, since there is nothing to clean up + dec.client.get.assert_not_called() + self.assertFalse(mc.azure_monitor_profile.container_insights.enabled) + + def test_disable_azure_monitor_logs_runs_dcra_cleanup_on_aad_auth_cluster(self): + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"disable_azure_monitor_logs": True, "yes": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self._brownfield_legacy_mc("true") + dec.context.attach_mc(mc) + dec.client = Mock() + dec.client.get = Mock(return_value=mc) + + with patch.object( + dec.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec.context, "get_name", return_value="test-cluster" + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_mock: + dec._disable_azure_monitor_logs(mc) + + ensure_mock.assert_called_once() + self.assertTrue(ensure_mock.call_args.kwargs["aad_route"]) + self.assertTrue(ensure_mock.call_args.kwargs["remove_monitoring"]) + + def _enable_amp_logs_decorator(self, workspace_resource_id): + return AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_logs": True, + "workspace_resource_id": workspace_resource_id, + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + + def test_enable_azure_monitor_logs_rejected_on_legacy_auth_cluster(self): + """A cluster onboarded with legacy auth keeps that auth mode server-side, so the flag + cannot be honoured. Reject it and point at the migration doc.""" + for use_aad_auth in ("false", "False", "", None): + with self.subTest(use_aad_auth=use_aad_auth): + dec = self._enable_amp_logs_decorator("/subscriptions/test/workspace") + mc = self._brownfield_legacy_mc(use_aad_auth) + dec.context.attach_mc(mc) + + with patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda x: x, + ): + with self.assertRaises(ArgumentUsageError) as ctx: + dec._setup_azure_monitor_logs(mc) + + message = str(ctx.exception) + self.assertIn("managed identity", message) + self.assertIn( + "container-insights-authentication?tabs=cli" + "#migrate-to-managed-identity-authentication", + message, + ) + + def test_enable_azure_monitor_logs_rejected_before_creating_default_workspace(self): + """The rejection must happen before a Log Analytics workspace is provisioned.""" + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"enable_azure_monitor_logs": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self._brownfield_legacy_mc("false") + dec.context.attach_mc(mc) + + with patch.object( + dec.context.external_functions, + "ensure_default_log_analytics_workspace_for_monitoring", + ) as ensure_workspace_mock: + with self.assertRaises(ArgumentUsageError): + dec._setup_azure_monitor_logs(mc) + + ensure_workspace_mock.assert_not_called() + + def test_enable_azure_monitor_logs_allowed_when_legacy_addon_is_disabled(self): + """A disabled addon is a fresh onboarding as far as the RP is concerned, so the + re-enable defaults to managed identity and must not be blocked.""" + dec = self._enable_amp_logs_decorator("/subscriptions/test/workspace") + mc = self._brownfield_legacy_mc("false", addon_enabled=False) + dec.context.attach_mc(mc) + + with patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda x: x, + ): + dec._setup_azure_monitor_logs(mc) + + self.assertTrue(mc.azure_monitor_profile.container_insights.enabled) + + def test_workspace_change_allowed_on_aad_auth_cluster(self): + dec = self._enable_amp_logs_decorator("/subscriptions/test/other-workspace") + mc = self._brownfield_legacy_mc("true") + dec.context.attach_mc(mc) + + with patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda x: x, + ): + dec._setup_azure_monitor_logs(mc) + + self.assertEqual( + mc.azure_monitor_profile.container_insights.log_analytics_workspace_resource_id, + "/subscriptions/test/other-workspace", + ) + self.assertTrue( + dec.context.get_intermediate("monitoring_addon_postprocessing_required") + ) + + def test_hlsm_rejected_on_legacy_auth_cluster_with_amp_profile(self): + """HLSM needs a DCR, which legacy shared-key clusters do not have.""" + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"enable_high_log_scale_mode": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self._brownfield_legacy_mc("false") + dec.context.attach_mc(mc) + + with self.assertRaises(RequiredArgumentMissingError): + dec.update_monitoring_profile_flow_logs(mc) + + def test_hlsm_allowed_on_aad_auth_cluster(self): + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"enable_high_log_scale_mode": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self._brownfield_legacy_mc("true") + dec.context.attach_mc(mc) + + dec.update_monitoring_profile_flow_logs(mc) + + self.assertTrue( + dec.context.get_intermediate("monitoring_addon_postprocessing_required") + ) + + # ------------------------------------------------------------------ + # R4: Prometheus scraping and syslog port controls on the AMP path. + # ------------------------------------------------------------------ + + def _amp_enabled_mc(self): + """A cluster already onboarded to Azure Monitor logs through the AMP profile.""" + return self.models.ManagedCluster( + location="test_location", + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + enabled=True, + log_analytics_workspace_resource_id="/subscriptions/test/workspace", + ) + ), + ) + + def test_get_syslog_port_returns_none_when_unset(self): + # Unset must stay None so the server default (28330) is left in place. + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict({"syslog_port": None}), + self.models, + decorator_mode=DecoratorMode.CREATE, + ) + self.assertIsNone(ctx.get_syslog_port()) + + def test_get_syslog_port_rejects_out_of_range(self): + for bad_port in (0, -1, 65536): + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict( + {"syslog_port": bad_port, "enable_azure_monitor_logs": True} + ), + self.models, + decorator_mode=DecoratorMode.CREATE, + ) + with self.assertRaises(InvalidArgumentValueError): + ctx.get_syslog_port() + + def test_get_syslog_port_requires_azure_monitor_logs_on_create(self): + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict({"syslog_port": 28331}), + self.models, + decorator_mode=DecoratorMode.CREATE, + ) + with self.assertRaises(InvalidArgumentValueError): + ctx.get_syslog_port() + + def test_get_syslog_port_allowed_on_update_when_already_enabled(self): + # The flag must work on its own against a cluster that already has the AMP profile on. + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict({"syslog_port": 29000}), + self.models, + decorator_mode=DecoratorMode.UPDATE, + ) + ctx.attach_mc(self._amp_enabled_mc()) + self.assertEqual(ctx.get_syslog_port(), 29000) + + def test_get_syslog_port_rejected_on_update_when_not_enabled(self): + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict({"syslog_port": 29000}), + self.models, + decorator_mode=DecoratorMode.UPDATE, + ) + ctx.attach_mc(self.models.ManagedCluster(location="test_location")) + with self.assertRaises(InvalidArgumentValueError): + ctx.get_syslog_port() + + def test_container_insights_settings_rejected_with_disable_azure_monitor_logs(self): + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict( + {"syslog_port": 29000, "disable_azure_monitor_logs": True} + ), + self.models, + decorator_mode=DecoratorMode.UPDATE, + ) + ctx.attach_mc(self._amp_enabled_mc()) + with self.assertRaises(InvalidArgumentValueError): + ctx.get_syslog_port() + + def test_get_disable_prometheus_metrics_scraping_values(self): + # Neither flag leaves the field untouched. + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict({}), + self.models, + decorator_mode=DecoratorMode.CREATE, + ) + self.assertIsNone(ctx.get_disable_prometheus_metrics_scraping()) + + # --disable-prometheus-metrics-scraping sets the field to True. + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict( + { + "disable_prometheus_metrics_scraping": True, + "enable_azure_monitor_logs": True, + } + ), + self.models, + decorator_mode=DecoratorMode.CREATE, + ) + self.assertTrue(ctx.get_disable_prometheus_metrics_scraping()) + + # --enable-prometheus-metrics-scraping clears it. + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict( + { + "enable_prometheus_metrics_scraping": True, + "enable_azure_monitor_logs": True, + } + ), + self.models, + decorator_mode=DecoratorMode.CREATE, + ) + self.assertFalse(ctx.get_disable_prometheus_metrics_scraping()) + + def test_prometheus_metrics_scraping_flags_are_mutually_exclusive(self): + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict( + { + "enable_prometheus_metrics_scraping": True, + "disable_prometheus_metrics_scraping": True, + "enable_azure_monitor_logs": True, + } + ), + self.models, + decorator_mode=DecoratorMode.CREATE, + ) + with self.assertRaises(MutuallyExclusiveArgumentError): + ctx.get_disable_prometheus_metrics_scraping() + + def test_create_writes_syslog_port_and_scraping_to_amp_profile(self): + dec = AKSPreviewManagedClusterCreateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_logs": True, + "workspace_resource_id": "/subscriptions/test/workspace", + "syslog_port": 28331, + "disable_prometheus_metrics_scraping": True, + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster(location="test_location") + dec.context.attach_mc(mc) + with patch( + "azext_aks_preview.managed_cluster_decorator.AKSPreviewManagedClusterContext." + "get_subscription_id", + return_value="test-subscription-id", + ): + dec._setup_azure_monitor_logs(mc) + + container_insights = mc.azure_monitor_profile.container_insights + self.assertTrue(container_insights.enabled) + self.assertEqual(container_insights.syslog_port, 28331) + self.assertTrue(container_insights.disable_prometheus_metrics_scraping) + + def test_update_writes_container_insights_settings_standalone(self): + # No --enable-azure-monitor-logs: the flags apply to a cluster that already has it on. + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"syslog_port": 29000, "enable_prometheus_metrics_scraping": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self._amp_enabled_mc() + dec.context.attach_mc(mc) + + dec.update_azure_monitor_logs_settings(mc) + + container_insights = mc.azure_monitor_profile.container_insights + self.assertEqual(container_insights.syslog_port, 29000) + self.assertFalse(container_insights.disable_prometheus_metrics_scraping) + # The rest of the monitoring configuration is left alone. + self.assertTrue(container_insights.enabled) + self.assertEqual( + container_insights.log_analytics_workspace_resource_id, + "/subscriptions/test/workspace", + ) + + def test_update_container_insights_settings_is_noop_without_flags(self): + # Nothing specified must not fabricate an empty containerInsights profile. + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, self.client, {}, CUSTOM_MGMT_AKS_PREVIEW + ) + mc = self.models.ManagedCluster(location="test_location") + dec.context.attach_mc(mc) + + dec.update_azure_monitor_logs_settings(mc) + + self.assertIsNone(mc.azure_monitor_profile) + + def test_update_container_insights_settings_partial_write(self): + # Only the specified field is written; the other is left untouched. + mc = self._amp_enabled_mc() + mc.azure_monitor_profile.container_insights.disable_prometheus_metrics_scraping = True + + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, self.client, {"syslog_port": 29000}, CUSTOM_MGMT_AKS_PREVIEW + ) + dec.context.attach_mc(mc) + dec.update_azure_monitor_logs_settings(mc) + + container_insights = mc.azure_monitor_profile.container_insights + self.assertEqual(container_insights.syslog_port, 29000) + self.assertTrue(container_insights.disable_prometheus_metrics_scraping) + + # ------------------------------------------------------------------ + # R3: disableCustomMetrics was removed from the API and must not resurface. + # ------------------------------------------------------------------ + def test_disable_custom_metrics_is_not_part_of_container_insights(self): + container_insights = ( + self.models.ManagedClusterAzureMonitorProfileContainerInsights(enabled=True) + ) + self.assertFalse(hasattr(container_insights, "disable_custom_metrics")) + self.assertNotIn("disableCustomMetrics", container_insights) + + def test_enabling_azure_monitor_logs_emits_no_disable_custom_metrics(self): + dec = AKSPreviewManagedClusterCreateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_logs": True, + "workspace_resource_id": "/subscriptions/test/workspace", + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster(location="test_location") + dec.context.attach_mc(mc) + with patch( + "azext_aks_preview.managed_cluster_decorator.AKSPreviewManagedClusterContext." + "get_subscription_id", + return_value="test-subscription-id", + ): + dec._setup_azure_monitor_logs(mc) + + self.assertNotIn( + "disablecustommetrics", + str(dict(mc.azure_monitor_profile.container_insights)).lower(), + ) + + # ------------------------------------------------------------------ + # R0: OTLP gRPC ports are independent of the HTTP ports and default to unset. + # ------------------------------------------------------------------ + def test_opentelemetry_grpc_port_unset_leaves_server_default(self): + dec = AKSPreviewManagedClusterCreateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_metrics": True, + "enable_opentelemetry_metrics": True, + "opentelemetry_metrics_port": 8080, + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster( + location="test_location", + identity=self.models.ManagedClusterIdentity(type="SystemAssigned"), + ) + dec.context.attach_mc(mc) + dec_mc = dec.set_up_azure_monitor_profile(mc) + + otlp_metrics = dec_mc.azure_monitor_profile.app_monitoring.open_telemetry_metrics + self.assertEqual(otlp_metrics.http_port, 8080) + self.assertIsNone(otlp_metrics.grpc_port) + + def test_opentelemetry_logs_traces_grpc_port_set_independently(self): + # gRPC settable without an HTTP override, for the logs and traces signal. + dec = AKSPreviewManagedClusterCreateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_logs": True, + "workspace_resource_id": "/subscriptions/test/workspace", + "enable_opentelemetry_logs": True, + "opentelemetry_logs_traces_port_grpc": 8082, + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster( + location="test_location", + identity=self.models.ManagedClusterIdentity(type="SystemAssigned"), + ) + dec.context.attach_mc(mc) + dec._setup_opentelemetry_logs(mc) + + otlp_logs = mc.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces + self.assertTrue(otlp_logs.enabled) + self.assertEqual(otlp_logs.grpc_port, 8082) + self.assertIsNone(otlp_logs.http_port) if __name__ == "__main__": diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py b/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py index 918f0fc4742..2aa551f0e91 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py @@ -2802,3 +2802,80 @@ def test_nodepool_update_allows_windows2022_and_windows2025(self): if __name__ == "__main__": unittest.main() + + +class ContainerInsightsSettingsNamespace(SimpleNamespace): + """Namespace for the R4 containerInsights tuning flags, with CLI defaults.""" + + def __init__(self, **kwargs): + defaults = { + "syslog_port": None, + "enable_prometheus_metrics_scraping": False, + "disable_prometheus_metrics_scraping": False, + "enable_azure_monitor_logs": False, + "disable_azure_monitor_logs": False, + } + defaults.update(kwargs) + super().__init__(**defaults) + + +class TestValidateContainerInsightsSettings(unittest.TestCase): + def test_no_flags_is_valid(self): + namespace = ContainerInsightsSettingsNamespace() + validators.validate_container_insights_settings_for_create(namespace) + validators.validate_container_insights_settings_for_update(namespace) + + def test_scraping_flags_are_mutually_exclusive(self): + namespace = ContainerInsightsSettingsNamespace( + enable_prometheus_metrics_scraping=True, + disable_prometheus_metrics_scraping=True, + enable_azure_monitor_logs=True, + ) + with self.assertRaises(MutuallyExclusiveArgumentError): + validators.validate_container_insights_settings_for_create(namespace) + + def test_syslog_port_out_of_range(self): + for bad_port in (0, -1, 65536): + namespace = ContainerInsightsSettingsNamespace( + syslog_port=bad_port, enable_azure_monitor_logs=True + ) + with self.assertRaises(InvalidArgumentValueError): + validators.validate_container_insights_settings_for_create(namespace) + + def test_syslog_port_boundaries_are_valid(self): + for port in (1, 28330, 65535): + namespace = ContainerInsightsSettingsNamespace( + syslog_port=port, enable_azure_monitor_logs=True + ) + validators.validate_container_insights_settings_for_create(namespace) + + def test_create_requires_enable_azure_monitor_logs(self): + # Without the AMP profile these flags would be silently dropped, so fail fast. + namespace = ContainerInsightsSettingsNamespace(syslog_port=28331) + with self.assertRaises(ArgumentUsageError) as cm: + validators.validate_container_insights_settings_for_create(namespace) + self.assertIn("--syslog-port", str(cm.exception)) + self.assertIn("--enable-azure-monitor-logs", str(cm.exception)) + + def test_create_lists_every_specified_flag_in_the_error(self): + namespace = ContainerInsightsSettingsNamespace( + syslog_port=28331, disable_prometheus_metrics_scraping=True + ) + with self.assertRaises(ArgumentUsageError) as cm: + validators.validate_container_insights_settings_for_create(namespace) + self.assertIn("--syslog-port", str(cm.exception)) + self.assertIn("--disable-prometheus-metrics-scraping", str(cm.exception)) + + def test_update_defers_dependency_check_to_the_decorator(self): + # On update the cluster may already have Azure Monitor logs enabled, which the namespace + # validator cannot see, so it must not reject the flag on its own. + namespace = ContainerInsightsSettingsNamespace(syslog_port=29000) + validators.validate_container_insights_settings_for_update(namespace) + + def test_conflicts_with_disable_azure_monitor_logs(self): + namespace = ContainerInsightsSettingsNamespace( + syslog_port=29000, disable_azure_monitor_logs=True + ) + with self.assertRaises(ArgumentUsageError) as cm: + validators.validate_container_insights_settings_for_update(namespace) + self.assertIn("--disable-azure-monitor-logs", str(cm.exception)) diff --git a/src/aks-preview/azmon-logs-cli-enhancements-spec.md b/src/aks-preview/azmon-logs-cli-enhancements-spec.md new file mode 100644 index 00000000000..07704486d73 --- /dev/null +++ b/src/aks-preview/azmon-logs-cli-enhancements-spec.md @@ -0,0 +1,221 @@ +# PRD: Azure Monitor Logs (Container Insights) CLI Enhancements + +## 1. TL;DR + +We are moving Azure Monitor Logs (Container Insights) onboarding off the legacy **addon profile** and onto the first-class **Azure Monitor Profile (AMP)**. This PRD covers the two onboarding commands in scope, closes AMP feature gaps (syslog port, Prometheus scraping, OTLP gRPC ports), removes an obsolete auth flag from the new path, and adds a deprecation warning on the legacy path. + +**Two commands are in scope:** + +| Command | Profile it configures | Direction | +|---|---|---| +| az aks enable-addons -a monitoring | Addon profile (addonProfiles.omsagent) | Legacy — keep working, start deprecating | +| az aks create/update --enable-azure-monitor-logs | Azure Monitor Profile (azureMonitorProfile.containerInsights) | Strategic — become the only supported way | + +## 2. Problem & motivation + +- `--enable-azure-monitor-logs` was meant to be the modern, AMP-based onboarding path, but it is **currently implemented on the legacy addon profile** — so today both commands write the same `omsagent` addon object. This blocks us from using AMP-only capabilities and from deprecating the addon. +- The AMP `containerInsights` schema already supports **syslog port** and **Prometheus scraping** controls, and the app-monitoring schema supports **OTLP gRPC ports**, but the CLI exposes none of them on the new path. +- The legacy **auth flag** (`--enable-msi-auth-for-monitoring`) is meaningless for AMP (which is managed-identity only), yet it is still accepted on the new path. +- The latest API version **removed** `disableCustomMetrics` from `containerInsights`, so it must not surface anywhere in the CLI. + +## 3. Goals / Non-goals + +**Goals** + +- Make `--enable-azure-monitor-logs` write **only** the AMP profile. +- Reach parity + close gaps on the AMP path (syslog port, Prometheus scraping, OTLP gRPC). +- Remove the obsolete auth flag from the AMP path; warn about it on the legacy path. +- Keep `az aks enable-addons -a monitoring` working (backward compatible). +**Non-goals** + +- Removing or breaking the legacy `enable-addons monitoring` command. +- Exposing `disableCustomMetrics` (removed from the API). +- Changes to metrics-only (Managed Prometheus) onboarding. + +## 4. Background: the two profiles & their schema + +### 4.1 Addon profile — set by `az aks enable-addons -a monitoring` + +`properties.addonProfiles.omsagent`: + +```json +{ + "enabled": true, + "config": { + "logAnalyticsWorkspaceResourceID": "", + "useAADAuth": "true" | "false" // MSI auth vs legacy shared-key auth + } +} +``` + +- Auth: supports **both** managed-identity (`useAADAuth=true`) and legacy shared-key (`useAADAuth=false`), driven by `--enable-msi-auth-for-monitoring`. + +### 4.2 Azure Monitor Profile (AMP) — target of `--enable-azure-monitor-logs` + +`properties.azureMonitorProfile` (latest API `2026-07-02-preview`): + +```json +{ + "containerInsights": { + "enabled": true, + "logAnalyticsWorkspaceResourceId": "", + "syslogPort": 28330, // default 28330 + "disablePrometheusMetricsScraping": false, // default false + "containerNetworkLogs": "" + // NOTE: disableCustomMetrics was REMOVED in the latest API — do not use. + }, + "appMonitoring": { + "openTelemetryLogsAndTraces": { "enabled": true, "httpPort": 0, "grpcPort": 0 }, + "openTelemetryMetrics": { "enabled": true, "httpPort": 0, "grpcPort": 0 } + } +} +``` + +- Auth: **managed-identity only** — there is no shared-key/`useAADAuth` concept. + +### 4.3 Current CLI coverage of the AMP schema + +| AMP field | Flag today? | +|---|---| +| `containerInsights.enabled` | Yes (`--enable-azure-monitor-logs` / `--disable-azure-monitor-logs`) — *but writes the addon profile* | +| `containerInsights.logAnalyticsWorkspaceResourceId` | Yes (`--workspace-resource-id`) | +| `containerInsights.syslogPort` | **No** | +| `containerInsights.disablePrometheusMetricsScraping` | **No** | +| `appMonitoring.openTelemetry*.httpPort` | Yes (`--opentelemetry-metrics-port`, `--opentelemetry-logs-port`) | +| `appMonitoring.openTelemetry*.grpcPort` | **No** | + +## 5. Requirements + +| ID | Requirement | Priority | +|---|---|---| +| R0 | Add OTLP gRPC port overrides | P0 | +| R1 | `--enable-azure-monitor-logs` writes only the AMP profile | P0 | +| R2 | Remove legacy auth flag from `--enable-azure-monitor-logs` | P0 | +| R3 | Do not expose `disableCustomMetrics` (removed from API) | P0 | +| R4 | Add Prometheus-scraping + syslog-port controls to the AMP path | P0 | +| R5 | Warn when the legacy auth flag is used with `enable-addons monitoring` | P0 | + +### R0 — OTLP gRPC port overrides + +Today only the OTLP **HTTP** port is settable; the gRPC port cannot be overridden. + +- Allow overriding the gRPC port for both OTLP signals: + - metrics → `appMonitoring.openTelemetryMetrics.grpcPort` + - logs & traces → `appMonitoring.openTelemetryLogsAndTraces.grpcPort` +- HTTP and gRPC ports are independently settable per signal; make the HTTP-vs-gRPC mapping unambiguous. +**Acceptance criteria** + +- A gRPC override sets the corresponding `grpcPort`; unset leaves the server default. +- HTTP and gRPC ports settable independently for metrics and logs/traces. +- Values round-trip on `az aks update`. + +### R1 — Switch `--enable-azure-monitor-logs` to the AMP profile only + +**As an** AKS user, **when** I run `az aks create/update --enable-azure-monitor-logs`, **I want** it to configure `azureMonitorProfile.containerInsights` (and not the legacy `omsagent` addon), so onboarding uses the modern profile. + +- Today the flag writes `addonProfiles.omsagent`; it must instead write only `azureMonitorProfile.containerInsights.enabled = true` (+ workspace id). +- `--disable-azure-monitor-logs` must disable via the AMP profile (`containerInsights.enabled = false`) and no longer depend on the addon object. +- **Container network logs move with it.** On the legacy addon path this is the `omsagent` config key `enableRetinaNetworkFlags` ("True"/"False"); it lines up with the AMP field `containerInsights.containerNetworkLogs` (enum `Enabled`/`Disabled`, default `Disabled`). The **same** `--enable-container-network-logs` / `--disable-container-network-logs` commands must set `containerInsights.containerNetworkLogs` accordingly (`Enabled` / `Disabled`) instead of the addon config key. +- The legacy `enable-addons monitoring` command is unaffected and keeps using the addon profile. +**Parity — all monitoring options must keep working on the new command** + +Several legacy monitoring options are realized as **out-of-band ARM resources** (DCR / DCE / DCRA / AMPLS), not as `containerInsights` fields. They work today only because `--enable-azure-monitor-logs` writes the `omsagent` addon, and that provisioning is gated on the addon object existing. After the AMP switch, the **same** command must continue to honor them without the addon. + +| Legacy flag | Configures | Post-switch requirement | +|---|---|---| +| `--workspace-resource-id` | Log Analytics workspace | → `containerInsights.logAnalyticsWorkspaceResourceId` | +| `--enable-syslog` | Syslog collection (DCR syslog data source) | Provision the same DCR data source | +| `--data-collection-settings` | DCR tuning (interval, namespaces, streams, `enableContainerLogV2`) | Apply the same DCR settings | +| `--enable-high-log-scale-mode` | High log scale (ingestion DCE) | Provision the same DCE | +| `--ampls-resource-id` | AMPLS private-link scope (private cluster, MSI) | Provision the same AMPLS links | +| `--enable/disable-container-network-logs` | Container network logs | → `containerInsights.containerNetworkLogs` (see above) | +| `--enable-msi-auth-for-monitoring` | Auth mode | **Dropped** — AMP is MSI-only (see R2) | + +> The DCR/DCE/DCRA/AMPLS provisioning currently keys off the `omsagent` addon object; after the switch it must be driven off the AMP profile (workspace id from `containerInsights.logAnalyticsWorkspaceResourceId`) and must not require the addon to exist. + +**Acceptance criteria** + +- `--enable-azure-monitor-logs` results in `azureMonitorProfile.containerInsights.enabled=true` with the workspace id, and **no** `omsagent` addon entry authored by this flag. +- `--disable-azure-monitor-logs` sets `containerInsights.enabled=false`. +- `--enable-container-network-logs` sets `containerInsights.containerNetworkLogs="Enabled"` (and `--disable-container-network-logs` → `"Disabled"`); it no longer writes `enableRetinaNetworkFlags`. +- `--enable-syslog`, `--data-collection-settings`, `--enable-high-log-scale-mode`, and `--ampls-resource-id` produce the **same** DCR/DCE/DCRA/AMPLS artifacts on `--enable-azure-monitor-logs` as they do today on `--enable-addons monitoring`. +- That artifact provisioning no longer depends on the presence of the `omsagent` addon profile. +- `enable-addons -a monitoring` behavior is unchanged. + +### R2 — Remove the legacy auth flag from `--enable-azure-monitor-logs` + +The AMP profile is managed-identity only, so `--enable-msi-auth-for-monitoring` is meaningless here. + +- `--enable-azure-monitor-logs` must not accept or honor `--enable-msi-auth-for-monitoring`. +- Passing them together returns a clear, actionable error. +- No `useAADAuth`/shared-key concept is written on the AMP path. +**Acceptance criteria** + +- `--enable-azure-monitor-logs` works with no auth flag (MSI implicit). +- Combining the two flags errors out with a helpful message. + +### R3 — Do not expose `disableCustomMetrics` + +`containerInsights.disableCustomMetrics` was removed in the latest API version. + +- No CLI flag maps to it; it must not appear in any request the CLI sends. +- Schema/reference docs must reflect its removal. +**Acceptance criteria** + +- No CLI surface reads or writes `disableCustomMetrics`. + +### R4 — Prometheus-scraping & syslog-port controls on the AMP path + +Expose the two `containerInsights` controls that the schema already supports. + +- A flag to toggle Prometheus metrics scraping → `containerInsights.disablePrometheusMetricsScraping`. +- A flag to set the syslog host port → `containerInsights.syslogPort` (integer; server default 28330 when unset). +- Available on `az aks create` and `az aks update` for the AMP path; both are updatable. +**Suggested flags** + +| Flag | Type | Maps to | Notes | +|---|---|---|---| +| `--disable-prometheus-metrics-scraping` | switch | `disablePrometheusMetricsScraping = true` | default off (scraping on); mirrors the existing `--enable-…/--disable-…` metrics style | +| `--enable-prometheus-metrics-scraping` | switch | `disablePrometheusMetricsScraping = false` | to re-enable on `update`; mutually exclusive with the disable flag | +| `--syslog-port` | int | `syslogPort` | valid TCP port; unset ⇒ server default 28330 | + +> Naming note: the pre-existing `--enable-syslog` (three-state) toggles legacy syslog **collection** and is distinct from the new `--syslog-port` host-port control. + +**Example usage** + +```bash +# Enable Azure Monitor logs (AMP), disable Prometheus scraping, custom syslog port +az aks create -g -n \ + --enable-azure-monitor-logs \ + --disable-prometheus-metrics-scraping \ + --syslog-port 28331 + +# Later, re-enable scraping and change the port on an existing cluster +az aks update -g -n \ + --enable-prometheus-metrics-scraping \ + --syslog-port 29000 +``` + +**Acceptance criteria** + +- The scraping flag sets/clears `disablePrometheusMetricsScraping`. +- The syslog-port flag sets `syslogPort` (validated as a TCP port); unset leaves the default. +- Both round-trip on `az aks update` without disturbing other monitoring settings. + +### R5 — Deprecation warning on the legacy auth flag + +**As an** AKS user, **when** I pass `--enable-msi-auth-for-monitoring` on `az aks enable-addons -a monitoring`, **I want** a warning that the flag is deprecated and that managed-identity auth (and `--enable-azure-monitor-logs`) is the recommended path. + +- The command still succeeds (non-breaking); only a warning is added. +- Especially relevant when the flag is set to `false` (opting into legacy shared-key auth). +**Acceptance criteria** + +- Warning shown once when the flag is explicitly provided; not shown when omitted. +- Command completes successfully; resulting `useAADAuth` value is unchanged. + +## 6. Target end-state (the two commands) + +| Command | Writes | Auth | Extra controls | +|---|---|---|---| +| `az aks enable-addons -a monitoring` | `addonProfiles.omsagent` | MSI or shared-key (**warns** on explicit legacy auth flag) | unchanged | +| `az aks create/update --enable-azure-monitor-logs` | `azureMonitorProfile.containerInsights` **only** | MSI only (**no** auth flag) | `--syslog-port`, Prometheus-scraping toggle, `--enable/disable-container-network-logs` (→ `containerNetworkLogs`); OTLP gRPC ports via OTLP flags | From f3dc3e196997b33d89219fe4dbf4938939bf4440 Mon Sep 17 00:00:00 2001 From: Sunil Yadav Date: Thu, 17 Sep 2026 02:35:39 +0000 Subject: [PATCH 2/8] Fix DCR/DCRA creation in logs --- src/aks-preview/HISTORY.rst | 6 + src/aks-preview/azext_aks_preview/_consts.py | 2 + src/aks-preview/azext_aks_preview/_help.py | 13 +- .../managed_cluster_decorator.py | 170 +++- .../latest/test_managed_cluster_decorator.py | 725 ++++++++++++++++-- .../azmon-logs-cli-enhancements-spec.md | 86 +++ src/aks-preview/linter_exclusions.yml | 12 + 7 files changed, 934 insertions(+), 80 deletions(-) diff --git a/src/aks-preview/HISTORY.rst b/src/aks-preview/HISTORY.rst index fc91075c6a6..305e8e56aab 100644 --- a/src/aks-preview/HISTORY.rst +++ b/src/aks-preview/HISTORY.rst @@ -18,6 +18,12 @@ Pending * `az aks maintenanceconfiguration add` and `az aks maintenanceconfiguration update`: Preserve configuration-file fields with the typespec-generated SDK model. * Improve AKS live-test resilience for preview feature gates, transient resource and monitoring-table readiness, retired configurations, and service propagation delays. * `az aks create` and `az aks update`: Reject `--outbound-type managedNATGatewayV2` with an actionable error directing to `--outbound-type managedNATGateway --outbound-type-sku StandardV2` (the GA-aligned shape); the legacy value is no longer accepted on the target api-version. +* `az aks create` and `az aks update`: Reject `--enable-azure-monitor-logs` on clusters using service principal authentication, since the Azure Monitor profile onboards with managed identity only. +* `az aks update`: Reject `--enable-azure-monitor-logs` when Azure Monitor logs is already enabled on the cluster, matching `az aks enable-addons -a monitoring`. Run `--disable-azure-monitor-logs` first to change the configuration. +* `az aks update`: `--disable-azure-monitor-logs` now removes the data collection rule association and resets the Container Insights settings (syslog port, Prometheus scraping and container network logs) back to their defaults, and asks for confirmation when OpenTelemetry logs and traces are enabled. +* `az aks update`: Fix `--enable-azure-monitor-logs` not creating the data collection rule and association unless the Log Analytics workspace changed, which left the agent running with no data collection rule attached so no logs were ingested. +* `az aks update`: Create the data collection rule and association before the cluster update when enabling with `--enable-azure-monitor-logs`, matching `az aks enable-addons -a monitoring`. Provisioning them afterwards meant the agent started before the association existed and then stayed idle for several minutes before restarting once the configuration arrived. +* `az aks update`: Declining the OpenTelemetry confirmation prompt for `--disable-azure-monitor-metrics` now leaves the cluster unchanged and exits without an error, matching every other confirmation prompt, instead of failing the command. 22.0.0b6 +++++++++ diff --git a/src/aks-preview/azext_aks_preview/_consts.py b/src/aks-preview/azext_aks_preview/_consts.py index fe402022c89..11d9ce13a7d 100644 --- a/src/aks-preview/azext_aks_preview/_consts.py +++ b/src/aks-preview/azext_aks_preview/_consts.py @@ -193,6 +193,8 @@ CONST_CONTAINER_NETWORK_LOGS_DISABLED = "Disabled" # legacy omsagent addon config key, superseded by containerNetworkLogs on the AMP path CONST_MONITORING_ENABLE_RETINA_NETWORK_FLAGS = "enableRetinaNetworkFlags" +# server-side default for azureMonitorProfile.containerInsights.syslogPort +CONST_CONTAINER_INSIGHTS_DEFAULT_SYSLOG_PORT = 28330 # virtual node CONST_VIRTUAL_NODE_ADDON_NAME = "aciConnector" diff --git a/src/aks-preview/azext_aks_preview/_help.py b/src/aks-preview/azext_aks_preview/_help.py index 0dfdd6bfcb6..ddcebfb0fc4 100644 --- a/src/aks-preview/azext_aks_preview/_help.py +++ b/src/aks-preview/azext_aks_preview/_help.py @@ -197,7 +197,9 @@ - name: --enable-azure-monitor-logs type: bool short-summary: Enable Azure Monitor logs for the cluster. - long-summary: This is equivalent to using "--enable-addons monitoring". Turn on Log Analytics monitoring. Uses the Log Analytics Default Workspace if it exists, else creates one. Specify "--workspace-resource-id" to use an existing workspace. If monitoring addon is enabled --no-wait argument will have no effect + long-summary: | + Enables Log Analytics monitoring for the cluster through the Azure Monitor profile. Uses the Log Analytics Default Workspace if it exists, else creates one. Specify "--workspace-resource-id" to use an existing workspace. + Requires the cluster to use a managed identity; clusters created with service principal authentication are not supported. - name: --disable-rbac type: bool short-summary: Disable Kubernetes Role-Based Access Control. @@ -1155,11 +1157,16 @@ - name: --enable-azure-monitor-logs type: bool short-summary: Enable Azure Monitor logs for the cluster. - long-summary: This is equivalent to using "az aks enable-addons -a monitoring". Enables Log Analytics monitoring for the cluster. Uses the Log Analytics Default Workspace if it exists, else creates one. Specify "--workspace-resource-id" to use an existing workspace. If monitoring addon is enabled --no-wait argument will have no effect + long-summary: | + Enables Log Analytics monitoring for the cluster through the Azure Monitor profile. Uses the Log Analytics Default Workspace if it exists, else creates one. Specify "--workspace-resource-id" to use an existing workspace. + Requires the cluster to use a managed identity; clusters using service principal authentication are not supported. + Fails if Azure Monitor logs is already enabled on the cluster, or if the cluster was onboarded with legacy (non-managed-identity) authentication. To change the configuration, run "az aks update --disable-azure-monitor-logs" first. - name: --disable-azure-monitor-logs type: bool short-summary: Disable Azure Monitor logs for the cluster. - long-summary: This is equivalent to using "az aks disable-addons -a monitoring". Disables Log Analytics monitoring for the cluster. + long-summary: | + Disables Log Analytics monitoring for the cluster, removes the data collection rule association, and resets the Container Insights settings (syslog port, Prometheus scraping and container network logs) back to their defaults. The workspace is left recorded on the profile but is unused while disabled, and is replaced on the next enable. + If OpenTelemetry logs and traces are enabled they are disabled as well, and confirmation is requested first unless "--yes" is specified. - name: --workspace-resource-id type: string short-summary: The resource ID of an existing Log Analytics Workspace to use for storing monitoring data. If not specified, uses the default Log Analytics Workspace if it exists, otherwise creates one. diff --git a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py index 31f233a5673..a4518bf1180 100644 --- a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py @@ -57,6 +57,7 @@ CONST_CONTAINER_NETWORK_LOGS_ENABLED, CONST_CONTAINER_NETWORK_LOGS_DISABLED, CONST_MONITORING_ENABLE_RETINA_NETWORK_FLAGS, + CONST_CONTAINER_INSIGHTS_DEFAULT_SYSLOG_PORT, ) from azext_aks_preview.azurecontainerstorage._consts import ( CONST_ACSTOR_EXT_INSTALLATION_NAME, @@ -228,6 +229,59 @@ def _apply_container_insights_settings(container_insights, syslog_port, disable_ container_insights.disable_prometheus_metrics_scraping = disable_prometheus_scraping +def _reset_container_insights_to_defaults(container_insights): + """Reset the AMP containerInsights settings back to their documented defaults. + + The RP copies a containerInsights field from the request onto the cluster only when the + field is present (see ``ApplyAzureMonitorProfileContainerInsights``), so leaving a field + as ``None`` preserves whatever the cluster already has. Disabling therefore has to write + the defaults explicitly, otherwise a later re-enable silently inherits the old syslog port, + scraping choice and container network logs setting. + + ``logAnalyticsWorkspaceResourceId`` is deliberately left alone. It is a resource-id typed + field, and blanking it makes the RP mirror the empty string into + ``addonProfiles.omsagent.config.logAnalyticsWorkspaceResourceID``; ARM then rejects every + later write of the cluster with ``LinkedInvalidPropertyId``, which would break unrelated + ``az aks update`` calls too. The stale id is inert once ``enabled`` is false, and the enable + path always overwrites it with a freshly resolved workspace, so nothing is inherited. + """ + container_insights.enabled = False + container_insights.syslog_port = CONST_CONTAINER_INSIGHTS_DEFAULT_SYSLOG_PORT + container_insights.disable_prometheus_metrics_scraping = False + container_insights.container_network_logs = CONST_CONTAINER_NETWORK_LOGS_DISABLED + + +def _is_service_principal_cluster(mc): + """Whether the cluster authenticates to Azure with a service principal instead of an identity. + + Managed-identity clusters report ``servicePrincipalProfile.clientId == "msi"``, so any other + non-empty client id means a real service principal. A missing profile means managed identity. + """ + service_principal_profile = getattr(mc, "service_principal_profile", None) if mc is not None else None + if service_principal_profile is None: + return False + client_id = getattr(service_principal_profile, "client_id", None) + if not client_id: + return False + return client_id.lower() != "msi" + + +def _raise_if_service_principal_cluster(mc): + """Reject --enable-azure-monitor-logs on a service principal cluster. + + The Azure Monitor Profile has no shared-key/``useAADAuth`` concept: the agent authenticates + to the Log Analytics workspace with the cluster's managed identity. A service principal + cluster has no such identity, so the onboarding cannot work and is rejected up front. + """ + if _is_service_principal_cluster(mc): + raise ArgumentUsageError( + "'--enable-azure-monitor-logs' cannot be used on clusters with service principal " + "authentication. Azure Monitor logs onboards through the Azure Monitor Profile, " + "which requires the cluster to use a managed identity. Update the cluster to use a " + "managed identity with 'az aks update --enable-managed-identity', then retry." + ) + + def _is_container_network_logs_enabled_on_mc(mc, addon_consts): """Whether container network logs are on, via the AMP profile or the legacy addon config key.""" container_insights = _get_container_insights_profile(mc) @@ -5403,6 +5457,11 @@ def _setup_azure_monitor_app_monitoring(self, mc: ManagedCluster) -> None: def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: """Set up Azure Monitor logs configuration.""" + # The Azure Monitor Profile is managed-identity only, so a cluster created with a service + # principal can never authenticate to the workspace. Reject before a default workspace is + # created on the user's behalf. + _raise_if_service_principal_cluster(mc) + # Get or create workspace resource ID workspace_resource_id = self.context.raw_param.get("workspace_resource_id") if not workspace_resource_id: @@ -8906,14 +8965,17 @@ def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: """Set up Azure Monitor logs configuration.""" addon_consts = self.context.get_addon_consts() - CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID = addon_consts.get( - "CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID") + + # The Azure Monitor Profile is managed-identity only, so a service principal cluster can + # never authenticate to the workspace. Reject before a default workspace is created. + _raise_if_service_principal_cluster(mc) # --enable-azure-monitor-logs onboards through the Azure Monitor Profile, which is managed # identity only. A cluster already onboarded with legacy (shared key) authentication keeps # that authentication mode on the server side, so this flag cannot be honoured as asked. # Reject it up front, before a default workspace is created, rather than silently leaving - # the cluster on legacy auth. + # the cluster on legacy auth. This is checked before the "already enabled" guard below so + # the more actionable migration message wins for a legacy-auth cluster. if not _is_monitoring_aad_auth(mc, addon_consts): raise ArgumentUsageError( "Azure Monitor logs is already enabled on this cluster using legacy " @@ -8924,6 +8986,17 @@ def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: "container-insights-authentication?tabs=cli#migrate-to-managed-identity-authentication" ) + # Re-onboarding an already onboarded cluster is rejected, matching the behaviour of + # 'az aks enable-addons -a monitoring'. Silently re-running would otherwise recreate the + # default workspace and re-provision DCR/DCRA artifacts for a cluster that is already set + # up, which hides configuration mistakes such as a mistyped --workspace-resource-id. + if _is_monitoring_enabled_on_mc(mc, addon_consts): + raise ArgumentUsageError( + "Azure Monitor logs is already enabled for this managed cluster.\n" + "To change the Azure Monitor logs configuration, run " + "'az aks update --disable-azure-monitor-logs' before enabling it again." + ) + # Get or create workspace resource ID workspace_resource_id = self.context.raw_param.get("workspace_resource_id") if not workspace_resource_id: @@ -8939,30 +9012,19 @@ def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: sanitize_func = self.context.external_functions.sanitize_loganalytics_ws_resource_id workspace_resource_id = sanitize_func(workspace_resource_id) - # Detect a workspace change so the DCR destination gets rewritten in postprocessing. - # The previous workspace may live on the AMP profile or, for clusters onboarded before - # the AMP switch, on the legacy addon. - container_insights = self._ensure_container_insights(mc) - old_workspace = container_insights.log_analytics_workspace_resource_id or "" - if not old_workspace and mc.addon_profiles: - addon_key = _get_monitoring_addon_key_from_consts(mc.addon_profiles, addon_consts) - addon_profile = mc.addon_profiles.get(addon_key) - old_workspace = (addon_profile.config or {}).get( - CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID, "") if addon_profile else "" - if old_workspace and old_workspace.lower() != workspace_resource_id.lower(): - # A legacy-auth cluster cannot reach a different workspace, but that case is already - # rejected above, so reaching here means the cluster uses managed identity and the DCR - # destination simply has to be rewritten in postprocessing. - self.context.set_intermediate( - "monitoring_addon_postprocessing_required", True, overwrite_exists=True) - # Write the Azure Monitor Profile rather than the legacy omsagent addon. No auth mode is # recorded here: the RP derives it, defaulting new onboardings (and re-enables of a # disabled addon) to managed identity, while preserving the existing useAADAuth value on a # cluster that is already onboarded with legacy shared-key auth. + container_insights = self._ensure_container_insights(mc) container_insights.enabled = True container_insights.log_analytics_workspace_resource_id = workspace_resource_id + # Reaching here means a genuine onboarding: the guards above reject service principal + # clusters, legacy-auth clusters and clusters that are already enabled. The DCR and the + # DCRA therefore always have to be provisioned, otherwise the agent is deployed with no + # data collection rule attached and no logs are ever ingested. + # Container network logs are applied here as well as in update_monitoring_profile_flow_logs, # because that method may run before this one depending on the order the base class invokes # them. Both write the same value, so the result is order-independent. @@ -8974,8 +9036,55 @@ def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: else CONST_CONTAINER_NETWORK_LOGS_DISABLED ) + # Provisioned once the profile above is fully built, so the DCR reflects the final shape. + self._provision_azure_monitor_logs_dcr(mc, addon_consts) + self.context.set_intermediate("monitoring_addon_enabled", True, overwrite_exists=True) + def _provision_azure_monitor_logs_dcr(self, mc: ManagedCluster, addon_consts: dict) -> None: + """Create the DCR and DCRA before the cluster PUT. + + 'az aks enable-addons -a monitoring' provisions these artifacts first and only then + updates the cluster, so by the time the RP rolls out the ama-logs DaemonSet the data + collection rule is already associated and mdsd downloads it within seconds. + + Deferring the work to postprocessing_after_mc_created inverts that order: the agent + starts before the DCRA exists, finds no configuration to download, and then backs off + for several minutes before retrying. The agent ingests nothing for the whole of that + window and restarts once the configuration finally arrives, because its liveness probe + treats the newly appeared DCR as a configuration change. Provisioning up front keeps + the flag's behaviour identical to the addon it replaces. + """ + monitoring_profile = _build_monitoring_addon_shim(mc, self.models, addon_consts) + if not (monitoring_profile and monitoring_profile.enabled): + return + + data_collection_settings = self.context.get_data_collection_settings() + # Oversized settings are dropped rather than sent, to avoid the DCR call failing with + # "Request Header Fields Too Large". + if data_collection_settings and len(str(data_collection_settings)) > 10000: + data_collection_settings = None + + self.context.external_functions.ensure_container_insights_for_monitoring( + self.cmd, + monitoring_profile, + self.context.get_subscription_id(), + self.context.get_resource_group_name(), + self.context.get_name(), + mc.location or self.context.get_location(), + remove_monitoring=False, + # The legacy-auth guard in _setup_azure_monitor_logs has already rejected anything + # that is not managed identity, so the AAD route is the only reachable one here. + aad_route=True, + create_dcr=True, + create_dcra=True, + enable_syslog=self.context.get_enable_syslog(), + data_collection_settings=data_collection_settings, + is_private_cluster=self.context.get_enable_private_cluster(), + ampls_resource_id=self.context.get_ampls_resource_id(), + enable_high_log_scale_mode=self.context.get_enable_high_log_scale_mode(), + ) + def _disable_azure_monitor_logs(self, mc: ManagedCluster) -> None: """Disable Azure Monitor logs configuration.""" addon_consts = self.context.get_addon_consts() @@ -8988,7 +9097,8 @@ def _disable_azure_monitor_logs(self, mc: ManagedCluster) -> None: if not azure_monitor_logs_enabled: return - # Check if OpenTelemetry logs are enabled and prompt for confirmation + # OpenTelemetry logs and traces are collected by the Container Insights agent, so disabling + # Azure Monitor logs necessarily turns them off too. Confirm before doing that. opentelemetry_logs_enabled = ( mc.azure_monitor_profile and mc.azure_monitor_profile.app_monitoring and @@ -8998,11 +9108,12 @@ def _disable_azure_monitor_logs(self, mc: ManagedCluster) -> None: if opentelemetry_logs_enabled and not self.context.get_yes(): msg = ( - "Disabling Azure Monitor logs will also disable OpenTelemetry logs. " - "Do you want to continue?" + "OpenTelemetry logs and traces are enabled on this cluster and are collected by " + "Azure Monitor logs. Disabling Azure Monitor logs will also disable OpenTelemetry " + "logs and traces. Do you want to continue?" ) if not prompt_y_n(msg, default="n"): - raise CLIError("Operation cancelled.") + raise DecoratorEarlyExitException() # Perform DCR/DCRA cleanup BEFORE disabling (same as aks_disable_addons lines 2796-2822). # Only MSI-auth clusters have a DCR/DCRA to clean up, so decide from local state first to @@ -9040,11 +9151,10 @@ def _disable_azure_monitor_logs(self, mc: ManagedCluster) -> None: pass # Disable through the AMP profile. The RP keeps the legacy addon in sync, so the addon - # object is intentionally left untouched here. - container_insights = self._ensure_container_insights(mc) - container_insights.enabled = False - # Reset container network logs so a later re-enable does not silently carry CNL forward. - container_insights.container_network_logs = CONST_CONTAINER_NETWORK_LOGS_DISABLED + # object is intentionally left untouched here. Every containerInsights field is reset to + # its default so a later --enable-azure-monitor-logs starts from a clean profile instead + # of silently inheriting the old workspace, syslog port, scraping choice or CNL setting. + _reset_container_insights_to_defaults(self._ensure_container_insights(mc)) # Also disable OpenTelemetry logs when disabling Azure Monitor logs if opentelemetry_logs_enabled: @@ -9080,7 +9190,7 @@ def _disable_azure_monitor_metrics(self, mc: ManagedCluster) -> None: "Do you want to continue?" ) if not prompt_y_n(msg, default="n"): - raise CLIError("Operation cancelled.") + raise DecoratorEarlyExitException() # Disable Azure Monitor metrics if mc.azure_monitor_profile is None: diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py index 788c9faba20..7552784452f 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py @@ -115,6 +115,9 @@ CONST_OUTBOUND_TYPE_BLOCK, CONST_OUTBOUND_TYPE_MANAGED_NAT_GATEWAY_V2, CONST_OUTBOUND_TYPE_NONE, + CONST_CONTAINER_NETWORK_LOGS_ENABLED, + CONST_CONTAINER_NETWORK_LOGS_DISABLED, + CONST_CONTAINER_INSIGHTS_DEFAULT_SYSLOG_PORT, ) from dateutil.parser import parse from deepdiff import DeepDiff @@ -13473,6 +13476,48 @@ def test_update_disable_azure_monitor_metrics(self): dec_mc_3.azure_monitor_profile.app_monitoring.open_telemetry_metrics ) + def test_disable_azure_monitor_metrics_aborts_when_confirmation_declined(self): + """Declining the OpenTelemetry prompt must exit gracefully, not fail the command. + + Every other declined confirmation in this decorator raises + DecoratorEarlyExitException, which aks_update turns into a no-op `return None`. + Raising CLIError here instead made `--disable-azure-monitor-metrics` the only + prompt whose decline exited non-zero, which scripts read as a failure. + """ + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"disable_azure_monitor_metrics": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster( + location="test_location", + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + metrics=self.models.ManagedClusterAzureMonitorProfileMetrics(enabled=True), + app_monitoring=self.models.ManagedClusterAzureMonitorProfileAppMonitoring( + open_telemetry_metrics=self.models.ManagedClusterAzureMonitorProfileAppMonitoringOpenTelemetryMetrics( + enabled=True, + http_port=8080, + ) + ), + ), + ) + dec.context.attach_mc(mc) + + with patch( + "azext_aks_preview.managed_cluster_decorator.prompt_y_n", return_value=False + ) as prompt_mock: + with self.assertRaises(DecoratorEarlyExitException): + dec._disable_azure_monitor_metrics(mc) + + prompt_mock.assert_called_once() + self.assertIn("OpenTelemetry metrics", prompt_mock.call_args[0][0]) + # nothing was disabled + self.assertTrue(mc.azure_monitor_profile.metrics.enabled) + otlp_metrics = mc.azure_monitor_profile.app_monitoring.open_telemetry_metrics + self.assertTrue(otlp_metrics.enabled) + self.assertEqual(otlp_metrics.http_port, 8080) + def test_update_enable_control_plane_metrics_requires_parent_metrics(self): # Update path: --enable-control-plane-metrics on a cluster that has neither # Azure Monitor metrics already enabled nor --enable-azure-monitor-metrics in @@ -13632,12 +13677,13 @@ def test_setup_azure_monitor_logs_with_omsagent_camelcase(self): CUSTOM_MGMT_AKS_PREVIEW, ) - # Create MC with omsAgent (camelCase) - as Azure API returns it + # Create MC with omsAgent (camelCase) - as Azure API returns it. The addon is disabled + # because re-enabling an already onboarded cluster is rejected. mc_1 = self.models.ManagedCluster( location="test_location", addon_profiles={ "omsAgent": self.models.ManagedClusterAddonProfile( - enabled=True, + enabled=False, config={ "logAnalyticsWorkspaceResourceID": "/old/workspace", "useAADAuth": "true", @@ -13649,7 +13695,12 @@ def test_setup_azure_monitor_logs_with_omsagent_camelcase(self): dec_1.context.set_intermediate("subscription_id", "test-subscription-id") # Call _setup_azure_monitor_logs - dec_1._setup_azure_monitor_logs(mc_1) + with patch.object( + dec_1.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_ci: + dec_1._setup_azure_monitor_logs(mc_1) # Verify: the AMP profile carries the new workspace and the legacy addon is left untouched. self.assertTrue(mc_1.azure_monitor_profile.container_insights.enabled) @@ -13662,10 +13713,8 @@ def test_setup_azure_monitor_logs_with_omsagent_camelcase(self): mc_1.addon_profiles[actual_key].config["logAnalyticsWorkspaceResourceID"], "/old/workspace", ) - # The workspace changed, so DCR postprocessing must be requested - self.assertTrue( - dec_1.context.get_intermediate("monitoring_addon_postprocessing_required") - ) + # The workspace changed, so the DCR/DCRA must be provisioned before the cluster PUT + ensure_ci.assert_called_once() def test_setup_azure_monitor_logs_with_omsagent_lowercase(self): # Test that _setup_azure_monitor_logs handles existing omsagent (lowercase) correctly @@ -13684,7 +13733,7 @@ def test_setup_azure_monitor_logs_with_omsagent_lowercase(self): location="test_location", addon_profiles={ "omsagent": self.models.ManagedClusterAddonProfile( - enabled=True, + enabled=False, config={ "logAnalyticsWorkspaceResourceID": "/old/workspace", "useAADAuth": "true", @@ -13696,7 +13745,12 @@ def test_setup_azure_monitor_logs_with_omsagent_lowercase(self): dec_1.context.set_intermediate("subscription_id", "test-subscription-id") # Call _setup_azure_monitor_logs - dec_1._setup_azure_monitor_logs(mc_1) + with patch.object( + dec_1.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec_1._setup_azure_monitor_logs(mc_1) # Verify: the AMP profile is written and the legacy addon key is left as-is self.assertTrue(mc_1.azure_monitor_profile.container_insights.enabled) @@ -19074,11 +19128,16 @@ def test_setup_azure_monitor_logs_sets_retina_flags_when_cnl_enabled(self): addon_profiles={}, ) dec.context.attach_mc(mc) + dec.context.set_intermediate("subscription_id", "test-subscription-id") with patch.object( dec.context.external_functions, "sanitize_loganalytics_ws_resource_id", side_effect=lambda x: x, + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, ): dec._setup_azure_monitor_logs(mc) @@ -19105,11 +19164,16 @@ def test_setup_azure_monitor_logs_no_retina_flags_without_cnl(self): addon_profiles={}, ) dec.context.attach_mc(mc) + dec.context.set_intermediate("subscription_id", "test-subscription-id") with patch.object( dec.context.external_functions, "sanitize_loganalytics_ws_resource_id", side_effect=lambda x: x, + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, ): dec._setup_azure_monitor_logs(mc) @@ -19121,8 +19185,8 @@ def test_setup_azure_monitor_logs_no_retina_flags_without_cnl(self): # ------------------------------------------------------------------ # Tests for _setup_azure_monitor_logs workspace change detection # ------------------------------------------------------------------ - def test_setup_azure_monitor_logs_workspace_change_triggers_postprocessing(self): - """_setup_azure_monitor_logs sets monitoring_addon_postprocessing_required when workspace changes.""" + def test_setup_azure_monitor_logs_workspace_change_provisions_dcr(self): + """_setup_azure_monitor_logs provisions the DCR/DCRA up front when the workspace changes.""" old_ws = "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/old-ws" new_ws = "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/new-ws" dec = AKSPreviewManagedClusterUpdateDecorator( @@ -19134,11 +19198,13 @@ def test_setup_azure_monitor_logs_workspace_change_triggers_postprocessing(self) }, CUSTOM_MGMT_AKS_PREVIEW, ) + # The addon is disabled: re-enabling an already onboarded cluster is rejected, so a + # workspace change is only reachable on a cluster that still carries a stale workspace. mc = self.models.ManagedCluster( location="test_location", addon_profiles={ CONST_MONITORING_ADDON_NAME: self.models.ManagedClusterAddonProfile( - enabled=True, + enabled=False, config={ "logAnalyticsWorkspaceResourceID": old_ws, "useAADAuth": "true", @@ -19153,22 +19219,27 @@ def test_setup_azure_monitor_logs_workspace_change_triggers_postprocessing(self) dec.context.external_functions, "sanitize_loganalytics_ws_resource_id", side_effect=lambda x: x, - ): + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_ci: dec._setup_azure_monitor_logs(mc) - self.assertTrue( - dec.context.get_intermediate( - "monitoring_addon_postprocessing_required", default_value=False - ) - ) + # Provisioned before the cluster PUT rather than deferred to postprocessing, so the DCRA + # exists by the time the RP rolls out the agent. + ensure_ci.assert_called_once() + self.assertTrue(ensure_ci.call_args.kwargs["create_dcr"]) + self.assertTrue(ensure_ci.call_args.kwargs["create_dcra"]) # Verify the new workspace landed on the AMP profile self.assertEqual( mc.azure_monitor_profile.container_insights.log_analytics_workspace_resource_id, new_ws, ) - def test_setup_azure_monitor_logs_same_workspace_no_postprocessing(self): - """_setup_azure_monitor_logs does NOT set monitoring_addon_postprocessing_required when workspace is unchanged.""" + def test_setup_azure_monitor_logs_same_workspace_still_provisions_dcr(self): + """Re-enabling onto the same workspace must still provision the DCR/DCRA: the disable + removed the association, so skipping it would leave the agent unattached.""" ws = "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/same-ws" dec = AKSPreviewManagedClusterUpdateDecorator( self.cmd, @@ -19183,7 +19254,7 @@ def test_setup_azure_monitor_logs_same_workspace_no_postprocessing(self): location="test_location", addon_profiles={ CONST_MONITORING_ADDON_NAME: self.models.ManagedClusterAddonProfile( - enabled=True, + enabled=False, config={ "logAnalyticsWorkspaceResourceID": ws, "useAADAuth": "true", @@ -19198,17 +19269,18 @@ def test_setup_azure_monitor_logs_same_workspace_no_postprocessing(self): dec.context.external_functions, "sanitize_loganalytics_ws_resource_id", side_effect=lambda x: x, - ): + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_ci: dec._setup_azure_monitor_logs(mc) - self.assertFalse( - dec.context.get_intermediate( - "monitoring_addon_postprocessing_required", default_value=False - ) - ) + ensure_ci.assert_called_once() + self.assertTrue(ensure_ci.call_args.kwargs["create_dcra"]) def test_setup_azure_monitor_logs_workspace_change_case_insensitive(self): - """_setup_azure_monitor_logs compares workspaces case-insensitively (no false positives on casing).""" + """Casing differences must not change the outcome: postprocessing runs either way.""" ws_lower = "/subscriptions/test/resourcegroups/test/providers/microsoft.operationalinsights/workspaces/my-ws" ws_mixed = "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/my-ws" dec = AKSPreviewManagedClusterUpdateDecorator( @@ -19224,7 +19296,7 @@ def test_setup_azure_monitor_logs_workspace_change_case_insensitive(self): location="test_location", addon_profiles={ CONST_MONITORING_ADDON_NAME: self.models.ManagedClusterAddonProfile( - enabled=True, + enabled=False, config={ "logAnalyticsWorkspaceResourceID": ws_lower, "useAADAuth": "true", @@ -19239,18 +19311,20 @@ def test_setup_azure_monitor_logs_workspace_change_case_insensitive(self): dec.context.external_functions, "sanitize_loganalytics_ws_resource_id", side_effect=lambda x: x, - ): + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_ci: dec._setup_azure_monitor_logs(mc) - # Same workspace (different casing) should NOT trigger postprocessing - self.assertFalse( - dec.context.get_intermediate( - "monitoring_addon_postprocessing_required", default_value=False - ) - ) + # Same workspace (different casing) still needs the DCR/DCRA provisioned + ensure_ci.assert_called_once() + self.assertTrue(ensure_ci.call_args.kwargs["create_dcra"]) - def test_setup_azure_monitor_logs_new_addon_no_postprocessing(self): - """_setup_azure_monitor_logs does NOT trigger postprocessing when there is no existing addon (fresh enable).""" + def test_setup_azure_monitor_logs_new_addon_provisions_dcr(self): + """A fresh enable must provision the DCR/DCRA: without them the agent runs with no data + collection rule attached, so no logs are ingested.""" new_ws = "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/new-ws" dec = AKSPreviewManagedClusterUpdateDecorator( self.cmd, @@ -19272,10 +19346,58 @@ def test_setup_azure_monitor_logs_new_addon_no_postprocessing(self): dec.context.external_functions, "sanitize_loganalytics_ws_resource_id", side_effect=lambda x: x, - ): + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_ci: + dec._setup_azure_monitor_logs(mc) + + # Fresh enable — the DCR/DCRA still have to be provisioned + ensure_ci.assert_called_once() + self.assertTrue(ensure_ci.call_args.kwargs["create_dcr"]) + self.assertTrue(ensure_ci.call_args.kwargs["create_dcra"]) + + def test_setup_azure_monitor_logs_provisions_dcr_before_cluster_put(self): + """The DCR/DCRA must be provisioned while the profile is being built, not deferred to + postprocessing_after_mc_created. + + postprocessing runs after the cluster PUT, so the RP has already rolled out the ama-logs + DaemonSet by the time the DCRA appears. The agent then starts with nothing to download, + backs off for several minutes before retrying, ingests nothing for that whole window and + restarts once the configuration finally lands. 'az aks enable-addons -a monitoring' + creates the artifacts before its PUT, and this flag has to match that ordering. + """ + ws = "/subscriptions/test/resourceGroups/test/providers/Microsoft.OperationalInsights/workspaces/ws" + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"enable_azure_monitor_logs": True, "workspace_resource_id": ws}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster(location="test_location", addon_profiles={}) + dec.context.attach_mc(mc) + dec.context.set_intermediate("subscription_id", "test-subscription-id") + + with patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda x: x, + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_ci: dec._setup_azure_monitor_logs(mc) - # Fresh enable — no old workspace to compare, should NOT trigger postprocessing + # Provisioned synchronously as part of building the profile, i.e. before the PUT. + ensure_ci.assert_called_once() + self.assertTrue(ensure_ci.call_args.kwargs["create_dcr"]) + self.assertTrue(ensure_ci.call_args.kwargs["create_dcra"]) + self.assertTrue(ensure_ci.call_args.kwargs["aad_route"]) + self.assertFalse(ensure_ci.call_args.kwargs["remove_monitoring"]) + + # And not queued for postprocessing, which would repeat the same work after the PUT. self.assertFalse( dec.context.get_intermediate( "monitoring_addon_postprocessing_required", default_value=False @@ -19537,6 +19659,7 @@ def test_reenable_monitoring_after_disable_does_not_carry_cnl(self): CUSTOM_MGMT_AKS_PREVIEW, ) dec_disable.context.attach_mc(mc) + dec_disable.context.set_intermediate("subscription_id", "test-subscription-id") dec_disable.client = Mock() dec_disable.client.get = Mock(return_value=mc) dec_disable._disable_azure_monitor_logs(mc) @@ -19561,7 +19684,12 @@ def test_reenable_monitoring_after_disable_does_not_carry_cnl(self): ) dec_enable.context.attach_mc(mc) dec_enable.context.set_intermediate("subscription_id", "test-subscription-id") - dec_enable._setup_azure_monitor_logs(mc) + with patch.object( + dec_enable.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec_enable._setup_azure_monitor_logs(mc) # Monitoring is back on, but CNL is not silently carried forward container_insights = mc.azure_monitor_profile.container_insights @@ -19583,7 +19711,10 @@ def _addon_consts(self): return dec.context.get_addon_consts() def _brownfield_legacy_mc(self, use_aad_auth="false", addon_enabled=True): - """A cluster onboarded with legacy auth, which the RP has mirrored into the AMP profile.""" + """A cluster onboarded with legacy auth, which the RP has mirrored into the AMP profile. + + The RP keeps the two profiles in sync, so containerInsights.enabled tracks the addon. + """ config = {"logAnalyticsWorkspaceResourceID": "/subscriptions/test/workspace"} if use_aad_auth is not None: config[CONST_MONITORING_USING_AAD_MSI_AUTH] = use_aad_auth @@ -19597,7 +19728,7 @@ def _brownfield_legacy_mc(self, use_aad_auth="false", addon_enabled=True): }, azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( - enabled=True, + enabled=addon_enabled, log_analytics_workspace_resource_id="/subscriptions/test/workspace", ) ), @@ -19788,17 +19919,24 @@ def test_enable_azure_monitor_logs_allowed_when_legacy_addon_is_disabled(self): dec = self._enable_amp_logs_decorator("/subscriptions/test/workspace") mc = self._brownfield_legacy_mc("false", addon_enabled=False) dec.context.attach_mc(mc) + dec.context.set_intermediate("subscription_id", "test-subscription-id") with patch.object( dec.context.external_functions, "sanitize_loganalytics_ws_resource_id", side_effect=lambda x: x, + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, ): dec._setup_azure_monitor_logs(mc) self.assertTrue(mc.azure_monitor_profile.container_insights.enabled) - def test_workspace_change_allowed_on_aad_auth_cluster(self): + def test_workspace_change_rejected_on_already_onboarded_cluster(self): + """Re-running the enable flag on an onboarded cluster is rejected, so a workspace change + has to go through a disable first.""" dec = self._enable_amp_logs_decorator("/subscriptions/test/other-workspace") mc = self._brownfield_legacy_mc("true") dec.context.attach_mc(mc) @@ -19808,15 +19946,46 @@ def test_workspace_change_allowed_on_aad_auth_cluster(self): "sanitize_loganalytics_ws_resource_id", side_effect=lambda x: x, ): + with self.assertRaises(ArgumentUsageError) as ctx: + dec._setup_azure_monitor_logs(mc) + + self.assertIn("already enabled", str(ctx.exception)) + # the original workspace is left untouched + self.assertEqual( + mc.azure_monitor_profile.container_insights.log_analytics_workspace_resource_id, + "/subscriptions/test/workspace", + ) + + def test_workspace_change_allowed_after_disable(self): + """A cluster that was disabled is no longer onboarded, so re-enabling against a different + workspace is allowed and flags the DCR destination for a rewrite.""" + dec = self._enable_amp_logs_decorator("/subscriptions/test/other-workspace") + mc = self._brownfield_legacy_mc("true", addon_enabled=False) + # a cluster disabled before the reset behaviour existed still carries the stale workspace + mc.azure_monitor_profile.container_insights.log_analytics_workspace_resource_id = ( + "/subscriptions/test/workspace" + ) + dec.context.attach_mc(mc) + dec.context.set_intermediate("subscription_id", "test-subscription-id") + + with patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda x: x, + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_ci: dec._setup_azure_monitor_logs(mc) self.assertEqual( mc.azure_monitor_profile.container_insights.log_analytics_workspace_resource_id, "/subscriptions/test/other-workspace", ) - self.assertTrue( - dec.context.get_intermediate("monitoring_addon_postprocessing_required") - ) + # The DCR destination is rewritten up front, before the cluster PUT + ensure_ci.assert_called_once() + self.assertTrue(ensure_ci.call_args.kwargs["create_dcr"]) def test_hlsm_rejected_on_legacy_auth_cluster_with_amp_profile(self): """HLSM needs a DCR, which legacy shared-key clusters do not have.""" @@ -20146,6 +20315,468 @@ def test_opentelemetry_logs_traces_grpc_port_set_independently(self): self.assertEqual(otlp_logs.grpc_port, 8082) self.assertIsNone(otlp_logs.http_port) + # ------------------------------------------------------------------ + # Re-onboarding an already onboarded cluster must be rejected, mirroring + # 'az aks enable-addons -a monitoring'. + # ------------------------------------------------------------------ + def _amp_onboarded_mc(self, **container_insights_kwargs): + """A cluster onboarded through the AMP profile only (no legacy addon).""" + settings = { + "enabled": True, + "log_analytics_workspace_resource_id": "/subscriptions/test/workspace", + } + settings.update(container_insights_kwargs) + return self.models.ManagedCluster( + location="test_location", + service_principal_profile=self.models.ManagedClusterServicePrincipalProfile( + client_id="msi" + ), + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=( + self.models.ManagedClusterAzureMonitorProfileContainerInsights(**settings) + ) + ), + ) + + def test_enable_azure_monitor_logs_rejected_when_container_insights_already_enabled(self): + dec = self._enable_amp_logs_decorator("/subscriptions/test/workspace") + mc = self._amp_onboarded_mc() + dec.context.attach_mc(mc) + + with self.assertRaises(ArgumentUsageError) as ctx: + dec._setup_azure_monitor_logs(mc) + + message = str(ctx.exception) + self.assertIn("already enabled for this managed cluster", message) + self.assertIn("--disable-azure-monitor-logs", message) + + def test_enable_azure_monitor_logs_rejected_before_provisioning_default_workspace(self): + """The already-enabled rejection must happen before a workspace is created.""" + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"enable_azure_monitor_logs": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + dec.context.attach_mc(self._amp_onboarded_mc()) + + with patch.object( + dec.context.external_functions, + "ensure_default_log_analytics_workspace_for_monitoring", + ) as ensure_workspace_mock: + with self.assertRaises(ArgumentUsageError): + dec._setup_azure_monitor_logs(dec.context.mc) + + ensure_workspace_mock.assert_not_called() + + def test_enable_azure_monitor_logs_allowed_when_container_insights_disabled(self): + dec = self._enable_amp_logs_decorator("/subscriptions/test/workspace") + mc = self._amp_onboarded_mc(enabled=False, log_analytics_workspace_resource_id="") + dec.context.attach_mc(mc) + dec.context.set_intermediate("subscription_id", "test-subscription-id") + + with patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda x: x, + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec._setup_azure_monitor_logs(mc) + + self.assertTrue(mc.azure_monitor_profile.container_insights.enabled) + + def test_legacy_auth_rejection_wins_over_already_enabled_rejection(self): + """A legacy-auth cluster is also 'already enabled', but the migration message is the + actionable one, so it must be the error the user sees.""" + dec = self._enable_amp_logs_decorator("/subscriptions/test/workspace") + mc = self._brownfield_legacy_mc("false") + dec.context.attach_mc(mc) + + with self.assertRaises(ArgumentUsageError) as ctx: + dec._setup_azure_monitor_logs(mc) + + self.assertIn("legacy", str(ctx.exception)) + + # ------------------------------------------------------------------ + # The AMP path is managed-identity only, so service principal clusters are rejected. + # ------------------------------------------------------------------ + def test_enable_azure_monitor_logs_rejected_on_service_principal_cluster_update(self): + dec = self._enable_amp_logs_decorator("/subscriptions/test/workspace") + mc = self.models.ManagedCluster( + location="test_location", + service_principal_profile=self.models.ManagedClusterServicePrincipalProfile( + client_id="00000000-0000-0000-0000-000000000001" + ), + ) + dec.context.attach_mc(mc) + + with patch.object( + dec.context.external_functions, + "ensure_default_log_analytics_workspace_for_monitoring", + ) as ensure_workspace_mock: + with self.assertRaises(ArgumentUsageError) as ctx: + dec._setup_azure_monitor_logs(mc) + + self.assertIn("service principal", str(ctx.exception)) + # rejected before any workspace is provisioned on the user's behalf + ensure_workspace_mock.assert_not_called() + + def test_enable_azure_monitor_logs_allowed_on_msi_cluster_update(self): + dec = self._enable_amp_logs_decorator("/subscriptions/test/workspace") + mc = self.models.ManagedCluster( + location="test_location", + service_principal_profile=self.models.ManagedClusterServicePrincipalProfile( + client_id="msi" + ), + ) + dec.context.attach_mc(mc) + dec.context.set_intermediate("subscription_id", "test-subscription-id") + + with patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda x: x, + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec._setup_azure_monitor_logs(mc) + + self.assertTrue(mc.azure_monitor_profile.container_insights.enabled) + + def test_enable_azure_monitor_logs_rejected_on_service_principal_cluster_create(self): + """On create the service principal profile is populated from --service-principal before + the addon profiles are set up, so the rejection applies there too.""" + dec = AKSPreviewManagedClusterCreateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_logs": True, + "workspace_resource_id": "/subscriptions/test/workspace", + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster( + location="test_location", + service_principal_profile=self.models.ManagedClusterServicePrincipalProfile( + client_id="00000000-0000-0000-0000-000000000001", secret="secret" + ), + ) + dec.context.attach_mc(mc) + + with patch.object( + dec.context.external_functions, + "ensure_default_log_analytics_workspace_for_monitoring", + ) as ensure_workspace_mock: + with self.assertRaises(ArgumentUsageError) as ctx: + dec._setup_azure_monitor_logs(mc) + + self.assertIn("service principal", str(ctx.exception)) + ensure_workspace_mock.assert_not_called() + + def test_enable_azure_monitor_logs_allowed_without_service_principal_profile_on_create(self): + """A managed identity cluster has no service principal profile at all.""" + dec = AKSPreviewManagedClusterCreateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_logs": True, + "workspace_resource_id": "/subscriptions/test/workspace", + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster(location="test_location") + dec.context.attach_mc(mc) + + with patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda x: x, + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec._setup_azure_monitor_logs(mc) + + self.assertTrue(mc.azure_monitor_profile.container_insights.enabled) + + # ------------------------------------------------------------------ + # Disabling resets every containerInsights field back to its default. + # ------------------------------------------------------------------ + def _disable_amp_logs_decorator(self, yes=True): + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"disable_azure_monitor_logs": True, "yes": yes}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + dec.client = Mock() + return dec + + def test_disable_azure_monitor_logs_resets_container_insights_to_defaults(self): + dec = self._disable_amp_logs_decorator() + mc = self._amp_onboarded_mc( + syslog_port=29000, + disable_prometheus_metrics_scraping=True, + container_network_logs=CONST_CONTAINER_NETWORK_LOGS_ENABLED, + ) + dec.context.attach_mc(mc) + dec.client.get = Mock(return_value=mc) + + with patch.object( + dec.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec.context, "get_name", return_value="test-cluster" + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec._disable_azure_monitor_logs(mc) + + container_insights = mc.azure_monitor_profile.container_insights + self.assertFalse(container_insights.enabled) + # The workspace id is deliberately preserved: blanking it makes the RP mirror an empty + # string into the omsagent addon config, and ARM then rejects every later write of the + # cluster with LinkedInvalidPropertyId. + self.assertEqual( + container_insights.log_analytics_workspace_resource_id, "/subscriptions/test/workspace" + ) + self.assertEqual( + container_insights.syslog_port, CONST_CONTAINER_INSIGHTS_DEFAULT_SYSLOG_PORT + ) + self.assertFalse(container_insights.disable_prometheus_metrics_scraping) + self.assertEqual( + container_insights.container_network_logs, CONST_CONTAINER_NETWORK_LOGS_DISABLED + ) + + def test_disable_azure_monitor_logs_never_blanks_workspace_id(self): + """Regression: an empty logAnalyticsWorkspaceResourceID makes ARM reject every later + write of the cluster with LinkedInvalidPropertyId, breaking unrelated 'az aks update' + calls, because the RP mirrors the AMP value into the omsagent addon config.""" + dec = self._disable_amp_logs_decorator() + mc = self._amp_onboarded_mc() + dec.context.attach_mc(mc) + dec.client.get = Mock(return_value=mc) + + with patch.object( + dec.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec.context, "get_name", return_value="test-cluster" + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec._disable_azure_monitor_logs(mc) + + self.assertTrue( + mc.azure_monitor_profile.container_insights.log_analytics_workspace_resource_id + ) + + def test_disable_azure_monitor_logs_emits_every_default_field(self): + """Every reset field must be present in the payload: the RP only overwrites fields it + receives, so a field left out would survive the disable.""" + dec = self._disable_amp_logs_decorator() + mc = self._amp_onboarded_mc(syslog_port=29000) + dec.context.attach_mc(mc) + dec.client.get = Mock(return_value=mc) + + with patch.object( + dec.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec.context, "get_name", return_value="test-cluster" + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec._disable_azure_monitor_logs(mc) + + payload = dict(mc.azure_monitor_profile.container_insights) + for field in ( + "enabled", + "syslogPort", + "disablePrometheusMetricsScraping", + "containerNetworkLogs", + ): + self.assertIn(field, payload) + + def test_disable_azure_monitor_logs_reset_survives_reenable(self): + """After a disable, re-enabling starts from a clean profile.""" + dec_disable = self._disable_amp_logs_decorator() + mc = self._amp_onboarded_mc( + syslog_port=29000, + disable_prometheus_metrics_scraping=True, + container_network_logs=CONST_CONTAINER_NETWORK_LOGS_ENABLED, + ) + dec_disable.context.attach_mc(mc) + dec_disable.context.set_intermediate("subscription_id", "test-subscription-id") + dec_disable.client.get = Mock(return_value=mc) + + with patch.object( + dec_disable.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec_disable.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec_disable.context, "get_name", return_value="test-cluster" + ), patch.object( + dec_disable.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec_disable._disable_azure_monitor_logs(mc) + + dec_enable = self._enable_amp_logs_decorator("/subscriptions/test/new-workspace") + dec_enable.context.attach_mc(mc) + dec_enable.context.set_intermediate("subscription_id", "test-subscription-id") + with patch.object( + dec_enable.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda x: x, + ), patch.object( + dec_enable.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec_enable._setup_azure_monitor_logs(mc) + + container_insights = mc.azure_monitor_profile.container_insights + self.assertTrue(container_insights.enabled) + self.assertEqual( + container_insights.log_analytics_workspace_resource_id, + "/subscriptions/test/new-workspace", + ) + self.assertEqual( + container_insights.container_network_logs, CONST_CONTAINER_NETWORK_LOGS_DISABLED + ) + + # ------------------------------------------------------------------ + # Disabling also turns off OpenTelemetry logs and traces, so it must be confirmed. + # ------------------------------------------------------------------ + def _amp_onboarded_mc_with_otlp_logs(self, enabled=True): + mc = self._amp_onboarded_mc() + mc.azure_monitor_profile.app_monitoring = ( + self.models.ManagedClusterAzureMonitorProfileAppMonitoring( + open_telemetry_logs_and_traces=self.models. + ManagedClusterAzureMonitorProfileAppMonitoringOpenTelemetryLogsAndTraces( + enabled=enabled, http_port=8080, grpc_port=8081 + ) + ) + ) + return mc + + def test_disable_azure_monitor_logs_prompts_when_otlp_logs_and_traces_enabled(self): + dec = self._disable_amp_logs_decorator(yes=False) + mc = self._amp_onboarded_mc_with_otlp_logs() + dec.context.attach_mc(mc) + dec.client.get = Mock(return_value=mc) + + with patch( + "azext_aks_preview.managed_cluster_decorator.prompt_y_n", return_value=True + ) as prompt_mock, patch.object( + dec.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec.context, "get_name", return_value="test-cluster" + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec._disable_azure_monitor_logs(mc) + + prompt_mock.assert_called_once() + self.assertIn("OpenTelemetry logs and traces", prompt_mock.call_args[0][0]) + self.assertFalse(mc.azure_monitor_profile.container_insights.enabled) + otlp_logs = mc.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces + self.assertFalse(otlp_logs.enabled) + self.assertIsNone(otlp_logs.http_port) + self.assertIsNone(otlp_logs.grpc_port) + + def test_disable_azure_monitor_logs_aborts_when_confirmation_declined(self): + dec = self._disable_amp_logs_decorator(yes=False) + mc = self._amp_onboarded_mc_with_otlp_logs() + dec.context.attach_mc(mc) + dec.client.get = Mock(return_value=mc) + + with patch( + "azext_aks_preview.managed_cluster_decorator.prompt_y_n", return_value=False + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_mock: + with self.assertRaises(DecoratorEarlyExitException): + dec._disable_azure_monitor_logs(mc) + + # nothing was torn down and the profile is untouched + ensure_mock.assert_not_called() + self.assertTrue(mc.azure_monitor_profile.container_insights.enabled) + self.assertTrue( + mc.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces.enabled + ) + + def test_disable_azure_monitor_logs_skips_prompt_with_yes(self): + dec = self._disable_amp_logs_decorator(yes=True) + mc = self._amp_onboarded_mc_with_otlp_logs() + dec.context.attach_mc(mc) + dec.client.get = Mock(return_value=mc) + + with patch( + "azext_aks_preview.managed_cluster_decorator.prompt_y_n", return_value=True + ) as prompt_mock, patch.object( + dec.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec.context, "get_name", return_value="test-cluster" + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec._disable_azure_monitor_logs(mc) + + prompt_mock.assert_not_called() + self.assertFalse(mc.azure_monitor_profile.container_insights.enabled) + + def test_disable_azure_monitor_logs_no_prompt_when_otlp_logs_disabled(self): + dec = self._disable_amp_logs_decorator(yes=False) + mc = self._amp_onboarded_mc_with_otlp_logs(enabled=False) + dec.context.attach_mc(mc) + dec.client.get = Mock(return_value=mc) + + with patch( + "azext_aks_preview.managed_cluster_decorator.prompt_y_n", return_value=True + ) as prompt_mock, patch.object( + dec.context, "get_subscription_id", return_value="test-sub" + ), patch.object( + dec.context, "get_resource_group_name", return_value="test-rg" + ), patch.object( + dec.context, "get_name", return_value="test-cluster" + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ): + dec._disable_azure_monitor_logs(mc) + + prompt_mock.assert_not_called() + self.assertFalse(mc.azure_monitor_profile.container_insights.enabled) + if __name__ == "__main__": unittest.main() diff --git a/src/aks-preview/azmon-logs-cli-enhancements-spec.md b/src/aks-preview/azmon-logs-cli-enhancements-spec.md index 07704486d73..6201c3f22c2 100644 --- a/src/aks-preview/azmon-logs-cli-enhancements-spec.md +++ b/src/aks-preview/azmon-logs-cli-enhancements-spec.md @@ -94,6 +94,12 @@ We are moving Azure Monitor Logs (Container Insights) onboarding off the legacy | R3 | Do not expose `disableCustomMetrics` (removed from API) | P0 | | R4 | Add Prometheus-scraping + syslog-port controls to the AMP path | P0 | | R5 | Warn when the legacy auth flag is used with `enable-addons monitoring` | P0 | +| R6 | Reject `--enable-azure-monitor-logs` on an already-onboarded cluster | P0 | +| R7 | `--disable-azure-monitor-logs` resets `containerInsights` to defaults | P0 | +| R8 | Reject `--enable-azure-monitor-logs` on service principal clusters | P0 | +| R9 | Reject `--enable-azure-monitor-logs` on a legacy-auth onboarded cluster | P0 | +| R10 | Confirm before disabling when OTLP logs & traces are on | P0 | +| R11 | Every enable provisions the DCR and DCRA, before the cluster update | P0 | ### R0 — OTLP gRPC port overrides @@ -213,6 +219,86 @@ az aks update -g -n \ - Warning shown once when the flag is explicitly provided; not shown when omitted. - Command completes successfully; resulting `useAADAuth` value is unchanged. +### R6 — Reject re-onboarding an already-onboarded cluster + +`az aks enable-addons -a monitoring` fails when the addon is already enabled and tells the user to disable it first. The AMP path must behave the same way, so that a re-run cannot silently recreate a default workspace or re-provision DCR/DCE/DCRA artifacts and mask a mistake such as a mistyped `--workspace-resource-id`. + +- `az aks update --enable-azure-monitor-logs` errors when `containerInsights.enabled` is already `true` (or the legacy `omsagent` addon is enabled, which the RP mirrors into that field). +- The error names `--disable-azure-monitor-logs` as the way to change the configuration. + +**Acceptance criteria** + +- Re-running the flag on an onboarded cluster errors, and errors *before* a default workspace is provisioned. +- The flag still succeeds when `containerInsights.enabled` is `false` (a fresh enable, or a re-enable after a disable). +- Changing the workspace requires a disable first. + +### R7 — `--disable-azure-monitor-logs` resets `containerInsights` to defaults + +The RP copies a `containerInsights` field from the request onto the cluster only when the field is present (`ApplyAzureMonitorProfileContainerInsights`), so any field left unset survives the disable and is silently inherited by the next enable. + +- Disabling writes the behavioural fields explicitly: `enabled=false`, `syslogPort=28330`, `disablePrometheusMetricsScraping=false`, `containerNetworkLogs="Disabled"`. +- `logAnalyticsWorkspaceResourceId` is deliberately **not** blanked. It is a resource-id typed property, and the RP mirrors the AMP value into `addonProfiles.omsagent.config.logAnalyticsWorkspaceResourceID`; ARM then rejects every subsequent write of the cluster with `LinkedInvalidPropertyId`, breaking even updates unrelated to monitoring. The stale id is inert while `enabled` is false and the enable path always overwrites it with a freshly resolved workspace, so nothing is inherited. + +**Acceptance criteria** + +- The four behavioural fields are present in the PUT payload after a disable. +- The workspace id remains a valid resource id (never `""`), and a later `az aks update` of any kind still succeeds. +- A subsequent `--enable-azure-monitor-logs` starts from the defaults rather than inheriting the previous syslog port, scraping choice or CNL setting. + +### R11 — Every enable provisions the DCR and DCRA, before the cluster update + +Postprocessing previously ran only when the workspace changed, so a fresh enable — or a re-enable onto the same workspace after a disable removed the association — deployed the agent with no data collection rule attached and silently ingested nothing. + +Provisioning also has to happen *before* the cluster PUT, not in `postprocessing_after_mc_created`. The RP rolls out the `ama-logs` DaemonSet as part of the PUT, so creating the artifacts afterwards means the agent starts before the association exists. `dcr-config-parser.rb` then finds no configuration chunk, logs `Exception while parsing dcr : No JSON file found in the specified directory`, and mdsd backs off for several minutes before retrying; the agent ingests nothing for that window and is restarted by its own liveness probe once the configuration finally lands. `az aks enable-addons -a monitoring` creates the DCR/DCRA before its PUT, and this flag matches that ordering. + +- The enable path calls `ensure_container_insights_for_monitoring` directly, once the AMP profile has been fully built, since the guards above have already rejected the no-op cases. +- `monitoring_addon_postprocessing_required` is therefore *not* set by the enable path; leaving it set would repeat the same work after the PUT. +- `ensure_container_insights_for_monitoring` is idempotent and rewrites the DCR destination, so this also covers the workspace-change case. + +**Acceptance criteria** + +- After `--enable-azure-monitor-logs`, the cluster has a `ContainerInsightsExtension` DCRA pointing at an `MSCI--` DCR whose destination is the resolved workspace. +- The DCRA exists before the cluster update completes. +- The agent logs no `Exception while parsing dcr` on first start and does not restart to pick up the configuration. +- Disabling removes the association. + +### R8 — Reject `--enable-azure-monitor-logs` on service principal clusters + +The AMP profile has no shared-key/`useAADAuth` concept: the agent reaches the workspace with the cluster's managed identity, which a service principal cluster does not have. + +- Clusters whose `servicePrincipalProfile.clientId` is set to anything other than `msi` are rejected on both `az aks create` and `az aks update`. +- A missing `servicePrincipalProfile` means managed identity and is allowed. + +**Acceptance criteria** + +- The rejection happens before a default workspace is provisioned. +- The error points at `az aks update --enable-managed-identity` as the fix. + +### R9 — Reject `--enable-azure-monitor-logs` on a legacy-auth onboarded cluster + +A cluster onboarded with shared-key auth keeps that auth mode server-side, so the flag cannot be honoured as asked. Absent or empty `useAADAuth` counts as legacy, matching the RP's derivation. + +- Checked ahead of R6, so the actionable migration message wins for a legacy-auth cluster. +- A *disabled* `omsagent` addon is a fresh onboarding as far as the RP is concerned and must not be blocked. + +**Acceptance criteria** + +- `useAADAuth` of `false`, `""` or absent on an enabled addon errors with a link to the managed-identity migration doc. +- The rejection happens before a default workspace is provisioned. + +### R10 — Confirm before disabling when OTLP logs & traces are on + +OpenTelemetry logs and traces are collected by the Container Insights agent, so disabling Azure Monitor logs necessarily turns them off too. + +- Prompt for confirmation, defaulting to "no"; `--yes` skips the prompt. +- Declining exits cleanly without tearing down the DCR/DCRA or modifying the profile. +- Accepting disables `openTelemetryLogsAndTraces` and clears its HTTP and gRPC ports. + +**Acceptance criteria** + +- The prompt appears only when `openTelemetryLogsAndTraces.enabled` is true and `--yes` was not passed. +- Declining leaves both `containerInsights` and `appMonitoring` untouched. + ## 6. Target end-state (the two commands) | Command | Writes | Auth | Extra controls | diff --git a/src/aks-preview/linter_exclusions.yml b/src/aks-preview/linter_exclusions.yml index f5d57739834..a3f51615674 100644 --- a/src/aks-preview/linter_exclusions.yml +++ b/src/aks-preview/linter_exclusions.yml @@ -129,6 +129,12 @@ aks create: enable_azure_monitor_logs: rule_exclusions: - option_length_too_long + enable_prometheus_metrics_scraping: + rule_exclusions: + - option_length_too_long + disable_prometheus_metrics_scraping: + rule_exclusions: + - option_length_too_long enable_opentelemetry_metrics: rule_exclusions: - option_length_too_long @@ -326,6 +332,12 @@ aks update: enable_azure_monitor_logs: rule_exclusions: - option_length_too_long + enable_prometheus_metrics_scraping: + rule_exclusions: + - option_length_too_long + disable_prometheus_metrics_scraping: + rule_exclusions: + - option_length_too_long enable_msi_auth_for_monitoring: rule_exclusions: - option_length_too_long From 68c295f47af71628582865dc0ba8cffc4a209046 Mon Sep 17 00:00:00 2001 From: Sunil Yadav Date: Thu, 17 Sep 2026 02:37:15 +0000 Subject: [PATCH 3/8] Remove temp readmes --- REPO-GUIDE.md | 363 ------------------ .../azmon-logs-cli-enhancements-spec.md | 307 --------------- 2 files changed, 670 deletions(-) delete mode 100644 REPO-GUIDE.md delete mode 100644 src/aks-preview/azmon-logs-cli-enhancements-spec.md diff --git a/REPO-GUIDE.md b/REPO-GUIDE.md deleted file mode 100644 index a6ba844e86d..00000000000 --- a/REPO-GUIDE.md +++ /dev/null @@ -1,363 +0,0 @@ -# `azure-cli-extensions` — How This Repo Works - -A practical orientation guide for someone new to the repo. Everything below was verified against -the current checkout; file paths and line references point at real code. - -`aks-preview` is used as the worked example throughout, since that is where AKS / Azure Monitor -work lands. - ---- - -## 1. What this repo actually is - -Per `README.md`, the repo serves **two independent purposes**: - -| Purpose | Location | What it is | -|---|---|---| -| Extension **source code** | `src//` | 211 extension folders, each an installable Python wheel | -| The **extension index** | `src/index.json` | 6.5 MB catalogue of *published* wheels (220 extensions, 386 `aks-preview` versions) | - -These are decoupled. `src/index.json` is what `az extension add --name ` reads (synced to -`https://aka.ms/azure-cli-extension-index-v1` every few minutes). A wheel can be indexed without -its source living here, and source can live here without being indexed. - -> **Key mental model:** this repo does *not* ship the CLI itself. It ships **add-on command -> modules** that plug into `azure-cli` core at runtime. Core lives in the separate -> [`Azure/azure-cli`](https://github.com/Azure/azure-cli) repo, and you need a clone of it to -> develop here (see §5). - -### Top-level layout - -``` -azure-cli-extensions/ -├── src/ # all extensions + index.json -│ ├── index.json # published wheel catalogue (hash, version, metadata) -│ ├── service_name.json # top-level command group → Azure service name + docs URL -│ │ # (checked by scripts/ci/service_name.py) -│ └── aks-preview/ # one extension -├── scripts/ci/ # index + release automation (update_index.py, test_index.py, ...) -├── scripts/automation/ # build_package.py, create_release_tag.py -├── .azure-pipelines/ # ADO templates (azdev_setup.yml, variables.yml) -├── azure-pipelines.yml # the PR CI definition -├── .github/workflows/ # GitHub-side automation (release trigger, linter comments) -├── .github/CODEOWNERS # per-extension reviewers -└── docs/ # points at the canonical azure-cli authoring docs -``` - ---- - -## 2. Anatomy of a single extension - -``` -src/aks-preview/ -├── setup.py # VERSION = "22.0.0b6" ← the published version -├── setup.cfg # bdist_wheel config -├── HISTORY.rst # changelog; has a "Pending" section (see §6) -├── README.rst -├── linter_exclusions.yml # per-extension linter waivers -├── azcli_aks_live_test/ # live-test harness (ADO pipelines + configs) -└── azext_aks_preview/ # ← the actual Python package that gets imported - ├── azext_metadata.json # minCliCoreVersion, isPreview flag - ├── __init__.py # COMMAND_LOADER_CLS — the entry point - ├── commands.py # command name → Python function mapping - ├── _params.py # CLI argument definitions - ├── _help.py # help text - ├── _validators.py # argument validation - ├── _format.py # table output transformers - ├── _client_factory.py # builds SDK clients ← request plumbing - ├── custom.py # command implementations - ├── managed_cluster_decorator.py # builds the ManagedCluster request body - ├── agentpool_decorator.py # same, for node pools - ├── vendored_sdks/ # generated ARM SDK (excluded from static checks) - ├── aaz/ # aaz-dev-tools generated commands - └── tests/latest/ # tests + 289 HTTP recordings -``` - -The folder naming is a hard convention: directory `src//` contains package `azext_/`. - ---- - -## 3. The command flow — from `az` to ARM - -This is the core thing to understand. - -``` - $ az aks create -g rg -n mycluster --enable-addons monitoring - │ - │ 1. azure-cli core discovers installed extensions and calls COMMAND_LOADER_CLS - ▼ - azext_aks_preview/__init__.py → ContainerServiceCommandsLoader - │ load_command_table(): - │ a) load_aaz_command_table(...) ← AAZ-generated commands first - │ b) commands.load_command_table() ← handwritten ones, which OVERRIDE AAZ - │ load_arguments() → _params.py - ▼ - commands.py g.custom_command("create", "aks_create", supports_no_wait=True) - │ (inside a command_group bound to client_factory=cf_managed_clusters) - ▼ - custom.py::aks_create (line 1225) - │ builds a decorator, then: - │ mc = aks_create_decorator.construct_mc_profile_preview() (line 1518) - │ return aks_create_decorator.create_mc(mc) (line 1526) - ▼ - managed_cluster_decorator.py :: AKSPreviewManagedClusterCreateDecorator (line 4487) - │ set_up_network_profile(mc), set_up_addon_profiles(mc), - │ set_up_api_server_access_profile(mc), ... ← one method per feature - │ each mutates the ManagedCluster model object - ▼ - put_mc() (line 6336) - │ self.client.begin_create_or_update(...) (line 6342) - │ or sdk_no_wait(...) when --no-wait (line 6355) - ▼ - vendored_sdks/azure_mgmt_preview_aks → ContainerServiceClient - │ api_version = "2026-06-02-preview" (_configuration.py:52) - ▼ - HTTPS PUT to Azure Resource Manager -``` - -### 3a. Where requests are built — **two** distinct mechanisms - -The repo mixes two generations of tooling. Knowing which one a command uses tells you where to edit. - -#### Mechanism A — vendored SDK + decorators (handwritten; most of `aks-preview`) - -1. **`vendored_sdks/`** holds an AutoRest/TypeSpec-generated ARM SDK. It is checked in - deliberately, and `README.md` notes it is excluded from CI static checking precisely because - generated code fails those checks. Two clients are vendored here: - `azure_mgmt_preview_aks` and `azure_mgmt_preview_aks_pis`. - -2. **`_client_factory.py`** registers those SDKs as CLI "custom resource types" and exposes one - factory per operation group: - - ```python - CUSTOM_MGMT_AKS_PREVIEW = CustomResourceType( - 'azext_aks_preview.vendored_sdks.azure_mgmt_preview_aks', 'ContainerServiceClient') - - def cf_managed_clusters(cli_ctx, *_): - return get_container_service_client(cli_ctx).managed_clusters - ``` - - `commands.py` attaches these via `client_factory=cf_managed_clusters`, so the command - implementation receives an authenticated, subscription-scoped client for free. - -3. **Decorators build the request body.** For anything as large as `ManagedCluster`, the payload - is assembled by an ordered chain of `set_up_*` / `update_*` methods, each owning exactly one - feature. This is the pattern to follow when adding a flag: - - add the argument in `_params.py` - - add a `set_up_` (create) and `update_` (update) method in the decorator - - call it from `construct_mc_profile_preview()` / `update_mc_profile_preview()` (line 9027) - - the final `put_mc()` sends it - - Constants for feature names/addons live in `_consts.py`. - -#### Mechanism B — AAZ (`aaz/`, generated by `aaz-dev-tools`) - -Newer commands are **declarative and fully generated from the Swagger/TypeSpec spec** — no -handwritten SDK. Example: `aaz/latest/aks/safeguards/_show.py`. - -Here the HTTP request is expressed directly as a class: - -```python -class DeploymentSafeguardsGet(AAZHttpOperation): - @property - def url(self): ... # ARM template URL - @property - def method(self): ... # "GET" - @property - def url_parameters(self): ... # subscriptionId, resourceGroupName, ... - @property - def query_parameters(self): - return { "api-version", "2025-05-02-preview" } - @property - def header_parameters(self): ... - def on_200(self, session): ... # response deserialization schema -``` - -Note each mechanism pins its **own** api-version (`2026-06-02-preview` for the vendored SDK, -`2025-05-02-preview` for this AAZ command). They are independent. - -> ⚠️ Files under `aaz/` are marked `# Code generated by aaz-dev-tools` with `pylint: skip-file`. -> **Do not hand-edit them** — regenerate with `aaz-dev-tools`. To customize behaviour, use the -> `pre_operations()` / `post_operations()` hooks, or override the command in `commands.py` -> (remember the loader applies handwritten commands *after* AAZ, so they win). - ---- - -## 4. Adding new API surface - -When the AKS RP ships a new api-version and you need new fields: - -1. **Regenerate / re-vendor the SDK** into `vendored_sdks/` (AutoRest or TypeSpec). The api-version - default lives in `vendored_sdks/azure_mgmt_preview_aks/_configuration.py`. -2. **Expose the flag** in `_params.py` (+ `_validators.py`, `_help.py`, `_consts.py`). -3. **Wire the payload** in the relevant decorator (`managed_cluster_decorator.py` / - `agentpool_decorator.py`). -4. **Register** the command in `commands.py` if it is new. -5. **Test** (§5) and **changelog** (§6). - -`swagger_to_sdk_config.json` at the repo root drives spec→SDK automation. - ---- - -## 5. Development cycle - -### Setup (mirrors CI exactly — `.azure-pipelines/templates/azdev_setup.yml`) - -```bash -python -m venv env && source env/bin/activate -pip install -U pip -pip install --upgrade azdev==0.2.13 # CI pins this exact version -pip install build wheel - -git clone -q --single-branch -b dev https://github.com/Azure/azure-cli.git ../azure-cli -azdev setup -c ../azure-cli -r . # -c = CLI core repo, -r = this repo -``` - -`azdev setup -r .` installs every extension in `src/` in editable mode, so edits take effect -immediately with no reinstall. - -To work on just one extension: - -```bash -azdev extension add aks-preview -az extension list # confirm it resolves to your working tree -az aks create --help -``` - -### Inner loop - -```bash -# edit code, then run it straight away — no build step needed -az aks create -g rg -n c1 --enable-addons monitoring --debug - -# see the exact HTTP request/response ARM receives -az aks create ... --debug # or --verbose -``` - -`--debug` is the fastest way to confirm your decorator actually put the field on the wire. - -### Test - -Tests are `ScenarioTest`-based and replay recorded HTTP traffic (289 recordings under -`src/aks-preview/azext_aks_preview/tests/latest/recordings/`). - -```bash -azdev test aks-preview # whole extension (playback) -azdev test aks-preview --live # hit real Azure, re-record -azdev test --src aks-preview test_aks_commands # single module -pytest src/aks-preview/.../tests/latest/test_aks_commands.py -k mytest -``` - -Recording helpers live alongside the tests: `custom_preparers.py`, `recording_processors.py` -(scrubs secrets), `mocks.py`. Long-running end-to-end validation uses the separate -`azcli_aks_live_test/` ADO pipelines. - -`pytest.ini` defines markers: `e2e_packaging`, `smoke`, `slow`. - -### Lint / style — run these *before* pushing, they are enforced gates - -```bash -azdev linter --include-whl-extensions aks-preview -azdev style aks-preview -azdev scan --include-whl-extensions aks-preview # credential scan -``` - -Waivers go in `linter_exclusions.yml` (root and/or per-extension). - -### Build a wheel locally - -```bash -azdev extension build aks-preview # emits dist/*.whl -az extension add --source dist/aks_preview-22.0.0b6-py3-none-any.whl --yes -``` - ---- - -## 6. Release & publish flow - -### What you do in your PR - -1. Add your entry to the **`Pending`** section of `HISTORY.rst`. -2. When releasing, promote `Pending` into a new numbered section and bump `VERSION` in `setup.py` - (semver; `22.0.0b6` style for preview). -3. **Do not edit `src/index.json`** — automation owns it. `README.md` is explicit that the - precondition for auto-publish is *bump the version but do not touch the index*. - -`HISTORY.rst` states this guidance verbatim at the top of the file. - -### What automation does - -``` -PR merged to main - └─► .github/workflows/TriggerExtensionRelease.yml - (OIDC login → az pipelines build queue → ADO OneBranch release pipeline) - └─► builds the .whl, signs, uploads to the CDN/storage - └─► opens a follow-up PR updating src/index.json - (sha256 + metadata via scripts/ci/update_index.py) - └─► index syncs to aka.ms/azure-cli-extension-index-v1 - └─► `az extension add/update --name aks-preview` sees it -``` - -Supporting automation: `.github/workflows/CreateReleaseTag.yml`, -`scripts/automation/build_package.py`, `scripts/ci/idempotent_release.py`, -`scripts/ci/release_version_cal.py`. - -Manual fallback (external-hosted wheels): `azdev extension update-index ` computes -the sha256 and writes the entry. Hashes must be **lowercase** or CI fails. - ---- - -## 7. CI gates on every PR (`azure-pipelines.yml`) - -| Job | What it enforces | -|---|---| -| `CredScan` | no secrets committed | -| `PolicyCheck` | Microsoft policy compliance | -| `CheckLicenseHeader` | MIT header on every source file | -| `CheckInit` | `__init__.py` present where required | -| `IndexVerify` | `src/index.json` well-formed, hashes match published wheels | -| `SourceTests` | integration + build tests across a Python matrix; publishes wheel artifacts | -| `AzdevStyleModifiedExtensions` | `azdev style` on changed extensions only | -| `AzdevLinterModifiedExtensions` | `azdev linter` on changed extensions only | -| `AzdevScanModifiedExtensions{High,Medium}` | credential scanning by confidence tier | -| `CheckExternalUrls` | external links resolve (`external_url_exclusions.json` waives) | - -GitHub-side helpers post linter/style results as PR comments (`AddPRComment.yml`, -`AzdevLinter.yml`, `AzdevStyle.yml`) and `BlockPRMerge.yml` gates merges. - -Review routing is `.github/CODEOWNERS`; `aks-preview` is owned by `@fumingzhang @elvazhu521` -(line 63), so any change under `src/aks-preview/` requires their approval. - ---- - -## 8. Quick reference — "where do I change X?" - -| I want to… | File | -|---|---| -| Add a CLI flag | `_params.py` | -| Validate a flag | `_validators.py` | -| Write help text / examples | `_help.py` | -| Register a new command | `commands.py` | -| Implement command logic | `custom.py` | -| Put a field on the ARM request body | `managed_cluster_decorator.py` / `agentpool_decorator.py` | -| Change which SDK/api-version is called | `_client_factory.py` + `vendored_sdks/.../_configuration.py` | -| Change table output | `_format.py` | -| Add a constant / addon name | `_consts.py` | -| Edit a generated command | ❌ regenerate `aaz/` with `aaz-dev-tools`; don't hand-edit | -| Record a changelog entry | `HISTORY.rst` → `Pending` | -| Bump the shipped version | `setup.py` → `VERSION` | -| Add an index entry | ❌ automation does it; don't hand-edit `src/index.json` | - ---- - -## 9. Canonical upstream docs - -`docs/README.md` intentionally defers to `Azure/azure-cli`: - -- [Authoring an extension](https://github.com/Azure/azure-cli/blob/dev/doc/extensions/authoring.md) -- [Publishing](https://github.com/Azure/azure-cli/blob/dev/doc/extensions/authoring.md#publish) -- [Command guidelines](https://github.com/Azure/azure-cli/blob/dev/doc/command_guidelines.md) -- [Extension metadata](https://github.com/Azure/azure-cli/blob/dev/doc/extensions/metadata.md) -- [FAQ](https://github.com/Azure/azure-cli/blob/dev/doc/extensions/faq.md) -- [`azdev` tooling](https://github.com/Azure/azure-cli-dev-tools) -- [Migrating to `pyproject.toml`](docs/pyproject-migration.md) diff --git a/src/aks-preview/azmon-logs-cli-enhancements-spec.md b/src/aks-preview/azmon-logs-cli-enhancements-spec.md deleted file mode 100644 index 6201c3f22c2..00000000000 --- a/src/aks-preview/azmon-logs-cli-enhancements-spec.md +++ /dev/null @@ -1,307 +0,0 @@ -# PRD: Azure Monitor Logs (Container Insights) CLI Enhancements - -## 1. TL;DR - -We are moving Azure Monitor Logs (Container Insights) onboarding off the legacy **addon profile** and onto the first-class **Azure Monitor Profile (AMP)**. This PRD covers the two onboarding commands in scope, closes AMP feature gaps (syslog port, Prometheus scraping, OTLP gRPC ports), removes an obsolete auth flag from the new path, and adds a deprecation warning on the legacy path. - -**Two commands are in scope:** - -| Command | Profile it configures | Direction | -|---|---|---| -| az aks enable-addons -a monitoring | Addon profile (addonProfiles.omsagent) | Legacy — keep working, start deprecating | -| az aks create/update --enable-azure-monitor-logs | Azure Monitor Profile (azureMonitorProfile.containerInsights) | Strategic — become the only supported way | - -## 2. Problem & motivation - -- `--enable-azure-monitor-logs` was meant to be the modern, AMP-based onboarding path, but it is **currently implemented on the legacy addon profile** — so today both commands write the same `omsagent` addon object. This blocks us from using AMP-only capabilities and from deprecating the addon. -- The AMP `containerInsights` schema already supports **syslog port** and **Prometheus scraping** controls, and the app-monitoring schema supports **OTLP gRPC ports**, but the CLI exposes none of them on the new path. -- The legacy **auth flag** (`--enable-msi-auth-for-monitoring`) is meaningless for AMP (which is managed-identity only), yet it is still accepted on the new path. -- The latest API version **removed** `disableCustomMetrics` from `containerInsights`, so it must not surface anywhere in the CLI. - -## 3. Goals / Non-goals - -**Goals** - -- Make `--enable-azure-monitor-logs` write **only** the AMP profile. -- Reach parity + close gaps on the AMP path (syslog port, Prometheus scraping, OTLP gRPC). -- Remove the obsolete auth flag from the AMP path; warn about it on the legacy path. -- Keep `az aks enable-addons -a monitoring` working (backward compatible). -**Non-goals** - -- Removing or breaking the legacy `enable-addons monitoring` command. -- Exposing `disableCustomMetrics` (removed from the API). -- Changes to metrics-only (Managed Prometheus) onboarding. - -## 4. Background: the two profiles & their schema - -### 4.1 Addon profile — set by `az aks enable-addons -a monitoring` - -`properties.addonProfiles.omsagent`: - -```json -{ - "enabled": true, - "config": { - "logAnalyticsWorkspaceResourceID": "", - "useAADAuth": "true" | "false" // MSI auth vs legacy shared-key auth - } -} -``` - -- Auth: supports **both** managed-identity (`useAADAuth=true`) and legacy shared-key (`useAADAuth=false`), driven by `--enable-msi-auth-for-monitoring`. - -### 4.2 Azure Monitor Profile (AMP) — target of `--enable-azure-monitor-logs` - -`properties.azureMonitorProfile` (latest API `2026-07-02-preview`): - -```json -{ - "containerInsights": { - "enabled": true, - "logAnalyticsWorkspaceResourceId": "", - "syslogPort": 28330, // default 28330 - "disablePrometheusMetricsScraping": false, // default false - "containerNetworkLogs": "" - // NOTE: disableCustomMetrics was REMOVED in the latest API — do not use. - }, - "appMonitoring": { - "openTelemetryLogsAndTraces": { "enabled": true, "httpPort": 0, "grpcPort": 0 }, - "openTelemetryMetrics": { "enabled": true, "httpPort": 0, "grpcPort": 0 } - } -} -``` - -- Auth: **managed-identity only** — there is no shared-key/`useAADAuth` concept. - -### 4.3 Current CLI coverage of the AMP schema - -| AMP field | Flag today? | -|---|---| -| `containerInsights.enabled` | Yes (`--enable-azure-monitor-logs` / `--disable-azure-monitor-logs`) — *but writes the addon profile* | -| `containerInsights.logAnalyticsWorkspaceResourceId` | Yes (`--workspace-resource-id`) | -| `containerInsights.syslogPort` | **No** | -| `containerInsights.disablePrometheusMetricsScraping` | **No** | -| `appMonitoring.openTelemetry*.httpPort` | Yes (`--opentelemetry-metrics-port`, `--opentelemetry-logs-port`) | -| `appMonitoring.openTelemetry*.grpcPort` | **No** | - -## 5. Requirements - -| ID | Requirement | Priority | -|---|---|---| -| R0 | Add OTLP gRPC port overrides | P0 | -| R1 | `--enable-azure-monitor-logs` writes only the AMP profile | P0 | -| R2 | Remove legacy auth flag from `--enable-azure-monitor-logs` | P0 | -| R3 | Do not expose `disableCustomMetrics` (removed from API) | P0 | -| R4 | Add Prometheus-scraping + syslog-port controls to the AMP path | P0 | -| R5 | Warn when the legacy auth flag is used with `enable-addons monitoring` | P0 | -| R6 | Reject `--enable-azure-monitor-logs` on an already-onboarded cluster | P0 | -| R7 | `--disable-azure-monitor-logs` resets `containerInsights` to defaults | P0 | -| R8 | Reject `--enable-azure-monitor-logs` on service principal clusters | P0 | -| R9 | Reject `--enable-azure-monitor-logs` on a legacy-auth onboarded cluster | P0 | -| R10 | Confirm before disabling when OTLP logs & traces are on | P0 | -| R11 | Every enable provisions the DCR and DCRA, before the cluster update | P0 | - -### R0 — OTLP gRPC port overrides - -Today only the OTLP **HTTP** port is settable; the gRPC port cannot be overridden. - -- Allow overriding the gRPC port for both OTLP signals: - - metrics → `appMonitoring.openTelemetryMetrics.grpcPort` - - logs & traces → `appMonitoring.openTelemetryLogsAndTraces.grpcPort` -- HTTP and gRPC ports are independently settable per signal; make the HTTP-vs-gRPC mapping unambiguous. -**Acceptance criteria** - -- A gRPC override sets the corresponding `grpcPort`; unset leaves the server default. -- HTTP and gRPC ports settable independently for metrics and logs/traces. -- Values round-trip on `az aks update`. - -### R1 — Switch `--enable-azure-monitor-logs` to the AMP profile only - -**As an** AKS user, **when** I run `az aks create/update --enable-azure-monitor-logs`, **I want** it to configure `azureMonitorProfile.containerInsights` (and not the legacy `omsagent` addon), so onboarding uses the modern profile. - -- Today the flag writes `addonProfiles.omsagent`; it must instead write only `azureMonitorProfile.containerInsights.enabled = true` (+ workspace id). -- `--disable-azure-monitor-logs` must disable via the AMP profile (`containerInsights.enabled = false`) and no longer depend on the addon object. -- **Container network logs move with it.** On the legacy addon path this is the `omsagent` config key `enableRetinaNetworkFlags` ("True"/"False"); it lines up with the AMP field `containerInsights.containerNetworkLogs` (enum `Enabled`/`Disabled`, default `Disabled`). The **same** `--enable-container-network-logs` / `--disable-container-network-logs` commands must set `containerInsights.containerNetworkLogs` accordingly (`Enabled` / `Disabled`) instead of the addon config key. -- The legacy `enable-addons monitoring` command is unaffected and keeps using the addon profile. -**Parity — all monitoring options must keep working on the new command** - -Several legacy monitoring options are realized as **out-of-band ARM resources** (DCR / DCE / DCRA / AMPLS), not as `containerInsights` fields. They work today only because `--enable-azure-monitor-logs` writes the `omsagent` addon, and that provisioning is gated on the addon object existing. After the AMP switch, the **same** command must continue to honor them without the addon. - -| Legacy flag | Configures | Post-switch requirement | -|---|---|---| -| `--workspace-resource-id` | Log Analytics workspace | → `containerInsights.logAnalyticsWorkspaceResourceId` | -| `--enable-syslog` | Syslog collection (DCR syslog data source) | Provision the same DCR data source | -| `--data-collection-settings` | DCR tuning (interval, namespaces, streams, `enableContainerLogV2`) | Apply the same DCR settings | -| `--enable-high-log-scale-mode` | High log scale (ingestion DCE) | Provision the same DCE | -| `--ampls-resource-id` | AMPLS private-link scope (private cluster, MSI) | Provision the same AMPLS links | -| `--enable/disable-container-network-logs` | Container network logs | → `containerInsights.containerNetworkLogs` (see above) | -| `--enable-msi-auth-for-monitoring` | Auth mode | **Dropped** — AMP is MSI-only (see R2) | - -> The DCR/DCE/DCRA/AMPLS provisioning currently keys off the `omsagent` addon object; after the switch it must be driven off the AMP profile (workspace id from `containerInsights.logAnalyticsWorkspaceResourceId`) and must not require the addon to exist. - -**Acceptance criteria** - -- `--enable-azure-monitor-logs` results in `azureMonitorProfile.containerInsights.enabled=true` with the workspace id, and **no** `omsagent` addon entry authored by this flag. -- `--disable-azure-monitor-logs` sets `containerInsights.enabled=false`. -- `--enable-container-network-logs` sets `containerInsights.containerNetworkLogs="Enabled"` (and `--disable-container-network-logs` → `"Disabled"`); it no longer writes `enableRetinaNetworkFlags`. -- `--enable-syslog`, `--data-collection-settings`, `--enable-high-log-scale-mode`, and `--ampls-resource-id` produce the **same** DCR/DCE/DCRA/AMPLS artifacts on `--enable-azure-monitor-logs` as they do today on `--enable-addons monitoring`. -- That artifact provisioning no longer depends on the presence of the `omsagent` addon profile. -- `enable-addons -a monitoring` behavior is unchanged. - -### R2 — Remove the legacy auth flag from `--enable-azure-monitor-logs` - -The AMP profile is managed-identity only, so `--enable-msi-auth-for-monitoring` is meaningless here. - -- `--enable-azure-monitor-logs` must not accept or honor `--enable-msi-auth-for-monitoring`. -- Passing them together returns a clear, actionable error. -- No `useAADAuth`/shared-key concept is written on the AMP path. -**Acceptance criteria** - -- `--enable-azure-monitor-logs` works with no auth flag (MSI implicit). -- Combining the two flags errors out with a helpful message. - -### R3 — Do not expose `disableCustomMetrics` - -`containerInsights.disableCustomMetrics` was removed in the latest API version. - -- No CLI flag maps to it; it must not appear in any request the CLI sends. -- Schema/reference docs must reflect its removal. -**Acceptance criteria** - -- No CLI surface reads or writes `disableCustomMetrics`. - -### R4 — Prometheus-scraping & syslog-port controls on the AMP path - -Expose the two `containerInsights` controls that the schema already supports. - -- A flag to toggle Prometheus metrics scraping → `containerInsights.disablePrometheusMetricsScraping`. -- A flag to set the syslog host port → `containerInsights.syslogPort` (integer; server default 28330 when unset). -- Available on `az aks create` and `az aks update` for the AMP path; both are updatable. -**Suggested flags** - -| Flag | Type | Maps to | Notes | -|---|---|---|---| -| `--disable-prometheus-metrics-scraping` | switch | `disablePrometheusMetricsScraping = true` | default off (scraping on); mirrors the existing `--enable-…/--disable-…` metrics style | -| `--enable-prometheus-metrics-scraping` | switch | `disablePrometheusMetricsScraping = false` | to re-enable on `update`; mutually exclusive with the disable flag | -| `--syslog-port` | int | `syslogPort` | valid TCP port; unset ⇒ server default 28330 | - -> Naming note: the pre-existing `--enable-syslog` (three-state) toggles legacy syslog **collection** and is distinct from the new `--syslog-port` host-port control. - -**Example usage** - -```bash -# Enable Azure Monitor logs (AMP), disable Prometheus scraping, custom syslog port -az aks create -g -n \ - --enable-azure-monitor-logs \ - --disable-prometheus-metrics-scraping \ - --syslog-port 28331 - -# Later, re-enable scraping and change the port on an existing cluster -az aks update -g -n \ - --enable-prometheus-metrics-scraping \ - --syslog-port 29000 -``` - -**Acceptance criteria** - -- The scraping flag sets/clears `disablePrometheusMetricsScraping`. -- The syslog-port flag sets `syslogPort` (validated as a TCP port); unset leaves the default. -- Both round-trip on `az aks update` without disturbing other monitoring settings. - -### R5 — Deprecation warning on the legacy auth flag - -**As an** AKS user, **when** I pass `--enable-msi-auth-for-monitoring` on `az aks enable-addons -a monitoring`, **I want** a warning that the flag is deprecated and that managed-identity auth (and `--enable-azure-monitor-logs`) is the recommended path. - -- The command still succeeds (non-breaking); only a warning is added. -- Especially relevant when the flag is set to `false` (opting into legacy shared-key auth). -**Acceptance criteria** - -- Warning shown once when the flag is explicitly provided; not shown when omitted. -- Command completes successfully; resulting `useAADAuth` value is unchanged. - -### R6 — Reject re-onboarding an already-onboarded cluster - -`az aks enable-addons -a monitoring` fails when the addon is already enabled and tells the user to disable it first. The AMP path must behave the same way, so that a re-run cannot silently recreate a default workspace or re-provision DCR/DCE/DCRA artifacts and mask a mistake such as a mistyped `--workspace-resource-id`. - -- `az aks update --enable-azure-monitor-logs` errors when `containerInsights.enabled` is already `true` (or the legacy `omsagent` addon is enabled, which the RP mirrors into that field). -- The error names `--disable-azure-monitor-logs` as the way to change the configuration. - -**Acceptance criteria** - -- Re-running the flag on an onboarded cluster errors, and errors *before* a default workspace is provisioned. -- The flag still succeeds when `containerInsights.enabled` is `false` (a fresh enable, or a re-enable after a disable). -- Changing the workspace requires a disable first. - -### R7 — `--disable-azure-monitor-logs` resets `containerInsights` to defaults - -The RP copies a `containerInsights` field from the request onto the cluster only when the field is present (`ApplyAzureMonitorProfileContainerInsights`), so any field left unset survives the disable and is silently inherited by the next enable. - -- Disabling writes the behavioural fields explicitly: `enabled=false`, `syslogPort=28330`, `disablePrometheusMetricsScraping=false`, `containerNetworkLogs="Disabled"`. -- `logAnalyticsWorkspaceResourceId` is deliberately **not** blanked. It is a resource-id typed property, and the RP mirrors the AMP value into `addonProfiles.omsagent.config.logAnalyticsWorkspaceResourceID`; ARM then rejects every subsequent write of the cluster with `LinkedInvalidPropertyId`, breaking even updates unrelated to monitoring. The stale id is inert while `enabled` is false and the enable path always overwrites it with a freshly resolved workspace, so nothing is inherited. - -**Acceptance criteria** - -- The four behavioural fields are present in the PUT payload after a disable. -- The workspace id remains a valid resource id (never `""`), and a later `az aks update` of any kind still succeeds. -- A subsequent `--enable-azure-monitor-logs` starts from the defaults rather than inheriting the previous syslog port, scraping choice or CNL setting. - -### R11 — Every enable provisions the DCR and DCRA, before the cluster update - -Postprocessing previously ran only when the workspace changed, so a fresh enable — or a re-enable onto the same workspace after a disable removed the association — deployed the agent with no data collection rule attached and silently ingested nothing. - -Provisioning also has to happen *before* the cluster PUT, not in `postprocessing_after_mc_created`. The RP rolls out the `ama-logs` DaemonSet as part of the PUT, so creating the artifacts afterwards means the agent starts before the association exists. `dcr-config-parser.rb` then finds no configuration chunk, logs `Exception while parsing dcr : No JSON file found in the specified directory`, and mdsd backs off for several minutes before retrying; the agent ingests nothing for that window and is restarted by its own liveness probe once the configuration finally lands. `az aks enable-addons -a monitoring` creates the DCR/DCRA before its PUT, and this flag matches that ordering. - -- The enable path calls `ensure_container_insights_for_monitoring` directly, once the AMP profile has been fully built, since the guards above have already rejected the no-op cases. -- `monitoring_addon_postprocessing_required` is therefore *not* set by the enable path; leaving it set would repeat the same work after the PUT. -- `ensure_container_insights_for_monitoring` is idempotent and rewrites the DCR destination, so this also covers the workspace-change case. - -**Acceptance criteria** - -- After `--enable-azure-monitor-logs`, the cluster has a `ContainerInsightsExtension` DCRA pointing at an `MSCI--` DCR whose destination is the resolved workspace. -- The DCRA exists before the cluster update completes. -- The agent logs no `Exception while parsing dcr` on first start and does not restart to pick up the configuration. -- Disabling removes the association. - -### R8 — Reject `--enable-azure-monitor-logs` on service principal clusters - -The AMP profile has no shared-key/`useAADAuth` concept: the agent reaches the workspace with the cluster's managed identity, which a service principal cluster does not have. - -- Clusters whose `servicePrincipalProfile.clientId` is set to anything other than `msi` are rejected on both `az aks create` and `az aks update`. -- A missing `servicePrincipalProfile` means managed identity and is allowed. - -**Acceptance criteria** - -- The rejection happens before a default workspace is provisioned. -- The error points at `az aks update --enable-managed-identity` as the fix. - -### R9 — Reject `--enable-azure-monitor-logs` on a legacy-auth onboarded cluster - -A cluster onboarded with shared-key auth keeps that auth mode server-side, so the flag cannot be honoured as asked. Absent or empty `useAADAuth` counts as legacy, matching the RP's derivation. - -- Checked ahead of R6, so the actionable migration message wins for a legacy-auth cluster. -- A *disabled* `omsagent` addon is a fresh onboarding as far as the RP is concerned and must not be blocked. - -**Acceptance criteria** - -- `useAADAuth` of `false`, `""` or absent on an enabled addon errors with a link to the managed-identity migration doc. -- The rejection happens before a default workspace is provisioned. - -### R10 — Confirm before disabling when OTLP logs & traces are on - -OpenTelemetry logs and traces are collected by the Container Insights agent, so disabling Azure Monitor logs necessarily turns them off too. - -- Prompt for confirmation, defaulting to "no"; `--yes` skips the prompt. -- Declining exits cleanly without tearing down the DCR/DCRA or modifying the profile. -- Accepting disables `openTelemetryLogsAndTraces` and clears its HTTP and gRPC ports. - -**Acceptance criteria** - -- The prompt appears only when `openTelemetryLogsAndTraces.enabled` is true and `--yes` was not passed. -- Declining leaves both `containerInsights` and `appMonitoring` untouched. - -## 6. Target end-state (the two commands) - -| Command | Writes | Auth | Extra controls | -|---|---|---|---| -| `az aks enable-addons -a monitoring` | `addonProfiles.omsagent` | MSI or shared-key (**warns** on explicit legacy auth flag) | unchanged | -| `az aks create/update --enable-azure-monitor-logs` | `azureMonitorProfile.containerInsights` **only** | MSI only (**no** auth flag) | `--syslog-port`, Prometheus-scraping toggle, `--enable/disable-container-network-logs` (→ `containerNetworkLogs`); OTLP gRPC ports via OTLP flags | From c4225f689380aba7538551ec382c8970c3d4bbfd Mon Sep 17 00:00:00 2001 From: Sunil Yadav Date: Thu, 17 Sep 2026 02:41:56 +0000 Subject: [PATCH 4/8] nit --- .../azext_aks_preview/tests/latest/test_custom.py | 6 +++--- .../tests/latest/test_managed_cluster_decorator.py | 8 ++++---- .../azext_aks_preview/tests/latest/test_validators.py | 2 +- 3 files changed, 8 insertions(+), 8 deletions(-) diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py index 0577a963fe9..3ff2b4cd8d3 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py @@ -909,7 +909,7 @@ def test_other_error_after_table_readiness_retry_still_bounded(self): class TestLegacyMonitoringAuthDeprecation(unittest.TestCase): - """R5: warn when the legacy auth flag is used with enable-addons monitoring.""" + """Warn when the legacy auth flag is used with enable-addons monitoring.""" def _warn(self, value, addons="monitoring"): from azext_aks_preview.addonconfiguration import warn_on_legacy_monitoring_auth @@ -948,7 +948,7 @@ def test_does_not_change_the_auth_value(self): class TestMonitoringArgumentRegistration(unittest.TestCase): - """Argument registration for R4 (containerInsights controls) and R5 (legacy auth warning).""" + """Argument registration for the containerInsights controls and the legacy auth warning.""" def setUp(self): register_aks_preview_resource_type() @@ -990,7 +990,7 @@ def test_legacy_auth_flag_is_deprecated_on_addon_commands(self): self.assertIn("--enable-azure-monitor-logs", message) def test_legacy_auth_flag_is_not_deprecated_on_create_and_update(self): - # R5 scopes the warning to the legacy addon commands. + # The deprecation is scoped to the legacy addon commands. for command_name in ("aks create", "aks update"): arguments = self._arguments(command_name) self.assertIsNone( diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py index 7552784452f..e613601b381 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py @@ -13117,7 +13117,7 @@ def test_update_enable_azure_monitor_logs(self): 8080, ) - # R2: --enable-msi-auth-for-monitoring is rejected on the Azure Monitor logs path, + # --enable-msi-auth-for-monitoring is rejected on the Azure Monitor logs path, # which is managed-identity only. dec_4 = AKSPreviewManagedClusterUpdateDecorator( self.cmd, @@ -20018,7 +20018,7 @@ def test_hlsm_allowed_on_aad_auth_cluster(self): ) # ------------------------------------------------------------------ - # R4: Prometheus scraping and syslog port controls on the AMP path. + # Prometheus scraping and syslog port controls on the AMP path. # ------------------------------------------------------------------ def _amp_enabled_mc(self): @@ -20232,7 +20232,7 @@ def test_update_container_insights_settings_partial_write(self): self.assertTrue(container_insights.disable_prometheus_metrics_scraping) # ------------------------------------------------------------------ - # R3: disableCustomMetrics was removed from the API and must not resurface. + # disableCustomMetrics was removed from the API and must not resurface. # ------------------------------------------------------------------ def test_disable_custom_metrics_is_not_part_of_container_insights(self): container_insights = ( @@ -20266,7 +20266,7 @@ def test_enabling_azure_monitor_logs_emits_no_disable_custom_metrics(self): ) # ------------------------------------------------------------------ - # R0: OTLP gRPC ports are independent of the HTTP ports and default to unset. + # OTLP gRPC ports are independent of the HTTP ports and default to unset. # ------------------------------------------------------------------ def test_opentelemetry_grpc_port_unset_leaves_server_default(self): dec = AKSPreviewManagedClusterCreateDecorator( diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py b/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py index 2aa551f0e91..9e2be666d35 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py @@ -2805,7 +2805,7 @@ def test_nodepool_update_allows_windows2022_and_windows2025(self): class ContainerInsightsSettingsNamespace(SimpleNamespace): - """Namespace for the R4 containerInsights tuning flags, with CLI defaults.""" + """Namespace for the containerInsights tuning flags, with CLI defaults.""" def __init__(self, **kwargs): defaults = { From dc048460fc736110cb23c95e7cdfc3e2a2c2bfbb Mon Sep 17 00:00:00 2001 From: Sunil Yadav Date: Fri, 18 Sep 2026 00:18:19 +0000 Subject: [PATCH 5/8] resolved comments --- src/aks-preview/HISTORY.rst | 5 + .../azext_aks_preview/addonconfiguration.py | 4 +- src/aks-preview/azext_aks_preview/custom.py | 7 + .../managed_cluster_decorator.py | 249 +++++++++--- .../tests/latest/test_custom.py | 50 +++ .../latest/test_managed_cluster_decorator.py | 355 ++++++++++++++++++ .../tests/latest/test_validators.py | 8 +- 7 files changed, 623 insertions(+), 55 deletions(-) diff --git a/src/aks-preview/HISTORY.rst b/src/aks-preview/HISTORY.rst index 305e8e56aab..f3426b29d7d 100644 --- a/src/aks-preview/HISTORY.rst +++ b/src/aks-preview/HISTORY.rst @@ -24,6 +24,11 @@ Pending * `az aks update`: Fix `--enable-azure-monitor-logs` not creating the data collection rule and association unless the Log Analytics workspace changed, which left the agent running with no data collection rule attached so no logs were ingested. * `az aks update`: Create the data collection rule and association before the cluster update when enabling with `--enable-azure-monitor-logs`, matching `az aks enable-addons -a monitoring`. Provisioning them afterwards meant the agent started before the association existed and then stayed idle for several minutes before restarting once the configuration arrived. * `az aks update`: Declining the OpenTelemetry confirmation prompt for `--disable-azure-monitor-metrics` now leaves the cluster unchanged and exits without an error, matching every other confirmation prompt, instead of failing the command. +* `az aks enable-addons`, `az aks addon enable` and `az aks addon update`: Warn that shared key authentication for the monitoring addon is deprecated when `--enable-msi-auth-for-monitoring false` is passed, and point to `--enable-azure-monitor-logs`. The warning reflects the value you supply, so it stays silent when the flag is omitted on a cluster using service principal authentication. +* `az aks update`: Fix `--enable-syslog`, `--data-collection-settings` and `--ampls-resource-id` being silently ignored when supplied on their own, as none of them re-provisioned the data collection rule that carries them, so the command reported success while the agent kept using the previous rule. +* `az aks update`: Reject the OpenTelemetry port flags (`--opentelemetry-metrics-port-http`, `--opentelemetry-metrics-port-grpc`, `--opentelemetry-logs-traces-port-http` and `--opentelemetry-logs-traces-port-grpc`) when the matching receiver is being disabled in the same command, instead of accepting the port and then silently dropping it. +* `az aks update`: Collect every monitoring disable confirmation before any of them deletes collection resources. Combining `--disable-azure-monitor-logs` with `--disable-azure-monitor-metrics` used to delete the logs data collection rule and association before asking about metrics, so declining that prompt aborted the command with Container Insights still enabled on the cluster but its collection objects already removed. +* `az aks create` and `az aks update`: Fix the `--data-collection-settings` size limit being applied to the file path instead of the settings it holds, which let an oversized file through to fail the data collection rule call with `Request Header Fields Too Large`. 22.0.0b6 +++++++++ diff --git a/src/aks-preview/azext_aks_preview/addonconfiguration.py b/src/aks-preview/azext_aks_preview/addonconfiguration.py index f9fdb107cda..d4df76a4f70 100644 --- a/src/aks-preview/azext_aks_preview/addonconfiguration.py +++ b/src/aks-preview/azext_aks_preview/addonconfiguration.py @@ -72,7 +72,8 @@ def warn_on_legacy_monitoring_auth(enable_msi_auth_for_monitoring, addons): """ if enable_msi_auth_for_monitoring is not False: return - if "monitoring" not in (addons or ""): + requested_addons = {addon.strip().lower() for addon in (addons or "").split(",")} + if "monitoring" not in requested_addons: return logger.warning( "--enable-msi-auth-for-monitoring false configures Container Insights with legacy shared " @@ -111,7 +112,6 @@ def enable_addons( ampls_resource_id=None, enable_high_log_scale_mode=False, ): - warn_on_legacy_monitoring_auth(enable_msi_auth_for_monitoring, addons) instance = client.get(resource_group_name, name) # this is overwritten by _update_addons(), so the value needs to be recorded here msi_auth = False diff --git a/src/aks-preview/azext_aks_preview/custom.py b/src/aks-preview/azext_aks_preview/custom.py index e64799e323a..1ecf4d79d2c 100644 --- a/src/aks-preview/azext_aks_preview/custom.py +++ b/src/aks-preview/azext_aks_preview/custom.py @@ -3543,6 +3543,10 @@ def aks_addon_enable( ampls_resource_id=None, enable_high_log_scale_mode=None ): + # Warn with the value the user supplied. aks_addon_update normalizes an omitted flag to False + # on service principal clusters, so the shared helper cannot tell "not supplied" from + # "explicitly false" by the time it runs. + warn_on_legacy_monitoring_auth(enable_msi_auth_for_monitoring, addon) return enable_addons( cmd, client, @@ -3601,6 +3605,9 @@ def aks_addon_update( ampls_resource_id=None, enable_high_log_scale_mode=None ): + # Warn before the service principal normalization below turns an omitted flag into False, + # which would otherwise warn users who never passed --enable-msi-auth-for-monitoring. + warn_on_legacy_monitoring_auth(enable_msi_auth_for_monitoring, addon) instance = client.get(resource_group_name, name) addon_profiles = instance.addon_profiles diff --git a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py index a4518bf1180..f2226e9f95a 100644 --- a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py @@ -6,6 +6,7 @@ # pylint: disable=too-many-lines import copy import datetime +import json import os from types import SimpleNamespace from typing import Any, Dict, List, Optional, Tuple, TypeVar, Union @@ -155,6 +156,11 @@ logger = get_logger(__name__) +# Maximum size, in characters of the serialized JSON, of the --data-collection-settings payload. +# The settings are embedded in the data collection rule request, which the service rejects with +# "Request Header Fields Too Large" beyond roughly this size. +CONST_DATA_COLLECTION_SETTINGS_MAX_CHARS = 10000 + def _get_etag_match_condition(if_match, if_none_match): """Convert if_match/if_none_match to etag/match_condition for the new SDK.""" @@ -285,9 +291,10 @@ def _raise_if_service_principal_cluster(mc): def _is_container_network_logs_enabled_on_mc(mc, addon_consts): """Whether container network logs are on, via the AMP profile or the legacy addon config key.""" container_insights = _get_container_insights_profile(mc) - if container_insights and str(container_insights.container_network_logs or "").lower() == \ - CONST_CONTAINER_NETWORK_LOGS_ENABLED.lower(): - return True + amp_value = str(getattr(container_insights, "container_network_logs", None) or "") \ + if container_insights else "" + if amp_value: + return amp_value.lower() == CONST_CONTAINER_NETWORK_LOGS_ENABLED.lower() addon_profiles = getattr(mc, "addon_profiles", None) if mc is not None else None if not addon_profiles: return False @@ -297,6 +304,38 @@ def _is_container_network_logs_enabled_on_mc(mc, addon_consts): return str(config.get(CONST_MONITORING_ENABLE_RETINA_NETWORK_FLAGS, "")).lower() == "true" +def _is_azure_monitor_metrics_enabled(mc): + """Whether managed Prometheus (Azure Monitor metrics) is enabled on the cluster.""" + return bool( + mc and + mc.azure_monitor_profile and + mc.azure_monitor_profile.metrics and + mc.azure_monitor_profile.metrics.enabled + ) + + +def _is_opentelemetry_metrics_enabled(mc): + """Whether the OpenTelemetry metrics receiver is enabled on the cluster.""" + return bool( + mc and + mc.azure_monitor_profile and + mc.azure_monitor_profile.app_monitoring and + mc.azure_monitor_profile.app_monitoring.open_telemetry_metrics and + mc.azure_monitor_profile.app_monitoring.open_telemetry_metrics.enabled + ) + + +def _is_opentelemetry_logs_traces_enabled(mc): + """Whether the OpenTelemetry logs and traces receiver is enabled on the cluster.""" + return bool( + mc and + mc.azure_monitor_profile and + mc.azure_monitor_profile.app_monitoring and + mc.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces and + mc.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces.enabled + ) + + def _get_addon_config_value(config, key): """Case-insensitive lookup of an addon config value. @@ -3208,6 +3247,33 @@ def get_syslog_port(self) -> Union[int, None]: self._validate_container_insights_setting("--syslog-port") return syslog_port + def get_data_collection_settings(self) -> Union[str, None]: + """Obtain the value of data_collection_settings, rejecting an oversized settings file. + + The base implementation returns the *path* the settings were read from. The settings + themselves are serialized into the data collection rule request, which the service rejects + with "Request Header Fields Too Large" past roughly this size, so the length that matters + is that of the parsed contents. Measuring the path instead would never trigger and would + let an oversized file through to fail the DCR call with that opaque error. + + Raised rather than silently dropped: dropping the settings falls back to the default + collection settings, which changes what the cluster ingests without telling the caller. + + :return: string or None + """ + data_collection_settings_file_path = super().get_data_collection_settings() + if not data_collection_settings_file_path: + return data_collection_settings_file_path + + serialized_length = len(json.dumps(get_file_json(data_collection_settings_file_path))) + if serialized_length > CONST_DATA_COLLECTION_SETTINGS_MAX_CHARS: + raise InvalidArgumentValueError( + f"--data-collection-settings is too large: {serialized_length} characters once " + f"serialized, the limit is {CONST_DATA_COLLECTION_SETTINGS_MAX_CHARS}. Reduce " + "the number of namespaces or streams in the file and retry." + ) + return data_collection_settings_file_path + def get_disable_prometheus_metrics_scraping(self) -> Union[bool, None]: """Obtain the value to write to containerInsights.disablePrometheusMetricsScraping. @@ -3331,6 +3397,13 @@ def get_opentelemetry_metrics_port(self) -> Union[int, None]: raise InvalidArgumentValueError( "--opentelemetry-metrics-port-http cannot be specified when --disable-azure-monitor-metrics is used." ) + # Disabling the receiver in the same command clears its ports, so a port supplied + # alongside it would be accepted and then silently dropped. + if self.get_disable_opentelemetry_metrics(): + raise InvalidArgumentValueError( + "--opentelemetry-metrics-port-http cannot be specified when " + "--disable-opentelemetry-metrics is used." + ) # For CREATE: --enable-opentelemetry-metrics must be explicitly specified if self.decorator_mode == DecoratorMode.CREATE: @@ -3376,6 +3449,12 @@ def get_opentelemetry_metrics_port_grpc(self) -> Union[int, None]: raise InvalidArgumentValueError( "--opentelemetry-metrics-port-grpc cannot be specified when --disable-azure-monitor-metrics is used." ) + # See get_opentelemetry_metrics_port for why a disable in the same command is rejected. + if self.get_disable_opentelemetry_metrics(): + raise InvalidArgumentValueError( + "--opentelemetry-metrics-port-grpc cannot be specified when " + "--disable-opentelemetry-metrics is used." + ) # For CREATE: --enable-opentelemetry-metrics must be explicitly specified if self.decorator_mode == DecoratorMode.CREATE: @@ -3497,6 +3576,12 @@ def get_opentelemetry_logs_port(self) -> Union[int, None]: raise InvalidArgumentValueError( "--opentelemetry-logs-traces-port-http cannot be specified when --disable-azure-monitor-logs is used." ) + # See get_opentelemetry_metrics_port for why a disable in the same command is rejected. + if self.get_disable_opentelemetry_logs(): + raise InvalidArgumentValueError( + "--opentelemetry-logs-traces-port-http cannot be specified when " + "--disable-opentelemetry-logs-traces is used." + ) # For CREATE: --enable-opentelemetry-logs-traces must be explicitly specified if self.decorator_mode == DecoratorMode.CREATE: @@ -3542,6 +3627,12 @@ def get_opentelemetry_logs_traces_port_grpc(self) -> Union[int, None]: raise InvalidArgumentValueError( "--opentelemetry-logs-traces-port-grpc cannot be specified when --disable-azure-monitor-logs is used." ) + # See get_opentelemetry_metrics_port for why a disable in the same command is rejected. + if self.get_disable_opentelemetry_logs(): + raise InvalidArgumentValueError( + "--opentelemetry-logs-traces-port-grpc cannot be specified when " + "--disable-opentelemetry-logs-traces is used." + ) # For CREATE: --enable-opentelemetry-logs-traces must be explicitly specified if self.decorator_mode == DecoratorMode.CREATE: @@ -9060,10 +9151,6 @@ def _provision_azure_monitor_logs_dcr(self, mc: ManagedCluster, addon_consts: di return data_collection_settings = self.context.get_data_collection_settings() - # Oversized settings are dropped rather than sent, to avoid the DCR call failing with - # "Request Header Fields Too Large". - if data_collection_settings and len(str(data_collection_settings)) > 10000: - data_collection_settings = None self.context.external_functions.ensure_container_insights_for_monitoring( self.cmd, @@ -9098,22 +9185,11 @@ def _disable_azure_monitor_logs(self, mc: ManagedCluster) -> None: return # OpenTelemetry logs and traces are collected by the Container Insights agent, so disabling - # Azure Monitor logs necessarily turns them off too. Confirm before doing that. - opentelemetry_logs_enabled = ( - mc.azure_monitor_profile and - mc.azure_monitor_profile.app_monitoring and - mc.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces and - mc.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces.enabled - ) - - if opentelemetry_logs_enabled and not self.context.get_yes(): - msg = ( - "OpenTelemetry logs and traces are enabled on this cluster and are collected by " - "Azure Monitor logs. Disabling Azure Monitor logs will also disable OpenTelemetry " - "logs and traces. Do you want to continue?" - ) - if not prompt_y_n(msg, default="n"): - raise DecoratorEarlyExitException() + # Azure Monitor logs necessarily turns them off too. The confirmation is normally taken up + # front by confirm_monitoring_disables, before any cleanup has run; this call only prompts + # when the handler is driven directly. + opentelemetry_logs_enabled = _is_opentelemetry_logs_traces_enabled(mc) + self._confirm_disable_azure_monitor_logs(mc) # Perform DCR/DCRA cleanup BEFORE disabling (same as aks_disable_addons lines 2796-2822). # Only MSI-auth clusters have a DCR/DCRA to clean up, so decide from local state first to @@ -9176,21 +9252,12 @@ def _disable_azure_monitor_metrics(self, mc: ManagedCluster) -> None: if not azure_monitor_metrics_enabled: return - # Check if OpenTelemetry metrics are enabled and prompt for confirmation - opentelemetry_metrics_enabled = ( - mc.azure_monitor_profile and - mc.azure_monitor_profile.app_monitoring and - mc.azure_monitor_profile.app_monitoring.open_telemetry_metrics and - mc.azure_monitor_profile.app_monitoring.open_telemetry_metrics.enabled - ) - - if opentelemetry_metrics_enabled and not self.context.get_yes(): - msg = ( - "Disabling Azure Monitor metrics will also disable OpenTelemetry metrics. " - "Do you want to continue?" - ) - if not prompt_y_n(msg, default="n"): - raise DecoratorEarlyExitException() + # OpenTelemetry metrics are ingested through the managed Prometheus pipeline that Azure + # Monitor metrics sets up, so disabling the parent necessarily turns them off too. The + # confirmation is normally taken up front by confirm_monitoring_disables, before any + # cleanup has run; this call only prompts when the handler is driven directly. + opentelemetry_metrics_enabled = _is_opentelemetry_metrics_enabled(mc) + self._confirm_disable_azure_monitor_metrics(mc) # Disable Azure Monitor metrics if mc.azure_monitor_profile is None: @@ -9206,6 +9273,72 @@ def _disable_azure_monitor_metrics(self, mc: ManagedCluster) -> None: mc.azure_monitor_profile.app_monitoring.open_telemetry_metrics.http_port = None mc.azure_monitor_profile.app_monitoring.open_telemetry_metrics.grpc_port = None + def _monitoring_disables_confirmed(self) -> bool: + return self.context.get_intermediate("monitoring_disables_confirmed", default_value=False) + + def _confirm_disable_azure_monitor_logs(self, mc: ManagedCluster) -> None: + """Take the confirmation for a logs disable that also turns OpenTelemetry logs/traces off. + + Returns without prompting once confirm_monitoring_disables has already collected it, so + the question is asked exactly once per command no matter which path reaches here. + """ + if not self.context.get_disable_azure_monitor_logs(): + return + if not _is_monitoring_enabled_on_mc(mc, self.context.get_addon_consts()): + return + if not _is_opentelemetry_logs_traces_enabled(mc): + return + if self.context.get_yes() or self._monitoring_disables_confirmed(): + return + + msg = ( + "OpenTelemetry logs and traces are enabled on this cluster and are collected by " + "Azure Monitor logs. Disabling Azure Monitor logs will also disable OpenTelemetry " + "logs and traces. Do you want to continue?" + ) + if not prompt_y_n(msg, default="n"): + raise DecoratorEarlyExitException() + + def _confirm_disable_azure_monitor_metrics(self, mc: ManagedCluster) -> None: + """Take the confirmation for a metrics disable that also turns OpenTelemetry metrics off. + + See _confirm_disable_azure_monitor_logs for why this can be a no-op. + """ + if not self.context.get_disable_azure_monitor_metrics(): + return + # Mirror the handler's own guards so no question is asked for a disable that would not + # actually turn anything off. + if not _is_azure_monitor_metrics_enabled(mc) or not _is_opentelemetry_metrics_enabled(mc): + return + if self.context.get_yes() or self._monitoring_disables_confirmed(): + return + + msg = ( + "Disabling Azure Monitor metrics will also disable OpenTelemetry metrics. " + "Do you want to continue?" + ) + if not prompt_y_n(msg, default="n"): + raise DecoratorEarlyExitException() + + def confirm_monitoring_disables(self, mc: ManagedCluster) -> None: + """Collect every monitoring disable confirmation before any cleanup runs. + + --disable-azure-monitor-logs deletes the data collection rule and association in Azure + before the cluster PUT happens, and it runs from update_addon_profiles, which the base + update flow calls well before the metrics profile is touched. Prompting from inside each + handler therefore means the logs collection resources are already gone by the time the + metrics question is asked, so declining it aborts the command with Container Insights + still reporting as enabled on the cluster while its DCR and DCRA no longer exist. Asking + everything up front keeps the command all or nothing. + + :return: None + """ + self._ensure_mc(mc) + + self._confirm_disable_azure_monitor_logs(mc) + self._confirm_disable_azure_monitor_metrics(mc) + self.context.set_intermediate("monitoring_disables_confirmed", True, overwrite_exists=True) + def update_addon_profiles(self, mc: ManagedCluster) -> ManagedCluster: """Update addon profiles for the ManagedCluster object. @@ -9213,6 +9346,10 @@ def update_addon_profiles(self, mc: ManagedCluster) -> ManagedCluster: """ self._ensure_mc(mc) + # Take every monitoring disable confirmation before the logs disable below starts deleting + # collection resources, so a later declined prompt cannot leave the cluster half torn down. + self.confirm_monitoring_disables(mc) + # Call the parent class method to handle base addon profile updates # (including Azure Keyvault Secrets Provider secret rotation settings) mc = super().update_addon_profiles(mc) @@ -9231,13 +9368,32 @@ def update_azure_monitor_logs_settings(self, mc: ManagedCluster) -> ManagedClust """Update the AMP containerInsights tuning settings for the ManagedCluster object. These flags are independent of --enable-azure-monitor-logs, so they also apply to a - cluster where Azure Monitor logs is already enabled. When neither flag is given nothing - is touched, which keeps the rest of the monitoring configuration intact. + cluster where Azure Monitor logs is already enabled. When none of the flags are given + nothing is touched, which keeps the rest of the monitoring configuration intact. :return: the ManagedCluster object """ self._ensure_mc(mc) + # These flags require DCR reprovisioning. Skip it when enabling logs (the DCR is created + # inline) or disabling logs (the DCR is removed). Read them before the cluster PUT so + # invalid data collection settings fail before postprocessing. + enable_syslog = self.context.get_enable_syslog() + data_collection_settings = self.context.get_data_collection_settings() + ampls_resource_id = self.context.get_ampls_resource_id() + if ( + ( + enable_syslog is not None or + data_collection_settings is not None or + ampls_resource_id is not None + ) and + not self.context.raw_param.get("enable_azure_monitor_logs") and + not self.context.raw_param.get("disable_azure_monitor_logs") + ): + self.context.set_intermediate( + "monitoring_addon_postprocessing_required", True, overwrite_exists=True + ) + syslog_port = self.context.get_syslog_port() disable_prometheus_scraping = self.context.get_disable_prometheus_metrics_scraping() if syslog_port is None and disable_prometheus_scraping is None: @@ -9462,16 +9618,11 @@ def postprocessing_after_mc_created(self, cluster: ManagedCluster) -> None: msi_auth_enabled = _is_monitoring_aad_auth(cluster, addon_consts) if monitoring_profile and monitoring_profile.enabled and msi_auth_enabled: - # Check parameter sizes to identify what might be causing large headers + # Oversized settings are rejected by the getter rather than dropped here, so that + # the command fails with an actionable error instead of quietly collecting the + # default set of data. data_collection_settings = self.context.get_data_collection_settings() - # Try to limit data_collection_settings size to avoid "Request Header Fields Too Large" error - safe_data_collection_settings = None - if data_collection_settings and len(str(data_collection_settings)) > 10000: - safe_data_collection_settings = None - else: - safe_data_collection_settings = data_collection_settings - self.context.external_functions.ensure_container_insights_for_monitoring( self.cmd, monitoring_profile, @@ -9484,7 +9635,7 @@ def postprocessing_after_mc_created(self, cluster: ManagedCluster) -> None: create_dcr=True, create_dcra=True, enable_syslog=self.context.get_enable_syslog(), - data_collection_settings=safe_data_collection_settings, + data_collection_settings=data_collection_settings, is_private_cluster=self.context.get_enable_private_cluster(), ampls_resource_id=self.context.get_ampls_resource_id(), enable_high_log_scale_mode=self.context.get_enable_high_log_scale_mode(), diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py index 3ff2b4cd8d3..b614ebb0b3e 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py @@ -935,6 +935,14 @@ def test_silent_for_non_monitoring_addons(self): self._warn(False, addons="azure-policy").assert_not_called() self._warn(False, addons=None).assert_not_called() + def test_silent_for_addon_names_that_merely_contain_monitoring(self): + # A substring check would misfire on these, so the list is matched token by token. + self._warn(False, addons="monitoring-preview").assert_not_called() + self._warn(False, addons="notmonitoring").assert_not_called() + + def test_tolerates_whitespace_and_casing_in_the_addon_list(self): + self._warn(False, addons=" azure-policy , Monitoring ").assert_called_once() + def test_warns_when_monitoring_is_one_of_several_addons(self): self._warn(False, addons="azure-policy,monitoring").assert_called_once() @@ -947,6 +955,48 @@ def test_does_not_change_the_auth_value(self): self.assertIsNone(warn_on_legacy_monitoring_auth(value, "monitoring")) +class TestAddonUpdateLegacyAuthWarning(unittest.TestCase): + """`aks addon update` must judge the warning on the value the user actually supplied.""" + + def _run(self, supplied_value, client_id): + from azext_aks_preview import custom + + instance = Mock() + instance.service_principal_profile.client_id = client_id + instance.addon_profiles = {"omsagent": Mock(enabled=True, config={})} + client = Mock() + client.get.return_value = instance + + with patch.object(custom, "warn_on_legacy_monitoring_auth") as warn, patch.object( + custom, "enable_addons", return_value=instance + ) as enable: + custom.aks_addon_update( + cmd=Mock(), + client=client, + resource_group_name="rg", + name="cluster", + addon="monitoring", + enable_msi_auth_for_monitoring=supplied_value, + ) + return warn, enable + + def test_omitted_flag_is_silent_on_a_service_principal_cluster(self): + # Service principal clusters cannot use managed identity auth, so the command forces the + # flag to False. That rewrite must not be mistaken for the user opting into shared keys. + warn, enable = self._run(None, client_id="a-service-principal") + warn.assert_called_once_with(None, "monitoring") + self.assertIs(enable.call_args.kwargs["enable_msi_auth_for_monitoring"], False) + + def test_explicitly_disabling_still_warns(self): + warn, _ = self._run(False, client_id="a-service-principal") + warn.assert_called_once_with(False, "monitoring") + + def test_omitted_flag_defaults_to_managed_identity_on_msi_clusters(self): + warn, enable = self._run(None, client_id="msi") + warn.assert_called_once_with(None, "monitoring") + self.assertIs(enable.call_args.kwargs["enable_msi_auth_for_monitoring"], True) + + class TestMonitoringArgumentRegistration(unittest.TestCase): """Argument registration for the containerInsights controls and the legacy auth warning.""" diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py index e613601b381..e6797155700 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py @@ -5,6 +5,9 @@ import datetime import importlib +import json +import os +import tempfile import unittest from unittest import mock from unittest.mock import Mock, patch @@ -6003,6 +6006,43 @@ def test_get_enable_high_log_scale_mode_update_error_disable_hlsm_with_existing_ with self.assertRaises(MutuallyExclusiveArgumentError): ctx.get_enable_high_log_scale_mode() + def test_get_enable_high_log_scale_mode_update_allows_disable_when_amp_profile_says_disabled( + self, + ): + """An explicit Disabled on the Azure Monitor profile outranks the stale legacy mirror. + + The legacy ``enableRetinaNetworkFlags`` addon config key is a server-side mirror of the + Azure Monitor profile, so it still reports the previous value while an update is in + flight. Disabling container network logs and high log scale mode in one command must not + be rejected on the strength of that stale value. + """ + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict( + { + "enable_high_log_scale_mode": False, + } + ), + self.models, + decorator_mode=DecoratorMode.UPDATE, + ) + mc = self.models.ManagedCluster( + location="test_location", + addon_profiles={ + "omsagent": self.models.ManagedClusterAddonProfile( + enabled=True, + config={"enableRetinaNetworkFlags": "True"}, + ) + }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + container_network_logs=CONST_CONTAINER_NETWORK_LOGS_DISABLED, + ) + ), + ) + ctx.attach_mc(mc) + self.assertFalse(ctx.get_enable_high_log_scale_mode()) + def test_get_enable_high_log_scale_mode_update_monitoring_camelcase_key(self): """Test auto-enable HLSM in update mode when monitoring uses camelCase 'omsAgent' key.""" ctx = AKSPreviewManagedClusterContext( @@ -16979,6 +17019,56 @@ def test_update_monitoring_profile_flow_logs_no_flags_noop(self): ) ) + def test_update_disable_cnl_and_hlsm_together_is_not_rejected(self): + """Disabling container network logs and high log scale mode in one command must work. + + Regression test: the legacy ``enableRetinaNetworkFlags`` addon config key mirrors the + Azure Monitor profile server-side, so it still reads "True" for a cluster that has + container network logs on. Once ``--disable-container-network-logs`` has written + Disabled to the profile, the high log scale mode guard must read that value rather than + the stale mirror, otherwise the pair of flags can never be used together. + """ + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + { + "disable_container_network_logs": True, + "enable_high_log_scale_mode": False, + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster( + location="test_location", + addon_profiles={ + "omsagent": self.models.ManagedClusterAddonProfile( + enabled=True, + config={ + CONST_MONITORING_USING_AAD_MSI_AUTH: "true", + "enableRetinaNetworkFlags": "True", + }, + ) + }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + enabled=True, + container_network_logs=CONST_CONTAINER_NETWORK_LOGS_ENABLED, + ) + ), + ) + dec.context.attach_mc(mc) + dec_mc = dec.update_monitoring_profile_flow_logs(mc) + + self.assertEqual( + dec_mc.azure_monitor_profile.container_insights.container_network_logs, + CONST_CONTAINER_NETWORK_LOGS_DISABLED, + ) + # The DCR still has to be rewritten to drop the high-scale and flow log streams. + self.assertTrue( + dec.context.get_intermediate( + "monitoring_addon_postprocessing_required", default_value=False + ) + ) + def test_update_enable_cnl_with_azure_monitor_logs_on_cluster(self): """Test enabling CNL on update when monitoring was enabled via enable_azure_monitor_logs on existing cluster.""" dec = AKSPreviewManagedClusterUpdateDecorator( @@ -20778,5 +20868,270 @@ def test_disable_azure_monitor_logs_no_prompt_when_otlp_logs_disabled(self): self.assertFalse(mc.azure_monitor_profile.container_insights.enabled) +class AKSPreviewCoreMonitoringFixPortsTestCase(unittest.TestCase): + """Ports of the Azure Monitor fixes made in azure-cli core. + + The aks-preview extension shadows the core acs module for `az aks`, and implements the whole + Azure Monitor logs feature itself, so fixes made in core do not reach anyone with the + extension installed unless they are ported here too. + """ + + def setUp(self): + register_aks_preview_resource_type() + self.cli_ctx = MockCLI() + self.cmd = MockCmd(self.cli_ctx) + self.models = AKSPreviewManagedClusterModels(self.cmd, CUSTOM_MGMT_AKS_PREVIEW) + self.client = MockClient() + + def _write_settings(self, payload): + handle = tempfile.NamedTemporaryFile("w", suffix=".json", delete=False) + json.dump(payload, handle) + handle.close() + self.addCleanup(os.unlink, handle.name) + return handle.name + + def test_data_collection_settings_size_is_measured_on_the_contents(self): + # The getter returns the file path, so a check against the path length can never fire. + # An oversized file has to be rejected on the size of what gets serialized into the DCR. + oversized = self._write_settings({"namespaces": ["n" * 20000]}) + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict({"data_collection_settings": oversized}), + self.models, + decorator_mode=DecoratorMode.UPDATE, + ) + with self.assertRaises(InvalidArgumentValueError) as cm: + ctx.get_data_collection_settings() + self.assertIn("--data-collection-settings is too large", str(cm.exception)) + # The path itself is far below the limit, which is why the original check never triggered. + self.assertLess(len(oversized), 10000) + + def test_data_collection_settings_of_normal_size_are_accepted(self): + path = self._write_settings({"interval": "1m"}) + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict({"data_collection_settings": path}), + self.models, + decorator_mode=DecoratorMode.UPDATE, + ) + self.assertEqual(ctx.get_data_collection_settings(), path) + + def _flags_trigger_dcr_reprovisioning(self, raw_params): + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, self.client, raw_params, CUSTOM_MGMT_AKS_PREVIEW + ) + mc = self.models.ManagedCluster( + location="test_location", + addon_profiles={ + "omsagent": self.models.ManagedClusterAddonProfile( + enabled=True, config={CONST_MONITORING_USING_AAD_MSI_AUTH: "true"} + ) + }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + enabled=True, + ) + ), + ) + dec.context.attach_mc(mc) + dec.update_azure_monitor_logs_settings(mc) + return dec.context.get_intermediate( + "monitoring_addon_postprocessing_required", default_value=False + ) + + def test_syslog_and_dcr_carried_flags_reprovision_the_data_collection_rule(self): + # These three only reach the cluster through the DCR, so without re-provisioning the + # command reports success while the agent keeps using the previous rule. + self.assertTrue(self._flags_trigger_dcr_reprovisioning({"enable_syslog": True})) + self.assertTrue(self._flags_trigger_dcr_reprovisioning({"ampls_resource_id": "/amp/ls/id"})) + + def test_agent_side_settings_do_not_reprovision_the_data_collection_rule(self): + # The syslog port and the scraping toggle never appear in the DCR. + self.assertFalse(self._flags_trigger_dcr_reprovisioning({"syslog_port": 28330})) + self.assertFalse(self._flags_trigger_dcr_reprovisioning({})) + + def test_no_reprovisioning_while_enabling_or_disabling_azure_monitor_logs(self): + # Enabling provisions the DCR inline before the PUT, and disabling has just torn it down. + self.assertFalse( + self._flags_trigger_dcr_reprovisioning( + {"enable_syslog": True, "enable_azure_monitor_logs": True} + ) + ) + self.assertFalse( + self._flags_trigger_dcr_reprovisioning( + {"enable_syslog": True, "disable_azure_monitor_logs": True} + ) + ) + + def _mc_with_logs_and_otlp(self): + return self.models.ManagedCluster( + location="test_location", + addon_profiles={ + "omsagent": self.models.ManagedClusterAddonProfile( + enabled=True, config={CONST_MONITORING_USING_AAD_MSI_AUTH: "true"} + ) + }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + metrics=self.models.ManagedClusterAzureMonitorProfileMetrics(enabled=True), + app_monitoring=self.models.ManagedClusterAzureMonitorProfileAppMonitoring( + open_telemetry_metrics=( + self.models.ManagedClusterAzureMonitorProfileAppMonitoringOpenTelemetryMetrics( + enabled=True + ) + ), + open_telemetry_logs_and_traces=( + self.models.ManagedClusterAzureMonitorProfileAppMonitoringOpenTelemetryLogsAndTraces( + enabled=True + ) + ), + ), + ), + ) + + def test_declining_the_metrics_prompt_aborts_before_the_logs_cleanup_runs(self): + """Regression test: both confirmations are collected before anything is deleted. + + --disable-azure-monitor-logs deletes the DCR and DCRA from Azure before the cluster PUT, + and it runs from update_addon_profiles, well before the metrics profile is touched. If the + metrics question were still asked from its own handler, declining it would abort with the + logs collection resources already gone but Container Insights still enabled. + """ + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"disable_azure_monitor_logs": True, "disable_azure_monitor_metrics": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self._mc_with_logs_and_otlp() + dec.context.attach_mc(mc) + + # Say yes to the logs question, no to the metrics one. + with patch( + "azext_aks_preview.managed_cluster_decorator.prompt_y_n", side_effect=[True, False] + ) as prompt, patch.object( + dec, "_disable_azure_monitor_logs" + ) as disable_logs, patch.object( + dec.client, "get" + ) as client_get: + with self.assertRaises(DecoratorEarlyExitException): + dec.update_addon_profiles(mc) + + self.assertEqual(prompt.call_count, 2) + # Nothing was torn down: neither the cleanup handler nor the ARM read that precedes it ran. + disable_logs.assert_not_called() + client_get.assert_not_called() + + def test_each_disable_question_is_asked_exactly_once(self): + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"disable_azure_monitor_logs": True, "disable_azure_monitor_metrics": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self._mc_with_logs_and_otlp() + dec.context.attach_mc(mc) + + with patch( + "azext_aks_preview.managed_cluster_decorator.prompt_y_n", return_value=True + ) as prompt: + dec.confirm_monitoring_disables(mc) + # The handlers still call the confirmation helpers, which must now be no-ops. + dec._confirm_disable_azure_monitor_logs(mc) + dec._confirm_disable_azure_monitor_metrics(mc) + + self.assertEqual(prompt.call_count, 2) + + def test_opentelemetry_port_cannot_be_set_while_disabling_the_receiver(self): + # Disabling clears the ports, so a port supplied alongside would be accepted and then + # silently dropped. The cluster has the receivers already on, so the "must also be + # enabled" error cannot fire and the disable guard is the only thing under test. + cases = [ + ("get_opentelemetry_metrics_port", "opentelemetry_metrics_port", + "disable_opentelemetry_metrics", "--opentelemetry-metrics-port-http", + "--disable-opentelemetry-metrics"), + ("get_opentelemetry_metrics_port_grpc", "opentelemetry_metrics_port_grpc", + "disable_opentelemetry_metrics", "--opentelemetry-metrics-port-grpc", + "--disable-opentelemetry-metrics"), + ("get_opentelemetry_logs_port", "opentelemetry_logs_port", + "disable_opentelemetry_logs", "--opentelemetry-logs-traces-port-http", + "--disable-opentelemetry-logs-traces"), + ("get_opentelemetry_logs_traces_port_grpc", "opentelemetry_logs_traces_port_grpc", + "disable_opentelemetry_logs", "--opentelemetry-logs-traces-port-grpc", + "--disable-opentelemetry-logs-traces"), + ] + for getter, port_param, disable_param, port_flag, disable_flag in cases: + with self.subTest(getter=getter): + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict({port_param: 4318, disable_param: True}), + self.models, + decorator_mode=DecoratorMode.UPDATE, + ) + ctx.attach_mc(self._mc_with_logs_and_otlp()) + with self.assertRaises(InvalidArgumentValueError) as cm: + getattr(ctx, getter)() + message = str(cm.exception) + self.assertIn(port_flag, message) + self.assertIn(disable_flag, message) + + def test_opentelemetry_port_alone_is_accepted_on_an_enabled_cluster(self): + # Guard against over-rejecting: a port update on its own must still go through. + ctx = AKSPreviewManagedClusterContext( + self.cmd, + AKSManagedClusterParamDict({"opentelemetry_metrics_port": 4318}), + self.models, + decorator_mode=DecoratorMode.UPDATE, + ) + ctx.attach_mc(self._mc_with_logs_and_otlp()) + self.assertEqual(ctx.get_opentelemetry_metrics_port(), 4318) + + def test_a_port_only_update_reaches_the_cluster(self): + """aks-preview already implements core's port-only update; this locks the behaviour in. + + Supplying just a port flag, with no --enable-opentelemetry-* alongside it, must actually + change the port on an already-enabled receiver rather than being parsed and dropped. + """ + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"opentelemetry_metrics_port": 4318, "opentelemetry_logs_port": 4319}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster( + location="test_location", + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + app_monitoring=self.models.ManagedClusterAzureMonitorProfileAppMonitoring( + open_telemetry_metrics=( + self.models.ManagedClusterAzureMonitorProfileAppMonitoringOpenTelemetryMetrics( + enabled=True, http_port=1111 + ) + ), + open_telemetry_logs_and_traces=( + self.models.ManagedClusterAzureMonitorProfileAppMonitoringOpenTelemetryLogsAndTraces( + enabled=True, http_port=2222 + ) + ), + ), + ), + ) + dec.context.attach_mc(mc) + + # An OpenTelemetry metrics port legitimately drives the Prometheus prerequisite path, + # which talks to ARM; the port plumbing is what is under test here. + with patch( + "azext_aks_preview.managed_cluster_decorator.ensure_azure_monitor_profile_prerequisites" + ), patch.object( + dec.context, "get_subscription_id", return_value="test_sub_id" + ): + dec.update_azure_monitor_profile(mc) + + app_monitoring = mc.azure_monitor_profile.app_monitoring + self.assertEqual(app_monitoring.open_telemetry_metrics.http_port, 4318) + self.assertEqual(app_monitoring.open_telemetry_logs_and_traces.http_port, 4319) + # The receivers stay on: a port change must not alter the enabled state. + self.assertTrue(app_monitoring.open_telemetry_metrics.enabled) + self.assertTrue(app_monitoring.open_telemetry_logs_and_traces.enabled) + + if __name__ == "__main__": unittest.main() diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py b/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py index 9e2be666d35..acc17e1cdbb 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py @@ -2800,10 +2800,6 @@ def test_nodepool_update_allows_windows2022_and_windows2025(self): self.assertIn("Windows2025", node_os_skus_update) -if __name__ == "__main__": - unittest.main() - - class ContainerInsightsSettingsNamespace(SimpleNamespace): """Namespace for the containerInsights tuning flags, with CLI defaults.""" @@ -2879,3 +2875,7 @@ def test_conflicts_with_disable_azure_monitor_logs(self): with self.assertRaises(ArgumentUsageError) as cm: validators.validate_container_insights_settings_for_update(namespace) self.assertIn("--disable-azure-monitor-logs", str(cm.exception)) + + +if __name__ == "__main__": + unittest.main() From 77cf7115b307c1ad194905d70cc0551ebc73e080 Mon Sep 17 00:00:00 2001 From: Sunil Yadav Date: Tue, 22 Sep 2026 00:05:35 +0000 Subject: [PATCH 6/8] bug fixes --- src/aks-preview/HISTORY.rst | 9 + src/aks-preview/azext_aks_preview/_params.py | 6 + .../azext_aks_preview/_validators.py | 33 ++ src/aks-preview/azext_aks_preview/custom.py | 193 +++++++++++- .../managed_cluster_decorator.py | 14 + .../tests/latest/test_custom.py | 285 ++++++++++++++++++ .../latest/test_managed_cluster_decorator.py | 241 ++++++++++++++- .../tests/latest/test_validators.py | 95 ++++++ 8 files changed, 865 insertions(+), 11 deletions(-) diff --git a/src/aks-preview/HISTORY.rst b/src/aks-preview/HISTORY.rst index f3426b29d7d..9010e409b86 100644 --- a/src/aks-preview/HISTORY.rst +++ b/src/aks-preview/HISTORY.rst @@ -28,7 +28,16 @@ Pending * `az aks update`: Fix `--enable-syslog`, `--data-collection-settings` and `--ampls-resource-id` being silently ignored when supplied on their own, as none of them re-provisioned the data collection rule that carries them, so the command reported success while the agent kept using the previous rule. * `az aks update`: Reject the OpenTelemetry port flags (`--opentelemetry-metrics-port-http`, `--opentelemetry-metrics-port-grpc`, `--opentelemetry-logs-traces-port-http` and `--opentelemetry-logs-traces-port-grpc`) when the matching receiver is being disabled in the same command, instead of accepting the port and then silently dropping it. * `az aks update`: Collect every monitoring disable confirmation before any of them deletes collection resources. Combining `--disable-azure-monitor-logs` with `--disable-azure-monitor-metrics` used to delete the logs data collection rule and association before asking about metrics, so declining that prompt aborted the command with Container Insights still enabled on the cluster but its collection objects already removed. +* `az aks disable-addons`: Disabling the `monitoring` addon now resets the Container Insights settings (syslog port, Prometheus scraping and container network logs) back to their defaults and turns off OpenTelemetry logs and traces, matching `az aks update --disable-azure-monitor-logs`. The addon settings used to survive the disable and were silently inherited by the next onboarding. +* `az aks disable-addons`: Ask for confirmation before disabling the `monitoring` addon when OpenTelemetry logs and traces are enabled, since that collection is disabled along with it. Add `--yes` to skip the prompt. +* `az aks disable-addons`: Fix the monitoring cleanup being skipped when `monitoring` was passed alongside other addons, for example `--addons monitoring,azure-policy`. +* `az aks update`: Reset the Container Insights settings to their defaults when `--enable-azure-monitor-logs` re-onboards a cluster, so a previous onboarding's syslog port, Prometheus scraping and container network logs settings are no longer inherited. +* `az aks update`: Preserve the existing data collection rule settings when reconfiguring a cluster that is already onboarded, so that changing one setting no longer drops the others. `--enable-syslog` on its own used to rebuild the rule without the cluster's high log scale mode, custom data collection settings and ingestion data collection endpoint, and `--data-collection-settings` on its own used to drop syslog. Enabling Azure Monitor logs still starts from the documented defaults. +* `az aks create` and `az aks update`: Reject the OpenTelemetry port flags at argument validation time when the matching receiver is being disabled in the same command. The conflict was previously caught only after the Azure Monitor collection resources had already been removed, so the command failed with the cluster partially torn down. +* `az aks disable-addons`: Validate every addon name and its installed state before any cleanup runs. Disabling an unknown or not-installed addon alongside `monitoring` used to delete the monitoring data collection rule association first and only then fail, skipping the cluster update and leaving monitoring enabled with nothing to collect into. +* Fix `--enable-high-log-scale-mode` mutating the shared list of Container Insights streams, so the stream set leaked between data collection rules built in the same command invocation. * `az aks create` and `az aks update`: Fix the `--data-collection-settings` size limit being applied to the file path instead of the settings it holds, which let an oversized file through to fail the data collection rule call with `Request Header Fields Too Large`. +* `az aks update`: Fix `--enable-syslog false` being rejected with `Please specify one or more of "--enable-syslog"` after prompting to reconcile the cluster. Explicitly turning syslog collection off is a real update request, but the falsy value made the command treat it as though no argument had been supplied. 22.0.0b6 +++++++++ diff --git a/src/aks-preview/azext_aks_preview/_params.py b/src/aks-preview/azext_aks_preview/_params.py index b6aa4ef39bc..6c6283bb6f9 100644 --- a/src/aks-preview/azext_aks_preview/_params.py +++ b/src/aks-preview/azext_aks_preview/_params.py @@ -3262,6 +3262,12 @@ def load_arguments(self, _): with self.argument_context("aks disable-addons") as c: c.argument("addons", options_list=["--addons", "-a"], validator=validate_addons) + c.argument( + "yes", + options_list=["--yes", "-y"], + help="Do not prompt for confirmation.", + action="store_true", + ) with self.argument_context("aks enable-addons") as c: c.argument("addons", options_list=["--addons", "-a"], validator=validate_addons) diff --git a/src/aks-preview/azext_aks_preview/_validators.py b/src/aks-preview/azext_aks_preview/_validators.py index 867126bac43..0e0108e565a 100644 --- a/src/aks-preview/azext_aks_preview/_validators.py +++ b/src/aks-preview/azext_aks_preview/_validators.py @@ -1227,10 +1227,42 @@ def validate_opentelemetry_logs_dependencies_for_update(namespace): # to the cluster's Azure Monitor profile +def validate_opentelemetry_ports_not_disabled(namespace): + """Reject OpenTelemetry port flags that target a receiver being disabled. + + The decorator's port getters enforce this too, but only after cleanup of the + Azure Monitor collection resources has already run. Validating here fails the + command before any destructive work happens. + """ + metrics_disables = ( + ("--disable-azure-monitor-metrics", "disable_azure_monitor_metrics"), + ("--disable-opentelemetry-metrics", "disable_opentelemetry_metrics"), + ) + logs_traces_disables = ( + ("--disable-azure-monitor-logs", "disable_azure_monitor_logs"), + ("--disable-opentelemetry-logs-traces", "disable_opentelemetry_logs"), + ) + ports = ( + ("--opentelemetry-metrics-port-http", "opentelemetry_metrics_port", metrics_disables), + ("--opentelemetry-metrics-port-grpc", "opentelemetry_metrics_port_grpc", metrics_disables), + ("--opentelemetry-logs-traces-port-http", "opentelemetry_logs_port", logs_traces_disables), + ("--opentelemetry-logs-traces-port-grpc", "opentelemetry_logs_traces_port_grpc", logs_traces_disables), + ) + for port_flag, port_attr, disables in ports: + if getattr(namespace, port_attr, None) is None: + continue + for disable_flag, disable_attr in disables: + if getattr(namespace, disable_attr, False): + raise InvalidArgumentValueError( + f"{port_flag} cannot be specified when {disable_flag} is used." + ) + + def validate_azure_monitor_and_opentelemetry_for_create(namespace): """Main validator for Azure Monitor and OpenTelemetry configurations for create operations.""" # Run all OpenTelemetry-related validations validate_opentelemetry_ports(namespace) + validate_opentelemetry_ports_not_disabled(namespace) validate_opentelemetry_metrics_dependencies(namespace) validate_opentelemetry_logs_dependencies(namespace) @@ -1239,6 +1271,7 @@ def validate_azure_monitor_and_opentelemetry_for_update(namespace): """Main validator for Azure Monitor and OpenTelemetry configurations for update operations.""" # Run all OpenTelemetry-related validations validate_opentelemetry_ports(namespace) + validate_opentelemetry_ports_not_disabled(namespace) validate_opentelemetry_metrics_dependencies_for_update(namespace) validate_opentelemetry_logs_dependencies_for_update(namespace) diff --git a/src/aks-preview/azext_aks_preview/custom.py b/src/aks-preview/azext_aks_preview/custom.py index 1ecf4d79d2c..accd0e17fe6 100644 --- a/src/aks-preview/azext_aks_preview/custom.py +++ b/src/aks-preview/azext_aks_preview/custom.py @@ -159,7 +159,6 @@ ensure_container_insights_for_monitoring, ensure_default_log_analytics_workspace_for_monitoring, sanitize_loganalytics_ws_resource_id, - get_existing_container_insights_extension_dcr_tags, validate_data_collection_settings, create_data_collection_endpoint, create_or_delete_dcr_association, @@ -287,6 +286,66 @@ def _create_or_update_dcr_with_table_readiness_retry(resources, dcr_resource_id, # pylint: disable=too-many-locals,too-many-branches,too-many-statements,line-too-long +def get_existing_container_insights_extension_dcr(cmd, dcr_url): + """Fetch the Container Insights extension DCR that already exists, or {} if there is none.""" + _MAX_RETRY_TIMES = 3 + for retry_count in range(0, _MAX_RETRY_TIMES): + try: + resp = send_raw_request( + cmd.cli_ctx, "GET", dcr_url + ) + return json.loads(resp.text) + except CLIError as e: + if "ResourceNotFound" in str(e): + break + if retry_count >= (_MAX_RETRY_TIMES - 1): + raise e + return {} + + +def _resolve_dcr_settings_from_existing( + existing_dcr, enable_syslog, data_collection_settings, enable_high_log_scale_mode +): + """Carry over DCR settings the caller did not specify from the DCR that already exists. + + The DCR is rebuilt from scratch on every reconfiguration, so a setting that is not supplied on + the command line is otherwise silently dropped: 'az aks update --enable-syslog' on a cluster + using high log scale mode would rebuild the DCR without the high scale streams, its custom data + collection settings and its ingestion DCE, and '--data-collection-settings' on its own would + drop syslog. Only explicitly supplied settings should change anything. + + None means "not specified"; an explicit True or False always wins. Returns the resolved + (enable_syslog, existing_data_collection_settings, enable_high_log_scale_mode), where the + middle value is an already-parsed settings dict to fall back on, or None. + """ + properties = (existing_dcr or {}).get("properties") or {} + data_sources = properties.get("dataSources") or {} + ci_extension = next( + ( + extension + for extension in (data_sources.get("extensions") or []) + if extension.get("extensionName") == "ContainerInsights" + ), + {}, + ) + + if enable_syslog is None: + enable_syslog = bool(data_sources.get("syslog")) + + if enable_high_log_scale_mode is None: + enable_high_log_scale_mode = "Microsoft-ContainerLogV2-HighScale" in ( + ci_extension.get("streams") or [] + ) + + existing_data_collection_settings = None + if data_collection_settings is None: + existing_data_collection_settings = ( + ci_extension.get("extensionSettings") or {} + ).get("dataCollectionSettings") or None + + return enable_syslog, existing_data_collection_settings, enable_high_log_scale_mode + + def ensure_container_insights_for_monitoring_preview( cmd, addon, @@ -303,6 +362,7 @@ def ensure_container_insights_for_monitoring_preview( is_private_cluster=False, ampls_resource_id=None, enable_high_log_scale_mode=None, + preserve_existing_dcr_settings=False, ): """ Preview extension version of ensure_container_insights_for_monitoring that uses REST API @@ -430,6 +490,30 @@ def __init__(self, location, resource_id): f"/subscriptions/{cluster_subscription}/resourceGroups/{cluster_resource_group_name}/" f"providers/Microsoft.Insights/dataCollectionRules/{dataCollectionRuleName}" ) + dcr_url = cmd.cli_ctx.cloud.endpoints.resource_manager + \ + f"{dcr_resource_id}?api-version=2022-06-01" + + existing_dcr = {} + existing_data_collection_settings = None + if create_dcr: + # The DCR is rebuilt from scratch below, so read the one that already exists to keep + # customer-added tags. + existing_dcr = get_existing_container_insights_extension_dcr(cmd, dcr_url) + if preserve_existing_dcr_settings: + # Reconfiguring a cluster that is already onboarded: carry over any setting the + # caller did not supply instead of dropping it. A fresh onboarding deliberately + # skips this. Disabling monitoring leaves the DCR behind, so inheriting from it + # would make re-enabling pick up the previous onboarding's syslog, high log scale + # mode and custom data collection settings, when re-enabling is meant to start + # from the documented defaults just like the container insights profile does. + ( + enable_syslog, + existing_data_collection_settings, + enable_high_log_scale_mode, + ) = _resolve_dcr_settings_from_existing( + existing_dcr, enable_syslog, data_collection_settings, enable_high_log_scale_mode + ) + existing_tags = existing_dcr.get("tags") or {} # ingestion DCE MUST be in workspace region ingestionDataCollectionEndpointName = f"MSCI-ingest-{location}-{cluster_name}" @@ -476,22 +560,35 @@ def __init__(self, location, resource_id): "name" ] - dcr_url = cmd.cli_ctx.cloud.endpoints.resource_manager + \ - f"{dcr_resource_id}?api-version=2022-06-01" - # get existing tags on the container insights extension DCR if the customer added any - existing_tags = get_existing_container_insights_extension_dcr_tags( - cmd, dcr_url) # get data collection settings extensionSettings = {} cistreams = ["Microsoft-ContainerInsights-Group-Default"] if enable_high_log_scale_mode: - cistreams = ContainerInsightsStreams + cistreams = list(ContainerInsightsStreams) if data_collection_settings is not None: dataCollectionSettings = _get_data_collection_settings(data_collection_settings) validate_data_collection_settings(dataCollectionSettings) dataCollectionSettings.setdefault("enableContainerLogV2", True) extensionSettings["dataCollectionSettings"] = dataCollectionSettings cistreams = dataCollectionSettings["streams"] + elif existing_data_collection_settings is not None: + # inherited from the existing DCR, which was validated when it was first written + dataCollectionSettings = dict(existing_data_collection_settings) + dataCollectionSettings.setdefault("enableContainerLogV2", True) + if dataCollectionSettings.get("streams"): + # the stored streams were rewritten to match the high log scale mode the DCR + # was written with, so normalise them back to whichever mode applies now + inherited_streams = list(dataCollectionSettings["streams"]) + if not enable_high_log_scale_mode: + inherited_streams = [ + "Microsoft-ContainerLogV2" + if stream == "Microsoft-ContainerLogV2-HighScale" + else stream + for stream in inherited_streams + ] + dataCollectionSettings["streams"] = inherited_streams + cistreams = inherited_streams + extensionSettings["dataCollectionSettings"] = dataCollectionSettings else: # If data_collection_settings is None, set default dataCollectionSettings dataCollectionSettings = { @@ -3661,15 +3758,73 @@ def aks_addon_update( ) -def aks_disable_addons(cmd, client, resource_group_name, name, addons, no_wait=False): +def _validate_addons_for_disable(instance, addons): + """Validate every addon name and installed state before any cleanup runs. + + aks_disable_addons deletes the monitoring data collection rule association before it builds the + updated cluster payload. _update_addons rejects unknown or not-installed addons, but by then the + association is already gone, and the raised error skips the cluster PUT — leaving monitoring + enabled on the cluster with nothing for the agent to collect into. Fail here instead, before + anything is deleted. The checks mirror _update_addons so the two cannot drift. + """ + addon_profiles = instance.addon_profiles or {} + installed = {key.lower() for key in addon_profiles} + for addon_arg in addons.split(','): + # These two live in the ingress profile rather than the addon profiles, and _update_addons + # handles them before it validates anything, so neither check below applies to them. + if addon_arg in ("applicationloadbalancer", "web_application_routing"): + continue + if addon_arg not in ADDONS: + raise CLIError("Invalid addon name: {}.".format(addon_arg)) + addon = ADDONS[addon_arg] + if addon == CONST_VIRTUAL_NODE_ADDON_NAME: + # Only Linux is supported for now, matching _update_addons. + addon += 'Linux' + # kube-dashboard is exempt: _update_addons synthesizes a disabled profile for it rather + # than failing, so disabling it on a cluster that never had it is not an error. + if addon.lower() not in installed and addon != CONST_KUBE_DASHBOARD_ADDON_NAME: + raise CLIError("The addon {} is not installed.".format(addon)) + + +def aks_disable_addons(cmd, client, resource_group_name, name, addons, no_wait=False, yes=False): + from azext_aks_preview.managed_cluster_decorator import ( + _is_opentelemetry_logs_traces_enabled, + _reset_container_insights_to_defaults, + ) + instance = client.get(resource_group_name, name) subscription_id = get_subscription_id(cmd.cli_ctx) + # Every addon is validated up front, because the cleanup below is destructive and a later + # failure would skip the cluster PUT that is supposed to accompany it. + _validate_addons_for_disable(instance, addons) + + # --addons takes a comma separated list, so an exact string compare silently skips the + # monitoring cleanup for 'monitoring,azure-policy' and leaves the DCR association behind. + disabling_monitoring = CONST_MONITORING_ADDON_NAME in [ + ADDONS.get(addon) for addon in addons.split(",") + ] + + # OpenTelemetry logs and traces are collected by the Container Insights agent, so disabling the + # monitoring addon necessarily turns them off too. The confirmation is taken before any cleanup + # runs, otherwise declining the prompt would leave the DCR association already deleted. + opentelemetry_logs_traces_enabled = ( + disabling_monitoring and _is_opentelemetry_logs_traces_enabled(instance) + ) + if opentelemetry_logs_traces_enabled and not yes: + msg = ( + "OpenTelemetry logs and traces are enabled on this cluster and are collected by " + "Azure Monitor logs. Disabling the monitoring addon will also disable OpenTelemetry " + "logs and traces. Do you want to continue?" + ) + if not prompt_y_n(msg, default="n"): + return None + try: addon_profiles = instance.addon_profiles or {} monitoring_addon_key = get_monitoring_addon_key(addon_profiles, CONST_MONITORING_ADDON_NAME) if ( - addons == "monitoring" and + disabling_monitoring and monitoring_addon_key in addon_profiles and addon_profiles[monitoring_addon_key].enabled and CONST_MONITORING_USING_AAD_MSI_AUTH in @@ -3711,6 +3866,26 @@ def aks_disable_addons(cmd, client, resource_group_name, name, addons, no_wait=F no_wait=no_wait ) + if disabling_monitoring: + # The legacy addon and the AMP containerInsights profile are two views of the same feature, + # and the RP only copies a containerInsights field onto the cluster when that field is + # present on the request. Disabling therefore has to reset them explicitly, otherwise a + # later --enable-azure-monitor-logs silently re-onboards with the old syslog port, scraping + # choice and container network logs setting. + if instance.azure_monitor_profile and instance.azure_monitor_profile.container_insights: + _reset_container_insights_to_defaults( + instance.azure_monitor_profile.container_insights + ) + # OpenTelemetry logs and traces ride on the Container Insights agent, so they go down with + # it. The confirmation for this was taken above. + if opentelemetry_logs_traces_enabled: + open_telemetry_logs_and_traces = ( + instance.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces + ) + open_telemetry_logs_and_traces.enabled = False + open_telemetry_logs_and_traces.http_port = None + open_telemetry_logs_and_traces.grpc_port = None + # send the managed cluster representation to update the addon profiles return sdk_no_wait(no_wait, client.begin_create_or_update, resource_group_name, name, instance) diff --git a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py index f2226e9f95a..666cb3aeb4c 100644 --- a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py @@ -6772,6 +6772,7 @@ def get_special_parameter_default_value_pairs_list(self) -> List[Tuple[Any, Any] (self.context.get_nat_gateway_outbound_ip_ids(), None), (self.context.get_nat_gateway_outbound_ip_prefix_ids(), None), (self.context.raw_param.get("enable_high_log_scale_mode"), None), + (self.context.raw_param.get("enable_syslog"), None), ] def check_raw_parameters(self): @@ -9108,6 +9109,15 @@ def _setup_azure_monitor_logs(self, mc: ManagedCluster) -> None: # disabled addon) to managed identity, while preserving the existing useAADAuth value on a # cluster that is already onboarded with legacy shared-key auth. container_insights = self._ensure_container_insights(mc) + + # The guards above reject clusters that are already enabled, so reaching here is always a + # genuine onboarding rather than a reconfigure. Start from the documented defaults so the + # result does not depend on what a previous onboarding left behind: the RP copies a + # containerInsights field onto the cluster only when that field is present on the request, + # so a stale syslog port, scraping choice or container network logs setting would otherwise + # be inherited silently. Values the user asked for are applied on top of this below and by + # update_azure_monitor_logs_settings, which runs later in the update flow. + _reset_container_insights_to_defaults(container_insights) container_insights.enabled = True container_insights.log_analytics_workspace_resource_id = workspace_resource_id @@ -9639,6 +9649,10 @@ def postprocessing_after_mc_created(self, cluster: ManagedCluster) -> None: is_private_cluster=self.context.get_enable_private_cluster(), ampls_resource_id=self.context.get_ampls_resource_id(), enable_high_log_scale_mode=self.context.get_enable_high_log_scale_mode(), + # This is the reconfigure path: the cluster is already onboarded and only + # the settings named on the command line should change. Everything else is + # carried over from the existing DCR rather than reset to its default. + preserve_existing_dcr_settings=True, ) # Monitoring addon disable cleanup is now done upfront in _disable_azure_monitor_logs (not in postprocessing) diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py index b614ebb0b3e..976512c147b 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py @@ -9,6 +9,10 @@ from azext_aks_preview import ContainerServiceCommandsLoader from azext_aks_preview._client_factory import CUSTOM_MGMT_AKS_PREVIEW from azext_aks_preview._consts import CONST_FLEX_NODES +from azext_aks_preview._consts import ( + CONST_CONTAINER_INSIGHTS_DEFAULT_SYSLOG_PORT, + CONST_CONTAINER_NETWORK_LOGS_DISABLED, +) from azext_aks_preview.agentpool_decorator import AKSPreviewAgentPoolModels from azext_aks_preview.managed_cluster_decorator import ( AKSPreviewManagedClusterModels, @@ -23,6 +27,8 @@ aks_upgrade, aks_enable_addons, aks_list_vm_skus, + _validate_addons_for_disable, + _resolve_dcr_settings_from_existing, ) from azext_aks_preview.tests.latest.mocks import MockCLI, MockClient, MockCmd from azure.cli.command_modules.acs._consts import AgentPoolDecoratorMode @@ -1083,5 +1089,284 @@ def test_container_insights_controls_registered_on_create_and_update(self): self.assertIsNotNone(arguments["syslog_port"].get("validator"), command_name) +class TestDisableAddonsMonitoringCleanup(unittest.TestCase): + """`az aks disable-addons -a monitoring` must behave like `--disable-azure-monitor-logs`. + + The two commands are the legacy and the first-class view of the same feature, so a cluster + disabled through either one has to end up in the same state. + """ + + def setUp(self): + register_aks_preview_resource_type() + self.cli_ctx = MockCLI() + self.cmd = MockCmd(self.cli_ctx) + + def _instance(self, otlp_enabled=False, stale=True): + container_insights = Mock( + enabled=True, + syslog_port=51400 if stale else 28330, + disable_prometheus_metrics_scraping=True if stale else False, + container_network_logs="Enabled" if stale else "Disabled", + ) + otlp = Mock(enabled=otlp_enabled, http_port=4318, grpc_port=4317) + instance = Mock() + instance.azure_monitor_profile = Mock( + container_insights=container_insights, + app_monitoring=Mock(open_telemetry_logs_and_traces=otlp), + ) + # Both addons are installed: disabling an addon that is not installed is rejected up front. + instance.addon_profiles = { + "omsagent": Mock(enabled=True, config={}), + "azurepolicy": Mock(enabled=True, config={}), + } + instance.location = "westus2" + return instance + + def _run(self, addons, instance, yes=False, answer=True): + from azext_aks_preview import custom + + client = Mock() + client.get.return_value = instance + with patch.object(custom, "get_subscription_id", return_value="sub"), patch.object( + custom, "_update_addons", side_effect=lambda *a, **k: instance + ), patch.object(custom, "sdk_no_wait") as put, patch.object( + custom, "prompt_y_n", return_value=answer + ) as prompt: + result = custom.aks_disable_addons( + cmd=self.cmd, + client=client, + resource_group_name="rg", + name="cluster", + addons=addons, + yes=yes, + ) + return result, put, prompt + + def test_container_insights_settings_are_reset_to_defaults(self): + # The RP only copies a containerInsights field when it is present on the request, so a + # stale value left behind here is silently inherited by the next onboarding. + instance = self._instance() + _, put, _ = self._run("monitoring", instance) + + ci = instance.azure_monitor_profile.container_insights + self.assertFalse(ci.enabled) + self.assertEqual(ci.syslog_port, CONST_CONTAINER_INSIGHTS_DEFAULT_SYSLOG_PORT) + self.assertFalse(ci.disable_prometheus_metrics_scraping) + self.assertEqual(ci.container_network_logs, CONST_CONTAINER_NETWORK_LOGS_DISABLED) + put.assert_called_once() + + def test_comma_separated_addon_list_still_disables_monitoring(self): + # An exact string compare skips the cleanup for 'monitoring,azure-policy'. + instance = self._instance() + self._run("monitoring,azure-policy", instance) + self.assertFalse(instance.azure_monitor_profile.container_insights.enabled) + + def test_unrelated_addon_leaves_container_insights_alone(self): + instance = self._instance() + self._run("azure-policy", instance) + self.assertTrue(instance.azure_monitor_profile.container_insights.enabled) + + def test_declining_the_opentelemetry_prompt_makes_no_changes(self): + # Declining must abort before the cluster is written, not after. + instance = self._instance(otlp_enabled=True) + result, put, prompt = self._run("monitoring", instance, answer=False) + + self.assertIsNone(result) + put.assert_not_called() + prompt.assert_called_once() + self.assertTrue(instance.azure_monitor_profile.container_insights.enabled) + + def test_accepting_the_prompt_also_turns_opentelemetry_off(self): + instance = self._instance(otlp_enabled=True) + _, put, prompt = self._run("monitoring", instance, answer=True) + + prompt.assert_called_once() + otlp = instance.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces + self.assertFalse(otlp.enabled) + self.assertIsNone(otlp.http_port) + self.assertIsNone(otlp.grpc_port) + put.assert_called_once() + + def test_yes_skips_the_prompt(self): + instance = self._instance(otlp_enabled=True) + _, put, prompt = self._run("monitoring", instance, yes=True) + + prompt.assert_not_called() + put.assert_called_once() + self.assertFalse( + instance.azure_monitor_profile.app_monitoring.open_telemetry_logs_and_traces.enabled + ) + + def test_no_prompt_when_opentelemetry_is_off(self): + instance = self._instance(otlp_enabled=False) + _, put, prompt = self._run("monitoring", instance) + + prompt.assert_not_called() + put.assert_called_once() + + def test_bad_addon_is_rejected_before_any_cleanup_runs(self): + """The whole point of validating up front is that nothing is deleted on the way to the error. + + Calling the validator directly proves it rejects the addon, but not that aks_disable_addons + actually calls it, nor that it does so before the destructive DCR association cleanup. + """ + from azext_aks_preview import custom + + instance = self._instance() + client = Mock() + client.get.return_value = instance + with patch.object(custom, "get_subscription_id", return_value="sub"), patch.object( + custom, "_update_addons" + ) as update_addons, patch.object(custom, "sdk_no_wait") as put, patch.object( + custom, "ensure_container_insights_for_monitoring" + ) as cleanup: + with self.assertRaises(CLIError): + custom.aks_disable_addons( + cmd=self.cmd, + client=client, + resource_group_name="rg", + name="cluster", + addons="monitoring,bogusaddon", + yes=True, + ) + + cleanup.assert_not_called() + update_addons.assert_not_called() + put.assert_not_called() + # The cluster is left exactly as it was found. + self.assertTrue(instance.azure_monitor_profile.container_insights.enabled) + + +class TestValidateAddonsForDisable(unittest.TestCase): + """`aks disable-addons` deletes the DCR association before the cluster PUT, so a bad addon + name has to be rejected before that cleanup runs rather than after it.""" + + def _instance(self, *installed): + instance = Mock() + instance.addon_profiles = {key: Mock(enabled=True) for key in installed} + return instance + + def test_unknown_addon_rejected(self): + with self.assertRaises(CLIError) as cm: + _validate_addons_for_disable(self._instance("omsagent"), "bogusaddon") + self.assertIn("Invalid addon name", str(cm.exception)) + + def test_not_installed_addon_rejected(self): + with self.assertRaises(CLIError) as cm: + _validate_addons_for_disable(self._instance("azurepolicy"), "monitoring") + self.assertIn("is not installed", str(cm.exception)) + + def test_installed_addon_accepted(self): + _validate_addons_for_disable(self._instance("omsagent", "azurepolicy"), "monitoring,azure-policy") + + def test_installed_check_is_case_insensitive(self): + # _update_addons normalizes the casing of the stored key before checking, and the addon + # key has been seen as 'omsAgent' in the wild. + _validate_addons_for_disable(self._instance("omsAgent"), "monitoring") + + def test_bad_addon_alongside_a_good_one_is_still_rejected(self): + # The whole list has to be validated, otherwise the monitoring cleanup runs and only then + # does _update_addons reject the rest, skipping the PUT. + with self.assertRaises(CLIError): + _validate_addons_for_disable(self._instance("omsagent"), "monitoring,bogusaddon") + + def test_kube_dashboard_exempt_from_installed_check(self): + # _update_addons synthesizes a disabled profile for it instead of failing. + _validate_addons_for_disable(self._instance("omsagent"), "kube-dashboard") + + def test_virtual_node_uses_the_os_suffixed_key(self): + _validate_addons_for_disable(self._instance("aciConnectorLinux"), "virtual-node") + with self.assertRaises(CLIError): + _validate_addons_for_disable(self._instance("aciConnector"), "virtual-node") + + def test_ingress_profile_addons_are_exempt(self): + # These two live in the ingress profile, not addon_profiles. _update_addons handles them + # before it validates anything, so rejecting them here would break disabling them. + instance = self._instance() + _validate_addons_for_disable(instance, "web_application_routing") + _validate_addons_for_disable(instance, "applicationloadbalancer") + + +class TestResolveDcrSettingsFromExisting(unittest.TestCase): + """Reconfiguring rebuilds the DCR from scratch, so settings not named on the command line have + to be read back off the existing DCR or they are silently dropped.""" + + @staticmethod + def _dcr(streams=None, syslog=False, data_collection_settings=None): + extension = {"extensionName": "ContainerInsights", "streams": streams or []} + if data_collection_settings is not None: + extension["extensionSettings"] = {"dataCollectionSettings": data_collection_settings} + data_sources = {"extensions": [{"extensionName": "Unrelated"}, extension]} + if syslog: + data_sources["syslog"] = [{"name": "sysLogsDataSource"}] + return {"properties": {"dataSources": data_sources}} + + def test_unspecified_settings_are_inherited(self): + dcr = self._dcr( + streams=["Microsoft-ContainerLogV2-HighScale"], + syslog=True, + data_collection_settings={"interval": "5m"}, + ) + syslog, settings, hlsm = _resolve_dcr_settings_from_existing(dcr, None, None, None) + self.assertTrue(syslog) + self.assertTrue(hlsm) + self.assertEqual(settings, {"interval": "5m"}) + + def test_explicit_false_beats_the_existing_dcr(self): + # An explicitly supplied value always wins, including a falsy one. + dcr = self._dcr(streams=["Microsoft-ContainerLogV2-HighScale"], syslog=True) + syslog, _, hlsm = _resolve_dcr_settings_from_existing(dcr, False, None, False) + self.assertFalse(syslog) + self.assertFalse(hlsm) + + def test_explicit_data_collection_settings_are_not_overridden(self): + dcr = self._dcr(data_collection_settings={"interval": "5m"}) + _, settings, _ = _resolve_dcr_settings_from_existing(dcr, None, "{}", None) + self.assertIsNone(settings) + + def test_absent_dcr_resolves_to_defaults(self): + for existing in ({}, None, {"properties": {}}): + with self.subTest(existing=existing): + syslog, settings, hlsm = _resolve_dcr_settings_from_existing( + existing, None, None, None + ) + self.assertFalse(syslog) + self.assertFalse(hlsm) + self.assertIsNone(settings) + + def test_preserving_is_opt_in(self): + """Every call site that does not ask for preservation must get a fresh onboarding. + + Most call sites omit the flag entirely, so a default of True would silently make them all + inherit whatever the leftover DCR happens to contain. + """ + import inspect + + from azext_aks_preview.custom import ensure_container_insights_for_monitoring_preview + + default = inspect.signature( + ensure_container_insights_for_monitoring_preview + ).parameters["preserve_existing_dcr_settings"].default + self.assertIs(default, False) + + def test_high_log_scale_mode_read_from_the_container_insights_extension_only(self): + # The high scale stream on some other extension must not be mistaken for this one. + dcr = { + "properties": { + "dataSources": { + "extensions": [ + { + "extensionName": "Unrelated", + "streams": ["Microsoft-ContainerLogV2-HighScale"], + }, + {"extensionName": "ContainerInsights", "streams": ["Microsoft-ContainerLogV2"]}, + ] + } + } + } + _, _, hlsm = _resolve_dcr_settings_from_existing(dcr, None, None, None) + self.assertFalse(hlsm) + + if __name__ == '__main__': unittest.main() diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py index e6797155700..64db488cc5a 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py @@ -9895,6 +9895,31 @@ def test_postprocessing_create_dcr_true_when_enable_azure_monitor_logs(self): _, kwargs = mock_ecifm.call_args self.assertTrue(kwargs["create_dcr"]) + def test_postprocessing_amp_onboarding_skips_monitoring_role_assignment(self): + """--enable-azure-monitor-logs must never attempt the Monitoring Metrics Publisher grant. + + The grant only applies to legacy service-principal/addon-MSI auth. AMP onboarding is + always AAD-authenticated, and such a cluster may come back with no addon profiles at + all, so the call site is guarded by `if not aad_route`. Core hardened the body of + add_monitoring_role_assignment against that shape; here the guard is what keeps it + unreachable, so the guard itself is what has to be pinned. + """ + dec = self._make_postprocessing_decorator({"enable_azure_monitor_logs": True}) + cluster = self.models.ManagedCluster(location="test_location") + self.assertIsNone(cluster.addon_profiles) + external_functions = dec.context.external_functions + with patch.object( + external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ), patch.object( + external_functions, + "add_monitoring_role_assignment", + return_value=None, + ) as mock_role: + dec.postprocessing_after_mc_created(cluster) + mock_role.assert_not_called() + # ------------------------------------------------------------------ # Tests for _should_create_dcra and _is_cnl_or_hlsm_changing helpers # ------------------------------------------------------------------ @@ -19239,7 +19264,13 @@ def test_setup_azure_monitor_logs_sets_retina_flags_when_cnl_enabled(self): self.assertNotIn(CONST_MONITORING_ADDON_NAME, mc.addon_profiles or {}) def test_setup_azure_monitor_logs_no_retina_flags_without_cnl(self): - """_setup_azure_monitor_logs leaves containerNetworkLogs unset when CNL is not specified.""" + """_setup_azure_monitor_logs disables containerNetworkLogs when CNL is not specified. + + Leaving the field unset is not equivalent: the RP backfills an unspecified + containerNetworkLogs from the omsagent addon config, so a stale 'retinaNetworkFlowLogs=true' + left behind by a previous onboarding would turn CNL back on. Writing the default + explicitly keeps it off. + """ dec = AKSPreviewManagedClusterUpdateDecorator( self.cmd, self.client, @@ -19270,7 +19301,9 @@ def test_setup_azure_monitor_logs_no_retina_flags_without_cnl(self): container_insights = mc.azure_monitor_profile.container_insights self.assertIsNotNone(container_insights) self.assertTrue(container_insights.enabled) - self.assertIsNone(container_insights.container_network_logs) + self.assertEqual( + container_insights.container_network_logs, CONST_CONTAINER_NETWORK_LOGS_DISABLED + ) # ------------------------------------------------------------------ # Tests for _setup_azure_monitor_logs workspace change detection @@ -20890,6 +20923,26 @@ def _write_settings(self, payload): self.addCleanup(os.unlink, handle.name) return handle.name + def test_check_raw_parameters_explicit_false_three_state_flags(self): + # three-state flags explicitly set to false are a real update request, + # they must not be mistaken for "no argument specified". `False` is falsy, so the + # generic `any(...)` scan cannot see it; the flag has to be declared in + # get_special_parameter_default_value_pairs_list to be noticed. + for param in ("enable_syslog", "enable_high_log_scale_mode"): + with self.subTest(param=param): + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {param: False}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + with patch( + "azext_aks_preview.managed_cluster_decorator.prompt_y_n", + return_value=False, + ) as prompt: + dec.check_raw_parameters() + prompt.assert_not_called() + def test_data_collection_settings_size_is_measured_on_the_contents(self): # The getter returns the file path, so a check against the path length can never fire. # An oversized file has to be rejected on the size of what gets serialized into the DCR. @@ -21132,6 +21185,190 @@ def test_a_port_only_update_reaches_the_cluster(self): self.assertTrue(app_monitoring.open_telemetry_metrics.enabled) self.assertTrue(app_monitoring.open_telemetry_logs_and_traces.enabled) + def test_re_enabling_starts_from_defaults_not_the_previous_onboarding(self): + """Regression test: a re-onboard must not inherit the previous configuration. + + The RP copies a containerInsights field onto the cluster only when the field is present on + the request, so any value left behind by an earlier onboarding survives a disable/enable + cycle unless the enable path writes the defaults explicitly. + """ + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_logs": True, + "workspace_resource_id": "/subscriptions/s/resourceGroups/rg/providers/" + "Microsoft.OperationalInsights/workspaces/ws", + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + # A disabled cluster still carrying the settings of its previous onboarding. + mc = self.models.ManagedCluster( + location="test_location", + addon_profiles={ + "omsagent": self.models.ManagedClusterAddonProfile( + enabled=False, config={CONST_MONITORING_USING_AAD_MSI_AUTH: "true"} + ) + }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + enabled=False, + syslog_port=51400, + disable_prometheus_metrics_scraping=True, + container_network_logs="Enabled", + ) + ), + ) + dec.context.attach_mc(mc) + + with patch.object(dec, "_provision_azure_monitor_logs_dcr"), patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda ws: ws, + ): + dec._setup_azure_monitor_logs(mc) + + container_insights = mc.azure_monitor_profile.container_insights + self.assertTrue(container_insights.enabled) + self.assertEqual( + container_insights.syslog_port, CONST_CONTAINER_INSIGHTS_DEFAULT_SYSLOG_PORT + ) + self.assertFalse(container_insights.disable_prometheus_metrics_scraping) + self.assertEqual( + container_insights.container_network_logs, CONST_CONTAINER_NETWORK_LOGS_DISABLED + ) + + def test_re_enabling_keeps_settings_asked_for_in_the_same_command(self): + """Starting from the defaults must not discard values supplied alongside the enable.""" + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_logs": True, + "workspace_resource_id": "/subscriptions/s/resourceGroups/rg/providers/" + "Microsoft.OperationalInsights/workspaces/ws", + "enable_syslog": True, + "syslog_port": 2833, + "disable_prometheus_metrics_scraping": True, + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster( + location="test_location", + addon_profiles={ + "omsagent": self.models.ManagedClusterAddonProfile( + enabled=False, config={CONST_MONITORING_USING_AAD_MSI_AUTH: "true"} + ) + }, + azure_monitor_profile=self.models.ManagedClusterAzureMonitorProfile( + container_insights=self.models.ManagedClusterAzureMonitorProfileContainerInsights( + enabled=False, syslog_port=51400 + ) + ), + ) + dec.context.attach_mc(mc) + + with patch.object(dec, "_provision_azure_monitor_logs_dcr"), patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda ws: ws, + ): + dec._setup_azure_monitor_logs(mc) + # The real pipeline applies these after the enable, so the reset must not win. + dec.update_azure_monitor_logs_settings(mc) + + container_insights = mc.azure_monitor_profile.container_insights + self.assertTrue(container_insights.enabled) + self.assertEqual(container_insights.syslog_port, 2833) + self.assertTrue(container_insights.disable_prometheus_metrics_scraping) + + def test_enable_azure_monitor_logs_does_not_inherit_leftover_dcr_settings(self): + """Re-enabling is a fresh onboarding, so the leftover DCR must not be carried over. + + Disabling monitoring removes the association but leaves the DCR itself behind. If this + path preserved its settings, '--enable-azure-monitor-logs' would silently come back with + the previous onboarding's syslog, high log scale mode and custom data collection settings, + instead of the documented defaults the container insights profile is reset to. + """ + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + { + "enable_azure_monitor_logs": True, + "workspace_resource_id": "/subscriptions/s/resourceGroups/rg/providers/" + "Microsoft.OperationalInsights/workspaces/ws", + }, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster( + location="test_location", + addon_profiles={ + "omsagent": self.models.ManagedClusterAddonProfile( + enabled=False, config={CONST_MONITORING_USING_AAD_MSI_AUTH: "true"} + ) + }, + ) + dec.context.attach_mc(mc) + dec.context.set_intermediate("subscription_id", "test-subscription-id") + + with patch.object( + dec.context.external_functions, + "sanitize_loganalytics_ws_resource_id", + side_effect=lambda ws: ws, + ), patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_mock: + dec._setup_azure_monitor_logs(mc) + + ensure_mock.assert_called_once() + self.assertFalse( + ensure_mock.call_args.kwargs.get("preserve_existing_dcr_settings", False) + ) + + def test_reconfiguring_an_onboarded_cluster_preserves_existing_dcr_settings(self): + """The postprocessing path only changes what was named on the command line. + + '--enable-high-log-scale-mode' on its own rebuilds the DCR, so without this the rebuild + would drop the cluster's syslog and custom data collection settings. + """ + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, + self.client, + {"enable_high_log_scale_mode": True}, + CUSTOM_MGMT_AKS_PREVIEW, + ) + mc = self.models.ManagedCluster( + location="test_location", + addon_profiles={ + "omsagent": self.models.ManagedClusterAddonProfile( + enabled=True, + config={ + CONST_MONITORING_USING_AAD_MSI_AUTH: "true", + CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID: "/ws", + }, + ) + }, + ) + dec.context.attach_mc(mc) + dec.context.set_intermediate("subscription_id", "test-subscription-id") + dec.context.set_intermediate( + "monitoring_addon_postprocessing_required", True, overwrite_exists=True + ) + + with patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as ensure_mock, patch.object( + dec.context.external_functions, "add_monitoring_role_assignment", return_value=None + ): + dec.postprocessing_after_mc_created(mc) + + ensure_mock.assert_called_once() + self.assertTrue(ensure_mock.call_args.kwargs["preserve_existing_dcr_settings"]) + if __name__ == "__main__": unittest.main() diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py b/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py index acc17e1cdbb..1ebe7d8a637 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_validators.py @@ -2877,5 +2877,100 @@ def test_conflicts_with_disable_azure_monitor_logs(self): self.assertIn("--disable-azure-monitor-logs", str(cm.exception)) +class TestOpenTelemetryPortsNotDisabled(unittest.TestCase): + """A port flag naming a receiver that is being disabled must be rejected at validation time. + + The decorator's port getters catch this too, but only after the Azure Monitor collection + resources have already been deleted, so the command fails with the cluster half torn down. + """ + + PORT_DISABLE_COMBOS = [ + ("opentelemetry_metrics_port", "--opentelemetry-metrics-port-http", + "disable_azure_monitor_metrics", "--disable-azure-monitor-metrics"), + ("opentelemetry_metrics_port", "--opentelemetry-metrics-port-http", + "disable_opentelemetry_metrics", "--disable-opentelemetry-metrics"), + ("opentelemetry_metrics_port_grpc", "--opentelemetry-metrics-port-grpc", + "disable_azure_monitor_metrics", "--disable-azure-monitor-metrics"), + ("opentelemetry_metrics_port_grpc", "--opentelemetry-metrics-port-grpc", + "disable_opentelemetry_metrics", "--disable-opentelemetry-metrics"), + ("opentelemetry_logs_port", "--opentelemetry-logs-traces-port-http", + "disable_azure_monitor_logs", "--disable-azure-monitor-logs"), + ("opentelemetry_logs_port", "--opentelemetry-logs-traces-port-http", + "disable_opentelemetry_logs", "--disable-opentelemetry-logs-traces"), + ("opentelemetry_logs_traces_port_grpc", "--opentelemetry-logs-traces-port-grpc", + "disable_azure_monitor_logs", "--disable-azure-monitor-logs"), + ("opentelemetry_logs_traces_port_grpc", "--opentelemetry-logs-traces-port-grpc", + "disable_opentelemetry_logs", "--disable-opentelemetry-logs-traces"), + ] + + @staticmethod + def _namespace(**kwargs): + defaults = { + "opentelemetry_metrics_port": None, + "opentelemetry_metrics_port_grpc": None, + "opentelemetry_logs_port": None, + "opentelemetry_logs_traces_port_grpc": None, + "disable_azure_monitor_metrics": False, + "disable_azure_monitor_logs": False, + "disable_opentelemetry_metrics": False, + "disable_opentelemetry_logs": False, + "enable_azure_monitor_metrics": False, + "enable_azure_monitor_logs": False, + "enable_opentelemetry_metrics": False, + "enable_opentelemetry_logs": False, + "enable_addons": None, + } + defaults.update(kwargs) + return SimpleNamespace(**defaults) + + def test_port_with_matching_disable_rejected(self): + for port_attr, port_flag, disable_attr, disable_flag in self.PORT_DISABLE_COMBOS: + with self.subTest(port=port_flag, disable=disable_flag): + namespace = self._namespace(**{port_attr: 4318, disable_attr: True}) + with self.assertRaises(InvalidArgumentValueError) as cm: + validators.validate_opentelemetry_ports_not_disabled(namespace) + self.assertIn(port_flag, str(cm.exception)) + self.assertIn(disable_flag, str(cm.exception)) + + def test_port_with_unrelated_disable_allowed(self): + # Disabling metrics must not reject a logs/traces port, and vice versa. + namespace = self._namespace( + opentelemetry_logs_port=4318, + disable_azure_monitor_metrics=True, + disable_opentelemetry_metrics=True, + ) + validators.validate_opentelemetry_ports_not_disabled(namespace) + + namespace = self._namespace( + opentelemetry_metrics_port=4318, + disable_azure_monitor_logs=True, + disable_opentelemetry_logs=True, + ) + validators.validate_opentelemetry_ports_not_disabled(namespace) + + def test_disable_without_ports_allowed(self): + namespace = self._namespace( + disable_azure_monitor_metrics=True, + disable_azure_monitor_logs=True, + disable_opentelemetry_metrics=True, + disable_opentelemetry_logs=True, + ) + validators.validate_opentelemetry_ports_not_disabled(namespace) + + def test_port_disable_conflict_rejected_by_aggregate_validators(self): + # The aggregate validators are the entry points the commands register. The conflict has to + # be reachable through them, otherwise it is only caught in the decorator's port getters, + # which run after the Azure Monitor collection resources have already been deleted. + for aggregate in ( + validators.validate_azure_monitor_and_opentelemetry_for_create, + validators.validate_azure_monitor_and_opentelemetry_for_update, + ): + for port_attr, port_flag, disable_attr, disable_flag in self.PORT_DISABLE_COMBOS: + with self.subTest(aggregate=aggregate.__name__, port=port_flag, disable=disable_flag): + namespace = self._namespace(**{port_attr: 4318, disable_attr: True}) + with self.assertRaises(InvalidArgumentValueError): + aggregate(namespace) + + if __name__ == "__main__": unittest.main() From 4b10020eafdf2619938d23d39312318e47ab3746 Mon Sep 17 00:00:00 2001 From: Sunil Yadav Date: Wed, 23 Sep 2026 01:31:31 +0000 Subject: [PATCH 7/8] ampls security fixes Port of the AMPLS fixes made in azure-cli core. The aks-preview extension shadows the core acs module, so fixes made there do not reach anyone with the extension installed unless they are ported here too. * The ingestion data collection endpoint is created with create_or_update on every reconfiguration, and is_ampls is derived purely from whether --ampls-resource-id was supplied on the current command. Since that flag is only passed on the command that links the scope, an unrelated update such as 'az aks update --enable-syslog' arrived with is_ampls False and flipped an existing private endpoint back to publicNetworkAccess Enabled. The existing network configuration is now read back and preserved unless the caller explicitly asks to change it. * ensure_container_insights_for_monitoring rejects an AMPLS resource id unless is_private_cluster is true, but get_enable_private_cluster() only falls back to the ManagedCluster in create mode. In update mode it returns the command line flag, so 'az aks update --ampls-resource-id ' on an already-private cluster reached the guard with None and failed. The private state is now read from the cluster as well, at both update call sites. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 0d309873-2a9e-46f6-8472-9db05f31137b --- src/aks-preview/HISTORY.rst | 2 + src/aks-preview/azext_aks_preview/custom.py | 70 +++++++- .../managed_cluster_decorator.py | 25 ++- .../tests/latest/test_custom.py | 82 +++++++++ .../latest/test_managed_cluster_decorator.py | 169 ++++++++++++++++++ 5 files changed, 345 insertions(+), 3 deletions(-) diff --git a/src/aks-preview/HISTORY.rst b/src/aks-preview/HISTORY.rst index 9010e409b86..3970a543d1b 100644 --- a/src/aks-preview/HISTORY.rst +++ b/src/aks-preview/HISTORY.rst @@ -38,6 +38,8 @@ Pending * Fix `--enable-high-log-scale-mode` mutating the shared list of Container Insights streams, so the stream set leaked between data collection rules built in the same command invocation. * `az aks create` and `az aks update`: Fix the `--data-collection-settings` size limit being applied to the file path instead of the settings it holds, which let an oversized file through to fail the data collection rule call with `Request Header Fields Too Large`. * `az aks update`: Fix `--enable-syslog false` being rejected with `Please specify one or more of "--enable-syslog"` after prompting to reconcile the cluster. Explicitly turning syslog collection off is a real update request, but the falsy value made the command treat it as though no argument had been supplied. +* `az aks update`: Stop reopening public network access on the ingestion data collection endpoint of a cluster that is linked to an Azure Monitor Private Link Scope. The endpoint is created or updated on every reconfiguration, but `--ampls-resource-id` is only supplied on the command that links the scope, so an unrelated update such as `--enable-syslog` used to flip an existing private endpoint back to public. The existing network configuration is now preserved unless the caller explicitly asks to change it. +* `az aks update`: Fix `--ampls-resource-id` being rejected with `--ampls-resource-id can only be used with private cluster in MSI mode.` on a cluster that is already private. The private state was only read from the command line, so it was invisible unless `--enable-private-cluster` happened to be supplied again in the same command; it is now read from the cluster as well. 22.0.0b6 +++++++++ diff --git a/src/aks-preview/azext_aks_preview/custom.py b/src/aks-preview/azext_aks_preview/custom.py index accd0e17fe6..50a4b4d036e 100644 --- a/src/aks-preview/azext_aks_preview/custom.py +++ b/src/aks-preview/azext_aks_preview/custom.py @@ -160,7 +160,6 @@ ensure_default_log_analytics_workspace_for_monitoring, sanitize_loganalytics_ws_resource_id, validate_data_collection_settings, - create_data_collection_endpoint, create_or_delete_dcr_association, create_dce_association, create_ampls_scope, @@ -303,6 +302,75 @@ def get_existing_container_insights_extension_dcr(cmd, dcr_url): return {} +def get_existing_data_collection_endpoint(cmd, dce_resource_id): + """Fetch the data collection endpoint that already exists, or {} if there is none.""" + dce_url = cmd.cli_ctx.cloud.endpoints.resource_manager + \ + f"{dce_resource_id}?api-version=2022-06-01" + _MAX_RETRY_TIMES = 3 + for retry_count in range(0, _MAX_RETRY_TIMES): + try: + resp = send_raw_request(cmd.cli_ctx, "GET", dce_url) + return json.loads(resp.text) + except CLIError as e: + if "ResourceNotFound" in str(e): + break + if retry_count >= (_MAX_RETRY_TIMES - 1): + raise e + return {} + + +def create_data_collection_endpoint(cmd, subscription, resource_group, region, endpoint_name, is_ampls): + """Create or update a data collection endpoint. + + Defined here rather than imported from the acs module so that the network access fix below + reaches anyone running this extension, whose azure-cli core may predate it. + """ + dce_resource_id = ( + f"/subscriptions/{subscription}/resourceGroups/{resource_group}/" + f"providers/Microsoft.Insights/dataCollectionEndpoints/{endpoint_name}" + ) + public_network_access = "Disabled" if is_ampls else "Enabled" + if not is_ampls: + # This is a create_or_update, and an endpoint that is already private has to stay private. + # --ampls-resource-id is only supplied on the command that links the scope, so any later + # reconfiguration of an onboarded cluster ('az aks update --enable-syslog', for example) + # arrives here with is_ampls False and would otherwise reopen public network access on an + # existing private ingestion endpoint. The network configuration is only changed when the + # caller explicitly asks for it. + existing_dce = get_existing_data_collection_endpoint(cmd, dce_resource_id) + existing_network_acls = (existing_dce.get("properties") or {}).get("networkAcls") or {} + existing_public_network_access = existing_network_acls.get("publicNetworkAccess") + if existing_public_network_access: + public_network_access = existing_public_network_access + + # create the DCE + dce_creation_body_common = { + "location": region, + "kind": "Linux", + "properties": { + "networkAcls": { + "publicNetworkAccess": public_network_access + } + } + } + dce_creation_body_ = json.dumps(dce_creation_body_common) + resources = get_resources_client(cmd.cli_ctx, subscription) + for _ in range(3): + try: + resources.begin_create_or_update_by_id( + dce_resource_id, + "2022-06-01", + json.loads(dce_creation_body_) + ) + error = None + break + except CLIError as e: + error = e + else: + raise error + return dce_resource_id + + def _resolve_dcr_settings_from_existing( existing_dcr, enable_syslog, data_collection_settings, enable_high_log_scale_mode ): diff --git a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py index 666cb3aeb4c..d780f7651b7 100644 --- a/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/managed_cluster_decorator.py @@ -325,6 +325,22 @@ def _is_opentelemetry_metrics_enabled(mc): ) +def _is_private_cluster_on_mc(mc): + """Whether the cluster is private according to its own state. + + The AMPLS provisioning guard needs to know whether the cluster is private, but + get_enable_private_cluster() only falls back to the ManagedCluster in create mode; in update + mode it returns the command line flag, which is None unless --enable-private-cluster was + passed again. Callers combine this with that getter so that 'az aks update + --ampls-resource-id ' works on a cluster that is already private. + """ + return bool( + mc and + mc.api_server_access_profile and + mc.api_server_access_profile.enable_private_cluster + ) + + def _is_opentelemetry_logs_traces_enabled(mc): """Whether the OpenTelemetry logs and traces receiver is enabled on the cluster.""" return bool( @@ -9177,7 +9193,9 @@ def _provision_azure_monitor_logs_dcr(self, mc: ManagedCluster, addon_consts: di create_dcra=True, enable_syslog=self.context.get_enable_syslog(), data_collection_settings=data_collection_settings, - is_private_cluster=self.context.get_enable_private_cluster(), + is_private_cluster=( + self.context.get_enable_private_cluster() or _is_private_cluster_on_mc(mc) + ), ampls_resource_id=self.context.get_ampls_resource_id(), enable_high_log_scale_mode=self.context.get_enable_high_log_scale_mode(), ) @@ -9646,7 +9664,10 @@ def postprocessing_after_mc_created(self, cluster: ManagedCluster) -> None: create_dcra=True, enable_syslog=self.context.get_enable_syslog(), data_collection_settings=data_collection_settings, - is_private_cluster=self.context.get_enable_private_cluster(), + is_private_cluster=( + self.context.get_enable_private_cluster() or + _is_private_cluster_on_mc(cluster) + ), ampls_resource_id=self.context.get_ampls_resource_id(), enable_high_log_scale_mode=self.context.get_enable_high_log_scale_mode(), # This is the reconfigure path: the cluster is already onboarded and only diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py index 976512c147b..41192c972f9 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_custom.py @@ -2,6 +2,7 @@ # Copyright (c) Microsoft Corporation. All rights reserved. # Licensed under the MIT License. See License.txt in the project root for license information. # -------------------------------------------------------------------------------------------- +import json import unittest from unittest.mock import Mock, patch @@ -29,6 +30,7 @@ aks_list_vm_skus, _validate_addons_for_disable, _resolve_dcr_settings_from_existing, + create_data_collection_endpoint, ) from azext_aks_preview.tests.latest.mocks import MockCLI, MockClient, MockCmd from azure.cli.command_modules.acs._consts import AgentPoolDecoratorMode @@ -1368,5 +1370,85 @@ def test_high_log_scale_mode_read_from_the_container_insights_extension_only(sel self.assertFalse(hlsm) +class TestCreateDataCollectionEndpointNetworkAccess(unittest.TestCase): + """The ingestion DCE must not have its network configuration reopened by an unrelated update. + + create_data_collection_endpoint is a create_or_update, and is_ampls is derived purely from + whether --ampls-resource-id was supplied on the current command. Since that flag is only + passed on the command that links the scope, every later reconfiguration of an onboarded + cluster ('az aks update --enable-syslog', for example) arrives with is_ampls False. Without + the read-back below that would flip an existing private endpoint to publicNetworkAccess + Enabled, silently exposing it. + """ + + DCE_ID = ( + "/subscriptions/sub-id/resourceGroups/rg" + "/providers/Microsoft.Insights/dataCollectionEndpoints/MSCI-ingest-eastus-cluster" + ) + + def _run(self, existing_access, is_ampls=False): + """Drive the real function and return (written body, GET mock).""" + cmd = Mock() + cmd.cli_ctx.cloud.endpoints.resource_manager = "https://management.azure.com" + + def fake_send_raw_request(cli_ctx, method, url, **_): + self.assertEqual(method, "GET") + self.assertIn("dataCollectionEndpoints", url) + if existing_access is None: + raise CLIError("ResourceNotFound: no such DCE") + resp = Mock() + resp.text = json.dumps( + {"properties": {"networkAcls": {"publicNetworkAccess": existing_access}}} + ) + return resp + + resources = Mock() + base = "azext_aks_preview.custom." + with patch( + base + "send_raw_request", side_effect=fake_send_raw_request + ) as mock_get, patch(base + "get_resources_client", return_value=resources): + returned_id = create_data_collection_endpoint( + cmd, "sub-id", "rg", "eastus", "MSCI-ingest-eastus-cluster", is_ampls + ) + + self.assertEqual(returned_id, self.DCE_ID) + resources.begin_create_or_update_by_id.assert_called_once() + body = resources.begin_create_or_update_by_id.call_args[0][2] + return body, mock_get + + @staticmethod + def _access(body): + return body["properties"]["networkAcls"]["publicNetworkAccess"] + + def test_existing_private_endpoint_is_not_reopened(self): + # the regression: a syslog-only update must leave an AMPLS endpoint private + body, _ = self._run("Disabled", is_ampls=False) + self.assertEqual(self._access(body), "Disabled") + + def test_existing_public_endpoint_stays_public(self): + body, _ = self._run("Enabled", is_ampls=False) + self.assertEqual(self._access(body), "Enabled") + + def test_missing_endpoint_defaults_to_public(self): + # first onboarding without AMPLS: nothing to preserve, keep the documented default + body, _ = self._run(None, is_ampls=False) + self.assertEqual(self._access(body), "Enabled") + + def test_unrecognised_access_value_is_preserved_verbatim(self): + body, _ = self._run("SecuredByPerimeter", is_ampls=False) + self.assertEqual(self._access(body), "SecuredByPerimeter") + + def test_ampls_forces_private_without_reading_the_existing_endpoint(self): + # an explicit --ampls-resource-id is the one case that may change the configuration + body, mock_get = self._run("Enabled", is_ampls=True) + self.assertEqual(self._access(body), "Disabled") + mock_get.assert_not_called() + + def test_location_and_kind_are_unchanged(self): + body, _ = self._run("Disabled", is_ampls=False) + self.assertEqual(body["location"], "eastus") + self.assertEqual(body["kind"], "Linux") + + if __name__ == '__main__': unittest.main() diff --git a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py index 64db488cc5a..335622e5030 100644 --- a/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py +++ b/src/aks-preview/azext_aks_preview/tests/latest/test_managed_cluster_decorator.py @@ -79,6 +79,7 @@ _get_monitoring_addon_key_from_consts, _build_monitoring_addon_shim, _is_monitoring_aad_auth, + _is_private_cluster_on_mc, ) from azext_aks_preview.tests.latest.utils import get_test_data_file_path from azure.cli.command_modules.acs._consts import ( @@ -21370,5 +21371,173 @@ def test_reconfiguring_an_onboarded_cluster_preserves_existing_dcr_settings(self self.assertTrue(ensure_mock.call_args.kwargs["preserve_existing_dcr_settings"]) +class AKSPreviewMonitoringAmplsPrivateClusterTestCase(unittest.TestCase): + """--ampls-resource-id has to work on a cluster that is already private. + + ensure_container_insights_for_monitoring rejects an AMPLS resource id unless is_private_cluster + is true, but get_enable_private_cluster() only falls back to the ManagedCluster in create mode. + In update mode it returns the command line flag, so 'az aks update --ampls-resource-id ' on + an already-private cluster reached the guard with None and failed. The private state is now + taken from the cluster as well. + """ + + def setUp(self): + register_aks_preview_resource_type() + self.cli_ctx = MockCLI() + self.cmd = MockCmd(self.cli_ctx) + self.models = AKSPreviewManagedClusterModels(self.cmd, CUSTOM_MGMT_AKS_PREVIEW) + self.client = MockClient() + + def _private_cluster(self, enable_private_cluster): + monitoring_addon_profile = self.models.ManagedClusterAddonProfile( + enabled=True, + config={CONST_MONITORING_USING_AAD_MSI_AUTH: "true"}, + ) + return self.models.ManagedCluster( + location="test_location", + addon_profiles={CONST_MONITORING_ADDON_NAME: monitoring_addon_profile}, + api_server_access_profile=self.models.ManagedClusterAPIServerAccessProfile( + enable_private_cluster=enable_private_cluster, + ), + ) + + def _update_decorator(self, raw_params): + params = { + "resource_group_name": "test_rg_name", + "name": "test_name", + } + params.update(raw_params) + dec = AKSPreviewManagedClusterUpdateDecorator( + self.cmd, self.client, params, CUSTOM_MGMT_AKS_PREVIEW + ) + dec.context.set_intermediate("subscription_id", "test_subscription_id") + return dec + + def _postprocess(self, mc, raw_params): + """Drive the real postprocessing and return the is_private_cluster it provisioned with. + + Only monitoring_addon_postprocessing_required is set, which is what update_azure_monitor_profile + does for an --ampls-resource-id update. monitoring_addon_enabled is deliberately left unset: + it gates the core base class's own monitoring block, which would otherwise run its (unfixed) + call as well. + """ + params = {"enable_msi_auth_for_monitoring": True} + params.update(raw_params) + dec = self._update_decorator(params) + dec.context.attach_mc(mc) + dec.context.set_intermediate("monitoring_addon_postprocessing_required", True) + with patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as mock_ensure: + dec.postprocessing_after_mc_created(mc) + mock_ensure.assert_called_once() + return mock_ensure.call_args[1] + + def test_helper_reads_private_state_from_the_cluster(self): + self.assertTrue(_is_private_cluster_on_mc(self._private_cluster(True))) + self.assertFalse(_is_private_cluster_on_mc(self._private_cluster(False))) + self.assertFalse(_is_private_cluster_on_mc(self._private_cluster(None))) + + def test_helper_tolerates_missing_profile_and_cluster(self): + self.assertFalse( + _is_private_cluster_on_mc(self.models.ManagedCluster(location="test_location")) + ) + self.assertFalse(_is_private_cluster_on_mc(None)) + + def test_ampls_update_on_already_private_cluster(self): + # the regression: --ampls-resource-id alone, no --enable-private-cluster re-supplied + kwargs = self._postprocess( + self._private_cluster(True), {"ampls_resource_id": "/subscriptions/s/ampls/a"} + ) + self.assertTrue(kwargs["is_private_cluster"]) + self.assertEqual(kwargs["ampls_resource_id"], "/subscriptions/s/ampls/a") + + def test_public_cluster_is_still_reported_as_public(self): + # the guard in ensure_container_insights_for_monitoring must still reject this case + kwargs = self._postprocess( + self._private_cluster(False), {"ampls_resource_id": "/subscriptions/s/ampls/a"} + ) + self.assertFalse(kwargs["is_private_cluster"]) + + def test_command_line_flag_is_still_honoured(self): + # --enable-private-cluster on update additionally requires apiserver vnet integration + kwargs = self._postprocess( + self._private_cluster(False), + { + "ampls_resource_id": "/subscriptions/s/ampls/a", + "enable_private_cluster": True, + "enable_apiserver_vnet_integration": True, + }, + ) + self.assertTrue(kwargs["is_private_cluster"]) + + def _provision(self, enable_private_cluster): + """Drive the up-front DCR provisioning done by --enable-azure-monitor-logs.""" + dec = self._update_decorator( + { + "enable_azure_monitor_logs": True, + "ampls_resource_id": "/subscriptions/s/ampls/a", + } + ) + mc = self._private_cluster(enable_private_cluster) + mc.addon_profiles[CONST_MONITORING_ADDON_NAME].config[ + CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID + ] = "/subscriptions/s/workspaces/w" + dec.context.attach_mc(mc) + with patch.object( + dec.context.external_functions, + "ensure_container_insights_for_monitoring", + return_value=None, + ) as mock_ensure: + dec._provision_azure_monitor_logs_dcr(mc, dec.context.get_addon_consts()) + mock_ensure.assert_called_once() + return mock_ensure.call_args[1] + + def test_provisioning_reads_private_state_from_the_cluster(self): + self.assertTrue(self._provision(True)["is_private_cluster"]) + self.assertFalse(self._provision(False)["is_private_cluster"]) + + def _run_real_guard(self, enable_private_cluster): + """Drive the real ensure_container_insights_for_monitoring, AMPLS guard included. + + Mocking the provisioning call shows which value was passed; running the real function + shows that the value is actually accepted. Execution is cut short at the credential + lookup, which is the first thing past the guard that would reach ARM. + """ + reached_workspace_lookup = RuntimeError("reached the workspace lookup") + mc = self._private_cluster(enable_private_cluster) + mc.addon_profiles[CONST_MONITORING_ADDON_NAME].config[ + CONST_MONITORING_LOG_ANALYTICS_WORKSPACE_RESOURCE_ID + ] = ( + "/subscriptions/1234-5678-9012/resourceGroups/rg/providers" + "/microsoft.operationalinsights/workspaces/test_workspace" + ) + dec = self._update_decorator( + { + "enable_msi_auth_for_monitoring": True, + "ampls_resource_id": "/subscriptions/s/ampls/a", + } + ) + dec.context.attach_mc(mc) + dec.context.set_intermediate("monitoring_addon_postprocessing_required", True) + with patch( + "azure.cli.core._profile.Profile", + side_effect=reached_workspace_lookup, + ): + dec.postprocessing_after_mc_created(mc) + + def test_real_guard_accepts_an_already_private_cluster(self): + # pre-fix this raised ArgumentUsageError rather than getting past the guard at all + with self.assertRaises(RuntimeError) as ctx: + self._run_real_guard(True) + self.assertIn("reached the workspace lookup", str(ctx.exception)) + + def test_real_guard_still_rejects_a_public_cluster(self): + with self.assertRaises(ArgumentUsageError): + self._run_real_guard(False) + + if __name__ == "__main__": unittest.main() From 74ab160554012c6fd015a2f9192c458bf8215a3d Mon Sep 17 00:00:00 2001 From: Sunil Yadav Date: Wed, 23 Sep 2026 01:39:01 +0000 Subject: [PATCH 8/8] Move this PR's changelog entries back into Pending after the merge The merge with main resolved HISTORY.rst cleanly but silently filed all 22 of this PR's entries under the already-released 22.0.0b7 heading, because main released b7 and b8 while this branch was open. The 22.0.0b7 and 22.0.0b8 sections now match main exactly and every entry belonging to this PR is back under Pending. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 0d309873-2a9e-46f6-8472-9db05f31137b --- src/aks-preview/HISTORY.rst | 31 ++++++++++++++----------------- 1 file changed, 14 insertions(+), 17 deletions(-) diff --git a/src/aks-preview/HISTORY.rst b/src/aks-preview/HISTORY.rst index 613d79527d6..864585da9ab 100644 --- a/src/aks-preview/HISTORY.rst +++ b/src/aks-preview/HISTORY.rst @@ -14,20 +14,6 @@ Pending * `az aks nodepool update`: Preserve the existing GPU management mode when `--enable-managed-gpu` is omitted, including when enabling, updating, or disabling the cluster autoscaler. * `az aks alert-config add`: Reject an empty `--name` before looking up existing configurations instead of reporting that it already exists. * `az aks nodepool scale`: add `--use-patch-api` to optionally scale a VMSS node pool via the new dedicated PATCH agent pool API (scales to the target count without triggering full reconciliation). The default behavior continues to use the PUT agent pool API. - -22.0.0b8 -+++++++++ -* `az aks nodepool add`: Omit `nodeTaints` when `--node-taints` is not specified for FlexNodes pools, and reject explicitly empty values. - -22.0.0b7 -+++++++++ -* `az aks machine add`: Add preview `--capacity-reservation-group` support to associate a machine with a Capacity Reservation Group. -* Add `az aks alert-config` commands to manage AKS-managed alert configurations. -* `az aks create`: Honor `--enable-osdisk-full-caching` for the default agent pool. -* `az aks kollect` and `az aks kanalyze`: Fix compatibility with the keyword-only credential SDK parameters. -* `az aks maintenanceconfiguration add` and `az aks maintenanceconfiguration update`: Preserve configuration-file fields with the typespec-generated SDK model. -* Improve AKS live-test resilience for preview feature gates, transient resource and monitoring-table readiness, retired configurations, and service propagation delays. -* `az aks create` and `az aks update`: Reject `--outbound-type managedNATGatewayV2` with an actionable error directing to `--outbound-type managedNATGateway --outbound-type-sku StandardV2` (the GA-aligned shape); the legacy value is no longer accepted on the target api-version. * `az aks create` and `az aks update`: Reject `--enable-azure-monitor-logs` on clusters using service principal authentication, since the Azure Monitor profile onboards with managed identity only. * `az aks update`: Reject `--enable-azure-monitor-logs` when Azure Monitor logs is already enabled on the cluster, matching `az aks enable-addons -a monitoring`. Run `--disable-azure-monitor-logs` first to change the configuration. * `az aks update`: `--disable-azure-monitor-logs` now removes the data collection rule association and resets the Container Insights settings (syslog port, Prometheus scraping and container network logs) back to their defaults, and asks for confirmation when OpenTelemetry logs and traces are enabled. @@ -51,6 +37,20 @@ Pending * `az aks update`: Stop reopening public network access on the ingestion data collection endpoint of a cluster that is linked to an Azure Monitor Private Link Scope. The endpoint is created or updated on every reconfiguration, but `--ampls-resource-id` is only supplied on the command that links the scope, so an unrelated update such as `--enable-syslog` used to flip an existing private endpoint back to public. The existing network configuration is now preserved unless the caller explicitly asks to change it. * `az aks update`: Fix `--ampls-resource-id` being rejected with `--ampls-resource-id can only be used with private cluster in MSI mode.` on a cluster that is already private. The private state was only read from the command line, so it was invisible unless `--enable-private-cluster` happened to be supplied again in the same command; it is now read from the cluster as well. +22.0.0b8 ++++++++++ +* `az aks nodepool add`: Omit `nodeTaints` when `--node-taints` is not specified for FlexNodes pools, and reject explicitly empty values. + +22.0.0b7 ++++++++++ +* `az aks machine add`: Add preview `--capacity-reservation-group` support to associate a machine with a Capacity Reservation Group. +* Add `az aks alert-config` commands to manage AKS-managed alert configurations. +* `az aks create`: Honor `--enable-osdisk-full-caching` for the default agent pool. +* `az aks kollect` and `az aks kanalyze`: Fix compatibility with the keyword-only credential SDK parameters. +* `az aks maintenanceconfiguration add` and `az aks maintenanceconfiguration update`: Preserve configuration-file fields with the typespec-generated SDK model. +* Improve AKS live-test resilience for preview feature gates, transient resource and monitoring-table readiness, retired configurations, and service propagation delays. +* `az aks create` and `az aks update`: Reject `--outbound-type managedNATGatewayV2` with an actionable error directing to `--outbound-type managedNATGateway --outbound-type-sku StandardV2` (the GA-aligned shape); the legacy value is no longer accepted on the target api-version. + 22.0.0b6 +++++++++ * `az aks nodepool add/update`: Add preview `--enable-managed-dranet` to enable Managed DRANET on a node pool. @@ -815,7 +815,6 @@ Pending * Update --enable-advanced-network-observability description to note additional costs and add missing flag to create command. * Change default value of `--vm-set-type` to VirtualMachines when `--vm-sizes` is set. - 4.0.0b5 ++++++++ * Add warnings to `az aks mesh` commands for out of support asm revision in use. @@ -910,7 +909,6 @@ Pending * Add `--sku` to the `az aks update` command. * Support cluster service health probe mode by `--cluster-service-load-balancer-health-probe-mode {Shared, Servicenodeport}` - 3.0.0b1 +++++++ * [BREAKING CHANGE] Remove support for nodeSelector for egress gateway for `az aks mesh` command. @@ -1032,7 +1030,6 @@ Pending * Add `--node-soak-duration` to the `az aks nodepool add/update/upgrade` commands. * Add `--drain-timeout` to the `az aks nodepool add/update/upgrade` commands (already in [azure-cli](https://github.com/Azure/azure-cli/pull/27475)). - 0.5.168 +++++++ * Add `--enable-image-integrity` to the `az aks update` command.