From 2f9acc570cf4cdb948e3abd92d5aeb76e4ca98cd Mon Sep 17 00:00:00 2001 From: thenav56 Date: Wed, 23 Sep 2026 09:08:00 +0000 Subject: [PATCH 1/5] chore(monitoring): upgrade production loki chart to 6.55.0 --- applications/argocd/production/platform/monitoring/loki.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/applications/argocd/production/platform/monitoring/loki.yaml b/applications/argocd/production/platform/monitoring/loki.yaml index b4610f0c..4a8c312d 100644 --- a/applications/argocd/production/platform/monitoring/loki.yaml +++ b/applications/argocd/production/platform/monitoring/loki.yaml @@ -20,7 +20,7 @@ spec: sources: - chart: loki repoURL: https://grafana.github.io/helm-charts - targetRevision: 6.28.0 # https://github.com/grafana/loki/blob/main/production/helm/loki/Chart.yaml + targetRevision: 6.55.0 # https://github.com/grafana/loki/blob/main/production/helm/loki/Chart.yaml helm: valuesObject: # Copied from https://grafana.com/docs/loki/latest/setup/install/helm/install-monolithic/ - Single Replica From 8fdae32ae464560c6f563ee9c99765dfdcd2d13e Mon Sep 17 00:00:00 2001 From: thenav56 Date: Wed, 23 Sep 2026 09:08:08 +0000 Subject: [PATCH 2/5] feat(monitoring): back production loki with azure blob via thanos objstore Retention 2160h -> 4380h to clear the Cold tier's 90 day minimum. --- .../production/platform/monitoring/loki.yaml | 53 +++++++++++++++---- 1 file changed, 43 insertions(+), 10 deletions(-) diff --git a/applications/argocd/production/platform/monitoring/loki.yaml b/applications/argocd/production/platform/monitoring/loki.yaml index 4a8c312d..c1744708 100644 --- a/applications/argocd/production/platform/monitoring/loki.yaml +++ b/applications/argocd/production/platform/monitoring/loki.yaml @@ -40,9 +40,11 @@ spec: replication_factor: 1 schemaConfig: configs: + # object_store names the thanos backend, so it matches + # storage.object_store.type rather than a cloud vendor - from: "2024-04-01" store: tsdb - object_store: s3 + object_store: azure schema: v13 index: prefix: loki_index_ @@ -52,21 +54,53 @@ spec: limits_config: allow_structured_metadata: true volume_enabled: true - max_query_lookback: 2160h # ~3 months (TODO: Is this enough or to much?) - retention_period: 2160h # ~3 months (TODO: Is this enough or to much?) + max_query_lookback: 4380h # ~6 months + retention_period: 4380h # ~6 months # XXX: We need this disabled during loki init https://github.com/grafana/loki/issues/9634#issuecomment-2188215203 # Also, comment out the limits_config max_query_lookback and retention_period compactor: working_directory: /var/loki/data/retention - delete_request_store: s3 + delete_request_store: azure retention_enabled: true ruler: enable_api: true + # Chunks and the tsdb index live in azure blob storage, so the only + # disk the statefulset needs is for the WAL and the compactor's + # working directory. Retention costs blob storage, not PVC size. + # + # `use_thanos_objstore` selects loki's thanos object store client, + # which the chart says will become the default. It reaches azure + # through DefaultAzureCredential, so with no account_key set it uses + # the workload identity token projected into the pod. + storage: + use_thanos_objstore: true + bucketNames: + chunks: loki-chunks + ruler: loki-ruler + object_store: + type: azure + azure: + # terraform: module.resources.monitoring_storage_account_name + account_name: ifrcpgoaksmonitoring + deploymentMode: SingleBinary singleBinary: replicas: 1 + podLabels: + # Opts the pod in to the azure workload identity webhook, which + # injects the token DefaultAzureCredential reads + azure.workload.identity/use: "true" + + serviceAccount: + # Pinned because the federated identity credential's subject in + # resources/monitoring.tf names it. The chart's generated default + # changed between 6.28 and 6.55, and a rename breaks blob auth. + name: monitoring-loki + annotations: + # terraform: module.resources.loki_workload_identity_client_id + azure.workload.identity/client-id: "e0f3b9b3-f5d0-46d1-8994-3b1746f45b37" # XXX: Loki startup slow - https://github.com/grafana/loki/issues/7907#issuecomment-1445336799 memberlist: @@ -80,12 +114,8 @@ spec: cpu: "0.1" memory: 512Mi - # FIXME: Replace this with azure blob storage minio: - enabled: true - persistence: - enabled: true - size: 32Gi # Cost is defined at tier level + enabled: false # Zero out replica counts of other deployment modes backend: @@ -306,7 +336,10 @@ spec: // publish data to loki loki.write "default" { endpoint { - url = "http://monitoring-loki-gateway/loki/api/v1/push" + url = "http://monitoring-loki-gateway/loki/api/v1/push" + // Ignored while loki runs with auth_enabled: false, which + // files everything under the tenant `fake`. The blob lifecycle + // rule in resources/monitoring.tf matches on that prefix. tenant_id = "1" } } From a53980b8519b5f133c1ec5d90ea89abac58df7e0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Szab=C3=B3=2C=20Zolt=C3=A1n?= Date: Wed, 23 Sep 2026 12:12:29 +0200 Subject: [PATCH 3/5] Appeal Admin fix --- applications/argocd/staging/applications/go-api.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/applications/argocd/staging/applications/go-api.yaml b/applications/argocd/staging/applications/go-api.yaml index 97d7a813..743d2013 100644 --- a/applications/argocd/staging/applications/go-api.yaml +++ b/applications/argocd/staging/applications/go-api.yaml @@ -10,7 +10,7 @@ spec: source: repoURL: ghcr.io/ifrcgo chart: ifrcgo-helm - targetRevision: 0.0.2-develop.cb8e9907 + targetRevision: 0.0.2-develop.c9035a3a helm: valueFiles: - values/traefik.yaml From 5b0a51d3afaa1b94000ae071a7d6f018d623fc78 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Szab=C3=B3=2C=20Zolt=C3=A1n?= Date: Wed, 23 Sep 2026 12:41:04 +0200 Subject: [PATCH 4/5] Appeal Admin fix --- applications/argocd/production/applications/go-api.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/applications/argocd/production/applications/go-api.yaml b/applications/argocd/production/applications/go-api.yaml index 889d2abd..3f7b5484 100644 --- a/applications/argocd/production/applications/go-api.yaml +++ b/applications/argocd/production/applications/go-api.yaml @@ -11,7 +11,7 @@ spec: repoURL: ghcr.io/ifrcgo chart: ifrcgo-helm # TODO: do we need to switch to a master-build chart tag - targetRevision: 0.0.2-develop.cb8e9907 + targetRevision: 0.0.2-develop.c9035a3a helm: valueFiles: - values/traefik.yaml From 014529b0bff14352718a0ffc20c993e12df9fc2d Mon Sep 17 00:00:00 2001 From: Navin Ayer Date: Fri, 25 Sep 2026 15:35:14 +0545 Subject: [PATCH 5/5] feat(network): route cluster egress through an owned public IP (#251) --- base-infrastructure/terraform/main.tf | 5 +++++ base-infrastructure/terraform/resources/aks.tf | 12 ++++++++++++ base-infrastructure/terraform/resources/ip.tf | 15 +++++++++++++++ base-infrastructure/terraform/resources/output.tf | 5 +++++ 4 files changed, 37 insertions(+) diff --git a/base-infrastructure/terraform/main.tf b/base-infrastructure/terraform/main.tf index cc46fe13..b66f4aaf 100644 --- a/base-infrastructure/terraform/main.tf +++ b/base-infrastructure/terraform/main.tf @@ -23,6 +23,11 @@ terraform { } } +# Surfaced on its own since `resources` is sensitive and hidden in plan/apply output +output "egress_public_ip" { + value = module.resources.egress_public_ip +} + output "resources" { value = module.resources sensitive = true diff --git a/base-infrastructure/terraform/resources/aks.tf b/base-infrastructure/terraform/resources/aks.tf index bad36c7a..e01c3b59 100644 --- a/base-infrastructure/terraform/resources/aks.tf +++ b/base-infrastructure/terraform/resources/aks.tf @@ -37,6 +37,18 @@ resource "azurerm_kubernetes_cluster" "ifrcgo" { ManagedBy = "IFRCGo" } + # Mirrors the live cluster (kubenet, no network policy) so only the outbound IP changes; + # a plugin mismatch here would replace the cluster. + network_profile { + network_plugin = "kubenet" + load_balancer_sku = "standard" + outbound_type = "loadBalancer" + + load_balancer_profile { + outbound_ip_address_ids = [azurerm_public_ip.egress.id] + } + } + key_vault_secrets_provider { secret_rotation_enabled = true secret_rotation_interval = var.secret_rotation_interval diff --git a/base-infrastructure/terraform/resources/ip.tf b/base-infrastructure/terraform/resources/ip.tf index 576027b6..80be3448 100644 --- a/base-infrastructure/terraform/resources/ip.tf +++ b/base-infrastructure/terraform/resources/ip.tf @@ -28,6 +28,21 @@ resource "azurerm_public_ip" "traefik" { } } +# Cluster egress Public IP (see aks.tf network_profile). Every node SNATs outbound traffic +# through it. Owned here instead of AKS's node resource group so it survives a cluster +# recreation and can be handed out for allowlisting. +resource "azurerm_public_ip" "egress" { + name = "${local.prefix}EgressPublicIP" + resource_group_name = data.azurerm_resource_group.ifrcgo.name + location = data.azurerm_resource_group.ifrcgo.location + allocation_method = "Static" + sku = "Standard" + + tags = { + Environment = var.environment + } +} + # SSH bastion Public IP (see bastion.tf) — reserved so the bastion endpoint is # stable across recreations (fixed IP / DNS can be put in front later). resource "azurerm_public_ip" "bastion" { diff --git a/base-infrastructure/terraform/resources/output.tf b/base-infrastructure/terraform/resources/output.tf index 66b4a219..264a125e 100644 --- a/base-infrastructure/terraform/resources/output.tf +++ b/base-infrastructure/terraform/resources/output.tf @@ -18,6 +18,11 @@ output "cluster_kubelet_identity" { value = azurerm_kubernetes_cluster.ifrcgo.kubelet_identity[0].object_id } +# Cluster outbound (egress) IP, the source address external services see (see ip.tf) +output "egress_public_ip" { + value = azurerm_public_ip.egress.ip_address +} + output "resource_group" { value = data.azurerm_resource_group.ifrcgo.name }