Compare commits

...
Author SHA1 Message Date
Renovate Botanddanijel.simeunovic da51f5a89d chore(deps): update helm release prometheus to v29
AI Code Review / ai-review (pull_request) Skipped
scan.yaml / test (pull_request) Successful in 7s
2026-10-07 20:33:45 +00:00
danijel.simeunovic 52e384694c opencost
scan.yaml / test (push) Successful in 9s
2026-10-07 20:25:49 +02:00
danijel.simeunovicandClaude Opus 4.7 34444fdcd2 chore(prometheus): tighten retention and drop high-cardinality series
scan.yaml / test (push) Successful in 8s
- Cap retention at 7d / 6GB so WAL cannot fill the 8Gi PV
- Raise kyverno/traefik scrape interval 15s -> 60s
- Drop request-duration histogram buckets and go_*/process_* runtime
  metrics from both jobs

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-10-07 13:34:38 +02:00
danijel.simeunovicandClaude Opus 4.7 cd3c38463a fix(prometheus): use Recreate strategy to avoid Multi-Attach on RWO PVC
scan.yaml / test (push) Successful in 9s
Prometheus server runs as a single-replica Deployment with an RWO PVC.
RollingUpdate starts the new pod before the old releases the volume,
which fails with Multi-Attach on rollouts and reschedules.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-10-07 12:31:26 +02:00
danijel.simeunovic d503795201 chore(deps): update helm release traefik to v41
scan.yaml / test (push) Successful in 7s
2026-10-07 08:43:05 +02:00
danijel.simeunovic 39c1015b6b chore(deps): update terraform aws to v6
scan.yaml / test (push) Successful in 11s
2026-10-07 08:41:00 +02:00
8 changed files with 37 additions and 14 deletions
+1 -1
View File
@@ -2,7 +2,7 @@ terraform {
required_providers { required_providers {
aws = { aws = {
source = "hashicorp/aws" source = "hashicorp/aws"
version = "~> 5.0" version = "~> 6.0"
} }
tls = { tls = {
source = "hashicorp/tls" source = "hashicorp/tls"
@@ -2,7 +2,7 @@ terraform {
required_providers { required_providers {
aws = { aws = {
source = "hashicorp/aws" source = "hashicorp/aws"
version = "~> 5.0" version = "~> 6.0"
} }
tls = { tls = {
source = "hashicorp/tls" source = "hashicorp/tls"
+1 -1
View File
@@ -2,7 +2,7 @@ terraform {
required_providers { required_providers {
aws = { aws = {
source = "hashicorp/aws" source = "hashicorp/aws"
version = "~> 5.0" version = "~> 6.0"
} }
tls = { tls = {
source = "hashicorp/tls" source = "hashicorp/tls"
+1 -1
View File
@@ -2,7 +2,7 @@ terraform {
required_providers { required_providers {
aws = { aws = {
source = "hashicorp/aws" source = "hashicorp/aws"
version = "~> 5.0" version = "~> 6.0"
} }
tls = { tls = {
source = "hashicorp/tls" source = "hashicorp/tls"
+1 -1
View File
@@ -17,7 +17,7 @@ spec:
sources: sources:
- repoURL: https://prometheus-community.github.io/helm-charts - repoURL: https://prometheus-community.github.io/helm-charts
chart: prometheus chart: prometheus
targetRevision: "28.16.0" targetRevision: "29.35.0"
helm: helm:
releaseName: prometheus releaseName: prometheus
valueFiles: valueFiles:
@@ -24,7 +24,7 @@ spec:
sources: sources:
- repoURL: https://traefik.github.io/charts - repoURL: https://traefik.github.io/charts
chart: traefik chart: traefik
targetRevision: "28.3.0" targetRevision: "41.6.0"
helm: helm:
releaseName: traefik releaseName: traefik
valueFiles: valueFiles:
+9 -6
View File
@@ -4,14 +4,17 @@ opencost:
extraEnv: extraEnv:
EMIT_KSM_V1_METRICS: "false" EMIT_KSM_V1_METRICS: "false"
EMIT_KSM_V1_METRICS_ONLY: "true" EMIT_KSM_V1_METRICS_ONLY: "true"
prometheus:
internal:
enabled: true
serviceName: prometheus-server
namespaceName: monitoring
port: 80
# Cloud-specific pricing is in per-cluster value overrides # Cloud-specific pricing is in per-cluster value overrides
# (e.g. infra/values/upc-dev/opencost-values.yaml) # (e.g. infra/values/upc-dev/opencost-values.yaml)
# NOTE: `prometheus` is a sibling of `exporter` under `opencost` in the
# chart's values schema - nesting it inside `exporter` silently falls back
# to the chart default namespace (prometheus-system).
prometheus:
internal:
enabled: true
serviceName: prometheus-server
namespaceName: monitoring
port: 80
ui: ui:
enabled: false enabled: false
service: service:
+22 -2
View File
@@ -4,6 +4,12 @@ server:
service: service:
servicePort: 80 servicePort: 80
strategy:
type: Recreate
retention: 7d
retentionSize: 6GB
resources: resources:
requests: requests:
cpu: 150m cpu: 150m
@@ -18,7 +24,7 @@ server:
extraScrapeConfigs: | extraScrapeConfigs: |
- job_name: kyverno - job_name: kyverno
scrape_interval: 15s scrape_interval: 60s
metrics_path: /metrics metrics_path: /metrics
kubernetes_sd_configs: kubernetes_sd_configs:
- role: endpoints - role: endpoints
@@ -35,9 +41,16 @@ extraScrapeConfigs: |
target_label: pod target_label: pod
- source_labels: [__meta_kubernetes_namespace] - source_labels: [__meta_kubernetes_namespace]
target_label: namespace target_label: namespace
metric_relabel_configs:
- source_labels: [__name__]
regex: 'kyverno_(http_requests_duration_seconds|controller_.+_duration_seconds|admission_review_duration_seconds)_bucket'
action: drop
- source_labels: [__name__]
regex: 'go_.*|process_.*'
action: drop
- job_name: traefik - job_name: traefik
scrape_interval: 15s scrape_interval: 60s
metrics_path: /metrics metrics_path: /metrics
kubernetes_sd_configs: kubernetes_sd_configs:
- role: endpoints - role: endpoints
@@ -54,6 +67,13 @@ extraScrapeConfigs: |
target_label: pod target_label: pod
- source_labels: [__meta_kubernetes_namespace] - source_labels: [__meta_kubernetes_namespace]
target_label: namespace target_label: namespace
metric_relabel_configs:
- source_labels: [__name__]
regex: 'traefik_(router|service|entrypoint)_request_duration_seconds_bucket'
action: drop
- source_labels: [__name__]
regex: 'go_.*|process_.*'
action: drop
alertmanager: alertmanager:
enabled: false enabled: false