Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions kubernetes/alertmanager/config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -27,11 +27,21 @@ route:

# The child route trees.
routes:
- match:
alertname: Watchdog
receiver: healthchecks-watchdog
group_wait: 0s
group_interval: 1m
repeat_interval: 5m
- match:
realert: daily
repeat_interval: 1d

receivers:
- name: healthchecks-watchdog
webhook_configs:
- url_file: /secrets/HEALTHCHECKS_WATCHDOG_URL
send_resolved: false
- name: default
slack_configs:
- api_url_file: /secrets/SLACK_WEBHOOK_URL
Expand Down
1 change: 1 addition & 0 deletions kubernetes/alertmanager/deploy.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@ spec:
app.kubernetes.io/name: alertmanager
app.kubernetes.io/instance: alertmanager
spec:
priorityClassName: monitoring
Comment thread
claytono marked this conversation as resolved.
containers:
- name: alertmanager
image: prom/alertmanager:v0.34.1@sha256:e9733bafb1bdef9b00e25a21f8f99dc26a22224bf16641ad754d1649f4c3357a
Expand Down
9 changes: 9 additions & 0 deletions kubernetes/alertmanager/externalsecret.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,11 @@ spec:
secretStoreRef:
name: production
kind: ClusterSecretStore
target:
template:
mergePolicy: Merge
data:
HEALTHCHECKS_WATCHDOG_URL: 'https://hc.k.oneill.net/ping/{{ .HEALTHCHECKS_PING_KEY }}/prometheus-watchdog'
data:
- secretKey: SMTP_PASSWORD
remoteRef:
Expand All @@ -18,3 +23,7 @@ spec:
remoteRef:
key: alertmanager
property: SLACK_WEBHOOK_URL
- secretKey: HEALTHCHECKS_PING_KEY
remoteRef:
key: healthchecks
property: PING_KEY
2 changes: 2 additions & 0 deletions kubernetes/karakeep/helm/meilisearch/configmap.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,8 @@ metadata:
app.kubernetes.io/managed-by: Helm
data:
MEILI_ENV: "production"
MEILI_EXPERIMENTAL_REDUCE_INDEXING_MEMORY_USAGE: "true"
MEILI_MAX_INDEXING_MEMORY: "2Gb"
MEILI_NO_ANALYTICS: "true"
MEILI_UPGRADE_DB: "true"

2 changes: 1 addition & 1 deletion kubernetes/karakeep/helm/meilisearch/statefulset.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,7 @@ spec:
app.kubernetes.io/part-of: meilisearch
app.kubernetes.io/managed-by: Helm
annotations:
checksum/config: 4cb7c9de031634b65307b783c37ea7c3e2595a65f3f88d75bc6e509d673328fd
checksum/config: fc496f8e7457de8c064d49afeac3c06280869f29dcd1e2ddc37f4040576a20b8
spec:
serviceAccountName: karakeep-meilisearch
securityContext:
Expand Down
2 changes: 2 additions & 0 deletions kubernetes/karakeep/values.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,8 @@ meilisearch:
enabled: true
environment:
MEILI_UPGRADE_DB: 'true'
MEILI_MAX_INDEXING_MEMORY: 2Gb
MEILI_EXPERIMENTAL_REDUCE_INDEXING_MEMORY_USAGE: 'true'
auth:
existingMasterKeySecret: karakeep-meilisearch
persistence:
Expand Down
4 changes: 4 additions & 0 deletions kubernetes/openarchiver/meilisearch-statefulset.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,10 @@ spec:
value: 'true'
- name: MEILI_UPGRADE_DB
value: 'true'
- name: MEILI_MAX_INDEXING_MEMORY
value: 2Gb
- name: MEILI_EXPERIMENTAL_REDUCE_INDEXING_MEMORY_USAGE
value: 'true'
Comment thread
coderabbitai[bot] marked this conversation as resolved.
- name: MEILI_MASTER_KEY
valueFrom:
secretKeyRef:
Expand Down
2 changes: 1 addition & 1 deletion kubernetes/openarchiver/namespace.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8,4 +8,4 @@ metadata:
goldilocks.fairwinds.com/enabled: 'true'
annotations:
goldilocks.fairwinds.com/vpa-update-mode: InPlaceOrRecreate
goldilocks.fairwinds.com/vpa-resource-policy: '{"containerPolicies":[{"containerName":"*","controlledValues":"RequestsOnly"}]}'
goldilocks.fairwinds.com/vpa-resource-policy: '{"containerPolicies":[{"containerName":"*","controlledValues":"RequestsOnly"},{"containerName":"meilisearch","controlledValues":"RequestsOnly","maxAllowed":{"memory":"8Gi"}}]}'
4 changes: 4 additions & 0 deletions kubernetes/prometheus/config/rules.yml
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,10 @@ groups:
expr: hass_sensor_unit_m{entity="sensor.desk_ultrasonic_sensor"} > bool 0.9
- name: alert.rules
rules:
- alert: Watchdog
expr: vector(1)
annotations:
message: Always firing; Alertmanager forwards it to Healthchecks as a heartbeat
- alert: NodeDown
expr: up{job="node resources", instance!="xtal:9100"} == 0
for: 5m
Expand Down
1 change: 1 addition & 0 deletions kubernetes/prometheus/deploy.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@ spec:
app.kubernetes.io/name: prometheus
app.kubernetes.io/instance: prometheus
spec:
priorityClassName: monitoring
serviceAccountName: prometheus
containers:
- name: prometheus
Expand Down
1 change: 1 addition & 0 deletions kubernetes/prometheus/kustomization.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@ commonAnnotations:
resources:
- deploy.yaml
- ingress.yaml
- priorityclass.yaml
- pvc.yaml
- rbac.yaml
- service.yaml
Expand Down
10 changes: 10 additions & 0 deletions kubernetes/prometheus/priorityclass.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
---
apiVersion: scheduling.k8s.io/v1
kind: PriorityClass

metadata:
name: monitoring

value: 1000000
globalDefault: false
description: Prometheus and Alertmanager
14 changes: 14 additions & 0 deletions opentofu/healthchecks.tf
Original file line number Diff line number Diff line change
Expand Up @@ -159,6 +159,20 @@ resource "healthchecksio_check" "velero_daily" {
channels = [data.healthchecksio_channel.selfhosted_email.id]
}

# Prometheus/Alertmanager heartbeat - Alertmanager forwards the always-firing
# Watchdog alert every 5 minutes; silence means alerting is down
resource "healthchecksio_check" "prometheus_watchdog" {
provider = healthchecksio.selfhosted

name = "prometheus-watchdog"
slug = "prometheus-watchdog"
desc = "Heartbeat from the Prometheus Watchdog alert via Alertmanager"
tags = ["monitoring", "kubernetes"]
timeout = 600 # 10 minutes
grace = 600 # 10 minutes
channels = [data.healthchecksio_channel.selfhosted_email.id]
}

# Kubernetes control-plane backup - runs daily
resource "healthchecksio_check" "cluster_backup" {
provider = healthchecksio.selfhosted
Expand Down
Loading