Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
57b13387cb
|
||
|
|
1deb024621
|
||
|
|
311a506982
|
||
|
|
a0d38fc9ab
|
||
|
|
fa293ffc80
|
||
|
|
77df686981
|
||
|
|
9e6515406f
|
||
|
|
f6c216ce93
|
@@ -21,7 +21,7 @@ env:
|
|||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
lint:
|
lint:
|
||||||
runs-on: [self-hosted, pod]
|
runs-on: self-hosted
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
@@ -51,11 +51,6 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
export PATH="$HOME/.local/bin:$PATH"
|
export PATH="$HOME/.local/bin:$PATH"
|
||||||
export ANSIBLE_COLLECTIONS_PATH="$PWD/infrastructure/samba-ad/ansible/collections:/root/.ansible/collections"
|
export ANSIBLE_COLLECTIONS_PATH="$PWD/infrastructure/samba-ad/ansible/collections:/root/.ansible/collections"
|
||||||
# 静态检查不应依赖生产 vault 凭据。一次性 checkout 可以去掉加密变量文件;
|
|
||||||
# syntax-check 只验证结构,不需要解析变量的运行时值。
|
|
||||||
rm -f \
|
|
||||||
infrastructure/openbao/ansible/group_vars/all/vault.yml \
|
|
||||||
infrastructure/samba-ad/ansible/group_vars/all/vault.yml
|
|
||||||
rc=0
|
rc=0
|
||||||
for p in infrastructure/openbao infrastructure/samba-ad infrastructure/proxmox; do
|
for p in infrastructure/openbao infrastructure/samba-ad infrastructure/proxmox; do
|
||||||
echo "::group::$p"
|
echo "::group::$p"
|
||||||
@@ -65,7 +60,7 @@ jobs:
|
|||||||
exit $rc
|
exit $rc
|
||||||
|
|
||||||
collection-test:
|
collection-test:
|
||||||
runs-on: [self-hosted, pod]
|
runs-on: self-hosted
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,63 @@
|
|||||||
|
---
|
||||||
|
name: kind-on-kata-smoke
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- poc/kind-on-kata
|
||||||
|
paths:
|
||||||
|
- .gitea/workflows/kind-on-kata-smoke.yml
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
smoke:
|
||||||
|
runs-on: kata-poc
|
||||||
|
steps:
|
||||||
|
- name: Create nested kind cluster
|
||||||
|
shell: sh
|
||||||
|
env:
|
||||||
|
KIND_VERSION: v0.27.0
|
||||||
|
KIND_NODE_IMAGE: kindest/node:v1.32.2@sha256:f226345927d7e348497136874b6d207e0b32cc52154ad8323129352923a3142f
|
||||||
|
run: |
|
||||||
|
set -eu
|
||||||
|
apk add --no-cache ca-certificates curl docker-cli
|
||||||
|
curl --retry 5 --retry-all-errors --connect-timeout 15 -fsSLo /tmp/kind \
|
||||||
|
"https://kind.sigs.k8s.io/dl/${KIND_VERSION}/kind-linux-amd64"
|
||||||
|
curl --retry 5 --retry-all-errors --connect-timeout 15 -fsSLo /tmp/kind.sha256sum \
|
||||||
|
"https://kind.sigs.k8s.io/dl/${KIND_VERSION}/kind-linux-amd64.sha256sum"
|
||||||
|
expected="$(awk '{print $1}' /tmp/kind.sha256sum)"
|
||||||
|
printf '%s %s\n' "$expected" /tmp/kind | sha256sum -c -
|
||||||
|
install -m 0755 /tmp/kind /usr/local/bin/kind
|
||||||
|
docker info --format 'kernel={{.KernelVersion}} driver={{.Driver}}'
|
||||||
|
test "$(docker info --format '{{.Driver}}')" = overlay2
|
||||||
|
|
||||||
|
cleanup() {
|
||||||
|
kind delete cluster --name nested >/dev/null 2>&1 || true
|
||||||
|
}
|
||||||
|
trap cleanup EXIT
|
||||||
|
|
||||||
|
cat >/tmp/kind-config.yaml <<'EOF'
|
||||||
|
kind: Cluster
|
||||||
|
apiVersion: kind.x-k8s.io/v1alpha4
|
||||||
|
name: nested
|
||||||
|
nodes:
|
||||||
|
- role: control-plane
|
||||||
|
extraMounts:
|
||||||
|
- hostPath: /dev/kmsg
|
||||||
|
containerPath: /dev/kmsg
|
||||||
|
EOF
|
||||||
|
kind create cluster -v 9 --retain --config /tmp/kind-config.yaml --image "$KIND_NODE_IMAGE" --wait 5m
|
||||||
|
docker exec nested-control-plane kubectl \
|
||||||
|
--kubeconfig=/etc/kubernetes/admin.conf wait \
|
||||||
|
--for=condition=Ready node/nested-control-plane --timeout=2m
|
||||||
|
docker exec nested-control-plane kubectl \
|
||||||
|
--kubeconfig=/etc/kubernetes/admin.conf run smoke \
|
||||||
|
--image=docker.io/library/busybox:1.37 --restart=Never \
|
||||||
|
--command -- sh -c 'echo kind-on-kata-ok'
|
||||||
|
docker exec nested-control-plane kubectl \
|
||||||
|
--kubeconfig=/etc/kubernetes/admin.conf wait \
|
||||||
|
--for=jsonpath='{.status.phase}'=Succeeded pod/smoke --timeout=2m
|
||||||
|
test "$(docker exec nested-control-plane kubectl \
|
||||||
|
--kubeconfig=/etc/kubernetes/admin.conf logs smoke)" = kind-on-kata-ok
|
||||||
|
kind delete cluster --name nested
|
||||||
|
trap - EXIT
|
||||||
@@ -24,7 +24,7 @@ on:
|
|||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
yaml:
|
yaml:
|
||||||
runs-on: [self-hosted, pod]
|
runs-on: self-hosted
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ on:
|
|||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
validate:
|
validate:
|
||||||
runs-on: [self-hosted, pod]
|
runs-on: self-hosted
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
- uses: hashicorp/setup-terraform@v3
|
- uses: hashicorp/setup-terraform@v3
|
||||||
|
|||||||
@@ -229,7 +229,7 @@ configMap:
|
|||||||
require_pkce: false
|
require_pkce: false
|
||||||
token_endpoint_auth_method: 'client_secret_basic'
|
token_endpoint_auth_method: 'client_secret_basic'
|
||||||
redirect_uris:
|
redirect_uris:
|
||||||
- 'https://grafana.ad.ddupan.top/login/generic_oauth'
|
- 'https://grafana.tail7e769.ts.net/login/generic_oauth'
|
||||||
scopes:
|
scopes:
|
||||||
- 'openid'
|
- 'openid'
|
||||||
- 'profile'
|
- 'profile'
|
||||||
|
|||||||
@@ -48,11 +48,6 @@ gitea:
|
|||||||
# github.com is reachable from this network (verified 2026-07-28) even when
|
# github.com is reachable from this network (verified 2026-07-28) even when
|
||||||
# pypi.org/Fastly is not — see the flaky-WAN notes in the lint workflow.
|
# pypi.org/Fastly is not — see the flaky-WAN notes in the lint workflow.
|
||||||
DEFAULT_ACTIONS_URL: github
|
DEFAULT_ACTIONS_URL: github
|
||||||
webhook:
|
|
||||||
# Keep the default public-internet access for existing hooks while allowing
|
|
||||||
# only the dynamic Runner controller's exact in-cluster DNS name. Do not
|
|
||||||
# broaden this to the built-in `private` network group.
|
|
||||||
ALLOWED_HOST_LIST: external,dynamic-runner-controller.dynamic-runner.svc.cluster.local
|
|
||||||
mailer:
|
mailer:
|
||||||
# Outbound mail via the in-cluster Postfix+OAuth relay (see ../smtp-relay/).
|
# Outbound mail via the in-cluster Postfix+OAuth relay (see ../smtp-relay/).
|
||||||
# Plain SMTP on :25 — the relay does STARTTLS + OAuth to M365. From must be the
|
# Plain SMTP on :25 — the relay does STARTTLS + OAuth to M365. From must be the
|
||||||
|
|||||||
@@ -0,0 +1,5 @@
|
|||||||
|
route:
|
||||||
|
receiver: blackhole
|
||||||
|
|
||||||
|
receivers:
|
||||||
|
- name: blackhole
|
||||||
@@ -0,0 +1,94 @@
|
|||||||
|
services:
|
||||||
|
# Metrics collector.
|
||||||
|
# It scrapes targets defined in --promscrape.config
|
||||||
|
# And forward them to --remoteWrite.url
|
||||||
|
vmagent:
|
||||||
|
image: victoriametrics/vmagent:v1.132.0
|
||||||
|
depends_on:
|
||||||
|
- "victoriametrics"
|
||||||
|
ports:
|
||||||
|
- 8429:8429
|
||||||
|
volumes:
|
||||||
|
- vmagentdata:/vmagentdata
|
||||||
|
- ./prometheus.yaml:/etc/prometheus/prometheus.yml
|
||||||
|
command:
|
||||||
|
- "--promscrape.config=/etc/prometheus/prometheus.yml"
|
||||||
|
- "--remoteWrite.url=http://victoriametrics:8428/api/v1/write"
|
||||||
|
restart: always
|
||||||
|
# VictoriaMetrics instance, a single process responsible for
|
||||||
|
# storing metrics and serve read requests.
|
||||||
|
victoriametrics:
|
||||||
|
image: victoriametrics/victoria-metrics:v1.132.0
|
||||||
|
ports:
|
||||||
|
- 8428:8428
|
||||||
|
- 8089:8089
|
||||||
|
- 8089:8089/udp
|
||||||
|
- 2003:2003
|
||||||
|
- 2003:2003/udp
|
||||||
|
- 4242:4242
|
||||||
|
volumes:
|
||||||
|
- vmdata:/storage
|
||||||
|
command:
|
||||||
|
- "--storageDataPath=/storage"
|
||||||
|
- "--graphiteListenAddr=:2003"
|
||||||
|
- "--opentsdbListenAddr=:4242"
|
||||||
|
- "--httpListenAddr=:8428"
|
||||||
|
- "--influxListenAddr=:8089"
|
||||||
|
- "--vmalert.proxyURL=http://vmalert:8880"
|
||||||
|
restart: always
|
||||||
|
|
||||||
|
grafana:
|
||||||
|
image: grafana/grafana:12.2.0
|
||||||
|
depends_on:
|
||||||
|
- "victoriametrics"
|
||||||
|
ports:
|
||||||
|
- 3000:3000
|
||||||
|
volumes:
|
||||||
|
- grafanadata:/var/lib/grafana
|
||||||
|
- ./provisioning/datasources/prometheus-datasource/single.yml:/etc/grafana/provisioning/datasources/single.yml
|
||||||
|
- ./provisioning/dashboards:/etc/grafana/provisioning/dashboards
|
||||||
|
- ./provisioning/dashboards/victoriametrics.json:/var/lib/grafana/dashboards/vm.json
|
||||||
|
- ./provisioning/dashboards/vmagent.json:/var/lib/grafana/dashboards/vmagent.json
|
||||||
|
- ./provisioning/dashboards/vmalert.json:/var/lib/grafana/dashboards/vmalert.json
|
||||||
|
restart: always
|
||||||
|
|
||||||
|
# vmalert executes alerting and recording rules
|
||||||
|
vmalert:
|
||||||
|
image: victoriametrics/vmalert:v1.132.0
|
||||||
|
depends_on:
|
||||||
|
- "victoriametrics"
|
||||||
|
- "alertmanager"
|
||||||
|
ports:
|
||||||
|
- 8880:8880
|
||||||
|
volumes:
|
||||||
|
- ./rules/alerts.yml:/etc/alerts/alerts.yml
|
||||||
|
- ./rules/alerts-health.yml:/etc/alerts/alerts-health.yml
|
||||||
|
- ./rules/alerts-vmagent.yml:/etc/alerts/alerts-vmagent.yml
|
||||||
|
- ./rules/alerts-vmalert.yml:/etc/alerts/alerts-vmalert.yml
|
||||||
|
command:
|
||||||
|
- "--datasource.url=http://victoriametrics:8428/"
|
||||||
|
- "--remoteRead.url=http://victoriametrics:8428/"
|
||||||
|
- "--remoteWrite.url=http://vmagent:8429/"
|
||||||
|
- "--notifier.url=http://alertmanager:9093/"
|
||||||
|
- "--rule=/etc/alerts/*.yml"
|
||||||
|
# display source of alerts in grafana
|
||||||
|
- "--external.url=http://127.0.0.1:3000" #grafana outside container
|
||||||
|
- '--external.alert.source=explore?orgId=1&left={"datasource":"VictoriaMetrics","queries":[{"expr":{{.Expr|jsonEscape|queryEscape}},"refId":"A"}],"range":{"from":"{{ .ActiveAt.UnixMilli }}","to":"now"}}'
|
||||||
|
restart: always
|
||||||
|
|
||||||
|
# alertmanager receives alerting notifications from vmalert
|
||||||
|
# and distributes them according to --config.file.
|
||||||
|
alertmanager:
|
||||||
|
image: prom/alertmanager:v0.28.1
|
||||||
|
volumes:
|
||||||
|
- ./alertmanager.yaml:/config/alertmanager.yml
|
||||||
|
command:
|
||||||
|
- "--config.file=/config/alertmanager.yml"
|
||||||
|
ports:
|
||||||
|
- 9093:9093
|
||||||
|
restart: always
|
||||||
|
|
||||||
|
volumes:
|
||||||
|
vmagentdata: {}
|
||||||
|
vmdata: {}
|
||||||
|
grafanadata: {}
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
global:
|
||||||
|
scrape_interval: 10s
|
||||||
|
|
||||||
|
scrape_configs:
|
||||||
|
- job_name: vmagent
|
||||||
|
static_configs:
|
||||||
|
- targets:
|
||||||
|
- vmagent:8429
|
||||||
|
- job_name: vmalert
|
||||||
|
static_configs:
|
||||||
|
- targets:
|
||||||
|
- vmalert:8880
|
||||||
|
- job_name: victoriametrics
|
||||||
|
static_configs:
|
||||||
|
- targets:
|
||||||
|
- victoriametrics:8428
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
apiVersion: 1
|
||||||
|
|
||||||
|
providers:
|
||||||
|
- name: Prometheus
|
||||||
|
orgId: 1
|
||||||
|
folder: ''
|
||||||
|
type: file
|
||||||
|
options:
|
||||||
|
path: /var/lib/grafana/dashboards
|
||||||
+11
@@ -0,0 +1,11 @@
|
|||||||
|
apiVersion: 1
|
||||||
|
|
||||||
|
datasources:
|
||||||
|
- name: VictoriaMetrics
|
||||||
|
type: prometheus
|
||||||
|
access: proxy
|
||||||
|
url: http://victoriametrics:8428
|
||||||
|
isDefault: true
|
||||||
|
jsonData:
|
||||||
|
prometheusType: Prometheus
|
||||||
|
prometheusVersion: 2.24.0
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,11 @@
|
|||||||
|
apiVersion: 1
|
||||||
|
|
||||||
|
datasources:
|
||||||
|
- name: VictoriaMetrics
|
||||||
|
type: prometheus
|
||||||
|
access: proxy
|
||||||
|
url: http://victoriametrics:8428
|
||||||
|
isDefault: true
|
||||||
|
jsonData:
|
||||||
|
prometheusType: Prometheus
|
||||||
|
prometheusVersion: 2.24.0
|
||||||
@@ -0,0 +1,149 @@
|
|||||||
|
# File contains default list of alerts for various VM components.
|
||||||
|
# The following alerts are recommended for use for any VM installation.
|
||||||
|
# The alerts below are just recommendations and may require some updates
|
||||||
|
# and threshold calibration according to every specific setup.
|
||||||
|
groups:
|
||||||
|
- name: vm-health
|
||||||
|
# note the `job` filter and update accordingly to your setup
|
||||||
|
rules:
|
||||||
|
- alert: TooManyRestarts
|
||||||
|
expr: changes(process_start_time_seconds{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"}[15m]) > 2
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "{{ $labels.job }} too many restarts (instance {{ $labels.instance }})"
|
||||||
|
description: >
|
||||||
|
Job {{ $labels.job }} (instance {{ $labels.instance }}) has restarted more than twice in the last 15 minutes.
|
||||||
|
It might be crashlooping.
|
||||||
|
|
||||||
|
- alert: ServiceDown
|
||||||
|
expr: up{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"} == 0
|
||||||
|
for: 2m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "Service {{ $labels.job }} is down on {{ $labels.instance }}"
|
||||||
|
description: "{{ $labels.instance }} of job {{ $labels.job }} has been down for more than 2 minutes."
|
||||||
|
|
||||||
|
- alert: ProcessNearFDLimits
|
||||||
|
expr: (process_max_fds - process_open_fds) < 100
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "Number of free file descriptors is less than 100 for \"{{ $labels.job }}\"(\"{{ $labels.instance }}\") for the last 5m"
|
||||||
|
description: |
|
||||||
|
Exhausting OS file descriptors limit can cause severe degradation of the process.
|
||||||
|
Consider to increase the limit as fast as possible.
|
||||||
|
|
||||||
|
- alert: TooHighMemoryUsage
|
||||||
|
expr: (min_over_time(process_resident_memory_anon_bytes[10m]) / vm_available_memory_bytes) > 0.8
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "It is more than 80% of memory used by \"{{ $labels.job }}\"(\"{{ $labels.instance }}\")"
|
||||||
|
description: |
|
||||||
|
Too high memory usage may result into multiple issues such as OOMs or degraded performance.
|
||||||
|
Consider to either increase available memory or decrease the load on the process.
|
||||||
|
|
||||||
|
- alert: TooHighCPUUsage
|
||||||
|
expr: rate(process_cpu_seconds_total[5m]) / process_cpu_cores_available > 0.9
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "More than 90% of CPU is used by \"{{ $labels.job }}\"(\"{{ $labels.instance }}\") during the last 5m"
|
||||||
|
description: >
|
||||||
|
Too high CPU usage may be a sign of insufficient resources and make process unstable.
|
||||||
|
Consider to either increase available CPU resources or decrease the load on the process.
|
||||||
|
|
||||||
|
- alert: TooHighGoroutineSchedulingLatency
|
||||||
|
expr: histogram_quantile(0.99, sum(rate(go_sched_latencies_seconds_bucket{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"}[5m])) by (le, job, instance)) > 0.1
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "\"{{ $labels.job }}\"(\"{{ $labels.instance }}\") has insufficient CPU resources for >15m"
|
||||||
|
description: >
|
||||||
|
Go runtime is unable to schedule goroutines execution in acceptable time. This is usually a sign of
|
||||||
|
insufficient CPU resources or CPU throttling. Verify that service has enough CPU resources. Otherwise,
|
||||||
|
the service could work unreliably with delays in processing.
|
||||||
|
|
||||||
|
- alert: TooManyLogs
|
||||||
|
expr: sum(increase(vm_log_messages_total{level="error"}[5m])) without (app_version, location) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Too many logs printed for job \"{{ $labels.job }}\" ({{ $labels.instance }})"
|
||||||
|
description: >
|
||||||
|
Logging rate for job \"{{ $labels.job }}\" ({{ $labels.instance }}) is {{ $value }} for last 15m.
|
||||||
|
Worth to check logs for specific error messages.
|
||||||
|
|
||||||
|
- alert: TooManyTSIDMisses
|
||||||
|
expr: increase(vm_missing_tsids_for_metric_id_total[5m]) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "Unexpected TSID misses for job \"{{ $labels.job }}\" ({{ $labels.instance }}) for the last 15 minutes"
|
||||||
|
description: |
|
||||||
|
Unexpected TSID misses for \"{{ $labels.job }}\" ({{ $labels.instance }}) for the last 15 minutes.
|
||||||
|
If this happens after unclean shutdown of VictoriaMetrics process (via \"kill -9\", OOM or power off),
|
||||||
|
then this is OK - the alert must go away in a few minutes after the restart.
|
||||||
|
Otherwise this may point to the corruption of index data.
|
||||||
|
|
||||||
|
- alert: ConcurrentInsertsHitTheLimit
|
||||||
|
expr: avg_over_time(vm_concurrent_insert_current[1m]) >= vm_concurrent_insert_capacity
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "{{ $labels.job }} on instance {{ $labels.instance }} is constantly hitting concurrent inserts limit"
|
||||||
|
description: |
|
||||||
|
The limit of concurrent inserts on instance {{ $labels.instance }} depends on the number of CPUs.
|
||||||
|
Usually, when component constantly hits the limit it is likely the component is overloaded and requires more CPU.
|
||||||
|
In some cases for components like vmagent or vminsert the alert might trigger if there are too many clients
|
||||||
|
making write attempts. If vmagent's or vminsert's CPU usage and network saturation are at normal level, then
|
||||||
|
it might be worth adjusting `-maxConcurrentInserts` cmd-line flag.
|
||||||
|
|
||||||
|
- alert: IndexDBRecordsDrop
|
||||||
|
expr: increase(vm_indexdb_items_dropped_total[5m]) > 0
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "IndexDB skipped registering items during data ingestion with reason={{ $labels.reason }}."
|
||||||
|
description: |
|
||||||
|
VictoriaMetrics could skip registering new timeseries during ingestion if they fail the validation process.
|
||||||
|
For example, `reason=too_long_item` means that time series cannot exceed 64KB. Please, reduce the number
|
||||||
|
of labels or label values for such series. Or enforce these limits via `-maxLabelsPerTimeseries` and
|
||||||
|
`-maxLabelValueLen` command-line flags.
|
||||||
|
|
||||||
|
- alert: RowsRejectedOnIngestion
|
||||||
|
expr: rate(vm_rows_ignored_total[5m]) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Some rows are rejected on \"{{ $labels.instance }}\" on ingestion attempt"
|
||||||
|
description: "Ingested rows on instance \"{{ $labels.instance }}\" are rejected due to the
|
||||||
|
following reason: \"{{ $labels.reason }}\""
|
||||||
|
|
||||||
|
- alert: TooHighQueryLoad
|
||||||
|
expr: increase(vm_concurrent_select_limit_timeout_total[5m]) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Read queries fail with timeout for {{ $labels.job }} on instance {{ $labels.instance }}"
|
||||||
|
description: |
|
||||||
|
Instance {{ $labels.instance }} ({{ $labels.job }}) is failing to serve read queries during last 15m.
|
||||||
|
Concurrency limit `-search.maxConcurrentRequests` was reached on this instance and extra queries were
|
||||||
|
put into the queue for `-search.maxQueueDuration` interval. But even after waiting in the queue these queries weren't served.
|
||||||
|
This happens if instance is overloaded with the current workload, or datasource is too slow to respond.
|
||||||
|
Possible solutions are the following:
|
||||||
|
* reduce the query load;
|
||||||
|
* increase compute resources or number of replicas;
|
||||||
|
* adjust limits `-search.maxConcurrentRequests` and `-search.maxQueueDuration`.
|
||||||
|
See more at https://docs.victoriametrics.com/victoriametrics/troubleshooting/#slow-queries
|
||||||
@@ -0,0 +1,172 @@
|
|||||||
|
# File contains default list of alerts for vmagent service.
|
||||||
|
# The alerts below are just recommendations and may require some updates
|
||||||
|
# and threshold calibration according to every specific setup.
|
||||||
|
groups:
|
||||||
|
# Alerts group for vmagent assumes that Grafana dashboard
|
||||||
|
# https://grafana.com/grafana/dashboards/12683 is installed.
|
||||||
|
# Pls update the `dashboard` annotation according to your setup.
|
||||||
|
- name: vmagent
|
||||||
|
interval: 30s
|
||||||
|
concurrency: 2
|
||||||
|
rules:
|
||||||
|
- alert: PersistentQueueIsDroppingData
|
||||||
|
expr: sum(increase(vm_persistentqueue_bytes_dropped_total[5m])) without (path) > 0
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=49&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Instance {{ $labels.instance }} is dropping data from persistent queue"
|
||||||
|
description: "Vmagent dropped {{ $value | humanize1024 }} from persistent queue
|
||||||
|
on instance {{ $labels.instance }} for the last 10m."
|
||||||
|
|
||||||
|
- alert: RejectedRemoteWriteDataBlocksAreDropped
|
||||||
|
expr: sum(increase(vmagent_remotewrite_packets_dropped_total[5m])) without (url) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=79&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Vmagent is dropping data blocks that are rejected by remote storage"
|
||||||
|
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} drops the rejected by
|
||||||
|
remote-write server data blocks. Check the logs to find the reason for rejects."
|
||||||
|
|
||||||
|
- alert: TooManyScrapeErrors
|
||||||
|
expr: increase(vm_promscrape_scrapes_failed_total[5m]) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=31&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Vmagent fails to scrape one or more targets"
|
||||||
|
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} fails to scrape targets for last 15m"
|
||||||
|
|
||||||
|
- alert: ScrapePoolHasNoTargets
|
||||||
|
expr: sum(vm_promscrape_scrape_pool_targets) without (status, instance, pod) == 0
|
||||||
|
for: 30m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Vmagent has scrape_pool with 0 configured/discovered targets"
|
||||||
|
description: "Vmagent \"{{ $labels.job }}\" has scrape_pool \"{{ $labels.scrape_job }}\"
|
||||||
|
with 0 discovered targets. It is likely a misconfiguration. Please follow https://docs.victoriametrics.com/victoriametrics/vmagent/#debugging-scrape-targets
|
||||||
|
to troubleshoot the scraping config."
|
||||||
|
|
||||||
|
- alert: TooManyWriteErrors
|
||||||
|
expr: |
|
||||||
|
(sum(increase(vm_ingestserver_request_errors_total[5m])) without (name,net,type)
|
||||||
|
+
|
||||||
|
sum(increase(vmagent_http_request_errors_total[5m])) without (path,protocol)) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=77&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Vmagent responds with too many errors on data ingestion protocols"
|
||||||
|
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} responds with errors to write requests for last 15m."
|
||||||
|
|
||||||
|
- alert: TooManyRemoteWriteErrors
|
||||||
|
expr: rate(vmagent_remotewrite_retries_count_total[5m]) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=61&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} fails to push to remote storage"
|
||||||
|
description: "Vmagent fails to push data via remote write protocol to destination \"{{ $labels.url }}\"\n
|
||||||
|
Ensure that destination is up and reachable."
|
||||||
|
|
||||||
|
- alert: RemoteWriteConnectionIsSaturated
|
||||||
|
expr: |
|
||||||
|
(
|
||||||
|
rate(vmagent_remotewrite_send_duration_seconds_total[5m])
|
||||||
|
/
|
||||||
|
vmagent_remotewrite_queues
|
||||||
|
) > 0.9
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=84&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Remote write connection from \"{{ $labels.job }}\" (instance {{ $labels.instance }}) to {{ $labels.url }} is saturated"
|
||||||
|
description: "The remote write connection between vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }}) and destination \"{{ $labels.url }}\"
|
||||||
|
is saturated by more than 90% and vmagent won't be able to keep up.\n
|
||||||
|
There could be the following reasons for this:\n
|
||||||
|
* vmagent can't send data fast enough through the existing network connections. Increase `-remoteWrite.queues` cmd-line flag value to establish more connections per destination.\n
|
||||||
|
* remote destination can't accept data fast enough. Check if remote destination has enough resources for processing."
|
||||||
|
|
||||||
|
- alert: PersistentQueueForWritesIsSaturated
|
||||||
|
expr: rate(vm_persistentqueue_write_duration_seconds_total[5m]) > 0.9
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=98&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Persistent queue writes for instance {{ $labels.instance }} are saturated"
|
||||||
|
description: "Persistent queue writes for vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }})
|
||||||
|
are saturated by more than 90% and vmagent won't be able to keep up with flushing data on disk.
|
||||||
|
In this case, consider to decrease load on the vmagent or improve the disk throughput."
|
||||||
|
|
||||||
|
- alert: PersistentQueueForReadsIsSaturated
|
||||||
|
expr: rate(vm_persistentqueue_read_duration_seconds_total[5m]) > 0.9
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=99&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Persistent queue reads for instance {{ $labels.instance }} are saturated"
|
||||||
|
description: "Persistent queue reads for vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }})
|
||||||
|
are saturated by more than 90% and vmagent won't be able to keep up with reading data from the disk.
|
||||||
|
In this case, consider to decrease load on the vmagent or improve the disk throughput."
|
||||||
|
|
||||||
|
- alert: SeriesLimitHourReached
|
||||||
|
expr: (vmagent_hourly_series_limit_current_series / vmagent_hourly_series_limit_max_series) > 0.9
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=88&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Instance {{ $labels.instance }} reached 90% of the limit"
|
||||||
|
description: "Max series limit set via -remoteWrite.maxHourlySeries flag is close to reaching the max value.
|
||||||
|
Then samples for new time series will be dropped instead of sending them to remote storage systems."
|
||||||
|
|
||||||
|
- alert: SeriesLimitDayReached
|
||||||
|
expr: (vmagent_daily_series_limit_current_series / vmagent_daily_series_limit_max_series) > 0.9
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=90&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Instance {{ $labels.instance }} reached 90% of the limit"
|
||||||
|
description: "Max series limit set via -remoteWrite.maxDailySeries flag is close to reaching the max value.
|
||||||
|
Then samples for new time series will be dropped instead of sending them to remote storage systems."
|
||||||
|
|
||||||
|
- alert: ConfigurationReloadFailure
|
||||||
|
expr: |
|
||||||
|
vm_promscrape_config_last_reload_successful != 1
|
||||||
|
or
|
||||||
|
vmagent_relabel_config_last_reload_successful != 1
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Configuration reload failed for vmagent instance {{ $labels.instance }}"
|
||||||
|
description: "Configuration hot-reload failed for vmagent on instance {{ $labels.instance }}.
|
||||||
|
Check vmagent's logs for detailed error message."
|
||||||
|
|
||||||
|
- alert: StreamAggrFlushTimeout
|
||||||
|
expr: |
|
||||||
|
increase(vm_streamaggr_flush_timeouts_total[5m]) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Streaming aggregation at \"{{ $labels.job }}\" (instance {{ $labels.instance }}) can't be finished within the configured aggregation interval."
|
||||||
|
description: "Stream aggregation process can't keep up with the load and might produce incorrect aggregation results. Check logs for more details.
|
||||||
|
Possible solutions: increase aggregation interval; aggregate smaller number of series; reduce samples' ingestion rate to stream aggregation."
|
||||||
|
|
||||||
|
- alert: StreamAggrDedupFlushTimeout
|
||||||
|
expr: |
|
||||||
|
increase(vm_streamaggr_dedup_flush_timeouts_total[5m]) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Deduplication \"{{ $labels.job }}\" (instance {{ $labels.instance }}) can't be finished within configured deduplication interval."
|
||||||
|
description: "Deduplication process can't keep up with the load and might produce incorrect results. Check docs https://docs.victoriametrics.com/victoriametrics/stream-aggregation/#deduplication and logs for more details.
|
||||||
|
Possible solutions: increase deduplication interval; deduplicate smaller number of series; reduce samples' ingestion rate."
|
||||||
@@ -0,0 +1,96 @@
|
|||||||
|
# File contains default list of alerts for vmalert service.
|
||||||
|
# The alerts below are just recommendations and may require some updates
|
||||||
|
# and threshold calibration according to every specific setup.
|
||||||
|
groups:
|
||||||
|
# Alerts group for vmalert assumes that Grafana dashboard
|
||||||
|
# https://grafana.com/grafana/dashboards/14950 is installed.
|
||||||
|
# Pls update the `dashboard` annotation according to your setup.
|
||||||
|
- name: vmalert
|
||||||
|
interval: 30s
|
||||||
|
rules:
|
||||||
|
- alert: ConfigurationReloadFailure
|
||||||
|
expr: vmalert_config_last_reload_successful != 1
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Configuration reload failed for vmalert instance {{ $labels.instance }}"
|
||||||
|
description: "Configuration hot-reload failed for vmalert on instance {{ $labels.instance }}.
|
||||||
|
Check vmalert's logs for detailed error message."
|
||||||
|
|
||||||
|
- alert: AlertingRulesError
|
||||||
|
expr: sum(increase(vmalert_alerting_rules_errors_total[5m])) without(id) > 0
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=13&var-instance={{ $labels.instance }}&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
|
||||||
|
summary: "Alerting rules are failing for vmalert instance {{ $labels.instance }}"
|
||||||
|
description: "Alerting rules execution is failing for \"{{ $labels.alertname }}\" from group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
|
||||||
|
Check vmalert's logs for detailed error message."
|
||||||
|
|
||||||
|
- alert: RecordingRulesError
|
||||||
|
expr: sum(increase(vmalert_recording_rules_errors_total[5m])) without(id) > 0
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=30&var-instance={{ $labels.instance }}&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
|
||||||
|
summary: "Recording rules are failing for vmalert instance {{ $labels.instance }}"
|
||||||
|
description: "Recording rules execution is failing for \"{{ $labels.recording }}\" from group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
|
||||||
|
Check vmalert's logs for detailed error message."
|
||||||
|
|
||||||
|
- alert: RecordingRulesNoData
|
||||||
|
expr: sum(vmalert_recording_rules_last_evaluation_samples) without(id) < 1
|
||||||
|
for: 30m
|
||||||
|
labels:
|
||||||
|
severity: info
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=33&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
|
||||||
|
summary: "Recording rule {{ $labels.recording }} ({{ $labels.group }}) produces no data"
|
||||||
|
description: "Recording rule \"{{ $labels.recording }}\" from group \"{{ $labels.group }}\ in file \"{{ $labels.file }}\"
|
||||||
|
produces 0 samples over the last 30min. It might be caused by a misconfiguration
|
||||||
|
or incorrect query expression."
|
||||||
|
|
||||||
|
- alert: TooManyMissedIterations
|
||||||
|
expr: increase(vmalert_iteration_missed_total[5m]) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "vmalert instance {{ $labels.instance }} is missing rules evaluations"
|
||||||
|
description: "vmalert instance {{ $labels.instance }} is missing rules evaluations for group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
|
||||||
|
The group evaluation time takes longer than the configured evaluation interval. This may result in missed
|
||||||
|
alerting notifications or recording rules samples. Try increasing evaluation interval or concurrency of
|
||||||
|
group \"{{ $labels.group }}\". See https://docs.victoriametrics.com/victoriametrics/vmalert/#groups.
|
||||||
|
If rule expressions are taking longer than expected, please see https://docs.victoriametrics.com/victoriametrics/troubleshooting/#slow-queries."
|
||||||
|
|
||||||
|
- alert: RemoteWriteErrors
|
||||||
|
expr: increase(vmalert_remotewrite_errors_total[5m]) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "vmalert instance {{ $labels.instance }} is failing to push metrics to remote write URL"
|
||||||
|
description: "vmalert instance {{ $labels.instance }} is failing to push metrics generated via alerting
|
||||||
|
or recording rules to the configured remote write URL. Check vmalert's logs for detailed error message."
|
||||||
|
|
||||||
|
- alert: RemoteWriteDroppingData
|
||||||
|
expr: increase(vmalert_remotewrite_dropped_rows_total[5m]) > 0
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "vmalert instance {{ $labels.instance }} is dropping data sent to remote write URL"
|
||||||
|
description: "vmalert instance {{ $labels.instance }} is failing to send results of alerting or recording rules
|
||||||
|
to the configured remote write URL. This may result into gaps in recording rules or alerts state.
|
||||||
|
Check vmalert's logs for detailed error message."
|
||||||
|
|
||||||
|
- alert: AlertmanagerErrors
|
||||||
|
expr: increase(vmalert_alerts_send_errors_total[5m]) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "vmalert instance {{ $labels.instance }} is failing to send notifications to Alertmanager"
|
||||||
|
description: "vmalert instance {{ $labels.instance }} is failing to send alert notifications to \"{{ $labels.addr }}\".
|
||||||
|
Check vmalert's logs for detailed error message."
|
||||||
@@ -0,0 +1,138 @@
|
|||||||
|
# File contains default list of alerts for VictoriaMetrics single server.
|
||||||
|
# The alerts below are just recommendations and may require some updates
|
||||||
|
# and threshold calibration according to every specific setup.
|
||||||
|
groups:
|
||||||
|
# Alerts group for VM single assumes that Grafana dashboard
|
||||||
|
# https://grafana.com/grafana/dashboards/10229 is installed.
|
||||||
|
# Pls update the `dashboard` annotation according to your setup.
|
||||||
|
- name: vmsingle
|
||||||
|
interval: 30s
|
||||||
|
concurrency: 2
|
||||||
|
rules:
|
||||||
|
- alert: DiskRunsOutOfSpaceIn3Days
|
||||||
|
expr: |
|
||||||
|
sum(vm_free_disk_space_bytes) without(path) /
|
||||||
|
(
|
||||||
|
(rate(vm_rows_added_to_storage_total[1d]) - sum(rate(vm_deduplicated_samples_total[1d])) without(type)) * (
|
||||||
|
sum(vm_data_size_bytes{type!~"indexdb.*"}) without(type) /
|
||||||
|
sum(vm_rows{type!~"indexdb.*"}) without(type)
|
||||||
|
)
|
||||||
|
+
|
||||||
|
rate(vm_new_timeseries_created_total[1d]) * (
|
||||||
|
sum(vm_data_size_bytes{type="indexdb/file"}) without(type)/
|
||||||
|
sum(vm_rows{type="indexdb/file"}) without(type)
|
||||||
|
)
|
||||||
|
) < 3 * 24 * 3600 > 0
|
||||||
|
for: 30m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=53&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Instance {{ $labels.instance }} will run out of disk space soon"
|
||||||
|
description: "Taking into account current ingestion rate, free disk space will be enough only
|
||||||
|
for {{ $value | humanizeDuration }} on instance {{ $labels.instance }}.\n
|
||||||
|
Consider to limit the ingestion rate, decrease retention or scale the disk space if possible."
|
||||||
|
|
||||||
|
- alert: NodeBecomesReadonlyIn3Days
|
||||||
|
expr: |
|
||||||
|
sum(vm_free_disk_space_bytes - vm_free_disk_space_limit_bytes) without(path) /
|
||||||
|
(
|
||||||
|
(rate(vm_rows_added_to_storage_total[1d]) - sum(rate(vm_deduplicated_samples_total[1d])) without(type)) * (
|
||||||
|
sum(vm_data_size_bytes{type!~"indexdb.*"}) without(type) /
|
||||||
|
sum(vm_rows{type!~"indexdb.*"}) without(type)
|
||||||
|
)
|
||||||
|
+
|
||||||
|
rate(vm_new_timeseries_created_total[1d]) * (
|
||||||
|
sum(vm_data_size_bytes{type="indexdb/file"}) without(type) /
|
||||||
|
sum(vm_rows{type="indexdb/file"}) without(type)
|
||||||
|
)
|
||||||
|
) < 3 * 24 * 3600 > 0
|
||||||
|
for: 30m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/oS7Bi_0Wz?viewPanel=53&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Instance {{ $labels.instance }} will become read-only in 3 days"
|
||||||
|
description: "Taking into account current ingestion rate and free disk space
|
||||||
|
instance {{ $labels.instance }} is writable for {{ $value | humanizeDuration }}.\n
|
||||||
|
Consider to limit the ingestion rate, decrease retention or scale the disk space up if possible."
|
||||||
|
|
||||||
|
- alert: DiskRunsOutOfSpace
|
||||||
|
expr: |
|
||||||
|
sum(vm_data_size_bytes) by(job, instance) /
|
||||||
|
(
|
||||||
|
sum(vm_free_disk_space_bytes) by(job, instance) +
|
||||||
|
sum(vm_data_size_bytes) by(job, instance)
|
||||||
|
) > 0.8
|
||||||
|
for: 30m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=53&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Instance {{ $labels.instance }} (job={{ $labels.job }}) will run out of disk space soon"
|
||||||
|
description: "Disk utilisation on instance {{ $labels.instance }} is more than 80%.\n
|
||||||
|
Having less than 20% of free disk space could cripple merge processes and overall performance.
|
||||||
|
Consider to limit the ingestion rate, decrease retention or scale the disk space if possible."
|
||||||
|
|
||||||
|
- alert: RequestErrorsToAPI
|
||||||
|
expr: increase(vm_http_request_errors_total[5m]) > 0
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=35&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Too many errors served for path {{ $labels.path }} (instance {{ $labels.instance }})"
|
||||||
|
description: "Requests to path {{ $labels.path }} are receiving errors.
|
||||||
|
Please verify if clients are sending correct requests."
|
||||||
|
|
||||||
|
- alert: TooHighChurnRate
|
||||||
|
expr: |
|
||||||
|
(
|
||||||
|
sum(rate(vm_new_timeseries_created_total[5m])) by(instance)
|
||||||
|
/
|
||||||
|
sum(rate(vm_rows_inserted_total[5m])) by(instance)
|
||||||
|
) > 0.1
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=66&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Churn rate is more than 10% on \"{{ $labels.instance }}\" for the last 15m"
|
||||||
|
description: "VM constantly creates new time series on \"{{ $labels.instance }}\".\n
|
||||||
|
This effect is known as Churn Rate.\n
|
||||||
|
High Churn Rate is tightly connected with database performance and may
|
||||||
|
result in unexpected OOM's or slow queries."
|
||||||
|
|
||||||
|
- alert: TooHighChurnRate24h
|
||||||
|
expr: |
|
||||||
|
sum(increase(vm_new_timeseries_created_total[24h])) by(instance)
|
||||||
|
>
|
||||||
|
(sum(vm_cache_entries{type="storage/hour_metric_ids"}) by(instance) * 3)
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=66&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Too high number of new series on \"{{ $labels.instance }}\" created over last 24h"
|
||||||
|
description: "The number of created new time series over last 24h is 3x times higher than
|
||||||
|
current number of active series on \"{{ $labels.instance }}\".\n
|
||||||
|
This effect is known as Churn Rate.\n
|
||||||
|
High Churn Rate is tightly connected with database performance and may
|
||||||
|
result in unexpected OOM's or slow queries."
|
||||||
|
|
||||||
|
- alert: TooHighSlowInsertsRate
|
||||||
|
expr: |
|
||||||
|
(
|
||||||
|
sum(rate(vm_slow_row_inserts_total[5m])) by(instance)
|
||||||
|
/
|
||||||
|
sum(rate(vm_rows_inserted_total[5m])) by(instance)
|
||||||
|
) > 0.05
|
||||||
|
for: 15m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=68&var-instance={{ $labels.instance }}"
|
||||||
|
summary: "Percentage of slow inserts is more than 5% on \"{{ $labels.instance }}\" for the last 15m"
|
||||||
|
description: "High rate of slow inserts on \"{{ $labels.instance }}\" may be a sign of resource exhaustion
|
||||||
|
for the current load. It is likely more RAM is needed for optimal handling of the current number of active time series.
|
||||||
|
See also https://github.com/VictoriaMetrics/VictoriaMetrics/issues/3976#issuecomment-1476883183"
|
||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+14
-80
@@ -1,32 +1,15 @@
|
|||||||
# zot OCI Registry
|
# zot OCI Registry
|
||||||
|
|
||||||
内网匿名拉取入口为 `https://zot.ad.ddupan.top`,SPIRE 鉴权推送入口为
|
内网入口为 `https://zot.ad.ddupan.top`。使用官方 Helm chart `0.1.124`,运行
|
||||||
`https://zot-push.ad.ddupan.top`。使用官方 Helm chart `0.1.124`,运行
|
|
||||||
zot `v2.1.21`,镜像固定到官方 linux/amd64 digest。
|
zot `v2.1.21`,镜像固定到官方 linux/amd64 digest。
|
||||||
|
|
||||||
## 当前工作状态(2026-09-16 核验)
|
|
||||||
|
|
||||||
| 项目 | 状态 |
|
|
||||||
|---|---|
|
|
||||||
| 匿名拉取 | `zot.ad.ddupan.top` 已上线;空 `DOCKER_CONFIG` 的 crane pull 通过 |
|
|
||||||
| SPIRE 鉴权入口 | `zot-push.ad.ddupan.top` 已上线;真实 JWT-SVID 推送后可匿名拉取同一 digest |
|
|
||||||
| GitOps | 双入口配置已合并;Flux `zot` Kustomization 已应用 `d15733c`,状态 Ready |
|
|
||||||
| 运行与凭据同步 | `zot`、`zot-reader` HelmRelease 均 Ready,Pod 均 1/1;ESO SecretSynced |
|
|
||||||
| 临时配置清理 | 两个 HelmRelease 均无 `spec.values` 临时覆盖;暂停回写标记、测试身份和临时写权限已清理 |
|
|
||||||
| 接管复验 | 匿名拉取成功;推送入口无凭据返回 401,token realm 指向推送域名;接管未触发 Pod 重启 |
|
|
||||||
|
|
||||||
后续工作是给实际 CI 的 SPIFFE ID 配置具体仓库的 `create`/`update` 权限。
|
|
||||||
SPIRE 认证链路已经验证,但当前没有常驻 publisher 或删除授权;认证成功本身不代表
|
|
||||||
可以推送。S3 侧仍使用 Bao 管理的静态 AK/SK,尚未接入 SPIRE/STS。
|
|
||||||
|
|
||||||
## 存储与凭据
|
## 存储与凭据
|
||||||
|
|
||||||
制品、manifest 和 OCI layout 保存在现有 SeaweedFS 的 `zot` bucket,前缀为
|
制品、manifest 和 OCI layout 保存在现有 SeaweedFS 的 `zot` bucket,前缀为
|
||||||
`registry/`,S3 endpoint 为 `https://s3.ad.ddupan.top`。**不创建 PVC**;chart 的
|
`registry/`,S3 endpoint 为 `https://s3.ad.ddupan.top`。**不创建 PVC**;chart 的
|
||||||
`/var/lib/registry` 是 `emptyDir`,仅用于运行时本地工作数据。
|
`/var/lib/registry` 是 `emptyDir`,仅用于运行时本地工作数据。
|
||||||
|
|
||||||
两个单副本实例共用同一 bucket 和前缀:`zot` 负责鉴权写入,`zot-reader` 负责匿名
|
首期单副本,关闭跨仓库 dedupe,不额外部署 Redis/DynamoDB 缓存。保留 zot GC,
|
||||||
读取。关闭跨仓库 dedupe,不额外部署 Redis/DynamoDB 缓存。只有写入实例启用 GC,
|
|
||||||
暂不配置自动删除已发布版本的 retention policy。增加副本、启用 dedupe 或搜索等
|
暂不配置自动删除已发布版本的 retention policy。增加副本、启用 dedupe 或搜索等
|
||||||
扩展前,需要重新检查共享元数据与缓存的持久化要求。
|
扩展前,需要重新检查共享元数据与缓存的持久化要求。
|
||||||
|
|
||||||
@@ -37,7 +20,7 @@ OpenBao kv/k8s/seaweedfs-s3
|
|||||||
→ 原有 S3 身份及基础配置 ─┐
|
→ 原有 S3 身份及基础配置 ─┐
|
||||||
├→ ESO 模板 → seaweedfs/seaweedfs-s3-config
|
├→ ESO 模板 → seaweedfs/seaweedfs-s3-config
|
||||||
OpenBao kv/k8s/zot-s3 ────┘ → 完整 s3.config
|
OpenBao kv/k8s/zot-s3 ────┘ → 完整 s3.config
|
||||||
└→ ESO → zot/zot-s3 → zot 与 zot-reader 的 AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY
|
└→ ESO → zot/zot-s3 → zot 的 AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY
|
||||||
|
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -54,7 +37,7 @@ Terraform 的 `tfstate` bucket。`kv/k8s/zot-s3` 的字段是 `access_key` 和
|
|||||||
本次归一没有轮换密钥,生成的完整配置与归一前语义一致。当前模板只有一组 zot
|
本次归一没有轮换密钥,生成的完整配置与归一前语义一致。当前模板只有一组 zot
|
||||||
凭据,尚未实现新旧密钥重叠轮换。后续轮换只修改 `kv/k8s/zot-s3`,但仍需协调
|
凭据,尚未实现新旧密钥重叠轮换。后续轮换只修改 `kv/k8s/zot-s3`,但仍需协调
|
||||||
两个 ExternalSecret 同步:确认 SeaweedFS Secret volume 更新后向 filer 的
|
两个 ExternalSecret 同步:确认 SeaweedFS Secret volume 更新后向 filer 的
|
||||||
`weed` 进程发送 SIGHUP,再确认 zot Secret 更新并重启 `zot` 和 `zot-reader`(环境变量不会热更新)。
|
`weed` 进程发送 SIGHUP,再确认 zot Secret 更新并重启 zot(环境变量不会热更新)。
|
||||||
两端异步更新期间可能短暂认证失败;需要无中断轮换时先扩展模板支持新旧凭据重叠。
|
两端异步更新期间可能短暂认证失败;需要无中断轮换时先扩展模板支持新旧凭据重叠。
|
||||||
|
|
||||||
## SPIRE 认证和授权
|
## SPIRE 认证和授权
|
||||||
@@ -64,40 +47,24 @@ Terraform 的 `tfstate` bucket。`kv/k8s/zot-s3` 的字段是 `access_key` 和
|
|||||||
| issuer | `https://spire-oidc.ad.ddupan.top` |
|
| issuer | `https://spire-oidc.ad.ddupan.top` |
|
||||||
| JWT audience | `zot` |
|
| JWT audience | `zot` |
|
||||||
| subject | `spiffe://ddupan.top/` 下的 workload SPIFFE ID |
|
| subject | `spiffe://ddupan.top/` 下的 workload SPIFFE ID |
|
||||||
| token endpoint | `https://zot-push.ad.ddupan.top/zot/auth/token` |
|
| token endpoint | `https://zot.ad.ddupan.top/zot/auth/token` |
|
||||||
| 拉取入口 | 内网匿名读取所有仓库,不要求 SPIRE 身份 |
|
| 当前权限 | 受信身份可以读取所有仓库;没有常驻写入或删除授权 |
|
||||||
| 推送入口当前权限 | 受信身份可以读取所有仓库;没有常驻写入或删除授权 |
|
|
||||||
|
|
||||||
zot 通过已配置的 issuer discovery/JWKS 验证 JWT-SVID,再以 `sub` 作为授权身份。
|
zot 通过已配置的 issuer discovery/JWKS 验证 JWT-SVID,再以 `sub` 作为授权身份。
|
||||||
不接受任意 issuer,不关闭 TLS/issuer 验证。新的 Kata CI 负责取得并更新自己的
|
不接受任意 issuer,不关闭 TLS/issuer 验证。新的 Kata CI 负责取得并更新自己的
|
||||||
JWT-SVID;确认其身份命名后,再添加针对具体 repository 的 `create`/`update`
|
JWT-SVID;确认其身份命名后,再添加针对具体 repository 的 `create`/`update`
|
||||||
授权,不能把整个 trust domain 都授予写权限。
|
授权,不能把整个 trust domain 都授予写权限。
|
||||||
|
|
||||||
zot `v2.1.21` 的 OIDC Bearer middleware 会在授权阶段之前拒绝无 token 请求。
|
现阶段拉取也需要 JWT-SVID。原定内网匿名拉取尚未启用:zot `v2.1.21` 的
|
||||||
因此使用两个官方 zot 实例与两个域名,避免修改上游镜像,也避免同域名下匿名
|
OIDC Bearer middleware 会在授权阶段之前拒绝无 token 请求,单独增加
|
||||||
`/v2/` 返回 200 导致标准客户端跳过 token 交换的问题。
|
`anonymousPolicy` 无法解决。匿名读取与 SPIRE 写入共存需后续单独验证方案。
|
||||||
|
|
||||||
- `zot-reader` 叠加 `reader-values.yaml`,没有认证 middleware,只有
|
|
||||||
`anonymousPolicy: [read]`。入口只转发 `/v2/` 的 GET/HEAD,并移除客户端遗留的
|
|
||||||
Authorization/Cookie;直接访问 reader Service 也不能写入。
|
|
||||||
- `zot` 保留 SPIRE issuer/audience/subject 校验及仓库授权,`externalUrl`、
|
|
||||||
Bearer realm、service 与 HTTPRoute 均使用 `zot-push.ad.ddupan.top`。
|
|
||||||
- reader 关闭 GC,没有同步或扫描扩展;读取同一份 S3 制品,不复制 bucket,
|
|
||||||
不新增 PVC 或 S3 密钥。镜像、安全上下文、资源和 Secret 引用由共用 values 继承。
|
|
||||||
- 两个配置的 `storageDriver` 必须保持一致;修改 S3 endpoint/bucket/prefix 时
|
|
||||||
同时更新 `values.yaml` 与 `reader-values.yaml`。
|
|
||||||
|
|
||||||
推送客户端应登录 `zot-push.ad.ddupan.top`;拉取客户端无需登录。
|
|
||||||
已有 SPIRE 身份的进程可以通过 Workload API 获取 `aud=zot` 的 JWT-SVID,然后
|
已有 SPIRE 身份的进程可以通过 Workload API 获取 `aud=zot` 的 JWT-SVID,然后
|
||||||
通过 `docker login` 或 `crane auth login` 的 `--password-stdin` 交给 Registry。
|
通过 `docker login` 或 `crane auth login` 的 `--password-stdin` 交给 Registry。
|
||||||
用户名可以使用 `zot`,实际权限取自已验证 JWT 的身份。使用独立、权限为 `0700`
|
用户名可以使用 `zot`,实际权限取自已验证 JWT 的身份。使用独立、权限为 `0700`
|
||||||
的临时 `DOCKER_CONFIG`,结束后删除;不要开启 shell tracing,不要打印 token,
|
的临时 `DOCKER_CONFIG`,结束后删除;不要开启 shell tracing,不要打印 token,
|
||||||
不要把 token 放进命令参数。token 接口不会延长 SVID 有效期。
|
不要把 token 放进命令参数。token 接口不会延长 SVID 有效期。
|
||||||
|
|
||||||
同一仓库在两个入口使用相同路径和 tag/digest,例如 CI 推送到
|
|
||||||
`zot-push.ad.ddupan.top/team/image:tag`,部署时使用
|
|
||||||
`zot.ad.ddupan.top/team/image:tag`;无需在两个仓库间复制。
|
|
||||||
|
|
||||||
## 部署与网络
|
## 部署与网络
|
||||||
|
|
||||||
- 官方 chart 管理 Deployment、Service、ConfigMap 和 HTTPRoute。
|
- 官方 chart 管理 Deployment、Service、ConfigMap 和 HTTPRoute。
|
||||||
@@ -105,17 +72,13 @@ zot `v2.1.21` 的 OIDC Bearer middleware 会在授权阶段之前拒绝无 token
|
|||||||
`https` listener 与内网通配符证书终止。
|
`https` listener 与内网通配符证书终止。
|
||||||
- 仅配置 Samba AD 内网 DNS;不创建公网 DNS 或 Cloudflare Tunnel route。
|
- 仅配置 Samba AD 内网 DNS;不创建公网 DNS 或 Cloudflare Tunnel route。
|
||||||
- NetworkPolicy 只允许现有 Envoy Gateway 数据面访问 zot 的 5000 端口。
|
- NetworkPolicy 只允许现有 Envoy Gateway 数据面访问 zot 的 5000 端口。
|
||||||
- 拉取域名仅暴露 `/v2/` 的 GET/HEAD;推送域名暴露 `/v2/` 和
|
- HTTPRoute 只暴露 `/v2/` 和 `/zot/auth/token`,不暴露内部健康检查或管理端点。
|
||||||
`/zot/auth/token`,均不暴露内部健康检查或管理端点。
|
|
||||||
- namespace 使用 restricted PodSecurity,容器非 root、只读根文件系统。
|
- namespace 使用 restricted PodSecurity,容器非 root、只读根文件系统。
|
||||||
|
|
||||||
`clusters/homelab/apps/zot.yaml` 已将 `zot` 和 `zot-reader` 一并纳入 Flux 管理。
|
首次已按用户授权从本地执行 `kubectl apply -k apps/zot`,由集群 Helm controller
|
||||||
两个 HelmRelease 通过共用 `zot-values` 继承基础配置,reader 再叠加
|
安装。`clusters/homelab/apps/zot.yaml` 是 GitOps composition;对应文件合并进入
|
||||||
`zot-reader-values`。当前由 main 分支持续管理,不依赖本地覆盖或暂停回写。
|
Flux 跟踪分支后,才由根 Kustomization 持续管理,不能把未提交的本地部署写成
|
||||||
|
已完成 Git 接管。
|
||||||
后续若需临时验收,收尾时先确认 Git 管理的配置与目标运行配置一致,再移除
|
|
||||||
`spec.values` 临时覆盖及 `kustomize.toolkit.fluxcd.io/reconcile=disabled` 标记,
|
|
||||||
触发 zot Kustomization reconcile 并复验。临时测试身份和写权限不得留在持久配置中。
|
|
||||||
|
|
||||||
检查与渲染:
|
检查与渲染:
|
||||||
|
|
||||||
@@ -151,35 +114,6 @@ sha256:b8d3b977a1235022759470903dab4a46b7cf8107958624f1f76a323eabe37c5e
|
|||||||
|
|
||||||
它是仅含验证文本的 OCI 测试镜像,没有可执行入口,不用于运行服务。
|
它是仅含验证文本的 OCI 测试镜像,没有可执行入口,不用于运行服务。
|
||||||
|
|
||||||
双域名验收还使用 `verification/anonymous-spire:smoke`:标准 crane 从
|
|
||||||
`zot-push.ad.ddupan.top` 登录、推送,再从 `zot.ad.ddupan.top` 使用空
|
|
||||||
`DOCKER_CONFIG` 拉取,两个入口的 digest 必须一致。验证匿名 blob HEAD、tags、
|
|
||||||
referrers,以及客户端保存旧凭据时的公共拉取。推送入口检查无凭据、错误签名、
|
|
||||||
错误 audience、过期 SVID、跨仓库写入和删除拒绝;公共入口拒绝所有写方法,
|
|
||||||
reader Service 直连也拒绝写入。测试完成后撤回临时单仓库写权限。
|
|
||||||
|
|
||||||
2026-09-16 上述双域名验收通过;SVID 过期后推送入口返回 401,匿名拉取不受
|
|
||||||
影响。临时写权限已撤销,两个 HelmRelease Ready;推送 DNS 第二次检查 changed=0。
|
|
||||||
|
|
||||||
匿名拉取示例:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
crane pull zot.ad.ddupan.top/verification/anonymous-spire:smoke image.tar --format oci
|
|
||||||
```
|
|
||||||
|
|
||||||
鉴权推送示例(先通过 Workload API 将短期 JWT-SVID 保存到当前进程的 `ZOT_JWT`,
|
|
||||||
不要启用 shell tracing;示例中的仓库仍需提前给具体 SPIFFE ID 授权):
|
|
||||||
|
|
||||||
```bash
|
|
||||||
export DOCKER_CONFIG="$(mktemp -d)"
|
|
||||||
printf '%s' "$ZOT_JWT" | crane auth login zot-push.ad.ddupan.top \
|
|
||||||
--username zot --password-stdin
|
|
||||||
crane push image.tar zot-push.ad.ddupan.top/team/image:tag
|
|
||||||
rm -rf -- "$DOCKER_CONFIG"
|
|
||||||
unset DOCKER_CONFIG ZOT_JWT
|
|
||||||
```
|
|
||||||
|
|
||||||
|
|
||||||
Registry 恢复需要完整的 SeaweedFS bucket 数据、Bao 专用凭据和此目录配置。
|
Registry 恢复需要完整的 SeaweedFS bucket 数据、Bao 专用凭据和此目录配置。
|
||||||
zot 的临时目录不是制品备份。独立异机/离线备份尚未在本次部署中建立;不能把同一
|
zot 的临时目录不是制品备份。独立异机/离线备份尚未在本次部署中建立;不能把同一
|
||||||
SeaweedFS 内的数据副本当作独立灾备。重装 zot 不得删除 `zot` bucket。
|
SeaweedFS 内的数据副本当作独立灾备。重装 zot 不得删除 `zot` bucket。
|
||||||
|
|||||||
@@ -1,32 +0,0 @@
|
|||||||
apiVersion: helm.toolkit.fluxcd.io/v2
|
|
||||||
kind: HelmRelease
|
|
||||||
metadata:
|
|
||||||
name: zot-reader
|
|
||||||
namespace: zot
|
|
||||||
spec:
|
|
||||||
chart:
|
|
||||||
spec:
|
|
||||||
chart: zot
|
|
||||||
version: 0.1.124
|
|
||||||
interval: 1h
|
|
||||||
sourceRef:
|
|
||||||
kind: HelmRepository
|
|
||||||
name: zot
|
|
||||||
releaseName: zot-reader
|
|
||||||
interval: 30m
|
|
||||||
timeout: 5m
|
|
||||||
driftDetection:
|
|
||||||
mode: enabled
|
|
||||||
install:
|
|
||||||
strategy:
|
|
||||||
name: RetryOnFailure
|
|
||||||
retryInterval: 5m
|
|
||||||
upgrade:
|
|
||||||
strategy:
|
|
||||||
name: RetryOnFailure
|
|
||||||
retryInterval: 5m
|
|
||||||
valuesFrom:
|
|
||||||
- kind: ConfigMap
|
|
||||||
name: zot-values
|
|
||||||
- kind: ConfigMap
|
|
||||||
name: zot-reader-values
|
|
||||||
@@ -6,7 +6,6 @@ resources:
|
|||||||
- external-secret.yaml
|
- external-secret.yaml
|
||||||
- helmrepository.yaml
|
- helmrepository.yaml
|
||||||
- helmrelease.yaml
|
- helmrelease.yaml
|
||||||
- helmrelease-reader.yaml
|
|
||||||
- networkpolicy.yaml
|
- networkpolicy.yaml
|
||||||
generatorOptions:
|
generatorOptions:
|
||||||
disableNameSuffixHash: true
|
disableNameSuffixHash: true
|
||||||
@@ -17,7 +16,3 @@ configMapGenerator:
|
|||||||
namespace: zot
|
namespace: zot
|
||||||
files:
|
files:
|
||||||
- values.yaml=values.yaml
|
- values.yaml=values.yaml
|
||||||
- name: zot-reader-values
|
|
||||||
namespace: zot
|
|
||||||
files:
|
|
||||||
- values.yaml=reader-values.yaml
|
|
||||||
|
|||||||
@@ -1,64 +0,0 @@
|
|||||||
# 叠加于共用 values.yaml;同一镜像、S3、Secret、安全设置,无制品副本。
|
|
||||||
# 无 Bearer middleware,仅 anonymousPolicy=read;关闭 GC 避免多个实例清理共享存储。
|
|
||||||
configFiles:
|
|
||||||
config.json: |
|
|
||||||
{
|
|
||||||
"distSpecVersion": "1.1.1",
|
|
||||||
"storage": {
|
|
||||||
"rootDirectory": "/var/lib/registry",
|
|
||||||
"dedupe": false,
|
|
||||||
"gc": false,
|
|
||||||
"storageDriver": {
|
|
||||||
"name": "s3",
|
|
||||||
"region": "us-east-1",
|
|
||||||
"regionendpoint": "https://s3.ad.ddupan.top",
|
|
||||||
"bucket": "zot",
|
|
||||||
"rootdirectory": "/registry",
|
|
||||||
"secure": true,
|
|
||||||
"skipverify": false,
|
|
||||||
"forcepathstyle": true
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"http": {
|
|
||||||
"address": "0.0.0.0",
|
|
||||||
"port": "5000",
|
|
||||||
"externalUrl": "https://zot.ad.ddupan.top",
|
|
||||||
"compat": [
|
|
||||||
"docker2s2"
|
|
||||||
],
|
|
||||||
"accessControl": {
|
|
||||||
"repositories": {
|
|
||||||
"**": {
|
|
||||||
"anonymousPolicy": [
|
|
||||||
"read"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"log": {
|
|
||||||
"level": "info"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
httproute:
|
|
||||||
hostnames:
|
|
||||||
- zot.ad.ddupan.top
|
|
||||||
rules:
|
|
||||||
- matches:
|
|
||||||
- path:
|
|
||||||
type: PathPrefix
|
|
||||||
value: /v2/
|
|
||||||
method: GET
|
|
||||||
- path:
|
|
||||||
type: PathPrefix
|
|
||||||
value: /v2/
|
|
||||||
method: HEAD
|
|
||||||
filters:
|
|
||||||
- type: RequestHeaderModifier
|
|
||||||
requestHeaderModifier:
|
|
||||||
remove:
|
|
||||||
- Cookie
|
|
||||||
- Authorization
|
|
||||||
timeouts:
|
|
||||||
request: 900s
|
|
||||||
backendRequest: 900s
|
|
||||||
+4
-51
@@ -41,14 +41,14 @@ configFiles:
|
|||||||
"http": {
|
"http": {
|
||||||
"address": "0.0.0.0",
|
"address": "0.0.0.0",
|
||||||
"port": "5000",
|
"port": "5000",
|
||||||
"externalUrl": "https://zot-push.ad.ddupan.top",
|
"externalUrl": "https://zot.ad.ddupan.top",
|
||||||
"compat": [
|
"compat": [
|
||||||
"docker2s2"
|
"docker2s2"
|
||||||
],
|
],
|
||||||
"auth": {
|
"auth": {
|
||||||
"bearer": {
|
"bearer": {
|
||||||
"realm": "https://zot-push.ad.ddupan.top/zot/auth/token",
|
"realm": "https://zot.ad.ddupan.top/zot/auth/token",
|
||||||
"service": "zot-push.ad.ddupan.top",
|
"service": "zot.ad.ddupan.top",
|
||||||
"oidc": [
|
"oidc": [
|
||||||
{
|
{
|
||||||
"issuer": "https://spire-oidc.ad.ddupan.top",
|
"issuer": "https://spire-oidc.ad.ddupan.top",
|
||||||
@@ -70,54 +70,7 @@ configFiles:
|
|||||||
},
|
},
|
||||||
"accessControl": {
|
"accessControl": {
|
||||||
"repositories": {
|
"repositories": {
|
||||||
"panxiao81/gitea-dynamic-runner-controller": {
|
|
||||||
"policies": [
|
|
||||||
{
|
|
||||||
"users": [
|
|
||||||
"spiffe://ddupan.top/ci/panxiao81/gitea-dynamic-runner/publish-images"
|
|
||||||
],
|
|
||||||
"actions": [
|
|
||||||
"read",
|
|
||||||
"create",
|
|
||||||
"update"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"defaultPolicy": [
|
|
||||||
"read"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"panxiao81/gitea-dynamic-runner-runner": {
|
|
||||||
"policies": [
|
|
||||||
{
|
|
||||||
"users": [
|
|
||||||
"spiffe://ddupan.top/ci/panxiao81/gitea-dynamic-runner/publish-images"
|
|
||||||
],
|
|
||||||
"actions": [
|
|
||||||
"read",
|
|
||||||
"create",
|
|
||||||
"update"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"defaultPolicy": [
|
|
||||||
"read"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"**": {
|
"**": {
|
||||||
"policies": [
|
|
||||||
{
|
|
||||||
"users": [
|
|
||||||
"spiffe://ddupan.top/dev/panxiao81"
|
|
||||||
],
|
|
||||||
"actions": [
|
|
||||||
"read",
|
|
||||||
"create",
|
|
||||||
"update",
|
|
||||||
"delete"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"defaultPolicy": [
|
"defaultPolicy": [
|
||||||
"read"
|
"read"
|
||||||
]
|
]
|
||||||
@@ -180,7 +133,7 @@ httproute:
|
|||||||
namespace: envoy-gateway-system
|
namespace: envoy-gateway-system
|
||||||
sectionName: https
|
sectionName: https
|
||||||
hostnames:
|
hostnames:
|
||||||
- zot-push.ad.ddupan.top
|
- zot.ad.ddupan.top
|
||||||
rules:
|
rules:
|
||||||
- matches:
|
- matches:
|
||||||
- path:
|
- path:
|
||||||
|
|||||||
@@ -1,18 +0,0 @@
|
|||||||
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
|
||||||
kind: Kustomization
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner
|
|
||||||
namespace: flux-system
|
|
||||||
spec:
|
|
||||||
dependsOn:
|
|
||||||
- name: external-secrets
|
|
||||||
- name: nats
|
|
||||||
- name: spire
|
|
||||||
interval: 10m
|
|
||||||
path: ./platform/dynamic-runner
|
|
||||||
prune: false
|
|
||||||
sourceRef:
|
|
||||||
kind: GitRepository
|
|
||||||
name: flux-system
|
|
||||||
timeout: 5m
|
|
||||||
wait: true
|
|
||||||
@@ -1,18 +0,0 @@
|
|||||||
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
|
||||||
kind: Kustomization
|
|
||||||
metadata:
|
|
||||||
name: nats
|
|
||||||
namespace: flux-system
|
|
||||||
spec:
|
|
||||||
dependsOn:
|
|
||||||
- name: cert-manager
|
|
||||||
- name: external-secrets
|
|
||||||
- name: openebs
|
|
||||||
interval: 10m
|
|
||||||
path: ./platform/nats
|
|
||||||
prune: false
|
|
||||||
sourceRef:
|
|
||||||
kind: GitRepository
|
|
||||||
name: flux-system
|
|
||||||
timeout: 10m
|
|
||||||
wait: true
|
|
||||||
@@ -10,8 +10,6 @@ resources:
|
|||||||
- apps/gitea-actions.yaml
|
- apps/gitea-actions.yaml
|
||||||
- apps/http-echo.yaml
|
- apps/http-echo.yaml
|
||||||
- apps/openebs.yaml
|
- apps/openebs.yaml
|
||||||
- apps/nats.yaml
|
|
||||||
- apps/dynamic-runner.yaml
|
|
||||||
- apps/spire.yaml
|
- apps/spire.yaml
|
||||||
- apps/observability.yaml
|
- apps/observability.yaml
|
||||||
- apps/zot.yaml
|
- apps/zot.yaml
|
||||||
|
|||||||
@@ -11,13 +11,10 @@ homelab_dns:
|
|||||||
- { zone: ad.ddupan.top, name: pve2, type: A, values: [192.168.10.7] }
|
- { zone: ad.ddupan.top, name: pve2, type: A, values: [192.168.10.7] }
|
||||||
- { zone: ad.ddupan.top, name: pve3, type: A, values: [192.168.10.9] }
|
- { zone: ad.ddupan.top, name: pve3, type: A, values: [192.168.10.9] }
|
||||||
- { zone: ad.ddupan.top, name: retrolab, type: A, values: [10.60.0.10] }
|
- { zone: ad.ddupan.top, name: retrolab, type: A, values: [10.60.0.10] }
|
||||||
- { zone: ad.ddupan.top, name: grafana, type: A, values: [192.168.10.127] }
|
|
||||||
- { zone: ad.ddupan.top, name: netbox, type: A, values: [192.168.10.127] }
|
- { zone: ad.ddupan.top, name: netbox, type: A, values: [192.168.10.127] }
|
||||||
- { zone: ad.ddupan.top, name: nats, type: A, values: [192.168.10.127] }
|
|
||||||
- { zone: ad.ddupan.top, name: s3, type: A, values: [192.168.10.127] }
|
- { zone: ad.ddupan.top, name: s3, type: A, values: [192.168.10.127] }
|
||||||
- { zone: ad.ddupan.top, name: spire-oidc, type: A, values: [192.168.10.127] }
|
- { zone: ad.ddupan.top, name: spire-oidc, type: A, values: [192.168.10.127] }
|
||||||
- { zone: ad.ddupan.top, name: zot, type: A, values: [192.168.10.127] }
|
- { zone: ad.ddupan.top, name: zot, type: A, values: [192.168.10.127] }
|
||||||
- { zone: ad.ddupan.top, name: zot-push, type: A, values: [192.168.10.127] }
|
|
||||||
|
|
||||||
split_horizon:
|
split_horizon:
|
||||||
# LAN and pod resolvers should eventually render the same set from here.
|
# LAN and pod resolvers should eventually render the same set from here.
|
||||||
|
|||||||
@@ -25,29 +25,3 @@ resource "vault_jwt_auth_backend_role" "spire_poc" {
|
|||||||
token_ttl = 300
|
token_ttl = 300
|
||||||
token_max_ttl = 900
|
token_max_ttl = 900
|
||||||
}
|
}
|
||||||
|
|
||||||
# Local development on the laptop. Keep the subject exact: possession of any
|
|
||||||
# other identity in the trust domain must not grant interactive host access.
|
|
||||||
resource "vault_jwt_auth_backend_role" "local_development" {
|
|
||||||
backend = vault_jwt_auth_backend.spire.path
|
|
||||||
role_name = "local-development"
|
|
||||||
role_type = "jwt"
|
|
||||||
|
|
||||||
user_claim = "sub"
|
|
||||||
bound_audiences = ["openbao"]
|
|
||||||
bound_claims = {
|
|
||||||
sub = "spiffe://ddupan.top/dev/panxiao81"
|
|
||||||
}
|
|
||||||
|
|
||||||
# local-development grants normal KV v2 read/write under kv/k8s and kv/infra,
|
|
||||||
# plus short-lived SSH certificate signing.
|
|
||||||
# certificate signing; spire-poc only permits lookup and revocation of the
|
|
||||||
# caller's own short-lived Bao token.
|
|
||||||
token_policies = [
|
|
||||||
vault_policy.local_development.name,
|
|
||||||
vault_policy.spire_poc.name,
|
|
||||||
]
|
|
||||||
token_no_default_policy = true
|
|
||||||
token_ttl = 300
|
|
||||||
token_max_ttl = 900
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -14,11 +14,6 @@ resource "vault_policy" "ai_agent_ssh" {
|
|||||||
policy = file("${path.module}/policies/ai-agent-ssh.hcl")
|
policy = file("${path.module}/policies/ai-agent-ssh.hcl")
|
||||||
}
|
}
|
||||||
|
|
||||||
resource "vault_policy" "local_development" {
|
|
||||||
name = "local-development"
|
|
||||||
policy = file("${path.module}/policies/local-development.hcl")
|
|
||||||
}
|
|
||||||
|
|
||||||
resource "vault_policy" "spire_poc" {
|
resource "vault_policy" "spire_poc" {
|
||||||
name = "spire-poc"
|
name = "spire-poc"
|
||||||
policy = file("${path.module}/policies/spire-poc.hcl")
|
policy = file("${path.module}/policies/spire-poc.hcl")
|
||||||
|
|||||||
@@ -1,32 +0,0 @@
|
|||||||
# Local development identity. Limit normal KV v2 reads and writes to the k8s
|
|
||||||
# subtree, and exclude soft-delete, metadata deletion, permanent version
|
|
||||||
# destruction, auth administration, and privileged operations.
|
|
||||||
path "kv/data/k8s/*" {
|
|
||||||
capabilities = ["create", "read", "update"]
|
|
||||||
}
|
|
||||||
|
|
||||||
path "kv/metadata/k8s" {
|
|
||||||
capabilities = ["read", "list"]
|
|
||||||
}
|
|
||||||
|
|
||||||
path "kv/metadata/k8s/*" {
|
|
||||||
capabilities = ["read", "list"]
|
|
||||||
}
|
|
||||||
|
|
||||||
# Future destination for infrastructure secrets migrated from Ansible Vault.
|
|
||||||
path "kv/data/infra/*" {
|
|
||||||
capabilities = ["create", "read", "update"]
|
|
||||||
}
|
|
||||||
|
|
||||||
path "kv/metadata/infra" {
|
|
||||||
capabilities = ["read", "list"]
|
|
||||||
}
|
|
||||||
|
|
||||||
path "kv/metadata/infra/*" {
|
|
||||||
capabilities = ["read", "list"]
|
|
||||||
}
|
|
||||||
|
|
||||||
# Sign disposable SSH public keys for short-lived development access.
|
|
||||||
path "ssh-client-signer/sign/ai-agent" {
|
|
||||||
capabilities = ["create", "update"]
|
|
||||||
}
|
|
||||||
@@ -119,10 +119,6 @@ issuerRef:
|
|||||||
kind: ClusterIssuer
|
kind: ClusterIssuer
|
||||||
```
|
```
|
||||||
|
|
||||||
`values.yaml` 必须保持 `config.gatewayAPI.enabled: true`。`bao-acme` 的 HTTP-01
|
|
||||||
solver 通过共享 Gateway 创建临时 HTTPRoute;关闭该项不会让 ClusterIssuer 变为
|
|
||||||
NotReady,而是会让每个 Challenge 卡在 `gateway api is not enabled`。
|
|
||||||
|
|
||||||
Issuance is capped by `default_directory_policy = role:bao-server`
|
Issuance is capped by `default_directory_policy = role:bao-server`
|
||||||
(`../../infrastructure/openbao/terraform/pki.tf`), which permits `ad.ddupan.top` subdomains only. Clients
|
(`../../infrastructure/openbao/terraform/pki.tf`), which permits `ad.ddupan.top` subdomains only. Clients
|
||||||
need the internal CA in their trust store — already true for the PVE nodes, the DC and
|
need the internal CA in their trust store — already true for the PVE nodes, the DC and
|
||||||
|
|||||||
@@ -44,13 +44,6 @@ cainjector:
|
|||||||
limits:
|
limits:
|
||||||
memory: 256Mi
|
memory: 256Mi
|
||||||
|
|
||||||
# bao-acme solves HTTP-01 through the shared Gateway. The ClusterIssuer can be
|
|
||||||
# accepted while this is disabled, but every Challenge then stays pending with
|
|
||||||
# "gateway api is not enabled". Gateway API CRDs are installed by Envoy Gateway.
|
|
||||||
config:
|
|
||||||
gatewayAPI:
|
|
||||||
enabled: true
|
|
||||||
|
|
||||||
# ⚠ DNS-01 self-check: cert-manager polls authoritative NS for the _acme-challenge
|
# ⚠ DNS-01 self-check: cert-manager polls authoritative NS for the _acme-challenge
|
||||||
# TXT record before telling the CA to validate. By default it asks the cluster's
|
# TXT record before telling the CA to validate. By default it asks the cluster's
|
||||||
# resolver, which for ad.ddupan.top is CoreDNS -> the Samba AD DC (k3s/coredns-custom.yaml).
|
# resolver, which for ad.ddupan.top is CoreDNS -> the Samba AD DC (k3s/coredns-custom.yaml).
|
||||||
|
|||||||
@@ -1,51 +0,0 @@
|
|||||||
# Gitea dynamic runner controller
|
|
||||||
|
|
||||||
此目录只管理 homelab 中的 controller 部署。controller、worker、Cloud Hypervisor
|
|
||||||
launcher 和 guest runner 的源码与发布位于独立仓库
|
|
||||||
`panxiao81/gitea-dynamic-runner`。
|
|
||||||
|
|
||||||
当前 bootstrap controller 接收 Gitea `workflow_job` webhook,将 `[self-hosted, pod]` 和
|
|
||||||
`[self-hosted, vm]` 的 queued job 分别发布到 NATS。Pod worker 在本集群创建一次性
|
|
||||||
privileged host runner;Docker、BuildKit 和 kind 由 workflow 自行 setup。内部
|
|
||||||
endpoint:
|
|
||||||
|
|
||||||
```text
|
|
||||||
http://dynamic-runner-controller.dynamic-runner.svc.cluster.local:8787/webhook
|
|
||||||
```
|
|
||||||
|
|
||||||
OpenBao 路径:
|
|
||||||
|
|
||||||
- `kv/k8s/nats.ci_producer_password`:已有 NATS producer 密码。
|
|
||||||
- `kv/k8s/nats.ci_worker_password`:已有 NATS worker 密码。
|
|
||||||
- `kv/k8s/dynamic-runner.webhook_secret`:Gitea webhook HMAC secret。
|
|
||||||
- `kv/k8s/gitea-runner.token`:现有 instance runner registration token。
|
|
||||||
|
|
||||||
首期 controller 与 runner 镜像由 laptop 本机构建后导入 k3s containerd,作为 CI
|
|
||||||
发布链路建立前的 bootstrap。部署使用 `imagePullPolicy: Never`。正式发布 workflow
|
|
||||||
获得专用 SPIFFE ID 后,必须将 image 改为 zot digest 并移除本地导入步骤。
|
|
||||||
|
|
||||||
## 身份绑定
|
|
||||||
|
|
||||||
queued webhook 只负责创建没有业务身份的 Pod。runner 实际领取任务后,Gitea 的
|
|
||||||
`in_progress` webhook 会携带实际 `runner_name`;controller 将 binding 消息发布到
|
|
||||||
NATS,Pod worker 再给对应 Pod 添加:
|
|
||||||
|
|
||||||
```text
|
|
||||||
ci.ddupan.top/identity-bound=true
|
|
||||||
ci.ddupan.top/spiffe-path=<owner>/<repository>/<percent-encoded-job-name>
|
|
||||||
```
|
|
||||||
|
|
||||||
`ClusterSPIFFEID/gitea-dynamic-runner` 只匹配已经绑定的 Pod,并签发
|
|
||||||
`spiffe://ddupan.top/ci/<owner>/<repository>/<job-name>`。runner 的 job-start hook 在
|
|
||||||
SVID 可用之前不会放行第一步,因此不能根据 queued 事件错配身份。
|
|
||||||
|
|
||||||
每个 runner Pod 使用 `gitea-dynamic-runner` ServiceAccount。该 ServiceAccount 没有
|
|
||||||
Kubernetes API 权限;只有 `dynamic-runner-pod-worker` ServiceAccount 能在本 namespace
|
|
||||||
create/get/patch/delete Pod。
|
|
||||||
|
|
||||||
长期实现将由兼容 Gitea RunnerService 的 scheduler 直接领取 task,再交给 Pod/VM
|
|
||||||
executor;届时删除 webhook、临时 runner 注册和 identity binding 消息。跟踪见
|
|
||||||
`panxiao81/gitea-dynamic-runner` issue #7。
|
|
||||||
|
|
||||||
Gitea webhook 只订阅 `workflow_job`,content type 使用 JSON,secret 与 Bao 中值
|
|
||||||
一致。不要启用 `send_everything`,否则 controller 会收到无关仓库事件。
|
|
||||||
@@ -1,17 +0,0 @@
|
|||||||
apiVersion: spire.spiffe.io/v1alpha1
|
|
||||||
kind: ClusterSPIFFEID
|
|
||||||
metadata:
|
|
||||||
name: gitea-dynamic-runner
|
|
||||||
spec:
|
|
||||||
className: spire-mgmt-spire
|
|
||||||
spiffeIDTemplate: 'spiffe://{{ .TrustDomain }}/ci/{{ index .PodMeta.Annotations "ci.ddupan.top/spiffe-path" }}'
|
|
||||||
namespaceSelector:
|
|
||||||
matchLabels:
|
|
||||||
kubernetes.io/metadata.name: dynamic-runner
|
|
||||||
podSelector:
|
|
||||||
matchLabels:
|
|
||||||
app.kubernetes.io/name: gitea-dynamic-runner
|
|
||||||
ci.ddupan.top/identity-bound: "true"
|
|
||||||
workloadSelectorTemplates:
|
|
||||||
- k8s:ns:dynamic-runner
|
|
||||||
- k8s:sa:gitea-dynamic-runner
|
|
||||||
@@ -1,106 +0,0 @@
|
|||||||
apiVersion: apps/v1
|
|
||||||
kind: Deployment
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner-controller
|
|
||||||
namespace: dynamic-runner
|
|
||||||
spec:
|
|
||||||
replicas: 1
|
|
||||||
selector:
|
|
||||||
matchLabels:
|
|
||||||
app.kubernetes.io/name: dynamic-runner-controller
|
|
||||||
template:
|
|
||||||
metadata:
|
|
||||||
labels:
|
|
||||||
app.kubernetes.io/name: dynamic-runner-controller
|
|
||||||
spec:
|
|
||||||
serviceAccountName: dynamic-runner-controller
|
|
||||||
automountServiceAccountToken: false
|
|
||||||
initContainers:
|
|
||||||
- name: fetch-internal-ca
|
|
||||||
image: curlimages/curl:8.16.0@sha256:463eaf6072688fe96ac64fa623fe73e1dbe25d8ad6c34404a669ad3ce1f104b6
|
|
||||||
args:
|
|
||||||
- --fail
|
|
||||||
- --silent
|
|
||||||
- --show-error
|
|
||||||
- --output
|
|
||||||
- /trust/ca.pem
|
|
||||||
- https://bao.ad.ddupan.top:8200/v1/pki/ca/pem
|
|
||||||
securityContext:
|
|
||||||
allowPrivilegeEscalation: false
|
|
||||||
capabilities:
|
|
||||||
drop: [ALL]
|
|
||||||
readOnlyRootFilesystem: true
|
|
||||||
runAsNonRoot: true
|
|
||||||
runAsUser: 101
|
|
||||||
runAsGroup: 102
|
|
||||||
seccompProfile:
|
|
||||||
type: RuntimeDefault
|
|
||||||
volumeMounts:
|
|
||||||
- name: trust
|
|
||||||
mountPath: /trust
|
|
||||||
containers:
|
|
||||||
- name: controller
|
|
||||||
# Bootstrap import on laptop. Replace with a zot digest after the
|
|
||||||
# repository's image publishing workflow has a dedicated identity.
|
|
||||||
image: gitea-dynamic-runner-controller:0.3.0-bootstrap
|
|
||||||
imagePullPolicy: Never
|
|
||||||
env:
|
|
||||||
- name: NATS_URL
|
|
||||||
value: tls://nats.ad.ddupan.top:4222
|
|
||||||
- name: NATS_CA_FILE
|
|
||||||
value: /run/trust/ca.pem
|
|
||||||
- name: NATS_PASSWORD_FILE
|
|
||||||
value: /run/dynamic-runner-secrets/nats-password
|
|
||||||
- name: WEBHOOK_SECRET_FILE
|
|
||||||
value: /run/dynamic-runner-secrets/webhook-secret
|
|
||||||
ports:
|
|
||||||
- name: http
|
|
||||||
containerPort: 8787
|
|
||||||
readinessProbe:
|
|
||||||
httpGet:
|
|
||||||
path: /healthz
|
|
||||||
port: http
|
|
||||||
periodSeconds: 5
|
|
||||||
livenessProbe:
|
|
||||||
httpGet:
|
|
||||||
path: /healthz
|
|
||||||
port: http
|
|
||||||
initialDelaySeconds: 10
|
|
||||||
periodSeconds: 10
|
|
||||||
resources:
|
|
||||||
requests:
|
|
||||||
cpu: 25m
|
|
||||||
memory: 32Mi
|
|
||||||
limits:
|
|
||||||
cpu: 250m
|
|
||||||
memory: 128Mi
|
|
||||||
securityContext:
|
|
||||||
allowPrivilegeEscalation: false
|
|
||||||
capabilities:
|
|
||||||
drop: [ALL]
|
|
||||||
readOnlyRootFilesystem: true
|
|
||||||
runAsNonRoot: true
|
|
||||||
runAsUser: 65532
|
|
||||||
runAsGroup: 65532
|
|
||||||
seccompProfile:
|
|
||||||
type: RuntimeDefault
|
|
||||||
volumeMounts:
|
|
||||||
- name: secret
|
|
||||||
mountPath: /run/dynamic-runner-secrets
|
|
||||||
readOnly: true
|
|
||||||
- name: trust
|
|
||||||
mountPath: /run/trust
|
|
||||||
readOnly: true
|
|
||||||
securityContext:
|
|
||||||
fsGroup: 65532
|
|
||||||
fsGroupChangePolicy: OnRootMismatch
|
|
||||||
seccompProfile:
|
|
||||||
type: RuntimeDefault
|
|
||||||
volumes:
|
|
||||||
- name: secret
|
|
||||||
secret:
|
|
||||||
secretName: dynamic-runner
|
|
||||||
defaultMode: 0400
|
|
||||||
- name: trust
|
|
||||||
emptyDir:
|
|
||||||
sizeLimit: 1Mi
|
|
||||||
@@ -1,30 +0,0 @@
|
|||||||
apiVersion: external-secrets.io/v1
|
|
||||||
kind: ExternalSecret
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner
|
|
||||||
namespace: dynamic-runner
|
|
||||||
spec:
|
|
||||||
refreshInterval: 1h
|
|
||||||
secretStoreRef:
|
|
||||||
kind: ClusterSecretStore
|
|
||||||
name: openbao
|
|
||||||
target:
|
|
||||||
creationPolicy: Owner
|
|
||||||
name: dynamic-runner
|
|
||||||
data:
|
|
||||||
- secretKey: nats-password
|
|
||||||
remoteRef:
|
|
||||||
key: k8s/nats
|
|
||||||
property: ci_producer_password
|
|
||||||
- secretKey: nats-worker-password
|
|
||||||
remoteRef:
|
|
||||||
key: k8s/nats
|
|
||||||
property: ci_worker_password
|
|
||||||
- secretKey: webhook-secret
|
|
||||||
remoteRef:
|
|
||||||
key: k8s/dynamic-runner
|
|
||||||
property: webhook_secret
|
|
||||||
- secretKey: token
|
|
||||||
remoteRef:
|
|
||||||
key: k8s/gitea-runner
|
|
||||||
property: token
|
|
||||||
@@ -1,10 +0,0 @@
|
|||||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
|
||||||
kind: Kustomization
|
|
||||||
resources:
|
|
||||||
- namespace.yaml
|
|
||||||
- external-secret.yaml
|
|
||||||
- rbac.yaml
|
|
||||||
- clusterspiffeid.yaml
|
|
||||||
- deployment.yaml
|
|
||||||
- pod-worker-deployment.yaml
|
|
||||||
- service.yaml
|
|
||||||
@@ -1,9 +0,0 @@
|
|||||||
apiVersion: v1
|
|
||||||
kind: Namespace
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner
|
|
||||||
labels:
|
|
||||||
# Disposable Pod runners may start dockerd or kind from the workflow.
|
|
||||||
pod-security.kubernetes.io/enforce: privileged
|
|
||||||
pod-security.kubernetes.io/audit: restricted
|
|
||||||
pod-security.kubernetes.io/warn: restricted
|
|
||||||
@@ -1,100 +0,0 @@
|
|||||||
apiVersion: apps/v1
|
|
||||||
kind: Deployment
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner-pod-worker
|
|
||||||
namespace: dynamic-runner
|
|
||||||
spec:
|
|
||||||
replicas: 1
|
|
||||||
selector:
|
|
||||||
matchLabels:
|
|
||||||
app.kubernetes.io/name: dynamic-runner-pod-worker
|
|
||||||
template:
|
|
||||||
metadata:
|
|
||||||
labels:
|
|
||||||
app.kubernetes.io/name: dynamic-runner-pod-worker
|
|
||||||
spec:
|
|
||||||
serviceAccountName: dynamic-runner-pod-worker
|
|
||||||
initContainers:
|
|
||||||
- name: fetch-internal-ca
|
|
||||||
image: curlimages/curl:8.16.0@sha256:463eaf6072688fe96ac64fa623fe73e1dbe25d8ad6c34404a669ad3ce1f104b6
|
|
||||||
args:
|
|
||||||
- --fail
|
|
||||||
- --silent
|
|
||||||
- --show-error
|
|
||||||
- --output
|
|
||||||
- /trust/ca.pem
|
|
||||||
- https://bao.ad.ddupan.top:8200/v1/pki/ca/pem
|
|
||||||
securityContext:
|
|
||||||
allowPrivilegeEscalation: false
|
|
||||||
capabilities:
|
|
||||||
drop: [ALL]
|
|
||||||
readOnlyRootFilesystem: true
|
|
||||||
runAsNonRoot: true
|
|
||||||
runAsUser: 101
|
|
||||||
runAsGroup: 102
|
|
||||||
seccompProfile:
|
|
||||||
type: RuntimeDefault
|
|
||||||
volumeMounts:
|
|
||||||
- name: trust
|
|
||||||
mountPath: /trust
|
|
||||||
containers:
|
|
||||||
- name: pod-worker
|
|
||||||
image: gitea-dynamic-runner-controller:0.3.0-bootstrap
|
|
||||||
imagePullPolicy: Never
|
|
||||||
command: [/venv/bin/gitea-dynamic-runner-pod-worker]
|
|
||||||
env:
|
|
||||||
- name: NATS_URL
|
|
||||||
value: tls://nats.ad.ddupan.top:4222
|
|
||||||
- name: NATS_CA_FILE
|
|
||||||
value: /run/trust/ca.pem
|
|
||||||
- name: NATS_PASSWORD_FILE
|
|
||||||
value: /run/dynamic-runner-secrets/nats-worker-password
|
|
||||||
- name: RUNNER_NAMESPACE
|
|
||||||
valueFrom:
|
|
||||||
fieldRef:
|
|
||||||
fieldPath: metadata.namespace
|
|
||||||
- name: RUNNER_IMAGE
|
|
||||||
value: zot.ad.ddupan.top/panxiao81/gitea-dynamic-runner-runner@sha256:4c61f6315453d68a827ee9542f1345ed86576eaaf62325eb803c8fe3f06ddf2a
|
|
||||||
- name: RUNNER_SERVICE_ACCOUNT
|
|
||||||
value: gitea-dynamic-runner
|
|
||||||
- name: RUNNER_TOKEN_SECRET
|
|
||||||
value: dynamic-runner
|
|
||||||
- name: RUNNER_CAPACITY
|
|
||||||
value: "4"
|
|
||||||
resources:
|
|
||||||
requests:
|
|
||||||
cpu: 25m
|
|
||||||
memory: 32Mi
|
|
||||||
limits:
|
|
||||||
cpu: 250m
|
|
||||||
memory: 128Mi
|
|
||||||
securityContext:
|
|
||||||
allowPrivilegeEscalation: false
|
|
||||||
capabilities:
|
|
||||||
drop: [ALL]
|
|
||||||
readOnlyRootFilesystem: true
|
|
||||||
runAsNonRoot: true
|
|
||||||
runAsUser: 65532
|
|
||||||
runAsGroup: 65532
|
|
||||||
seccompProfile:
|
|
||||||
type: RuntimeDefault
|
|
||||||
volumeMounts:
|
|
||||||
- name: secret
|
|
||||||
mountPath: /run/dynamic-runner-secrets
|
|
||||||
readOnly: true
|
|
||||||
- name: trust
|
|
||||||
mountPath: /run/trust
|
|
||||||
readOnly: true
|
|
||||||
securityContext:
|
|
||||||
fsGroup: 65532
|
|
||||||
fsGroupChangePolicy: OnRootMismatch
|
|
||||||
seccompProfile:
|
|
||||||
type: RuntimeDefault
|
|
||||||
volumes:
|
|
||||||
- name: secret
|
|
||||||
secret:
|
|
||||||
secretName: dynamic-runner
|
|
||||||
defaultMode: 0400
|
|
||||||
- name: trust
|
|
||||||
emptyDir:
|
|
||||||
sizeLimit: 1Mi
|
|
||||||
@@ -1,41 +0,0 @@
|
|||||||
apiVersion: v1
|
|
||||||
kind: ServiceAccount
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner-controller
|
|
||||||
namespace: dynamic-runner
|
|
||||||
---
|
|
||||||
apiVersion: v1
|
|
||||||
kind: ServiceAccount
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner-pod-worker
|
|
||||||
namespace: dynamic-runner
|
|
||||||
---
|
|
||||||
apiVersion: v1
|
|
||||||
kind: ServiceAccount
|
|
||||||
metadata:
|
|
||||||
name: gitea-dynamic-runner
|
|
||||||
namespace: dynamic-runner
|
|
||||||
---
|
|
||||||
apiVersion: rbac.authorization.k8s.io/v1
|
|
||||||
kind: Role
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner-pod-worker
|
|
||||||
namespace: dynamic-runner
|
|
||||||
rules:
|
|
||||||
- apiGroups: [""]
|
|
||||||
resources: [pods]
|
|
||||||
verbs: [create, get, patch, delete]
|
|
||||||
---
|
|
||||||
apiVersion: rbac.authorization.k8s.io/v1
|
|
||||||
kind: RoleBinding
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner-pod-worker
|
|
||||||
namespace: dynamic-runner
|
|
||||||
roleRef:
|
|
||||||
apiGroup: rbac.authorization.k8s.io
|
|
||||||
kind: Role
|
|
||||||
name: dynamic-runner-pod-worker
|
|
||||||
subjects:
|
|
||||||
- kind: ServiceAccount
|
|
||||||
name: dynamic-runner-pod-worker
|
|
||||||
namespace: dynamic-runner
|
|
||||||
@@ -1,12 +0,0 @@
|
|||||||
apiVersion: v1
|
|
||||||
kind: Service
|
|
||||||
metadata:
|
|
||||||
name: dynamic-runner-controller
|
|
||||||
namespace: dynamic-runner
|
|
||||||
spec:
|
|
||||||
selector:
|
|
||||||
app.kubernetes.io/name: dynamic-runner-controller
|
|
||||||
ports:
|
|
||||||
- name: http
|
|
||||||
port: 8787
|
|
||||||
targetPort: http
|
|
||||||
@@ -25,25 +25,6 @@ The runner registration token is authoritative in OpenBao at
|
|||||||
the `gitea-runner-token` Secret. Never put the token in this directory or a Helm
|
the `gitea-runner-token` Secret. Never put the token in this directory or a Helm
|
||||||
command line.
|
command line.
|
||||||
|
|
||||||
## SPIRE 与 OCI 发布
|
|
||||||
|
|
||||||
runner Pod 使用专用 ServiceAccount `gitea-actions`,并由精确匹配 namespace、
|
|
||||||
ServiceAccount 隐含的 Pod、以及 chart labels 的 `ClusterSPIFFEID` 获得:
|
|
||||||
|
|
||||||
```text
|
|
||||||
spiffe://ddupan.top/ci/gitea-actions
|
|
||||||
```
|
|
||||||
|
|
||||||
SPIFFE CSI socket 同时只读挂载到 runner 和 DinD。act 的 volume allowlist 只允许
|
|
||||||
`/run/spire/agent-sockets`;workflow 仍必须在 job container 中显式请求该 bind
|
|
||||||
mount。原因是 bind mount 由 DinD 内的 dockerd 解析,只挂 runner 容器无法让 job
|
|
||||||
访问 Workload API。
|
|
||||||
|
|
||||||
该身份不是通用 registry 管理员。zot 只对明确列出的 CI 镜像仓库授予
|
|
||||||
`read/create/update`,不授予 delete 或其他仓库写入。workflow 应获取
|
|
||||||
`aud=zot` 的短期 JWT-SVID,并经 stdin 传给 registry client,不得把 JWT、X.509
|
|
||||||
SVID 或 Docker auth 写入 workspace/artifact。
|
|
||||||
|
|
||||||
## Flux 接管状态
|
## Flux 接管状态
|
||||||
|
|
||||||
该 release 最初通过下述 review-first 流程手动 bootstrap。下一个 GitOps 阶段将
|
该 release 最初通过下述 review-first 流程手动 bootstrap。下一个 GitOps 阶段将
|
||||||
|
|||||||
@@ -1,17 +0,0 @@
|
|||||||
apiVersion: spire.spiffe.io/v1alpha1
|
|
||||||
kind: ClusterSPIFFEID
|
|
||||||
metadata:
|
|
||||||
name: gitea-actions
|
|
||||||
spec:
|
|
||||||
className: spire-mgmt-spire
|
|
||||||
spiffeIDTemplate: spiffe://{{ .TrustDomain }}/ci/gitea-actions
|
|
||||||
namespaceSelector:
|
|
||||||
matchLabels:
|
|
||||||
kubernetes.io/metadata.name: gitea-actions
|
|
||||||
podSelector:
|
|
||||||
matchLabels:
|
|
||||||
app.kubernetes.io/instance: gitea-actions
|
|
||||||
app.kubernetes.io/name: actions-runner
|
|
||||||
workloadSelectorTemplates:
|
|
||||||
- k8s:ns:gitea-actions
|
|
||||||
- k8s:sa:gitea-actions
|
|
||||||
@@ -11,8 +11,6 @@ configMapGenerator:
|
|||||||
- values.yaml=values.yaml
|
- values.yaml=values.yaml
|
||||||
resources:
|
resources:
|
||||||
- namespace.yaml
|
- namespace.yaml
|
||||||
- serviceaccount.yaml
|
|
||||||
- clusterspiffeid.yaml
|
|
||||||
- external-secret.yaml
|
- external-secret.yaml
|
||||||
- helmrepository.yaml
|
- helmrepository.yaml
|
||||||
- helmrelease.yaml
|
- helmrelease.yaml
|
||||||
|
|||||||
@@ -1,6 +0,0 @@
|
|||||||
apiVersion: v1
|
|
||||||
kind: ServiceAccount
|
|
||||||
metadata:
|
|
||||||
name: gitea-actions
|
|
||||||
namespace: gitea-actions
|
|
||||||
automountServiceAccountToken: false
|
|
||||||
@@ -7,12 +7,6 @@ existingSecretKey: token
|
|||||||
statefulset:
|
statefulset:
|
||||||
replicas: 1
|
replicas: 1
|
||||||
timezone: Etc/UTC
|
timezone: Etc/UTC
|
||||||
serviceAccountName: gitea-actions
|
|
||||||
extraVolumes:
|
|
||||||
- name: spiffe-workload-api
|
|
||||||
csi:
|
|
||||||
driver: csi.spiffe.io
|
|
||||||
readOnly: true
|
|
||||||
securityContext:
|
securityContext:
|
||||||
fsGroup: 1000
|
fsGroup: 1000
|
||||||
# Chart 0.1.1 applies this block to both runner and DinD containers.
|
# Chart 0.1.1 applies this block to both runner and DinD containers.
|
||||||
@@ -33,10 +27,6 @@ statefulset:
|
|||||||
repository: gitea/runner
|
repository: gitea/runner
|
||||||
tag: 2.3.0
|
tag: 2.3.0
|
||||||
pullPolicy: IfNotPresent
|
pullPolicy: IfNotPresent
|
||||||
extraVolumeMounts:
|
|
||||||
- name: spiffe-workload-api
|
|
||||||
mountPath: /run/spire/agent-sockets
|
|
||||||
readOnly: true
|
|
||||||
config: |
|
config: |
|
||||||
log:
|
log:
|
||||||
level: info
|
level: info
|
||||||
@@ -52,10 +42,6 @@ statefulset:
|
|||||||
container:
|
container:
|
||||||
require_docker: true
|
require_docker: true
|
||||||
docker_timeout: 300s
|
docker_timeout: 300s
|
||||||
# Workflows must still request this exact bind mount explicitly. The
|
|
||||||
# allowlist prevents arbitrary host paths from reaching job containers.
|
|
||||||
valid_volumes:
|
|
||||||
- /run/spire/agent-sockets
|
|
||||||
|
|
||||||
dind:
|
dind:
|
||||||
# The node enforces AppArmor's unprivileged-userns restriction, which blocks
|
# The node enforces AppArmor's unprivileged-userns restriction, which blocks
|
||||||
@@ -65,12 +51,6 @@ statefulset:
|
|||||||
repository: docker
|
repository: docker
|
||||||
tag: 29.7.1-dind
|
tag: 29.7.1-dind
|
||||||
pullPolicy: IfNotPresent
|
pullPolicy: IfNotPresent
|
||||||
# Bind mounts are resolved by dockerd, so the CSI socket must exist in the
|
|
||||||
# DinD container as well as in the runner container.
|
|
||||||
extraVolumeMounts:
|
|
||||||
- name: spiffe-workload-api
|
|
||||||
mountPath: /run/spire/agent-sockets
|
|
||||||
readOnly: true
|
|
||||||
# k3s uses a 1450-byte pod MTU. Without matching it here, nested Actions
|
# k3s uses a 1450-byte pod MTU. Without matching it here, nested Actions
|
||||||
# networks advertise 1500 and GitHub TLS packets disappear on the outer
|
# networks advertise 1500 and GitHub TLS packets disappear on the outer
|
||||||
# overlay path while direct pod traffic remains healthy.
|
# overlay path while direct pod traffic remains healthy.
|
||||||
|
|||||||
@@ -1,49 +0,0 @@
|
|||||||
# NATS
|
|
||||||
|
|
||||||
共享的轻量消息基础设施。首期为 Gitea microVM runner 提供 JetStream work queue,
|
|
||||||
但 Account、subject 与部署位置均不与 CI controller 绑定,其他服务可按独立 Account
|
|
||||||
复用。
|
|
||||||
|
|
||||||
## 当前拓扑
|
|
||||||
|
|
||||||
- 单节点 NATS;当前 homelab 没有资源运行有意义的三副本 JetStream quorum。
|
|
||||||
- JetStream file store 使用 `localpv-zfs-ceph`,PVC 2 GiB。
|
|
||||||
- 服务通过 k3s ServiceLB 在 `nats.ad.ddupan.top:4222` 暴露给内网;集群内客户端
|
|
||||||
使用 `nats.nats.svc.cluster.local:4222`。访问控制由 TLS、Account 与用户权限负责,
|
|
||||||
不额外维护易漂移的源 IP 白名单。
|
|
||||||
- TLS 证书由 `bao-acme` 签发。PVE 节点已信任内部 CA。
|
|
||||||
- `bao-server` PKI role 只接受 RSA CSR,因此 Certificate 使用 RSA 2048;不要改成
|
|
||||||
ECDSA,ACME challenge 会成功但 finalize 会以 `role requires keys of type rsa` 失败。
|
|
||||||
- `SYS` Account 用于管理;`CI` Account 启用 JetStream,存储上限 1 GiB。
|
|
||||||
|
|
||||||
Account 内的 JetStream 配额会原样进入 `nats.conf`,必须使用 NATS 的 `MB`/`GB`
|
|
||||||
格式;PVC 等 Kubernetes resource quantity 才使用 `Mi`/`Gi`。
|
|
||||||
|
|
||||||
首期使用静态用户,密码只存在 OpenBao `kv/k8s/nats`:
|
|
||||||
|
|
||||||
```text
|
|
||||||
sys_password
|
|
||||||
ci_producer_password
|
|
||||||
ci_worker_password
|
|
||||||
```
|
|
||||||
|
|
||||||
`ci-producer` 只能发布 `ci.runner.>` 并调用必要的 JetStream API;`ci-worker`
|
|
||||||
只能调用 JetStream pull/ACK API。二者都不能读取另一个 Account 的 subject。
|
|
||||||
|
|
||||||
后续 SPIRE/Auth Callout 动态认证见 homelab-infra issue #56。该迁移只替换连接
|
|
||||||
凭据,不改变 Account、stream、subject 或 consumer。
|
|
||||||
|
|
||||||
## CI stream 约定
|
|
||||||
|
|
||||||
controller 首次启动时幂等创建 `CI_RUNNER` stream:`ci.runner.>`、
|
|
||||||
`WorkQueuePolicy`、file storage、24h/10000 条/256 MiB 上限。每类 runner 使用独立
|
|
||||||
subject 和 durable pull consumer;`ci.runner.<backend>.binding` 传递 runner 实际
|
|
||||||
领取任务后的身份绑定。同类型的多个 worker 共享 durable consumer。ACK 后消息立即
|
|
||||||
删除,不保存 CI 历史。
|
|
||||||
|
|
||||||
## 验证
|
|
||||||
|
|
||||||
```bash
|
|
||||||
kubectl -n nats get helmrelease,pod,pvc,certificate,externalsecret
|
|
||||||
kubectl -n nats logs statefulset/nats -c nats
|
|
||||||
```
|
|
||||||
@@ -1,21 +0,0 @@
|
|||||||
apiVersion: cert-manager.io/v1
|
|
||||||
kind: Certificate
|
|
||||||
metadata:
|
|
||||||
name: nats-ad-ddupan-top
|
|
||||||
namespace: nats
|
|
||||||
spec:
|
|
||||||
secretName: nats-server-tls
|
|
||||||
issuerRef:
|
|
||||||
name: bao-acme
|
|
||||||
kind: ClusterIssuer
|
|
||||||
group: cert-manager.io
|
|
||||||
commonName: nats.ad.ddupan.top
|
|
||||||
dnsNames:
|
|
||||||
- nats.ad.ddupan.top
|
|
||||||
duration: 720h
|
|
||||||
renewBefore: 168h
|
|
||||||
privateKey:
|
|
||||||
# OpenBao's bao-server role intentionally accepts RSA keys only.
|
|
||||||
algorithm: RSA
|
|
||||||
size: 2048
|
|
||||||
rotationPolicy: Always
|
|
||||||
@@ -1,16 +0,0 @@
|
|||||||
apiVersion: external-secrets.io/v1
|
|
||||||
kind: ExternalSecret
|
|
||||||
metadata:
|
|
||||||
name: nats-auth
|
|
||||||
namespace: nats
|
|
||||||
spec:
|
|
||||||
refreshInterval: 1h
|
|
||||||
secretStoreRef:
|
|
||||||
kind: ClusterSecretStore
|
|
||||||
name: openbao
|
|
||||||
target:
|
|
||||||
creationPolicy: Owner
|
|
||||||
name: nats-auth
|
|
||||||
dataFrom:
|
|
||||||
- extract:
|
|
||||||
key: k8s/nats
|
|
||||||
@@ -1,31 +0,0 @@
|
|||||||
apiVersion: helm.toolkit.fluxcd.io/v2
|
|
||||||
kind: HelmRelease
|
|
||||||
metadata:
|
|
||||||
name: nats
|
|
||||||
namespace: nats
|
|
||||||
spec:
|
|
||||||
chart:
|
|
||||||
spec:
|
|
||||||
chart: nats
|
|
||||||
interval: 1h
|
|
||||||
sourceRef:
|
|
||||||
kind: HelmRepository
|
|
||||||
name: nats
|
|
||||||
version: 2.14.2
|
|
||||||
driftDetection:
|
|
||||||
mode: enabled
|
|
||||||
install:
|
|
||||||
strategy:
|
|
||||||
name: RetryOnFailure
|
|
||||||
retryInterval: 5m
|
|
||||||
interval: 30m
|
|
||||||
releaseName: nats
|
|
||||||
targetNamespace: nats
|
|
||||||
timeout: 10m
|
|
||||||
upgrade:
|
|
||||||
strategy:
|
|
||||||
name: RetryOnFailure
|
|
||||||
retryInterval: 5m
|
|
||||||
valuesFrom:
|
|
||||||
- kind: ConfigMap
|
|
||||||
name: nats-values
|
|
||||||
@@ -1,8 +0,0 @@
|
|||||||
apiVersion: source.toolkit.fluxcd.io/v1
|
|
||||||
kind: HelmRepository
|
|
||||||
metadata:
|
|
||||||
name: nats
|
|
||||||
namespace: nats
|
|
||||||
spec:
|
|
||||||
interval: 1h
|
|
||||||
url: https://nats-io.github.io/k8s/helm/charts/
|
|
||||||
@@ -1,17 +0,0 @@
|
|||||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
|
||||||
kind: Kustomization
|
|
||||||
generatorOptions:
|
|
||||||
disableNameSuffixHash: true
|
|
||||||
labels:
|
|
||||||
reconcile.fluxcd.io/watch: Enabled
|
|
||||||
configMapGenerator:
|
|
||||||
- name: nats-values
|
|
||||||
namespace: nats
|
|
||||||
files:
|
|
||||||
- values.yaml=values.yaml
|
|
||||||
resources:
|
|
||||||
- namespace.yaml
|
|
||||||
- helmrepository.yaml
|
|
||||||
- external-secret.yaml
|
|
||||||
- certificate.yaml
|
|
||||||
- helmrelease.yaml
|
|
||||||
@@ -1,4 +0,0 @@
|
|||||||
apiVersion: v1
|
|
||||||
kind: Namespace
|
|
||||||
metadata:
|
|
||||||
name: nats
|
|
||||||
@@ -1,76 +0,0 @@
|
|||||||
config:
|
|
||||||
jetstream:
|
|
||||||
enabled: true
|
|
||||||
fileStore:
|
|
||||||
enabled: true
|
|
||||||
maxSize: 2G
|
|
||||||
pvc:
|
|
||||||
enabled: true
|
|
||||||
size: 2Gi
|
|
||||||
storageClassName: localpv-zfs-ceph
|
|
||||||
memoryStore:
|
|
||||||
enabled: true
|
|
||||||
maxSize: 64M
|
|
||||||
nats:
|
|
||||||
tls:
|
|
||||||
enabled: true
|
|
||||||
secretName: nats-server-tls
|
|
||||||
merge:
|
|
||||||
system_account: SYS
|
|
||||||
accounts:
|
|
||||||
SYS:
|
|
||||||
users:
|
|
||||||
- user: sys
|
|
||||||
password: "<< $NATS_SYS_PASSWORD >>"
|
|
||||||
CI:
|
|
||||||
jetstream:
|
|
||||||
# Account limits are JSON-encoded by config.merge. Use explicit byte
|
|
||||||
# counts so NATS receives integers rather than quoted size strings.
|
|
||||||
max_memory: 33554432
|
|
||||||
max_file: 1073741824
|
|
||||||
max_streams: 16
|
|
||||||
max_consumers: 64
|
|
||||||
max_bytes_required: true
|
|
||||||
users:
|
|
||||||
- user: ci-producer
|
|
||||||
password: "<< $NATS_CI_PRODUCER_PASSWORD >>"
|
|
||||||
permissions:
|
|
||||||
publish:
|
|
||||||
allow: [ci.runner.>, $JS.API.>]
|
|
||||||
subscribe:
|
|
||||||
allow: [_INBOX.>]
|
|
||||||
- user: ci-worker
|
|
||||||
password: "<< $NATS_CI_WORKER_PASSWORD >>"
|
|
||||||
permissions:
|
|
||||||
publish:
|
|
||||||
allow: [$JS.API.>, $JS.ACK.>]
|
|
||||||
subscribe:
|
|
||||||
allow: [_INBOX.>]
|
|
||||||
|
|
||||||
container:
|
|
||||||
env:
|
|
||||||
NATS_SYS_PASSWORD:
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef: {name: nats-auth, key: sys_password}
|
|
||||||
NATS_CI_PRODUCER_PASSWORD:
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef: {name: nats-auth, key: ci_producer_password}
|
|
||||||
NATS_CI_WORKER_PASSWORD:
|
|
||||||
valueFrom:
|
|
||||||
secretKeyRef: {name: nats-auth, key: ci_worker_password}
|
|
||||||
resources:
|
|
||||||
requests: {cpu: 25m, memory: 64Mi}
|
|
||||||
limits: {memory: 192Mi}
|
|
||||||
|
|
||||||
natsBox:
|
|
||||||
enabled: false
|
|
||||||
|
|
||||||
promExporter:
|
|
||||||
enabled: true
|
|
||||||
podMonitor:
|
|
||||||
enabled: true
|
|
||||||
|
|
||||||
service:
|
|
||||||
merge:
|
|
||||||
spec:
|
|
||||||
type: LoadBalancer
|
|
||||||
@@ -2,25 +2,9 @@
|
|||||||
|
|
||||||
Metrics, logs, and traces for the cluster — one VictoriaMetrics-ecosystem stack,
|
Metrics, logs, and traces for the cluster — one VictoriaMetrics-ecosystem stack,
|
||||||
operator-driven, running in the `monitoring` namespace. Replaces the standalone
|
operator-driven, running in the `monitoring` namespace. Replaces the standalone
|
||||||
Docker Compose monitoring stack. The retired stack and its three Docker data volumes
|
Docker Compose stack in `../../apps/victoriametrics/`.
|
||||||
were removed on 2026-09-16 with the maintainer's explicit approval.
|
|
||||||
|
|
||||||
## Grafana 当前入口(2026-09-16)
|
## Status — DEPLOYED (2026-07-10)
|
||||||
|
|
||||||
统一使用 <https://grafana.ad.ddupan.top>,A 记录指向 `192.168.10.127`,由共享
|
|
||||||
Envoy Gateway 的 `https` listener 与 `*.ad.ddupan.top` 证书提供 TLS。
|
|
||||||
Grafana root_url 和 Authelia 回调均使用此域名;旧专用 Tailscale Ingress 已停用。
|
|
||||||
远程客户端仍需有到 LAN 的路由及已配置的内网 DNS 转发。
|
|
||||||
|
|
||||||
内存看板:`/d/homelab-memory`;采集配置及 AppArmor 规则见
|
|
||||||
[主机与进程内存采集](metrics/exporters/README.md)。
|
|
||||||
|
|
||||||
Grafana 由 Flux 管理;修改 values 后通过 Git 合并触发 HelmRelease,避免现场 Helm
|
|
||||||
修改被漂移检测回滚。Authelia 当前不由 Flux 管理,需单独 Helm upgrade 并先结构比较
|
|
||||||
live values。迁移时先添加 DNS 和回调,再切换 Grafana。Recreate 策略会短暂中断访问,
|
|
||||||
但保留原 PVC、用户、看板和数据源。回滚需同时恢复 root_url、回调和 Ingress 配置。
|
|
||||||
|
|
||||||
## 初始部署记录(2026-07-10)
|
|
||||||
|
|
||||||
Live and verified in the `monitoring` namespace (operator chart 0.66.2):
|
Live and verified in the `monitoring` namespace (operator chart 0.66.2):
|
||||||
metrics (data queryable), logs (pods ingesting), traces (VTSingle CRD, verified via
|
metrics (data queryable), logs (pods ingesting), traces (VTSingle CRD, verified via
|
||||||
@@ -72,10 +56,8 @@ hosts and `logs/vlogs-ingress.yaml` to push their logs.
|
|||||||
| Manage | **victoria-metrics-operator** | VMSingle/VMAgent/VMAlert/VMAlertmanager/VMRule **and** VLSingle as CRDs |
|
| Manage | **victoria-metrics-operator** | VMSingle/VMAgent/VMAlert/VMAlertmanager/VMRule **and** VLSingle as CRDs |
|
||||||
| Expose | **Tailscale ingress** (private) + **Authelia OIDC** | admin tool: private + SSO |
|
| Expose | **Tailscale ingress** (private) + **Authelia OIDC** | admin tool: private + SSO |
|
||||||
|
|
||||||
VictoriaMetrics Operator 的 Prometheus converter 已启用;官方
|
Grafana's Prometheus-operator converter is on, so any chart shipping a
|
||||||
`prometheus-operator-crds` chart 由 `operator/` 一并管理。因此应用 chart 可以原生
|
`ServiceMonitor`/`PodMonitor`/`PrometheusRule` is scraped automatically.
|
||||||
声明 `ServiceMonitor`、`PodMonitor` 或 `PrometheusRule`,再由 converter 转换为对应
|
|
||||||
VM 资源,不需要每个应用额外维护一份 `VM*Scrape`。
|
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
@@ -95,7 +77,7 @@ observability/
|
|||||||
operator/ victoria-metrics-operator (Helm)
|
operator/ victoria-metrics-operator (Helm)
|
||||||
metrics/
|
metrics/
|
||||||
vmsingle|vmagent|vmalert|vmalertmanager.yaml core CRs
|
vmsingle|vmagent|vmalert|vmalertmanager.yaml core CRs
|
||||||
rules/ VMRules migrated from the retired Compose stack (source in Git history)
|
rules/ VMRules ported from ../../apps/victoriametrics/rules
|
||||||
exporters/ node-exporter (+VMNodeScrape), kube-state-metrics (Helm)
|
exporters/ node-exporter (+VMNodeScrape), kube-state-metrics (Helm)
|
||||||
scrapes/ kubelet + cAdvisor VMNodeScrapes
|
scrapes/ kubelet + cAdvisor VMNodeScrapes
|
||||||
logs/ VLSingle CR + victoria-logs-collector (Helm)
|
logs/ VLSingle CR + victoria-logs-collector (Helm)
|
||||||
@@ -106,8 +88,8 @@ observability/
|
|||||||
|
|
||||||
## Docker hosts
|
## Docker hosts
|
||||||
|
|
||||||
The Compose services on Docker hosts (`../../apps/vlmcsd`, `../../apps/ps3netsrv`, `../../apps/netboot`, ...)
|
The Compose services on Docker hosts (`../../apps/vlmcsd`, `../../apps/ps3netsrv`, `../../apps/netboot`, ...
|
||||||
are monitored too:
|
and the legacy `../../apps/victoriametrics` stack) are monitored too:
|
||||||
|
|
||||||
- **Metrics — pull, nothing exposed.** Each host runs `docker-hosts/compose.yaml`
|
- **Metrics — pull, nothing exposed.** Each host runs `docker-hosts/compose.yaml`
|
||||||
(cAdvisor `:8080` + node-exporter `:9100`). The cluster's vmagent scrapes their LAN
|
(cAdvisor `:8080` + node-exporter `:9100`). The cluster's vmagent scrapes their LAN
|
||||||
@@ -178,9 +160,7 @@ cd grafana && ./helm.sh && cd ..
|
|||||||
vmalert (14950), node-exporter (1860), cAdvisor now that those exporters exist.
|
vmalert (14950), node-exporter (1860), cAdvisor now that those exporters exist.
|
||||||
- Wire real Alertmanager receivers in `metrics/vmalertmanager.yaml` (email via
|
- Wire real Alertmanager receivers in `metrics/vmalertmanager.yaml` (email via
|
||||||
`../../apps/smtp-relay/`), replacing the ported `blackhole`.
|
`../../apps/smtp-relay/`), replacing the ported `blackhole`.
|
||||||
- 旧 Compose 监控栈已于 2026-09-16 按维护者要求清理:原先已无容器,本次移除
|
- Once parity is confirmed, decommission the Compose stack: `../../apps/victoriametrics/`
|
||||||
旧配置及 `victoriametrics_vmdata`、`victoriametrics_vmagentdata`、
|
(`docker compose down`), and retire that folder.
|
||||||
`victoriametrics_grafanadata` 三个无引用 Docker 卷。没有创建备份或迁移旧数据,
|
|
||||||
Kubernetes 监控资源未修改。旧配置仍可从 Git 历史查阅,旧卷中的数据已删除。
|
|
||||||
- Add app instrumentation: point `OTEL_EXPORTER_OTLP_ENDPOINT` at
|
- Add app instrumentation: point `OTEL_EXPORTER_OTLP_ENDPOINT` at
|
||||||
`otel-collector.monitoring.svc:4317`.
|
`otel-collector.monitoring.svc:4317`.
|
||||||
|
|||||||
@@ -1,502 +0,0 @@
|
|||||||
{
|
|
||||||
"uid": "homelab-memory",
|
|
||||||
"title": "Homelab 内存与 Swap",
|
|
||||||
"schemaVersion": 39,
|
|
||||||
"version": 1,
|
|
||||||
"tags": [
|
|
||||||
"homelab",
|
|
||||||
"memory"
|
|
||||||
],
|
|
||||||
"timezone": "utc",
|
|
||||||
"refresh": "1m",
|
|
||||||
"time": {
|
|
||||||
"from": "now-6h",
|
|
||||||
"to": "now"
|
|
||||||
},
|
|
||||||
"templating": {
|
|
||||||
"list": [
|
|
||||||
{
|
|
||||||
"name": "DS_VM",
|
|
||||||
"label": "数据源",
|
|
||||||
"type": "datasource",
|
|
||||||
"query": "prometheus",
|
|
||||||
"current": {
|
|
||||||
"text": "VictoriaMetrics",
|
|
||||||
"value": "VictoriaMetrics"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"panels": [
|
|
||||||
{
|
|
||||||
"id": 1,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "主机物理内存",
|
|
||||||
"description": "ZFS ARC 是内存的一部分;不能与此面板或 Pod/PSS 再叠加。",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 0,
|
|
||||||
"y": 0,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "node_memory_MemTotal_bytes{job=\"node-exporter\"} - node_memory_MemAvailable_bytes{job=\"node-exporter\"}",
|
|
||||||
"legendFormat": "已用(total - available)"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "B",
|
|
||||||
"expr": "node_memory_MemAvailable_bytes{job=\"node-exporter\"}",
|
|
||||||
"legendFormat": "可用"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "C",
|
|
||||||
"expr": "node_memory_MemTotal_bytes{job=\"node-exporter\"}",
|
|
||||||
"legendFormat": "总量"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "bytes",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 2,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "ZFS ARC",
|
|
||||||
"description": "缓存可回收性取决于实际压力,ARC 不等同于 free 的 buff/cache。",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 12,
|
|
||||||
"y": 0,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "node_zfs_arc_size{job=\"node-exporter\"}",
|
|
||||||
"legendFormat": "当前 ARC"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "B",
|
|
||||||
"expr": "node_zfs_arc_c{job=\"node-exporter\"}",
|
|
||||||
"legendFormat": "目标"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "C",
|
|
||||||
"expr": "node_zfs_arc_c_max{job=\"node-exporter\"}",
|
|
||||||
"legendFormat": "上限"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "bytes",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 3,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "Swap 使用量",
|
|
||||||
"description": "",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 0,
|
|
||||||
"y": 8,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "node_memory_SwapTotal_bytes{job=\"node-exporter\"} - node_memory_SwapFree_bytes{job=\"node-exporter\"}",
|
|
||||||
"legendFormat": "已用"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "B",
|
|
||||||
"expr": "node_memory_SwapTotal_bytes{job=\"node-exporter\"}",
|
|
||||||
"legendFormat": "总量"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "bytes",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 4,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "Swap 换页速率",
|
|
||||||
"description": "单位为页/秒,不假定页大小。",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 12,
|
|
||||||
"y": 8,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "rate(node_vmstat_pswpin{job=\"node-exporter\"}[5m])",
|
|
||||||
"legendFormat": "读入 pages/s"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "B",
|
|
||||||
"expr": "rate(node_vmstat_pswpout{job=\"node-exporter\"}[5m])",
|
|
||||||
"legendFormat": "写出 pages/s"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "ops",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 5,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "程序与虚拟机 PSS:前 15",
|
|
||||||
"description": "PSS 按共享页比例分摊。vm: 表示 QEMU 在宿主机的占用,不是来宾内部应用占用。与 Pod working set、ARC 不能相加。",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 0,
|
|
||||||
"y": 16,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "topk(15, namedprocess_namegroup_memory_bytes{job=\"process-exporter\",memtype=\"proportionalResident\"})",
|
|
||||||
"legendFormat": "{{groupname}}"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "bytes",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 6,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "程序与虚拟机 SwapPss:前 15",
|
|
||||||
"description": "通过 smaps 对共享换出页按比例分摊。",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 12,
|
|
||||||
"y": 16,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "topk(15, namedprocess_namegroup_memory_bytes{job=\"process-exporter\",memtype=\"proportionalSwapped\"})",
|
|
||||||
"legendFormat": "{{groupname}}"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "bytes",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 7,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "Pod working set:前 15",
|
|
||||||
"description": "cgroup working set 与 PSS 口径不同,不相加。",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 0,
|
|
||||||
"y": 24,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "topk(15, sum by (namespace,pod) (container_memory_working_set_bytes{job=\"cadvisor\",container!=\"\",container!=\"POD\",pod!=\"\"}))",
|
|
||||||
"legendFormat": "{{namespace}}/{{pod}}"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "bytes",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 8,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "内存压力 PSI",
|
|
||||||
"description": "",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 12,
|
|
||||||
"y": 24,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "rate(node_pressure_memory_waiting_seconds_total{job=\"node-exporter\"}[5m])",
|
|
||||||
"legendFormat": "some"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "B",
|
|
||||||
"expr": "rate(node_pressure_memory_stalled_seconds_total{job=\"node-exporter\"}[5m])",
|
|
||||||
"legendFormat": "full"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "percentunit",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 9,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "Samba RPC worker 数量",
|
|
||||||
"description": "仅在 exporter 健康时将无 worker 解释为 0;见采集健康面板。",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 0,
|
|
||||||
"y": 32,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "namedprocess_namegroup_num_procs{job=\"process-exporter\",groupname=\"rpcd_lsad\"} or on() (0 * max(up{job=\"process-exporter\"} == 1))",
|
|
||||||
"legendFormat": "rpcd_lsad"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "short",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 10,
|
|
||||||
"type": "timeseries",
|
|
||||||
"title": "进程采集健康",
|
|
||||||
"description": "up 仅表示抓取成功;还需检查读取错误。进程退出等竞态可能造成偶发 partial errors,持续增长时检查 AppArmor 审计。",
|
|
||||||
"datasource": {
|
|
||||||
"type": "prometheus",
|
|
||||||
"uid": "${DS_VM}"
|
|
||||||
},
|
|
||||||
"gridPos": {
|
|
||||||
"x": 12,
|
|
||||||
"y": 32,
|
|
||||||
"w": 12,
|
|
||||||
"h": 8
|
|
||||||
},
|
|
||||||
"targets": [
|
|
||||||
{
|
|
||||||
"refId": "A",
|
|
||||||
"expr": "up{job=\"process-exporter\"}",
|
|
||||||
"legendFormat": "up"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "B",
|
|
||||||
"expr": "scrape_duration_seconds{job=\"process-exporter\"}",
|
|
||||||
"legendFormat": "scrape 秒"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "C",
|
|
||||||
"expr": "namedprocess_scrape_errors{job=\"process-exporter\"}",
|
|
||||||
"legendFormat": "采集错误"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"refId": "D",
|
|
||||||
"expr": "rate(namedprocess_scrape_partial_errors{job=\"process-exporter\"}[5m])",
|
|
||||||
"legendFormat": "部分字段读取失败/s"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"fieldConfig": {
|
|
||||||
"defaults": {
|
|
||||||
"unit": "short",
|
|
||||||
"min": 0
|
|
||||||
},
|
|
||||||
"overrides": []
|
|
||||||
},
|
|
||||||
"options": {
|
|
||||||
"legend": {
|
|
||||||
"displayMode": "table",
|
|
||||||
"placement": "bottom",
|
|
||||||
"calcs": [
|
|
||||||
"lastNotNull"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"tooltip": {
|
|
||||||
"mode": "multi"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
@@ -1,17 +0,0 @@
|
|||||||
# Grafana 自身使用 OIDC,不增加第二层 forward-auth。
|
|
||||||
apiVersion: gateway.networking.k8s.io/v1
|
|
||||||
kind: HTTPRoute
|
|
||||||
metadata:
|
|
||||||
name: grafana-lan
|
|
||||||
namespace: monitoring
|
|
||||||
spec:
|
|
||||||
parentRefs:
|
|
||||||
- name: eg
|
|
||||||
namespace: envoy-gateway-system
|
|
||||||
sectionName: https
|
|
||||||
hostnames:
|
|
||||||
- grafana.ad.ddupan.top
|
|
||||||
rules:
|
|
||||||
- backendRefs:
|
|
||||||
- name: grafana
|
|
||||||
port: 80
|
|
||||||
@@ -17,13 +17,5 @@ configMapGenerator:
|
|||||||
options:
|
options:
|
||||||
labels:
|
labels:
|
||||||
grafana_dashboard: "1"
|
grafana_dashboard: "1"
|
||||||
- name: grafana-dashboard-homelab-memory
|
|
||||||
namespace: monitoring
|
|
||||||
files:
|
|
||||||
- homelab-memory.json=dashboards/homelab-memory.json
|
|
||||||
options:
|
|
||||||
labels:
|
|
||||||
grafana_dashboard: "1"
|
|
||||||
resources:
|
resources:
|
||||||
- helmrelease.yaml
|
- helmrelease.yaml
|
||||||
- httproute.yaml
|
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
# Grafana — single pane over metrics (VictoriaMetrics), logs (VictoriaLogs) and
|
# Grafana — single pane over metrics (VictoriaMetrics), logs (VictoriaLogs) and
|
||||||
# traces (VictoriaTraces via Jaeger API). LAN: grafana.ad.ddupan.top,
|
# traces (VictoriaTraces via Jaeger API). Exposed privately on the tailnet
|
||||||
# authenticated via Authelia OIDC. Local admin is
|
# (grafana.tail7e769.ts.net) and authenticated via Authelia OIDC. Local admin is
|
||||||
# break-glass only.
|
# break-glass only.
|
||||||
#
|
#
|
||||||
# Chart: grafana/grafana (repo: https://grafana.github.io/helm-charts)
|
# Chart: grafana/grafana (repo: https://grafana.github.io/helm-charts)
|
||||||
@@ -54,10 +54,16 @@ persistence:
|
|||||||
deploymentStrategy:
|
deploymentStrategy:
|
||||||
type: Recreate
|
type: Recreate
|
||||||
|
|
||||||
# 内网入口由 httproute.yaml 接入 Envoy,使用现有通配 TLS 证书。
|
# --- Private exposure via the Tailscale ingress (like seaweedfs-admin) ---
|
||||||
# 客户端通过既有内网 DNS 转发解析,无需独立 Tailscale Ingress。
|
# The tailscale operator provisions grafana.<tailnet>.ts.net and a TLS cert.
|
||||||
ingress:
|
ingress:
|
||||||
enabled: false
|
enabled: true
|
||||||
|
ingressClassName: tailscale
|
||||||
|
hosts:
|
||||||
|
- grafana
|
||||||
|
tls:
|
||||||
|
- hosts:
|
||||||
|
- grafana
|
||||||
|
|
||||||
# --- OIDC via Authelia (AD groups -> Grafana roles) ---
|
# --- OIDC via Authelia (AD groups -> Grafana roles) ---
|
||||||
# client_secret is injected from the grafana-oidc Secret (see oidc-secret.yaml),
|
# client_secret is injected from the grafana-oidc Secret (see oidc-secret.yaml),
|
||||||
@@ -70,7 +76,7 @@ envValueFrom:
|
|||||||
|
|
||||||
grafana.ini:
|
grafana.ini:
|
||||||
server:
|
server:
|
||||||
root_url: "https://grafana.ad.ddupan.top" # 与 Authelia redirect_uri 一致
|
root_url: "https://grafana.tail7e769.ts.net" # must match the tailnet FQDN + Authelia redirect_uri
|
||||||
auth:
|
auth:
|
||||||
# Keep the local admin login available as break-glass; don't force OIDC-only.
|
# Keep the local admin login available as break-glass; don't force OIDC-only.
|
||||||
disable_login_form: false
|
disable_login_form: false
|
||||||
|
|||||||
@@ -3,7 +3,6 @@ kind: Kustomization
|
|||||||
resources:
|
resources:
|
||||||
- namespace.yaml
|
- namespace.yaml
|
||||||
- helmrepository.yaml
|
- helmrepository.yaml
|
||||||
- prometheus-helmrepository.yaml
|
|
||||||
- grafana-helmrepository.yaml
|
- grafana-helmrepository.yaml
|
||||||
- operator
|
- operator
|
||||||
- metrics
|
- metrics
|
||||||
|
|||||||
@@ -1,42 +0,0 @@
|
|||||||
# 主机与进程内存采集
|
|
||||||
|
|
||||||
2026-09-16 已增加 `process-exporter.yaml`,固定上游 0.8.7,以 DaemonSet 运行,
|
|
||||||
只读挂载宿主机 `/proc`,每 60 秒由 VMPodScrape 抓取,不开宿主机端口。
|
|
||||||
NetworkPolicy 仅放行同 namespace 的 vmagent。进程按名称分组,QEMU 按 guest 名,
|
|
||||||
NetBox 和 VS Code 按路径归组;不把完整命令行或 PID 放进指标标签。
|
|
||||||
|
|
||||||
当前 live 使用 `gather-smaps=true`,导出 RSS、VmSwap、PSS、SwapPss 与进程数。
|
|
||||||
`apparmor/homelab-process-exporter` 已安装到 `/etc/apparmor.d/homelab-process-exporter`
|
|
||||||
并以 enforce 加载。用户明确批准了跨进程读取:SYS_PTRACE、DAC_READ_SEARCH 与
|
|
||||||
`ptrace (read) peer=**`;文件权限限于指标需要的 proc 文件,不允许 ptrace trace,
|
|
||||||
不开放 `/proc/<pid>/mem`、`environ`,不关闭 AppArmor,也未修改虚拟机或容器默认 profile。
|
|
||||||
|
|
||||||
当前只部署到 laptop。其他节点必须先安装同名 profile,再扩展 nodeSelector:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
sudo install -m 0644 apparmor/homelab-process-exporter /etc/apparmor.d/homelab-process-exporter
|
|
||||||
sudo apparmor_parser -r /etc/apparmor.d/homelab-process-exporter
|
|
||||||
```
|
|
||||||
|
|
||||||
修改 profile 可原位重载,无需重启业务。禁用时先删除 exporter DaemonSet,再卸载
|
|
||||||
专用 profile;不要让引用 profile 的 Pod 在 profile 缺失时启动。
|
|
||||||
|
|
||||||
验证已读到三台虚拟机、NetBox、Codex 的非零 PSS/SwapPss。`namedprocess_scrape_errors`
|
|
||||||
和 `namedprocess_scrape_procread_errors` 为零。上游会先读取再匹配进程,内核线程没有
|
|
||||||
可读的 smaps_rollup 会计入 partial errors;现场有约 430 个这种线程。
|
|
||||||
partial errors 不保证为零,应结合业务组 PSS 和 AppArmor 审计判断。
|
|
||||||
RSS、PSS、cAdvisor working set、ARC 是不同口径,不能直接相加。
|
|
||||||
|
|
||||||
已有采集无需重复部署:
|
|
||||||
|
|
||||||
- node-exporter:主机内存、Swap、PSI、ZFS ARC。
|
|
||||||
- ARC 实际指标名为 `node_zfs_arc_size`、`node_zfs_arc_c`、`node_zfs_arc_c_max`。
|
|
||||||
- kubelet/cAdvisor:Pod/container working set、Swap 等。
|
|
||||||
|
|
||||||
Grafana 新增 `Homelab 内存与 Swap`(UID `homelab-memory`),由 sidecar ConfigMap 加载。
|
|
||||||
仅使用 `job="node-exporter"` 查询主机指标,避免目前 `docker-hosts` 对同一 9100 端口
|
|
||||||
的重复抓取。另有 `192.168.10.127:8080` Docker cAdvisor target 失效,本次未修改该旧配置。
|
|
||||||
|
|
||||||
所有新增 manifest 已纳入对应 Kustomization,并已单独应用到 live;尚未提交到远端,
|
|
||||||
Flux source 尚未包含这些新增资源。修改 exporter ConfigMap 后需滚动重启 DaemonSet,
|
|
||||||
程序不会自动重载进程匹配配置。
|
|
||||||
@@ -1,27 +0,0 @@
|
|||||||
#include <tunables/global>
|
|
||||||
|
|
||||||
# 限定为指标所需文件;跨 profile 读取由用户明确批准;不允许 trace 或读取进程 mem/environ。
|
|
||||||
profile homelab-process-exporter flags=(attach_disconnected,mediate_deleted) {
|
|
||||||
#include <abstractions/base>
|
|
||||||
network inet stream,
|
|
||||||
network inet6 stream,
|
|
||||||
capability sys_ptrace,
|
|
||||||
capability dac_read_search,
|
|
||||||
ptrace (read) peer=**,
|
|
||||||
signal (receive) peer=unconfined,
|
|
||||||
signal (receive) peer=cri-containerd.apparmor.d,
|
|
||||||
signal (receive) peer=runc,
|
|
||||||
/bin/process-exporter mr,
|
|
||||||
/process-exporter mr,
|
|
||||||
/config/** r,
|
|
||||||
/etc/{passwd,group,nsswitch.conf} r,
|
|
||||||
/{host/,}proc/ r,
|
|
||||||
/{host/,}proc/{stat,meminfo,cpuinfo,uptime,version,sys/kernel/random/boot_id} r,
|
|
||||||
/{host/,}proc/[0-9]*/ r,
|
|
||||||
/{host/,}proc/[0-9]*/{stat,status,cmdline,smaps,smaps_rollup,io,limits,cgroup,wchan} r,
|
|
||||||
/{host/,}proc/[0-9]*/fd/ r,
|
|
||||||
/{host/,}proc/[0-9]*/task/ r,
|
|
||||||
/{host/,}proc/[0-9]*/task/[0-9]*/{stat,status,io,cmdline,wchan,cgroup,limits,smaps,smaps_rollup} r,
|
|
||||||
/proc/sys/net/core/somaxconn r,
|
|
||||||
/proc/self/{stat,status,smaps,smaps_rollup,limits,cgroup,mountinfo} r,
|
|
||||||
}
|
|
||||||
@@ -1,124 +0,0 @@
|
|||||||
apiVersion: v1
|
|
||||||
kind: ConfigMap
|
|
||||||
metadata:
|
|
||||||
name: process-exporter-config
|
|
||||||
namespace: monitoring
|
|
||||||
data:
|
|
||||||
config.yml: |
|
|
||||||
process_names:
|
|
||||||
# 只暴露虚拟机名,不把完整命令行、PID 或凭据写入标签。
|
|
||||||
- name: 'vm:{{.Matches.VM}}'
|
|
||||||
comm: [qemu-system-x86]
|
|
||||||
cmdline: ['-name\s+guest=(?P<VM>[^,\s]+)']
|
|
||||||
- name: netbox
|
|
||||||
cmdline: ['/opt/netbox/']
|
|
||||||
- name: vscode
|
|
||||||
cmdline: ['\.vscode-server/']
|
|
||||||
- name: '{{.Comm}}'
|
|
||||||
cmdline: ['.+']
|
|
||||||
---
|
|
||||||
apiVersion: apps/v1
|
|
||||||
kind: DaemonSet
|
|
||||||
metadata:
|
|
||||||
name: process-exporter
|
|
||||||
namespace: monitoring
|
|
||||||
spec:
|
|
||||||
selector:
|
|
||||||
matchLabels:
|
|
||||||
app.kubernetes.io/name: process-exporter
|
|
||||||
template:
|
|
||||||
metadata:
|
|
||||||
labels:
|
|
||||||
app.kubernetes.io/name: process-exporter
|
|
||||||
spec:
|
|
||||||
nodeSelector:
|
|
||||||
kubernetes.io/hostname: laptop
|
|
||||||
hostPID: true
|
|
||||||
automountServiceAccountToken: false
|
|
||||||
tolerations:
|
|
||||||
- operator: Exists
|
|
||||||
containers:
|
|
||||||
- name: process-exporter
|
|
||||||
image: ncabatoff/process-exporter:0.8.7
|
|
||||||
args:
|
|
||||||
- -procfs=/host/proc
|
|
||||||
- -config.path=/config/config.yml
|
|
||||||
- -gather-smaps=true
|
|
||||||
- -threads=false
|
|
||||||
- -children=false
|
|
||||||
securityContext:
|
|
||||||
runAsUser: 0
|
|
||||||
readOnlyRootFilesystem: true
|
|
||||||
allowPrivilegeEscalation: false
|
|
||||||
appArmorProfile:
|
|
||||||
type: Localhost
|
|
||||||
localhostProfile: homelab-process-exporter
|
|
||||||
capabilities:
|
|
||||||
drop: [ALL]
|
|
||||||
add: [SYS_PTRACE, DAC_READ_SEARCH]
|
|
||||||
ports:
|
|
||||||
- name: metrics
|
|
||||||
containerPort: 9256
|
|
||||||
readinessProbe:
|
|
||||||
tcpSocket:
|
|
||||||
port: metrics
|
|
||||||
resources:
|
|
||||||
requests:
|
|
||||||
cpu: 25m
|
|
||||||
memory: 32Mi
|
|
||||||
limits:
|
|
||||||
cpu: 500m
|
|
||||||
memory: 128Mi
|
|
||||||
volumeMounts:
|
|
||||||
- name: proc
|
|
||||||
mountPath: /host/proc
|
|
||||||
readOnly: true
|
|
||||||
- name: config
|
|
||||||
mountPath: /config
|
|
||||||
readOnly: true
|
|
||||||
volumes:
|
|
||||||
- name: proc
|
|
||||||
hostPath:
|
|
||||||
path: /proc
|
|
||||||
type: Directory
|
|
||||||
- name: config
|
|
||||||
configMap:
|
|
||||||
name: process-exporter-config
|
|
||||||
---
|
|
||||||
apiVersion: operator.victoriametrics.com/v1beta1
|
|
||||||
kind: VMPodScrape
|
|
||||||
metadata:
|
|
||||||
name: process-exporter
|
|
||||||
namespace: monitoring
|
|
||||||
spec:
|
|
||||||
selector:
|
|
||||||
matchLabels:
|
|
||||||
app.kubernetes.io/name: process-exporter
|
|
||||||
podMetricsEndpoints:
|
|
||||||
- port: metrics
|
|
||||||
interval: 60s
|
|
||||||
scrapeTimeout: 30s
|
|
||||||
relabelConfigs:
|
|
||||||
- targetLabel: job
|
|
||||||
replacement: process-exporter
|
|
||||||
- sourceLabels: [__meta_kubernetes_pod_node_name]
|
|
||||||
targetLabel: node
|
|
||||||
---
|
|
||||||
apiVersion: networking.k8s.io/v1
|
|
||||||
kind: NetworkPolicy
|
|
||||||
metadata:
|
|
||||||
name: process-exporter
|
|
||||||
namespace: monitoring
|
|
||||||
spec:
|
|
||||||
podSelector:
|
|
||||||
matchLabels:
|
|
||||||
app.kubernetes.io/name: process-exporter
|
|
||||||
policyTypes: [Ingress]
|
|
||||||
ingress:
|
|
||||||
- from:
|
|
||||||
- podSelector:
|
|
||||||
matchLabels:
|
|
||||||
app.kubernetes.io/name: vmagent
|
|
||||||
ports:
|
|
||||||
- protocol: TCP
|
|
||||||
port: 9256
|
|
||||||
@@ -14,4 +14,3 @@ resources:
|
|||||||
- scrapes/docker-hosts.yaml
|
- scrapes/docker-hosts.yaml
|
||||||
- scrapes/kubelet.yaml
|
- scrapes/kubelet.yaml
|
||||||
- exporters/node-exporter.yaml
|
- exporters/node-exporter.yaml
|
||||||
- exporters/process-exporter.yaml
|
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
# 告警规则从已退役的 Compose 栈迁入,原始版本保留在 Git 历史。
|
# Ported from ../../../../apps/victoriametrics/rules/alerts-health.yml
|
||||||
apiVersion: operator.victoriametrics.com/v1beta1
|
apiVersion: operator.victoriametrics.com/v1beta1
|
||||||
kind: VMRule
|
kind: VMRule
|
||||||
metadata:
|
metadata:
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
# 告警规则从已退役的 Compose 栈迁入,原始版本保留在 Git 历史。
|
# Ported from ../../../../apps/victoriametrics/rules/alerts-vmagent.yml
|
||||||
apiVersion: operator.victoriametrics.com/v1beta1
|
apiVersion: operator.victoriametrics.com/v1beta1
|
||||||
kind: VMRule
|
kind: VMRule
|
||||||
metadata:
|
metadata:
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
# 告警规则从已退役的 Compose 栈迁入,原始版本保留在 Git 历史。
|
# Ported from ../../../../apps/victoriametrics/rules/alerts-vmalert.yml
|
||||||
apiVersion: operator.victoriametrics.com/v1beta1
|
apiVersion: operator.victoriametrics.com/v1beta1
|
||||||
kind: VMRule
|
kind: VMRule
|
||||||
metadata:
|
metadata:
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
# 告警规则从已退役的 Compose 栈迁入,原始版本保留在 Git 历史。
|
# Ported from ../../../../apps/victoriametrics/rules/alerts.yml
|
||||||
apiVersion: operator.victoriametrics.com/v1beta1
|
apiVersion: operator.victoriametrics.com/v1beta1
|
||||||
kind: VMRule
|
kind: VMRule
|
||||||
metadata:
|
metadata:
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
# Alert router/notifier. Migrated from the retired Compose stack (see Git history),
|
# Alert router/notifier. Ported 1:1 from ../../../apps/victoriametrics/alertmanager.yaml,
|
||||||
# which currently blackholes everything. Wire real receivers here (email via the
|
# which currently blackholes everything. Wire real receivers here (email via the
|
||||||
# in-cluster smtp-relay, or a webhook) when you want notifications.
|
# in-cluster smtp-relay, or a webhook) when you want notifications.
|
||||||
apiVersion: operator.victoriametrics.com/v1beta1
|
apiVersion: operator.victoriametrics.com/v1beta1
|
||||||
|
|||||||
@@ -10,5 +10,4 @@ configMapGenerator:
|
|||||||
files:
|
files:
|
||||||
- values.yaml=values.yaml
|
- values.yaml=values.yaml
|
||||||
resources:
|
resources:
|
||||||
- prometheus-crds-helmrelease.yaml
|
|
||||||
- helmrelease.yaml
|
- helmrelease.yaml
|
||||||
|
|||||||
@@ -1,29 +0,0 @@
|
|||||||
apiVersion: helm.toolkit.fluxcd.io/v2
|
|
||||||
kind: HelmRelease
|
|
||||||
metadata:
|
|
||||||
name: prometheus-operator-crds
|
|
||||||
namespace: monitoring
|
|
||||||
spec:
|
|
||||||
chart:
|
|
||||||
spec:
|
|
||||||
chart: prometheus-operator-crds
|
|
||||||
interval: 1h
|
|
||||||
sourceRef:
|
|
||||||
kind: HelmRepository
|
|
||||||
name: prometheus-community
|
|
||||||
namespace: monitoring
|
|
||||||
version: 32.0.0
|
|
||||||
install:
|
|
||||||
crds: CreateReplace
|
|
||||||
strategy:
|
|
||||||
name: RetryOnFailure
|
|
||||||
retryInterval: 5m
|
|
||||||
interval: 30m
|
|
||||||
releaseName: prometheus-operator-crds
|
|
||||||
targetNamespace: monitoring
|
|
||||||
timeout: 10m
|
|
||||||
upgrade:
|
|
||||||
crds: CreateReplace
|
|
||||||
strategy:
|
|
||||||
name: RetryOnFailure
|
|
||||||
retryInterval: 5m
|
|
||||||
@@ -1,8 +0,0 @@
|
|||||||
apiVersion: source.toolkit.fluxcd.io/v1
|
|
||||||
kind: HelmRepository
|
|
||||||
metadata:
|
|
||||||
name: prometheus-community
|
|
||||||
namespace: monitoring
|
|
||||||
spec:
|
|
||||||
interval: 1h
|
|
||||||
url: https://prometheus-community.github.io/helm-charts
|
|
||||||
@@ -47,37 +47,6 @@ PostgreSQL保存 registration state;SPIRE Server 的 disk KeyManager 仍使用
|
|||||||
`ClusterSPIFFEID`,并以 namespace、ServiceAccount、Pod label 等 selector 收窄。
|
`ClusterSPIFFEID`,并以 namespace、ServiceAccount、Pod label 等 selector 收窄。
|
||||||
不得仅因 Pod 能挂载 CSI socket 就给它签发身份。
|
不得仅因 Pod 能挂载 CSI socket 就给它签发身份。
|
||||||
|
|
||||||
## 宿主机本地开发身份
|
|
||||||
|
|
||||||
Kubernetes 节点 Agent 同时通过 hostPath 在宿主机发布 Workload API socket:
|
|
||||||
|
|
||||||
```text
|
|
||||||
/run/spire/agent-sockets/spire-agent.sock
|
|
||||||
```
|
|
||||||
|
|
||||||
Agent 已启用 Unix workload attestor,并为本机用户 `panxiao81`(UID `1000`)注册
|
|
||||||
`spiffe://ddupan.top/dev/panxiao81`。本地开发程序应设置:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
export SPIFFE_ENDPOINT_SOCKET=unix:///run/spire/agent-sockets/spire-agent.sock
|
|
||||||
```
|
|
||||||
|
|
||||||
该身份仅按 Unix UID 匹配,不是 SPIRE admin,也不会匹配 `sudo` 后以 root 运行的
|
|
||||||
进程。`ClusterStaticEntry.spec.parentID` 绑定当前 `laptop` Kubernetes node UID;若
|
|
||||||
节点被删除后重建,需从 `spire-server agent list` 取得新 Agent ID 并同步更新该字段。
|
|
||||||
|
|
||||||
下游授权仅绑定这个精确 SPIFFE ID:
|
|
||||||
|
|
||||||
- OpenBao `auth/jwt-spire/role/local-development` 接受 `aud=openbao`,签发 5 分钟
|
|
||||||
token;允许读取和写入 KV v2 的 `kv/k8s/*` 与 `kv/infra/*` 子树、列出对应
|
|
||||||
metadata、签发 `ai-agent` SSH
|
|
||||||
短证书,以及查询、撤销自身 token;不允许删除/永久销毁 KV 数据或管理 auth;
|
|
||||||
- zot 接受 `aud=zot`,允许本机开发身份对所有 repository 执行
|
|
||||||
`read/create/update/delete`;其他 SPIFFE 身份仍保持全仓库只读。
|
|
||||||
|
|
||||||
当前没有其他服务直接消费 SPIFFE 身份;Gitea runner 与 dynamic runner 是独立身份
|
|
||||||
使用方,NATS、SeaweedFS 等服务尚未通过 SPIFFE 做认证或授权。
|
|
||||||
|
|
||||||
稳定的 JWT issuer 预留为:
|
稳定的 JWT issuer 预留为:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
|
|||||||
@@ -15,4 +15,3 @@ resources:
|
|||||||
- helmrelease-crds.yaml
|
- helmrelease-crds.yaml
|
||||||
- helmrelease.yaml
|
- helmrelease.yaml
|
||||||
- httproute.yaml
|
- httproute.yaml
|
||||||
- local-development-identity.yaml
|
|
||||||
|
|||||||
@@ -1,12 +0,0 @@
|
|||||||
apiVersion: spire.spiffe.io/v1alpha1
|
|
||||||
kind: ClusterStaticEntry
|
|
||||||
metadata:
|
|
||||||
name: local-development-panxiao81
|
|
||||||
labels:
|
|
||||||
spire.spiffe.io/class-name: spire-mgmt-spire
|
|
||||||
spec:
|
|
||||||
className: spire-mgmt-spire
|
|
||||||
parentID: spiffe://ddupan.top/spire/agent/k8s_psat/homelab/cd2d0233-c4ea-4031-8327-e7e359e766dd
|
|
||||||
spiffeID: spiffe://ddupan.top/dev/panxiao81
|
|
||||||
selectors:
|
|
||||||
- unix:uid:1000
|
|
||||||
@@ -72,10 +72,7 @@ spire-agent:
|
|||||||
k8s:
|
k8s:
|
||||||
enabled: true
|
enabled: true
|
||||||
unix:
|
unix:
|
||||||
# The node Agent also exposes its Workload API socket on the host. Enable
|
enabled: false
|
||||||
# Unix attestation so local development processes can receive an identity
|
|
||||||
# through an explicitly scoped ClusterStaticEntry.
|
|
||||||
enabled: true
|
|
||||||
|
|
||||||
spiffe-csi-driver:
|
spiffe-csi-driver:
|
||||||
enabled: true
|
enabled: true
|
||||||
|
|||||||
Reference in New Issue
Block a user