Compare commits
45
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b617b4bc23 | ||
|
|
b6b4efe48f
|
||
|
|
d6518153c8 | ||
|
|
5b81719675
|
||
|
|
62c0ce1fff | ||
|
|
f3d8c38d28
|
||
|
|
61f0f864aa | ||
|
|
64b1acd6c3
|
||
|
|
bbf228de41 | ||
|
|
7703b5d21e
|
||
|
|
9c64d31d3d | ||
|
|
d9486b4c8c
|
||
|
|
f36e8a1733 | ||
|
|
926a90508a
|
||
|
|
cb2a66ded5 | ||
|
|
f767cd9a5e | ||
|
|
13279256a5
|
||
|
|
dc325f3a45 | ||
|
|
87bb556087
|
||
|
|
11f038794f
|
||
|
|
654087ef46 | ||
|
|
14fb5a323a
|
||
|
|
eb68e9f0c5 | ||
|
|
c4f425e9ea
|
||
|
|
d15733caac | ||
|
|
246b5023e7
|
||
|
|
21b7b48cdf | ||
|
|
94721c279b
|
||
|
|
31d89817b2
|
||
|
|
22bffe068c | ||
|
|
6078a06b99
|
||
|
|
47042d4df4 | ||
|
|
fd6bd62a4f | ||
|
|
3eb6f33dea
|
||
|
|
818192f453 | ||
|
|
387953c80a
|
||
|
|
298db6a745 | ||
|
|
f36a1cbf11
|
||
|
|
b953db199e | ||
|
|
b7b7975c9f
|
||
|
|
9604ff1004
|
||
|
|
819b521039 | ||
|
|
40703782ea
|
||
|
|
aae19850cf | ||
|
|
c11e1d5e6f
|
@@ -51,6 +51,11 @@ jobs:
|
||||
run: |
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
export ANSIBLE_COLLECTIONS_PATH="$PWD/infrastructure/samba-ad/ansible/collections:/root/.ansible/collections"
|
||||
# 静态检查不应依赖生产 vault 凭据。一次性 checkout 可以去掉加密变量文件;
|
||||
# syntax-check 只验证结构,不需要解析变量的运行时值。
|
||||
rm -f \
|
||||
infrastructure/openbao/ansible/group_vars/all/vault.yml \
|
||||
infrastructure/samba-ad/ansible/group_vars/all/vault.yml
|
||||
rc=0
|
||||
for p in infrastructure/openbao infrastructure/samba-ad infrastructure/proxmox; do
|
||||
echo "::group::$p"
|
||||
|
||||
@@ -229,7 +229,7 @@ configMap:
|
||||
require_pkce: false
|
||||
token_endpoint_auth_method: 'client_secret_basic'
|
||||
redirect_uris:
|
||||
- 'https://grafana.tail7e769.ts.net/login/generic_oauth'
|
||||
- 'https://grafana.ad.ddupan.top/login/generic_oauth'
|
||||
scopes:
|
||||
- 'openid'
|
||||
- 'profile'
|
||||
|
||||
@@ -48,6 +48,11 @@ gitea:
|
||||
# github.com is reachable from this network (verified 2026-07-28) even when
|
||||
# pypi.org/Fastly is not — see the flaky-WAN notes in the lint workflow.
|
||||
DEFAULT_ACTIONS_URL: github
|
||||
webhook:
|
||||
# Keep the default public-internet access for existing hooks while allowing
|
||||
# only the dynamic Runner controller's exact in-cluster DNS name. Do not
|
||||
# broaden this to the built-in `private` network group.
|
||||
ALLOWED_HOST_LIST: external,dynamic-runner-controller.dynamic-runner.svc.cluster.local
|
||||
mailer:
|
||||
# Outbound mail via the in-cluster Postfix+OAuth relay (see ../smtp-relay/).
|
||||
# Plain SMTP on :25 — the relay does STARTTLS + OAuth to M365. From must be the
|
||||
|
||||
@@ -1,5 +0,0 @@
|
||||
route:
|
||||
receiver: blackhole
|
||||
|
||||
receivers:
|
||||
- name: blackhole
|
||||
@@ -1,94 +0,0 @@
|
||||
services:
|
||||
# Metrics collector.
|
||||
# It scrapes targets defined in --promscrape.config
|
||||
# And forward them to --remoteWrite.url
|
||||
vmagent:
|
||||
image: victoriametrics/vmagent:v1.132.0
|
||||
depends_on:
|
||||
- "victoriametrics"
|
||||
ports:
|
||||
- 8429:8429
|
||||
volumes:
|
||||
- vmagentdata:/vmagentdata
|
||||
- ./prometheus.yaml:/etc/prometheus/prometheus.yml
|
||||
command:
|
||||
- "--promscrape.config=/etc/prometheus/prometheus.yml"
|
||||
- "--remoteWrite.url=http://victoriametrics:8428/api/v1/write"
|
||||
restart: always
|
||||
# VictoriaMetrics instance, a single process responsible for
|
||||
# storing metrics and serve read requests.
|
||||
victoriametrics:
|
||||
image: victoriametrics/victoria-metrics:v1.132.0
|
||||
ports:
|
||||
- 8428:8428
|
||||
- 8089:8089
|
||||
- 8089:8089/udp
|
||||
- 2003:2003
|
||||
- 2003:2003/udp
|
||||
- 4242:4242
|
||||
volumes:
|
||||
- vmdata:/storage
|
||||
command:
|
||||
- "--storageDataPath=/storage"
|
||||
- "--graphiteListenAddr=:2003"
|
||||
- "--opentsdbListenAddr=:4242"
|
||||
- "--httpListenAddr=:8428"
|
||||
- "--influxListenAddr=:8089"
|
||||
- "--vmalert.proxyURL=http://vmalert:8880"
|
||||
restart: always
|
||||
|
||||
grafana:
|
||||
image: grafana/grafana:12.2.0
|
||||
depends_on:
|
||||
- "victoriametrics"
|
||||
ports:
|
||||
- 3000:3000
|
||||
volumes:
|
||||
- grafanadata:/var/lib/grafana
|
||||
- ./provisioning/datasources/prometheus-datasource/single.yml:/etc/grafana/provisioning/datasources/single.yml
|
||||
- ./provisioning/dashboards:/etc/grafana/provisioning/dashboards
|
||||
- ./provisioning/dashboards/victoriametrics.json:/var/lib/grafana/dashboards/vm.json
|
||||
- ./provisioning/dashboards/vmagent.json:/var/lib/grafana/dashboards/vmagent.json
|
||||
- ./provisioning/dashboards/vmalert.json:/var/lib/grafana/dashboards/vmalert.json
|
||||
restart: always
|
||||
|
||||
# vmalert executes alerting and recording rules
|
||||
vmalert:
|
||||
image: victoriametrics/vmalert:v1.132.0
|
||||
depends_on:
|
||||
- "victoriametrics"
|
||||
- "alertmanager"
|
||||
ports:
|
||||
- 8880:8880
|
||||
volumes:
|
||||
- ./rules/alerts.yml:/etc/alerts/alerts.yml
|
||||
- ./rules/alerts-health.yml:/etc/alerts/alerts-health.yml
|
||||
- ./rules/alerts-vmagent.yml:/etc/alerts/alerts-vmagent.yml
|
||||
- ./rules/alerts-vmalert.yml:/etc/alerts/alerts-vmalert.yml
|
||||
command:
|
||||
- "--datasource.url=http://victoriametrics:8428/"
|
||||
- "--remoteRead.url=http://victoriametrics:8428/"
|
||||
- "--remoteWrite.url=http://vmagent:8429/"
|
||||
- "--notifier.url=http://alertmanager:9093/"
|
||||
- "--rule=/etc/alerts/*.yml"
|
||||
# display source of alerts in grafana
|
||||
- "--external.url=http://127.0.0.1:3000" #grafana outside container
|
||||
- '--external.alert.source=explore?orgId=1&left={"datasource":"VictoriaMetrics","queries":[{"expr":{{.Expr|jsonEscape|queryEscape}},"refId":"A"}],"range":{"from":"{{ .ActiveAt.UnixMilli }}","to":"now"}}'
|
||||
restart: always
|
||||
|
||||
# alertmanager receives alerting notifications from vmalert
|
||||
# and distributes them according to --config.file.
|
||||
alertmanager:
|
||||
image: prom/alertmanager:v0.28.1
|
||||
volumes:
|
||||
- ./alertmanager.yaml:/config/alertmanager.yml
|
||||
command:
|
||||
- "--config.file=/config/alertmanager.yml"
|
||||
ports:
|
||||
- 9093:9093
|
||||
restart: always
|
||||
|
||||
volumes:
|
||||
vmagentdata: {}
|
||||
vmdata: {}
|
||||
grafanadata: {}
|
||||
@@ -1,16 +0,0 @@
|
||||
global:
|
||||
scrape_interval: 10s
|
||||
|
||||
scrape_configs:
|
||||
- job_name: vmagent
|
||||
static_configs:
|
||||
- targets:
|
||||
- vmagent:8429
|
||||
- job_name: vmalert
|
||||
static_configs:
|
||||
- targets:
|
||||
- vmalert:8880
|
||||
- job_name: victoriametrics
|
||||
static_configs:
|
||||
- targets:
|
||||
- victoriametrics:8428
|
||||
@@ -1,9 +0,0 @@
|
||||
apiVersion: 1
|
||||
|
||||
providers:
|
||||
- name: Prometheus
|
||||
orgId: 1
|
||||
folder: ''
|
||||
type: file
|
||||
options:
|
||||
path: /var/lib/grafana/dashboards
|
||||
-11
@@ -1,11 +0,0 @@
|
||||
apiVersion: 1
|
||||
|
||||
datasources:
|
||||
- name: VictoriaMetrics
|
||||
type: prometheus
|
||||
access: proxy
|
||||
url: http://victoriametrics:8428
|
||||
isDefault: true
|
||||
jsonData:
|
||||
prometheusType: Prometheus
|
||||
prometheusVersion: 2.24.0
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,11 +0,0 @@
|
||||
apiVersion: 1
|
||||
|
||||
datasources:
|
||||
- name: VictoriaMetrics
|
||||
type: prometheus
|
||||
access: proxy
|
||||
url: http://victoriametrics:8428
|
||||
isDefault: true
|
||||
jsonData:
|
||||
prometheusType: Prometheus
|
||||
prometheusVersion: 2.24.0
|
||||
@@ -1,149 +0,0 @@
|
||||
# File contains default list of alerts for various VM components.
|
||||
# The following alerts are recommended for use for any VM installation.
|
||||
# The alerts below are just recommendations and may require some updates
|
||||
# and threshold calibration according to every specific setup.
|
||||
groups:
|
||||
- name: vm-health
|
||||
# note the `job` filter and update accordingly to your setup
|
||||
rules:
|
||||
- alert: TooManyRestarts
|
||||
expr: changes(process_start_time_seconds{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"}[15m]) > 2
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "{{ $labels.job }} too many restarts (instance {{ $labels.instance }})"
|
||||
description: >
|
||||
Job {{ $labels.job }} (instance {{ $labels.instance }}) has restarted more than twice in the last 15 minutes.
|
||||
It might be crashlooping.
|
||||
|
||||
- alert: ServiceDown
|
||||
expr: up{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"} == 0
|
||||
for: 2m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Service {{ $labels.job }} is down on {{ $labels.instance }}"
|
||||
description: "{{ $labels.instance }} of job {{ $labels.job }} has been down for more than 2 minutes."
|
||||
|
||||
- alert: ProcessNearFDLimits
|
||||
expr: (process_max_fds - process_open_fds) < 100
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Number of free file descriptors is less than 100 for \"{{ $labels.job }}\"(\"{{ $labels.instance }}\") for the last 5m"
|
||||
description: |
|
||||
Exhausting OS file descriptors limit can cause severe degradation of the process.
|
||||
Consider to increase the limit as fast as possible.
|
||||
|
||||
- alert: TooHighMemoryUsage
|
||||
expr: (min_over_time(process_resident_memory_anon_bytes[10m]) / vm_available_memory_bytes) > 0.8
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "It is more than 80% of memory used by \"{{ $labels.job }}\"(\"{{ $labels.instance }}\")"
|
||||
description: |
|
||||
Too high memory usage may result into multiple issues such as OOMs or degraded performance.
|
||||
Consider to either increase available memory or decrease the load on the process.
|
||||
|
||||
- alert: TooHighCPUUsage
|
||||
expr: rate(process_cpu_seconds_total[5m]) / process_cpu_cores_available > 0.9
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "More than 90% of CPU is used by \"{{ $labels.job }}\"(\"{{ $labels.instance }}\") during the last 5m"
|
||||
description: >
|
||||
Too high CPU usage may be a sign of insufficient resources and make process unstable.
|
||||
Consider to either increase available CPU resources or decrease the load on the process.
|
||||
|
||||
- alert: TooHighGoroutineSchedulingLatency
|
||||
expr: histogram_quantile(0.99, sum(rate(go_sched_latencies_seconds_bucket{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"}[5m])) by (le, job, instance)) > 0.1
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "\"{{ $labels.job }}\"(\"{{ $labels.instance }}\") has insufficient CPU resources for >15m"
|
||||
description: >
|
||||
Go runtime is unable to schedule goroutines execution in acceptable time. This is usually a sign of
|
||||
insufficient CPU resources or CPU throttling. Verify that service has enough CPU resources. Otherwise,
|
||||
the service could work unreliably with delays in processing.
|
||||
|
||||
- alert: TooManyLogs
|
||||
expr: sum(increase(vm_log_messages_total{level="error"}[5m])) without (app_version, location) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Too many logs printed for job \"{{ $labels.job }}\" ({{ $labels.instance }})"
|
||||
description: >
|
||||
Logging rate for job \"{{ $labels.job }}\" ({{ $labels.instance }}) is {{ $value }} for last 15m.
|
||||
Worth to check logs for specific error messages.
|
||||
|
||||
- alert: TooManyTSIDMisses
|
||||
expr: increase(vm_missing_tsids_for_metric_id_total[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Unexpected TSID misses for job \"{{ $labels.job }}\" ({{ $labels.instance }}) for the last 15 minutes"
|
||||
description: |
|
||||
Unexpected TSID misses for \"{{ $labels.job }}\" ({{ $labels.instance }}) for the last 15 minutes.
|
||||
If this happens after unclean shutdown of VictoriaMetrics process (via \"kill -9\", OOM or power off),
|
||||
then this is OK - the alert must go away in a few minutes after the restart.
|
||||
Otherwise this may point to the corruption of index data.
|
||||
|
||||
- alert: ConcurrentInsertsHitTheLimit
|
||||
expr: avg_over_time(vm_concurrent_insert_current[1m]) >= vm_concurrent_insert_capacity
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "{{ $labels.job }} on instance {{ $labels.instance }} is constantly hitting concurrent inserts limit"
|
||||
description: |
|
||||
The limit of concurrent inserts on instance {{ $labels.instance }} depends on the number of CPUs.
|
||||
Usually, when component constantly hits the limit it is likely the component is overloaded and requires more CPU.
|
||||
In some cases for components like vmagent or vminsert the alert might trigger if there are too many clients
|
||||
making write attempts. If vmagent's or vminsert's CPU usage and network saturation are at normal level, then
|
||||
it might be worth adjusting `-maxConcurrentInserts` cmd-line flag.
|
||||
|
||||
- alert: IndexDBRecordsDrop
|
||||
expr: increase(vm_indexdb_items_dropped_total[5m]) > 0
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "IndexDB skipped registering items during data ingestion with reason={{ $labels.reason }}."
|
||||
description: |
|
||||
VictoriaMetrics could skip registering new timeseries during ingestion if they fail the validation process.
|
||||
For example, `reason=too_long_item` means that time series cannot exceed 64KB. Please, reduce the number
|
||||
of labels or label values for such series. Or enforce these limits via `-maxLabelsPerTimeseries` and
|
||||
`-maxLabelValueLen` command-line flags.
|
||||
|
||||
- alert: RowsRejectedOnIngestion
|
||||
expr: rate(vm_rows_ignored_total[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Some rows are rejected on \"{{ $labels.instance }}\" on ingestion attempt"
|
||||
description: "Ingested rows on instance \"{{ $labels.instance }}\" are rejected due to the
|
||||
following reason: \"{{ $labels.reason }}\""
|
||||
|
||||
- alert: TooHighQueryLoad
|
||||
expr: increase(vm_concurrent_select_limit_timeout_total[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Read queries fail with timeout for {{ $labels.job }} on instance {{ $labels.instance }}"
|
||||
description: |
|
||||
Instance {{ $labels.instance }} ({{ $labels.job }}) is failing to serve read queries during last 15m.
|
||||
Concurrency limit `-search.maxConcurrentRequests` was reached on this instance and extra queries were
|
||||
put into the queue for `-search.maxQueueDuration` interval. But even after waiting in the queue these queries weren't served.
|
||||
This happens if instance is overloaded with the current workload, or datasource is too slow to respond.
|
||||
Possible solutions are the following:
|
||||
* reduce the query load;
|
||||
* increase compute resources or number of replicas;
|
||||
* adjust limits `-search.maxConcurrentRequests` and `-search.maxQueueDuration`.
|
||||
See more at https://docs.victoriametrics.com/victoriametrics/troubleshooting/#slow-queries
|
||||
@@ -1,172 +0,0 @@
|
||||
# File contains default list of alerts for vmagent service.
|
||||
# The alerts below are just recommendations and may require some updates
|
||||
# and threshold calibration according to every specific setup.
|
||||
groups:
|
||||
# Alerts group for vmagent assumes that Grafana dashboard
|
||||
# https://grafana.com/grafana/dashboards/12683 is installed.
|
||||
# Pls update the `dashboard` annotation according to your setup.
|
||||
- name: vmagent
|
||||
interval: 30s
|
||||
concurrency: 2
|
||||
rules:
|
||||
- alert: PersistentQueueIsDroppingData
|
||||
expr: sum(increase(vm_persistentqueue_bytes_dropped_total[5m])) without (path) > 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=49&var-instance={{ $labels.instance }}"
|
||||
summary: "Instance {{ $labels.instance }} is dropping data from persistent queue"
|
||||
description: "Vmagent dropped {{ $value | humanize1024 }} from persistent queue
|
||||
on instance {{ $labels.instance }} for the last 10m."
|
||||
|
||||
- alert: RejectedRemoteWriteDataBlocksAreDropped
|
||||
expr: sum(increase(vmagent_remotewrite_packets_dropped_total[5m])) without (url) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=79&var-instance={{ $labels.instance }}"
|
||||
summary: "Vmagent is dropping data blocks that are rejected by remote storage"
|
||||
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} drops the rejected by
|
||||
remote-write server data blocks. Check the logs to find the reason for rejects."
|
||||
|
||||
- alert: TooManyScrapeErrors
|
||||
expr: increase(vm_promscrape_scrapes_failed_total[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=31&var-instance={{ $labels.instance }}"
|
||||
summary: "Vmagent fails to scrape one or more targets"
|
||||
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} fails to scrape targets for last 15m"
|
||||
|
||||
- alert: ScrapePoolHasNoTargets
|
||||
expr: sum(vm_promscrape_scrape_pool_targets) without (status, instance, pod) == 0
|
||||
for: 30m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Vmagent has scrape_pool with 0 configured/discovered targets"
|
||||
description: "Vmagent \"{{ $labels.job }}\" has scrape_pool \"{{ $labels.scrape_job }}\"
|
||||
with 0 discovered targets. It is likely a misconfiguration. Please follow https://docs.victoriametrics.com/victoriametrics/vmagent/#debugging-scrape-targets
|
||||
to troubleshoot the scraping config."
|
||||
|
||||
- alert: TooManyWriteErrors
|
||||
expr: |
|
||||
(sum(increase(vm_ingestserver_request_errors_total[5m])) without (name,net,type)
|
||||
+
|
||||
sum(increase(vmagent_http_request_errors_total[5m])) without (path,protocol)) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=77&var-instance={{ $labels.instance }}"
|
||||
summary: "Vmagent responds with too many errors on data ingestion protocols"
|
||||
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} responds with errors to write requests for last 15m."
|
||||
|
||||
- alert: TooManyRemoteWriteErrors
|
||||
expr: rate(vmagent_remotewrite_retries_count_total[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=61&var-instance={{ $labels.instance }}"
|
||||
summary: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} fails to push to remote storage"
|
||||
description: "Vmagent fails to push data via remote write protocol to destination \"{{ $labels.url }}\"\n
|
||||
Ensure that destination is up and reachable."
|
||||
|
||||
- alert: RemoteWriteConnectionIsSaturated
|
||||
expr: |
|
||||
(
|
||||
rate(vmagent_remotewrite_send_duration_seconds_total[5m])
|
||||
/
|
||||
vmagent_remotewrite_queues
|
||||
) > 0.9
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=84&var-instance={{ $labels.instance }}"
|
||||
summary: "Remote write connection from \"{{ $labels.job }}\" (instance {{ $labels.instance }}) to {{ $labels.url }} is saturated"
|
||||
description: "The remote write connection between vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }}) and destination \"{{ $labels.url }}\"
|
||||
is saturated by more than 90% and vmagent won't be able to keep up.\n
|
||||
There could be the following reasons for this:\n
|
||||
* vmagent can't send data fast enough through the existing network connections. Increase `-remoteWrite.queues` cmd-line flag value to establish more connections per destination.\n
|
||||
* remote destination can't accept data fast enough. Check if remote destination has enough resources for processing."
|
||||
|
||||
- alert: PersistentQueueForWritesIsSaturated
|
||||
expr: rate(vm_persistentqueue_write_duration_seconds_total[5m]) > 0.9
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=98&var-instance={{ $labels.instance }}"
|
||||
summary: "Persistent queue writes for instance {{ $labels.instance }} are saturated"
|
||||
description: "Persistent queue writes for vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }})
|
||||
are saturated by more than 90% and vmagent won't be able to keep up with flushing data on disk.
|
||||
In this case, consider to decrease load on the vmagent or improve the disk throughput."
|
||||
|
||||
- alert: PersistentQueueForReadsIsSaturated
|
||||
expr: rate(vm_persistentqueue_read_duration_seconds_total[5m]) > 0.9
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=99&var-instance={{ $labels.instance }}"
|
||||
summary: "Persistent queue reads for instance {{ $labels.instance }} are saturated"
|
||||
description: "Persistent queue reads for vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }})
|
||||
are saturated by more than 90% and vmagent won't be able to keep up with reading data from the disk.
|
||||
In this case, consider to decrease load on the vmagent or improve the disk throughput."
|
||||
|
||||
- alert: SeriesLimitHourReached
|
||||
expr: (vmagent_hourly_series_limit_current_series / vmagent_hourly_series_limit_max_series) > 0.9
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=88&var-instance={{ $labels.instance }}"
|
||||
summary: "Instance {{ $labels.instance }} reached 90% of the limit"
|
||||
description: "Max series limit set via -remoteWrite.maxHourlySeries flag is close to reaching the max value.
|
||||
Then samples for new time series will be dropped instead of sending them to remote storage systems."
|
||||
|
||||
- alert: SeriesLimitDayReached
|
||||
expr: (vmagent_daily_series_limit_current_series / vmagent_daily_series_limit_max_series) > 0.9
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=90&var-instance={{ $labels.instance }}"
|
||||
summary: "Instance {{ $labels.instance }} reached 90% of the limit"
|
||||
description: "Max series limit set via -remoteWrite.maxDailySeries flag is close to reaching the max value.
|
||||
Then samples for new time series will be dropped instead of sending them to remote storage systems."
|
||||
|
||||
- alert: ConfigurationReloadFailure
|
||||
expr: |
|
||||
vm_promscrape_config_last_reload_successful != 1
|
||||
or
|
||||
vmagent_relabel_config_last_reload_successful != 1
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Configuration reload failed for vmagent instance {{ $labels.instance }}"
|
||||
description: "Configuration hot-reload failed for vmagent on instance {{ $labels.instance }}.
|
||||
Check vmagent's logs for detailed error message."
|
||||
|
||||
- alert: StreamAggrFlushTimeout
|
||||
expr: |
|
||||
increase(vm_streamaggr_flush_timeouts_total[5m]) > 0
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Streaming aggregation at \"{{ $labels.job }}\" (instance {{ $labels.instance }}) can't be finished within the configured aggregation interval."
|
||||
description: "Stream aggregation process can't keep up with the load and might produce incorrect aggregation results. Check logs for more details.
|
||||
Possible solutions: increase aggregation interval; aggregate smaller number of series; reduce samples' ingestion rate to stream aggregation."
|
||||
|
||||
- alert: StreamAggrDedupFlushTimeout
|
||||
expr: |
|
||||
increase(vm_streamaggr_dedup_flush_timeouts_total[5m]) > 0
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Deduplication \"{{ $labels.job }}\" (instance {{ $labels.instance }}) can't be finished within configured deduplication interval."
|
||||
description: "Deduplication process can't keep up with the load and might produce incorrect results. Check docs https://docs.victoriametrics.com/victoriametrics/stream-aggregation/#deduplication and logs for more details.
|
||||
Possible solutions: increase deduplication interval; deduplicate smaller number of series; reduce samples' ingestion rate."
|
||||
@@ -1,96 +0,0 @@
|
||||
# File contains default list of alerts for vmalert service.
|
||||
# The alerts below are just recommendations and may require some updates
|
||||
# and threshold calibration according to every specific setup.
|
||||
groups:
|
||||
# Alerts group for vmalert assumes that Grafana dashboard
|
||||
# https://grafana.com/grafana/dashboards/14950 is installed.
|
||||
# Pls update the `dashboard` annotation according to your setup.
|
||||
- name: vmalert
|
||||
interval: 30s
|
||||
rules:
|
||||
- alert: ConfigurationReloadFailure
|
||||
expr: vmalert_config_last_reload_successful != 1
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Configuration reload failed for vmalert instance {{ $labels.instance }}"
|
||||
description: "Configuration hot-reload failed for vmalert on instance {{ $labels.instance }}.
|
||||
Check vmalert's logs for detailed error message."
|
||||
|
||||
- alert: AlertingRulesError
|
||||
expr: sum(increase(vmalert_alerting_rules_errors_total[5m])) without(id) > 0
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=13&var-instance={{ $labels.instance }}&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
|
||||
summary: "Alerting rules are failing for vmalert instance {{ $labels.instance }}"
|
||||
description: "Alerting rules execution is failing for \"{{ $labels.alertname }}\" from group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
|
||||
Check vmalert's logs for detailed error message."
|
||||
|
||||
- alert: RecordingRulesError
|
||||
expr: sum(increase(vmalert_recording_rules_errors_total[5m])) without(id) > 0
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=30&var-instance={{ $labels.instance }}&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
|
||||
summary: "Recording rules are failing for vmalert instance {{ $labels.instance }}"
|
||||
description: "Recording rules execution is failing for \"{{ $labels.recording }}\" from group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
|
||||
Check vmalert's logs for detailed error message."
|
||||
|
||||
- alert: RecordingRulesNoData
|
||||
expr: sum(vmalert_recording_rules_last_evaluation_samples) without(id) < 1
|
||||
for: 30m
|
||||
labels:
|
||||
severity: info
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=33&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
|
||||
summary: "Recording rule {{ $labels.recording }} ({{ $labels.group }}) produces no data"
|
||||
description: "Recording rule \"{{ $labels.recording }}\" from group \"{{ $labels.group }}\ in file \"{{ $labels.file }}\"
|
||||
produces 0 samples over the last 30min. It might be caused by a misconfiguration
|
||||
or incorrect query expression."
|
||||
|
||||
- alert: TooManyMissedIterations
|
||||
expr: increase(vmalert_iteration_missed_total[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "vmalert instance {{ $labels.instance }} is missing rules evaluations"
|
||||
description: "vmalert instance {{ $labels.instance }} is missing rules evaluations for group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
|
||||
The group evaluation time takes longer than the configured evaluation interval. This may result in missed
|
||||
alerting notifications or recording rules samples. Try increasing evaluation interval or concurrency of
|
||||
group \"{{ $labels.group }}\". See https://docs.victoriametrics.com/victoriametrics/vmalert/#groups.
|
||||
If rule expressions are taking longer than expected, please see https://docs.victoriametrics.com/victoriametrics/troubleshooting/#slow-queries."
|
||||
|
||||
- alert: RemoteWriteErrors
|
||||
expr: increase(vmalert_remotewrite_errors_total[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "vmalert instance {{ $labels.instance }} is failing to push metrics to remote write URL"
|
||||
description: "vmalert instance {{ $labels.instance }} is failing to push metrics generated via alerting
|
||||
or recording rules to the configured remote write URL. Check vmalert's logs for detailed error message."
|
||||
|
||||
- alert: RemoteWriteDroppingData
|
||||
expr: increase(vmalert_remotewrite_dropped_rows_total[5m]) > 0
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "vmalert instance {{ $labels.instance }} is dropping data sent to remote write URL"
|
||||
description: "vmalert instance {{ $labels.instance }} is failing to send results of alerting or recording rules
|
||||
to the configured remote write URL. This may result into gaps in recording rules or alerts state.
|
||||
Check vmalert's logs for detailed error message."
|
||||
|
||||
- alert: AlertmanagerErrors
|
||||
expr: increase(vmalert_alerts_send_errors_total[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "vmalert instance {{ $labels.instance }} is failing to send notifications to Alertmanager"
|
||||
description: "vmalert instance {{ $labels.instance }} is failing to send alert notifications to \"{{ $labels.addr }}\".
|
||||
Check vmalert's logs for detailed error message."
|
||||
@@ -1,138 +0,0 @@
|
||||
# File contains default list of alerts for VictoriaMetrics single server.
|
||||
# The alerts below are just recommendations and may require some updates
|
||||
# and threshold calibration according to every specific setup.
|
||||
groups:
|
||||
# Alerts group for VM single assumes that Grafana dashboard
|
||||
# https://grafana.com/grafana/dashboards/10229 is installed.
|
||||
# Pls update the `dashboard` annotation according to your setup.
|
||||
- name: vmsingle
|
||||
interval: 30s
|
||||
concurrency: 2
|
||||
rules:
|
||||
- alert: DiskRunsOutOfSpaceIn3Days
|
||||
expr: |
|
||||
sum(vm_free_disk_space_bytes) without(path) /
|
||||
(
|
||||
(rate(vm_rows_added_to_storage_total[1d]) - sum(rate(vm_deduplicated_samples_total[1d])) without(type)) * (
|
||||
sum(vm_data_size_bytes{type!~"indexdb.*"}) without(type) /
|
||||
sum(vm_rows{type!~"indexdb.*"}) without(type)
|
||||
)
|
||||
+
|
||||
rate(vm_new_timeseries_created_total[1d]) * (
|
||||
sum(vm_data_size_bytes{type="indexdb/file"}) without(type)/
|
||||
sum(vm_rows{type="indexdb/file"}) without(type)
|
||||
)
|
||||
) < 3 * 24 * 3600 > 0
|
||||
for: 30m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=53&var-instance={{ $labels.instance }}"
|
||||
summary: "Instance {{ $labels.instance }} will run out of disk space soon"
|
||||
description: "Taking into account current ingestion rate, free disk space will be enough only
|
||||
for {{ $value | humanizeDuration }} on instance {{ $labels.instance }}.\n
|
||||
Consider to limit the ingestion rate, decrease retention or scale the disk space if possible."
|
||||
|
||||
- alert: NodeBecomesReadonlyIn3Days
|
||||
expr: |
|
||||
sum(vm_free_disk_space_bytes - vm_free_disk_space_limit_bytes) without(path) /
|
||||
(
|
||||
(rate(vm_rows_added_to_storage_total[1d]) - sum(rate(vm_deduplicated_samples_total[1d])) without(type)) * (
|
||||
sum(vm_data_size_bytes{type!~"indexdb.*"}) without(type) /
|
||||
sum(vm_rows{type!~"indexdb.*"}) without(type)
|
||||
)
|
||||
+
|
||||
rate(vm_new_timeseries_created_total[1d]) * (
|
||||
sum(vm_data_size_bytes{type="indexdb/file"}) without(type) /
|
||||
sum(vm_rows{type="indexdb/file"}) without(type)
|
||||
)
|
||||
) < 3 * 24 * 3600 > 0
|
||||
for: 30m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/oS7Bi_0Wz?viewPanel=53&var-instance={{ $labels.instance }}"
|
||||
summary: "Instance {{ $labels.instance }} will become read-only in 3 days"
|
||||
description: "Taking into account current ingestion rate and free disk space
|
||||
instance {{ $labels.instance }} is writable for {{ $value | humanizeDuration }}.\n
|
||||
Consider to limit the ingestion rate, decrease retention or scale the disk space up if possible."
|
||||
|
||||
- alert: DiskRunsOutOfSpace
|
||||
expr: |
|
||||
sum(vm_data_size_bytes) by(job, instance) /
|
||||
(
|
||||
sum(vm_free_disk_space_bytes) by(job, instance) +
|
||||
sum(vm_data_size_bytes) by(job, instance)
|
||||
) > 0.8
|
||||
for: 30m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=53&var-instance={{ $labels.instance }}"
|
||||
summary: "Instance {{ $labels.instance }} (job={{ $labels.job }}) will run out of disk space soon"
|
||||
description: "Disk utilisation on instance {{ $labels.instance }} is more than 80%.\n
|
||||
Having less than 20% of free disk space could cripple merge processes and overall performance.
|
||||
Consider to limit the ingestion rate, decrease retention or scale the disk space if possible."
|
||||
|
||||
- alert: RequestErrorsToAPI
|
||||
expr: increase(vm_http_request_errors_total[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=35&var-instance={{ $labels.instance }}"
|
||||
summary: "Too many errors served for path {{ $labels.path }} (instance {{ $labels.instance }})"
|
||||
description: "Requests to path {{ $labels.path }} are receiving errors.
|
||||
Please verify if clients are sending correct requests."
|
||||
|
||||
- alert: TooHighChurnRate
|
||||
expr: |
|
||||
(
|
||||
sum(rate(vm_new_timeseries_created_total[5m])) by(instance)
|
||||
/
|
||||
sum(rate(vm_rows_inserted_total[5m])) by(instance)
|
||||
) > 0.1
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=66&var-instance={{ $labels.instance }}"
|
||||
summary: "Churn rate is more than 10% on \"{{ $labels.instance }}\" for the last 15m"
|
||||
description: "VM constantly creates new time series on \"{{ $labels.instance }}\".\n
|
||||
This effect is known as Churn Rate.\n
|
||||
High Churn Rate is tightly connected with database performance and may
|
||||
result in unexpected OOM's or slow queries."
|
||||
|
||||
- alert: TooHighChurnRate24h
|
||||
expr: |
|
||||
sum(increase(vm_new_timeseries_created_total[24h])) by(instance)
|
||||
>
|
||||
(sum(vm_cache_entries{type="storage/hour_metric_ids"}) by(instance) * 3)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=66&var-instance={{ $labels.instance }}"
|
||||
summary: "Too high number of new series on \"{{ $labels.instance }}\" created over last 24h"
|
||||
description: "The number of created new time series over last 24h is 3x times higher than
|
||||
current number of active series on \"{{ $labels.instance }}\".\n
|
||||
This effect is known as Churn Rate.\n
|
||||
High Churn Rate is tightly connected with database performance and may
|
||||
result in unexpected OOM's or slow queries."
|
||||
|
||||
- alert: TooHighSlowInsertsRate
|
||||
expr: |
|
||||
(
|
||||
sum(rate(vm_slow_row_inserts_total[5m])) by(instance)
|
||||
/
|
||||
sum(rate(vm_rows_inserted_total[5m])) by(instance)
|
||||
) > 0.05
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=68&var-instance={{ $labels.instance }}"
|
||||
summary: "Percentage of slow inserts is more than 5% on \"{{ $labels.instance }}\" for the last 15m"
|
||||
description: "High rate of slow inserts on \"{{ $labels.instance }}\" may be a sign of resource exhaustion
|
||||
for the current load. It is likely more RAM is needed for optimal handling of the current number of active time series.
|
||||
See also https://github.com/VictoriaMetrics/VictoriaMetrics/issues/3976#issuecomment-1476883183"
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+80
-14
@@ -1,15 +1,32 @@
|
||||
# zot OCI Registry
|
||||
|
||||
内网入口为 `https://zot.ad.ddupan.top`。使用官方 Helm chart `0.1.124`,运行
|
||||
内网匿名拉取入口为 `https://zot.ad.ddupan.top`,SPIRE 鉴权推送入口为
|
||||
`https://zot-push.ad.ddupan.top`。使用官方 Helm chart `0.1.124`,运行
|
||||
zot `v2.1.21`,镜像固定到官方 linux/amd64 digest。
|
||||
|
||||
## 当前工作状态(2026-09-16 核验)
|
||||
|
||||
| 项目 | 状态 |
|
||||
|---|---|
|
||||
| 匿名拉取 | `zot.ad.ddupan.top` 已上线;空 `DOCKER_CONFIG` 的 crane pull 通过 |
|
||||
| SPIRE 鉴权入口 | `zot-push.ad.ddupan.top` 已上线;真实 JWT-SVID 推送后可匿名拉取同一 digest |
|
||||
| GitOps | 双入口配置已合并;Flux `zot` Kustomization 已应用 `d15733c`,状态 Ready |
|
||||
| 运行与凭据同步 | `zot`、`zot-reader` HelmRelease 均 Ready,Pod 均 1/1;ESO SecretSynced |
|
||||
| 临时配置清理 | 两个 HelmRelease 均无 `spec.values` 临时覆盖;暂停回写标记、测试身份和临时写权限已清理 |
|
||||
| 接管复验 | 匿名拉取成功;推送入口无凭据返回 401,token realm 指向推送域名;接管未触发 Pod 重启 |
|
||||
|
||||
后续工作是给实际 CI 的 SPIFFE ID 配置具体仓库的 `create`/`update` 权限。
|
||||
SPIRE 认证链路已经验证,但当前没有常驻 publisher 或删除授权;认证成功本身不代表
|
||||
可以推送。S3 侧仍使用 Bao 管理的静态 AK/SK,尚未接入 SPIRE/STS。
|
||||
|
||||
## 存储与凭据
|
||||
|
||||
制品、manifest 和 OCI layout 保存在现有 SeaweedFS 的 `zot` bucket,前缀为
|
||||
`registry/`,S3 endpoint 为 `https://s3.ad.ddupan.top`。**不创建 PVC**;chart 的
|
||||
`/var/lib/registry` 是 `emptyDir`,仅用于运行时本地工作数据。
|
||||
|
||||
首期单副本,关闭跨仓库 dedupe,不额外部署 Redis/DynamoDB 缓存。保留 zot GC,
|
||||
两个单副本实例共用同一 bucket 和前缀:`zot` 负责鉴权写入,`zot-reader` 负责匿名
|
||||
读取。关闭跨仓库 dedupe,不额外部署 Redis/DynamoDB 缓存。只有写入实例启用 GC,
|
||||
暂不配置自动删除已发布版本的 retention policy。增加副本、启用 dedupe 或搜索等
|
||||
扩展前,需要重新检查共享元数据与缓存的持久化要求。
|
||||
|
||||
@@ -20,7 +37,7 @@ OpenBao kv/k8s/seaweedfs-s3
|
||||
→ 原有 S3 身份及基础配置 ─┐
|
||||
├→ ESO 模板 → seaweedfs/seaweedfs-s3-config
|
||||
OpenBao kv/k8s/zot-s3 ────┘ → 完整 s3.config
|
||||
└→ ESO → zot/zot-s3 → zot 的 AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY
|
||||
└→ ESO → zot/zot-s3 → zot 与 zot-reader 的 AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY
|
||||
|
||||
```
|
||||
|
||||
@@ -37,7 +54,7 @@ Terraform 的 `tfstate` bucket。`kv/k8s/zot-s3` 的字段是 `access_key` 和
|
||||
本次归一没有轮换密钥,生成的完整配置与归一前语义一致。当前模板只有一组 zot
|
||||
凭据,尚未实现新旧密钥重叠轮换。后续轮换只修改 `kv/k8s/zot-s3`,但仍需协调
|
||||
两个 ExternalSecret 同步:确认 SeaweedFS Secret volume 更新后向 filer 的
|
||||
`weed` 进程发送 SIGHUP,再确认 zot Secret 更新并重启 zot(环境变量不会热更新)。
|
||||
`weed` 进程发送 SIGHUP,再确认 zot Secret 更新并重启 `zot` 和 `zot-reader`(环境变量不会热更新)。
|
||||
两端异步更新期间可能短暂认证失败;需要无中断轮换时先扩展模板支持新旧凭据重叠。
|
||||
|
||||
## SPIRE 认证和授权
|
||||
@@ -47,24 +64,40 @@ Terraform 的 `tfstate` bucket。`kv/k8s/zot-s3` 的字段是 `access_key` 和
|
||||
| issuer | `https://spire-oidc.ad.ddupan.top` |
|
||||
| JWT audience | `zot` |
|
||||
| subject | `spiffe://ddupan.top/` 下的 workload SPIFFE ID |
|
||||
| token endpoint | `https://zot.ad.ddupan.top/zot/auth/token` |
|
||||
| 当前权限 | 受信身份可以读取所有仓库;没有常驻写入或删除授权 |
|
||||
| token endpoint | `https://zot-push.ad.ddupan.top/zot/auth/token` |
|
||||
| 拉取入口 | 内网匿名读取所有仓库,不要求 SPIRE 身份 |
|
||||
| 推送入口当前权限 | 受信身份可以读取所有仓库;没有常驻写入或删除授权 |
|
||||
|
||||
zot 通过已配置的 issuer discovery/JWKS 验证 JWT-SVID,再以 `sub` 作为授权身份。
|
||||
不接受任意 issuer,不关闭 TLS/issuer 验证。新的 Kata CI 负责取得并更新自己的
|
||||
JWT-SVID;确认其身份命名后,再添加针对具体 repository 的 `create`/`update`
|
||||
授权,不能把整个 trust domain 都授予写权限。
|
||||
|
||||
现阶段拉取也需要 JWT-SVID。原定内网匿名拉取尚未启用:zot `v2.1.21` 的
|
||||
OIDC Bearer middleware 会在授权阶段之前拒绝无 token 请求,单独增加
|
||||
`anonymousPolicy` 无法解决。匿名读取与 SPIRE 写入共存需后续单独验证方案。
|
||||
zot `v2.1.21` 的 OIDC Bearer middleware 会在授权阶段之前拒绝无 token 请求。
|
||||
因此使用两个官方 zot 实例与两个域名,避免修改上游镜像,也避免同域名下匿名
|
||||
`/v2/` 返回 200 导致标准客户端跳过 token 交换的问题。
|
||||
|
||||
- `zot-reader` 叠加 `reader-values.yaml`,没有认证 middleware,只有
|
||||
`anonymousPolicy: [read]`。入口只转发 `/v2/` 的 GET/HEAD,并移除客户端遗留的
|
||||
Authorization/Cookie;直接访问 reader Service 也不能写入。
|
||||
- `zot` 保留 SPIRE issuer/audience/subject 校验及仓库授权,`externalUrl`、
|
||||
Bearer realm、service 与 HTTPRoute 均使用 `zot-push.ad.ddupan.top`。
|
||||
- reader 关闭 GC,没有同步或扫描扩展;读取同一份 S3 制品,不复制 bucket,
|
||||
不新增 PVC 或 S3 密钥。镜像、安全上下文、资源和 Secret 引用由共用 values 继承。
|
||||
- 两个配置的 `storageDriver` 必须保持一致;修改 S3 endpoint/bucket/prefix 时
|
||||
同时更新 `values.yaml` 与 `reader-values.yaml`。
|
||||
|
||||
推送客户端应登录 `zot-push.ad.ddupan.top`;拉取客户端无需登录。
|
||||
已有 SPIRE 身份的进程可以通过 Workload API 获取 `aud=zot` 的 JWT-SVID,然后
|
||||
通过 `docker login` 或 `crane auth login` 的 `--password-stdin` 交给 Registry。
|
||||
用户名可以使用 `zot`,实际权限取自已验证 JWT 的身份。使用独立、权限为 `0700`
|
||||
的临时 `DOCKER_CONFIG`,结束后删除;不要开启 shell tracing,不要打印 token,
|
||||
不要把 token 放进命令参数。token 接口不会延长 SVID 有效期。
|
||||
|
||||
同一仓库在两个入口使用相同路径和 tag/digest,例如 CI 推送到
|
||||
`zot-push.ad.ddupan.top/team/image:tag`,部署时使用
|
||||
`zot.ad.ddupan.top/team/image:tag`;无需在两个仓库间复制。
|
||||
|
||||
## 部署与网络
|
||||
|
||||
- 官方 chart 管理 Deployment、Service、ConfigMap 和 HTTPRoute。
|
||||
@@ -72,13 +105,17 @@ OIDC Bearer middleware 会在授权阶段之前拒绝无 token 请求,单独
|
||||
`https` listener 与内网通配符证书终止。
|
||||
- 仅配置 Samba AD 内网 DNS;不创建公网 DNS 或 Cloudflare Tunnel route。
|
||||
- NetworkPolicy 只允许现有 Envoy Gateway 数据面访问 zot 的 5000 端口。
|
||||
- HTTPRoute 只暴露 `/v2/` 和 `/zot/auth/token`,不暴露内部健康检查或管理端点。
|
||||
- 拉取域名仅暴露 `/v2/` 的 GET/HEAD;推送域名暴露 `/v2/` 和
|
||||
`/zot/auth/token`,均不暴露内部健康检查或管理端点。
|
||||
- namespace 使用 restricted PodSecurity,容器非 root、只读根文件系统。
|
||||
|
||||
首次已按用户授权从本地执行 `kubectl apply -k apps/zot`,由集群 Helm controller
|
||||
安装。`clusters/homelab/apps/zot.yaml` 是 GitOps composition;对应文件合并进入
|
||||
Flux 跟踪分支后,才由根 Kustomization 持续管理,不能把未提交的本地部署写成
|
||||
已完成 Git 接管。
|
||||
`clusters/homelab/apps/zot.yaml` 已将 `zot` 和 `zot-reader` 一并纳入 Flux 管理。
|
||||
两个 HelmRelease 通过共用 `zot-values` 继承基础配置,reader 再叠加
|
||||
`zot-reader-values`。当前由 main 分支持续管理,不依赖本地覆盖或暂停回写。
|
||||
|
||||
后续若需临时验收,收尾时先确认 Git 管理的配置与目标运行配置一致,再移除
|
||||
`spec.values` 临时覆盖及 `kustomize.toolkit.fluxcd.io/reconcile=disabled` 标记,
|
||||
触发 zot Kustomization reconcile 并复验。临时测试身份和写权限不得留在持久配置中。
|
||||
|
||||
检查与渲染:
|
||||
|
||||
@@ -114,6 +151,35 @@ sha256:b8d3b977a1235022759470903dab4a46b7cf8107958624f1f76a323eabe37c5e
|
||||
|
||||
它是仅含验证文本的 OCI 测试镜像,没有可执行入口,不用于运行服务。
|
||||
|
||||
双域名验收还使用 `verification/anonymous-spire:smoke`:标准 crane 从
|
||||
`zot-push.ad.ddupan.top` 登录、推送,再从 `zot.ad.ddupan.top` 使用空
|
||||
`DOCKER_CONFIG` 拉取,两个入口的 digest 必须一致。验证匿名 blob HEAD、tags、
|
||||
referrers,以及客户端保存旧凭据时的公共拉取。推送入口检查无凭据、错误签名、
|
||||
错误 audience、过期 SVID、跨仓库写入和删除拒绝;公共入口拒绝所有写方法,
|
||||
reader Service 直连也拒绝写入。测试完成后撤回临时单仓库写权限。
|
||||
|
||||
2026-09-16 上述双域名验收通过;SVID 过期后推送入口返回 401,匿名拉取不受
|
||||
影响。临时写权限已撤销,两个 HelmRelease Ready;推送 DNS 第二次检查 changed=0。
|
||||
|
||||
匿名拉取示例:
|
||||
|
||||
```bash
|
||||
crane pull zot.ad.ddupan.top/verification/anonymous-spire:smoke image.tar --format oci
|
||||
```
|
||||
|
||||
鉴权推送示例(先通过 Workload API 将短期 JWT-SVID 保存到当前进程的 `ZOT_JWT`,
|
||||
不要启用 shell tracing;示例中的仓库仍需提前给具体 SPIFFE ID 授权):
|
||||
|
||||
```bash
|
||||
export DOCKER_CONFIG="$(mktemp -d)"
|
||||
printf '%s' "$ZOT_JWT" | crane auth login zot-push.ad.ddupan.top \
|
||||
--username zot --password-stdin
|
||||
crane push image.tar zot-push.ad.ddupan.top/team/image:tag
|
||||
rm -rf -- "$DOCKER_CONFIG"
|
||||
unset DOCKER_CONFIG ZOT_JWT
|
||||
```
|
||||
|
||||
|
||||
Registry 恢复需要完整的 SeaweedFS bucket 数据、Bao 专用凭据和此目录配置。
|
||||
zot 的临时目录不是制品备份。独立异机/离线备份尚未在本次部署中建立;不能把同一
|
||||
SeaweedFS 内的数据副本当作独立灾备。重装 zot 不得删除 `zot` bucket。
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
apiVersion: helm.toolkit.fluxcd.io/v2
|
||||
kind: HelmRelease
|
||||
metadata:
|
||||
name: zot-reader
|
||||
namespace: zot
|
||||
spec:
|
||||
chart:
|
||||
spec:
|
||||
chart: zot
|
||||
version: 0.1.124
|
||||
interval: 1h
|
||||
sourceRef:
|
||||
kind: HelmRepository
|
||||
name: zot
|
||||
releaseName: zot-reader
|
||||
interval: 30m
|
||||
timeout: 5m
|
||||
driftDetection:
|
||||
mode: enabled
|
||||
install:
|
||||
strategy:
|
||||
name: RetryOnFailure
|
||||
retryInterval: 5m
|
||||
upgrade:
|
||||
strategy:
|
||||
name: RetryOnFailure
|
||||
retryInterval: 5m
|
||||
valuesFrom:
|
||||
- kind: ConfigMap
|
||||
name: zot-values
|
||||
- kind: ConfigMap
|
||||
name: zot-reader-values
|
||||
@@ -6,6 +6,7 @@ resources:
|
||||
- external-secret.yaml
|
||||
- helmrepository.yaml
|
||||
- helmrelease.yaml
|
||||
- helmrelease-reader.yaml
|
||||
- networkpolicy.yaml
|
||||
generatorOptions:
|
||||
disableNameSuffixHash: true
|
||||
@@ -16,3 +17,7 @@ configMapGenerator:
|
||||
namespace: zot
|
||||
files:
|
||||
- values.yaml=values.yaml
|
||||
- name: zot-reader-values
|
||||
namespace: zot
|
||||
files:
|
||||
- values.yaml=reader-values.yaml
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# 叠加于共用 values.yaml;同一镜像、S3、Secret、安全设置,无制品副本。
|
||||
# 无 Bearer middleware,仅 anonymousPolicy=read;关闭 GC 避免多个实例清理共享存储。
|
||||
configFiles:
|
||||
config.json: |
|
||||
{
|
||||
"distSpecVersion": "1.1.1",
|
||||
"storage": {
|
||||
"rootDirectory": "/var/lib/registry",
|
||||
"dedupe": false,
|
||||
"gc": false,
|
||||
"storageDriver": {
|
||||
"name": "s3",
|
||||
"region": "us-east-1",
|
||||
"regionendpoint": "https://s3.ad.ddupan.top",
|
||||
"bucket": "zot",
|
||||
"rootdirectory": "/registry",
|
||||
"secure": true,
|
||||
"skipverify": false,
|
||||
"forcepathstyle": true
|
||||
}
|
||||
},
|
||||
"http": {
|
||||
"address": "0.0.0.0",
|
||||
"port": "5000",
|
||||
"externalUrl": "https://zot.ad.ddupan.top",
|
||||
"compat": [
|
||||
"docker2s2"
|
||||
],
|
||||
"accessControl": {
|
||||
"repositories": {
|
||||
"**": {
|
||||
"anonymousPolicy": [
|
||||
"read"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"log": {
|
||||
"level": "info"
|
||||
}
|
||||
}
|
||||
httproute:
|
||||
hostnames:
|
||||
- zot.ad.ddupan.top
|
||||
rules:
|
||||
- matches:
|
||||
- path:
|
||||
type: PathPrefix
|
||||
value: /v2/
|
||||
method: GET
|
||||
- path:
|
||||
type: PathPrefix
|
||||
value: /v2/
|
||||
method: HEAD
|
||||
filters:
|
||||
- type: RequestHeaderModifier
|
||||
requestHeaderModifier:
|
||||
remove:
|
||||
- Cookie
|
||||
- Authorization
|
||||
timeouts:
|
||||
request: 900s
|
||||
backendRequest: 900s
|
||||
+17
-4
@@ -41,14 +41,14 @@ configFiles:
|
||||
"http": {
|
||||
"address": "0.0.0.0",
|
||||
"port": "5000",
|
||||
"externalUrl": "https://zot.ad.ddupan.top",
|
||||
"externalUrl": "https://zot-push.ad.ddupan.top",
|
||||
"compat": [
|
||||
"docker2s2"
|
||||
],
|
||||
"auth": {
|
||||
"bearer": {
|
||||
"realm": "https://zot.ad.ddupan.top/zot/auth/token",
|
||||
"service": "zot.ad.ddupan.top",
|
||||
"realm": "https://zot-push.ad.ddupan.top/zot/auth/token",
|
||||
"service": "zot-push.ad.ddupan.top",
|
||||
"oidc": [
|
||||
{
|
||||
"issuer": "https://spire-oidc.ad.ddupan.top",
|
||||
@@ -71,6 +71,19 @@ configFiles:
|
||||
"accessControl": {
|
||||
"repositories": {
|
||||
"**": {
|
||||
"policies": [
|
||||
{
|
||||
"users": [
|
||||
"spiffe://ddupan.top/dev/panxiao81"
|
||||
],
|
||||
"actions": [
|
||||
"read",
|
||||
"create",
|
||||
"update",
|
||||
"delete"
|
||||
]
|
||||
}
|
||||
],
|
||||
"defaultPolicy": [
|
||||
"read"
|
||||
]
|
||||
@@ -133,7 +146,7 @@ httproute:
|
||||
namespace: envoy-gateway-system
|
||||
sectionName: https
|
||||
hostnames:
|
||||
- zot.ad.ddupan.top
|
||||
- zot-push.ad.ddupan.top
|
||||
rules:
|
||||
- matches:
|
||||
- path:
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||
kind: Kustomization
|
||||
metadata:
|
||||
name: dynamic-runner
|
||||
namespace: flux-system
|
||||
spec:
|
||||
dependsOn:
|
||||
- name: external-secrets
|
||||
- name: nats
|
||||
- name: spire
|
||||
interval: 10m
|
||||
path: ./platform/dynamic-runner
|
||||
prune: false
|
||||
sourceRef:
|
||||
kind: GitRepository
|
||||
name: flux-system
|
||||
timeout: 5m
|
||||
wait: true
|
||||
@@ -0,0 +1,18 @@
|
||||
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||
kind: Kustomization
|
||||
metadata:
|
||||
name: nats
|
||||
namespace: flux-system
|
||||
spec:
|
||||
dependsOn:
|
||||
- name: cert-manager
|
||||
- name: external-secrets
|
||||
- name: openebs
|
||||
interval: 10m
|
||||
path: ./platform/nats
|
||||
prune: false
|
||||
sourceRef:
|
||||
kind: GitRepository
|
||||
name: flux-system
|
||||
timeout: 10m
|
||||
wait: true
|
||||
@@ -10,6 +10,8 @@ resources:
|
||||
- apps/gitea-actions.yaml
|
||||
- apps/http-echo.yaml
|
||||
- apps/openebs.yaml
|
||||
- apps/nats.yaml
|
||||
- apps/dynamic-runner.yaml
|
||||
- apps/spire.yaml
|
||||
- apps/observability.yaml
|
||||
- apps/zot.yaml
|
||||
|
||||
@@ -11,10 +11,13 @@ homelab_dns:
|
||||
- { zone: ad.ddupan.top, name: pve2, type: A, values: [192.168.10.7] }
|
||||
- { zone: ad.ddupan.top, name: pve3, type: A, values: [192.168.10.9] }
|
||||
- { zone: ad.ddupan.top, name: retrolab, type: A, values: [10.60.0.10] }
|
||||
- { zone: ad.ddupan.top, name: grafana, type: A, values: [192.168.10.127] }
|
||||
- { zone: ad.ddupan.top, name: netbox, type: A, values: [192.168.10.127] }
|
||||
- { zone: ad.ddupan.top, name: nats, type: A, values: [192.168.10.127] }
|
||||
- { zone: ad.ddupan.top, name: s3, type: A, values: [192.168.10.127] }
|
||||
- { zone: ad.ddupan.top, name: spire-oidc, type: A, values: [192.168.10.127] }
|
||||
- { zone: ad.ddupan.top, name: zot, type: A, values: [192.168.10.127] }
|
||||
- { zone: ad.ddupan.top, name: zot-push, type: A, values: [192.168.10.127] }
|
||||
|
||||
split_horizon:
|
||||
# LAN and pod resolvers should eventually render the same set from here.
|
||||
|
||||
@@ -25,3 +25,29 @@ resource "vault_jwt_auth_backend_role" "spire_poc" {
|
||||
token_ttl = 300
|
||||
token_max_ttl = 900
|
||||
}
|
||||
|
||||
# Local development on the laptop. Keep the subject exact: possession of any
|
||||
# other identity in the trust domain must not grant interactive host access.
|
||||
resource "vault_jwt_auth_backend_role" "local_development" {
|
||||
backend = vault_jwt_auth_backend.spire.path
|
||||
role_name = "local-development"
|
||||
role_type = "jwt"
|
||||
|
||||
user_claim = "sub"
|
||||
bound_audiences = ["openbao"]
|
||||
bound_claims = {
|
||||
sub = "spiffe://ddupan.top/dev/panxiao81"
|
||||
}
|
||||
|
||||
# local-development grants normal KV v2 read/write under kv/k8s and kv/infra,
|
||||
# plus short-lived SSH certificate signing.
|
||||
# certificate signing; spire-poc only permits lookup and revocation of the
|
||||
# caller's own short-lived Bao token.
|
||||
token_policies = [
|
||||
vault_policy.local_development.name,
|
||||
vault_policy.spire_poc.name,
|
||||
]
|
||||
token_no_default_policy = true
|
||||
token_ttl = 300
|
||||
token_max_ttl = 900
|
||||
}
|
||||
|
||||
@@ -14,6 +14,11 @@ resource "vault_policy" "ai_agent_ssh" {
|
||||
policy = file("${path.module}/policies/ai-agent-ssh.hcl")
|
||||
}
|
||||
|
||||
resource "vault_policy" "local_development" {
|
||||
name = "local-development"
|
||||
policy = file("${path.module}/policies/local-development.hcl")
|
||||
}
|
||||
|
||||
resource "vault_policy" "spire_poc" {
|
||||
name = "spire-poc"
|
||||
policy = file("${path.module}/policies/spire-poc.hcl")
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
# Local development identity. Limit normal KV v2 reads and writes to the k8s
|
||||
# subtree, and exclude soft-delete, metadata deletion, permanent version
|
||||
# destruction, auth administration, and privileged operations.
|
||||
path "kv/data/k8s/*" {
|
||||
capabilities = ["create", "read", "update"]
|
||||
}
|
||||
|
||||
path "kv/metadata/k8s" {
|
||||
capabilities = ["read", "list"]
|
||||
}
|
||||
|
||||
path "kv/metadata/k8s/*" {
|
||||
capabilities = ["read", "list"]
|
||||
}
|
||||
|
||||
# Future destination for infrastructure secrets migrated from Ansible Vault.
|
||||
path "kv/data/infra/*" {
|
||||
capabilities = ["create", "read", "update"]
|
||||
}
|
||||
|
||||
path "kv/metadata/infra" {
|
||||
capabilities = ["read", "list"]
|
||||
}
|
||||
|
||||
path "kv/metadata/infra/*" {
|
||||
capabilities = ["read", "list"]
|
||||
}
|
||||
|
||||
# Sign disposable SSH public keys for short-lived development access.
|
||||
path "ssh-client-signer/sign/ai-agent" {
|
||||
capabilities = ["create", "update"]
|
||||
}
|
||||
@@ -119,6 +119,10 @@ issuerRef:
|
||||
kind: ClusterIssuer
|
||||
```
|
||||
|
||||
`values.yaml` 必须保持 `config.gatewayAPI.enabled: true`。`bao-acme` 的 HTTP-01
|
||||
solver 通过共享 Gateway 创建临时 HTTPRoute;关闭该项不会让 ClusterIssuer 变为
|
||||
NotReady,而是会让每个 Challenge 卡在 `gateway api is not enabled`。
|
||||
|
||||
Issuance is capped by `default_directory_policy = role:bao-server`
|
||||
(`../../infrastructure/openbao/terraform/pki.tf`), which permits `ad.ddupan.top` subdomains only. Clients
|
||||
need the internal CA in their trust store — already true for the PVE nodes, the DC and
|
||||
|
||||
@@ -44,6 +44,13 @@ cainjector:
|
||||
limits:
|
||||
memory: 256Mi
|
||||
|
||||
# bao-acme solves HTTP-01 through the shared Gateway. The ClusterIssuer can be
|
||||
# accepted while this is disabled, but every Challenge then stays pending with
|
||||
# "gateway api is not enabled". Gateway API CRDs are installed by Envoy Gateway.
|
||||
config:
|
||||
gatewayAPI:
|
||||
enabled: true
|
||||
|
||||
# ⚠ DNS-01 self-check: cert-manager polls authoritative NS for the _acme-challenge
|
||||
# TXT record before telling the CA to validate. By default it asks the cluster's
|
||||
# resolver, which for ad.ddupan.top is CoreDNS -> the Samba AD DC (k3s/coredns-custom.yaml).
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
# Gitea dynamic runner controller
|
||||
|
||||
此目录只管理 homelab 中的 controller 部署。controller、worker、Cloud Hypervisor
|
||||
launcher 和 guest runner 的源码与发布位于独立仓库
|
||||
`panxiao81/gitea-dynamic-runner`。
|
||||
|
||||
当前 bootstrap controller 接收 Gitea `workflow_job` webhook,将 `[self-hosted, pod]` 和
|
||||
`[self-hosted, vm]` 的 queued job 分别发布到 NATS。Pod worker 在本集群创建一次性
|
||||
privileged host runner;Docker、BuildKit 和 kind 由 workflow 自行 setup。内部
|
||||
endpoint:
|
||||
|
||||
```text
|
||||
http://dynamic-runner-controller.dynamic-runner.svc.cluster.local:8787/webhook
|
||||
```
|
||||
|
||||
OpenBao 路径:
|
||||
|
||||
- `kv/k8s/nats.ci_producer_password`:已有 NATS producer 密码。
|
||||
- `kv/k8s/nats.ci_worker_password`:已有 NATS worker 密码。
|
||||
- `kv/k8s/dynamic-runner.webhook_secret`:Gitea webhook HMAC secret。
|
||||
- `kv/k8s/gitea-runner.token`:现有 instance runner registration token。
|
||||
|
||||
首期 controller 与 runner 镜像由 laptop 本机构建后导入 k3s containerd,作为 CI
|
||||
发布链路建立前的 bootstrap。部署使用 `imagePullPolicy: Never`。正式发布 workflow
|
||||
获得专用 SPIFFE ID 后,必须将 image 改为 zot digest 并移除本地导入步骤。
|
||||
|
||||
## 身份绑定
|
||||
|
||||
queued webhook 只负责创建没有业务身份的 Pod。runner 实际领取任务后,Gitea 的
|
||||
`in_progress` webhook 会携带实际 `runner_name`;controller 将 binding 消息发布到
|
||||
NATS,Pod worker 再给对应 Pod 添加:
|
||||
|
||||
```text
|
||||
ci.ddupan.top/identity-bound=true
|
||||
ci.ddupan.top/spiffe-path=<owner>/<repository>/<percent-encoded-job-name>
|
||||
```
|
||||
|
||||
`ClusterSPIFFEID/gitea-dynamic-runner` 只匹配已经绑定的 Pod,并签发
|
||||
`spiffe://ddupan.top/ci/<owner>/<repository>/<job-name>`。runner 的 job-start hook 在
|
||||
SVID 可用之前不会放行第一步,因此不能根据 queued 事件错配身份。
|
||||
|
||||
每个 runner Pod 使用 `gitea-dynamic-runner` ServiceAccount。该 ServiceAccount 没有
|
||||
Kubernetes API 权限;只有 `dynamic-runner-pod-worker` ServiceAccount 能在本 namespace
|
||||
create/get/patch/delete Pod。
|
||||
|
||||
长期实现将由兼容 Gitea RunnerService 的 scheduler 直接领取 task,再交给 Pod/VM
|
||||
executor;届时删除 webhook、临时 runner 注册和 identity binding 消息。跟踪见
|
||||
`panxiao81/gitea-dynamic-runner` issue #7。
|
||||
|
||||
Gitea webhook 只订阅 `workflow_job`,content type 使用 JSON,secret 与 Bao 中值
|
||||
一致。不要启用 `send_everything`,否则 controller 会收到无关仓库事件。
|
||||
@@ -0,0 +1,17 @@
|
||||
apiVersion: spire.spiffe.io/v1alpha1
|
||||
kind: ClusterSPIFFEID
|
||||
metadata:
|
||||
name: gitea-dynamic-runner
|
||||
spec:
|
||||
className: spire-mgmt-spire
|
||||
spiffeIDTemplate: 'spiffe://{{ .TrustDomain }}/ci/{{ index .PodMeta.Annotations "ci.ddupan.top/spiffe-path" }}'
|
||||
namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: dynamic-runner
|
||||
podSelector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: gitea-dynamic-runner
|
||||
ci.ddupan.top/identity-bound: "true"
|
||||
workloadSelectorTemplates:
|
||||
- k8s:ns:dynamic-runner
|
||||
- k8s:sa:gitea-dynamic-runner
|
||||
@@ -0,0 +1,106 @@
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: dynamic-runner-controller
|
||||
namespace: dynamic-runner
|
||||
spec:
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: dynamic-runner-controller
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/name: dynamic-runner-controller
|
||||
spec:
|
||||
serviceAccountName: dynamic-runner-controller
|
||||
automountServiceAccountToken: false
|
||||
initContainers:
|
||||
- name: fetch-internal-ca
|
||||
image: curlimages/curl:8.16.0@sha256:463eaf6072688fe96ac64fa623fe73e1dbe25d8ad6c34404a669ad3ce1f104b6
|
||||
args:
|
||||
- --fail
|
||||
- --silent
|
||||
- --show-error
|
||||
- --output
|
||||
- /trust/ca.pem
|
||||
- https://bao.ad.ddupan.top:8200/v1/pki/ca/pem
|
||||
securityContext:
|
||||
allowPrivilegeEscalation: false
|
||||
capabilities:
|
||||
drop: [ALL]
|
||||
readOnlyRootFilesystem: true
|
||||
runAsNonRoot: true
|
||||
runAsUser: 101
|
||||
runAsGroup: 102
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
volumeMounts:
|
||||
- name: trust
|
||||
mountPath: /trust
|
||||
containers:
|
||||
- name: controller
|
||||
# Bootstrap import on laptop. Replace with a zot digest after the
|
||||
# repository's image publishing workflow has a dedicated identity.
|
||||
image: gitea-dynamic-runner-controller:0.3.0-bootstrap
|
||||
imagePullPolicy: Never
|
||||
env:
|
||||
- name: NATS_URL
|
||||
value: tls://nats.ad.ddupan.top:4222
|
||||
- name: NATS_CA_FILE
|
||||
value: /run/trust/ca.pem
|
||||
- name: NATS_PASSWORD_FILE
|
||||
value: /run/dynamic-runner-secrets/nats-password
|
||||
- name: WEBHOOK_SECRET_FILE
|
||||
value: /run/dynamic-runner-secrets/webhook-secret
|
||||
ports:
|
||||
- name: http
|
||||
containerPort: 8787
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /healthz
|
||||
port: http
|
||||
periodSeconds: 5
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /healthz
|
||||
port: http
|
||||
initialDelaySeconds: 10
|
||||
periodSeconds: 10
|
||||
resources:
|
||||
requests:
|
||||
cpu: 25m
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: 250m
|
||||
memory: 128Mi
|
||||
securityContext:
|
||||
allowPrivilegeEscalation: false
|
||||
capabilities:
|
||||
drop: [ALL]
|
||||
readOnlyRootFilesystem: true
|
||||
runAsNonRoot: true
|
||||
runAsUser: 65532
|
||||
runAsGroup: 65532
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
volumeMounts:
|
||||
- name: secret
|
||||
mountPath: /run/dynamic-runner-secrets
|
||||
readOnly: true
|
||||
- name: trust
|
||||
mountPath: /run/trust
|
||||
readOnly: true
|
||||
securityContext:
|
||||
fsGroup: 65532
|
||||
fsGroupChangePolicy: OnRootMismatch
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
volumes:
|
||||
- name: secret
|
||||
secret:
|
||||
secretName: dynamic-runner
|
||||
defaultMode: 0400
|
||||
- name: trust
|
||||
emptyDir:
|
||||
sizeLimit: 1Mi
|
||||
@@ -0,0 +1,30 @@
|
||||
apiVersion: external-secrets.io/v1
|
||||
kind: ExternalSecret
|
||||
metadata:
|
||||
name: dynamic-runner
|
||||
namespace: dynamic-runner
|
||||
spec:
|
||||
refreshInterval: 1h
|
||||
secretStoreRef:
|
||||
kind: ClusterSecretStore
|
||||
name: openbao
|
||||
target:
|
||||
creationPolicy: Owner
|
||||
name: dynamic-runner
|
||||
data:
|
||||
- secretKey: nats-password
|
||||
remoteRef:
|
||||
key: k8s/nats
|
||||
property: ci_producer_password
|
||||
- secretKey: nats-worker-password
|
||||
remoteRef:
|
||||
key: k8s/nats
|
||||
property: ci_worker_password
|
||||
- secretKey: webhook-secret
|
||||
remoteRef:
|
||||
key: k8s/dynamic-runner
|
||||
property: webhook_secret
|
||||
- secretKey: token
|
||||
remoteRef:
|
||||
key: k8s/gitea-runner
|
||||
property: token
|
||||
@@ -0,0 +1,10 @@
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
resources:
|
||||
- namespace.yaml
|
||||
- external-secret.yaml
|
||||
- rbac.yaml
|
||||
- clusterspiffeid.yaml
|
||||
- deployment.yaml
|
||||
- pod-worker-deployment.yaml
|
||||
- service.yaml
|
||||
@@ -0,0 +1,9 @@
|
||||
apiVersion: v1
|
||||
kind: Namespace
|
||||
metadata:
|
||||
name: dynamic-runner
|
||||
labels:
|
||||
# Disposable Pod runners may start dockerd or kind from the workflow.
|
||||
pod-security.kubernetes.io/enforce: privileged
|
||||
pod-security.kubernetes.io/audit: restricted
|
||||
pod-security.kubernetes.io/warn: restricted
|
||||
@@ -0,0 +1,100 @@
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: dynamic-runner-pod-worker
|
||||
namespace: dynamic-runner
|
||||
spec:
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: dynamic-runner-pod-worker
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/name: dynamic-runner-pod-worker
|
||||
spec:
|
||||
serviceAccountName: dynamic-runner-pod-worker
|
||||
initContainers:
|
||||
- name: fetch-internal-ca
|
||||
image: curlimages/curl:8.16.0@sha256:463eaf6072688fe96ac64fa623fe73e1dbe25d8ad6c34404a669ad3ce1f104b6
|
||||
args:
|
||||
- --fail
|
||||
- --silent
|
||||
- --show-error
|
||||
- --output
|
||||
- /trust/ca.pem
|
||||
- https://bao.ad.ddupan.top:8200/v1/pki/ca/pem
|
||||
securityContext:
|
||||
allowPrivilegeEscalation: false
|
||||
capabilities:
|
||||
drop: [ALL]
|
||||
readOnlyRootFilesystem: true
|
||||
runAsNonRoot: true
|
||||
runAsUser: 101
|
||||
runAsGroup: 102
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
volumeMounts:
|
||||
- name: trust
|
||||
mountPath: /trust
|
||||
containers:
|
||||
- name: pod-worker
|
||||
image: gitea-dynamic-runner-controller:0.3.0-bootstrap
|
||||
imagePullPolicy: Never
|
||||
command: [/venv/bin/gitea-dynamic-runner-pod-worker]
|
||||
env:
|
||||
- name: NATS_URL
|
||||
value: tls://nats.ad.ddupan.top:4222
|
||||
- name: NATS_CA_FILE
|
||||
value: /run/trust/ca.pem
|
||||
- name: NATS_PASSWORD_FILE
|
||||
value: /run/dynamic-runner-secrets/nats-worker-password
|
||||
- name: RUNNER_NAMESPACE
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
fieldPath: metadata.namespace
|
||||
- name: RUNNER_IMAGE
|
||||
value: zot.ad.ddupan.top/panxiao81/gitea-dynamic-runner-runner@sha256:4c61f6315453d68a827ee9542f1345ed86576eaaf62325eb803c8fe3f06ddf2a
|
||||
- name: RUNNER_SERVICE_ACCOUNT
|
||||
value: gitea-dynamic-runner
|
||||
- name: RUNNER_TOKEN_SECRET
|
||||
value: dynamic-runner
|
||||
- name: RUNNER_CAPACITY
|
||||
value: "4"
|
||||
resources:
|
||||
requests:
|
||||
cpu: 25m
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: 250m
|
||||
memory: 128Mi
|
||||
securityContext:
|
||||
allowPrivilegeEscalation: false
|
||||
capabilities:
|
||||
drop: [ALL]
|
||||
readOnlyRootFilesystem: true
|
||||
runAsNonRoot: true
|
||||
runAsUser: 65532
|
||||
runAsGroup: 65532
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
volumeMounts:
|
||||
- name: secret
|
||||
mountPath: /run/dynamic-runner-secrets
|
||||
readOnly: true
|
||||
- name: trust
|
||||
mountPath: /run/trust
|
||||
readOnly: true
|
||||
securityContext:
|
||||
fsGroup: 65532
|
||||
fsGroupChangePolicy: OnRootMismatch
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
volumes:
|
||||
- name: secret
|
||||
secret:
|
||||
secretName: dynamic-runner
|
||||
defaultMode: 0400
|
||||
- name: trust
|
||||
emptyDir:
|
||||
sizeLimit: 1Mi
|
||||
@@ -0,0 +1,41 @@
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: dynamic-runner-controller
|
||||
namespace: dynamic-runner
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: dynamic-runner-pod-worker
|
||||
namespace: dynamic-runner
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: gitea-dynamic-runner
|
||||
namespace: dynamic-runner
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: Role
|
||||
metadata:
|
||||
name: dynamic-runner-pod-worker
|
||||
namespace: dynamic-runner
|
||||
rules:
|
||||
- apiGroups: [""]
|
||||
resources: [pods]
|
||||
verbs: [create, get, patch, delete]
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: RoleBinding
|
||||
metadata:
|
||||
name: dynamic-runner-pod-worker
|
||||
namespace: dynamic-runner
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: Role
|
||||
name: dynamic-runner-pod-worker
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: dynamic-runner-pod-worker
|
||||
namespace: dynamic-runner
|
||||
@@ -0,0 +1,12 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: dynamic-runner-controller
|
||||
namespace: dynamic-runner
|
||||
spec:
|
||||
selector:
|
||||
app.kubernetes.io/name: dynamic-runner-controller
|
||||
ports:
|
||||
- name: http
|
||||
port: 8787
|
||||
targetPort: http
|
||||
@@ -25,6 +25,25 @@ The runner registration token is authoritative in OpenBao at
|
||||
the `gitea-runner-token` Secret. Never put the token in this directory or a Helm
|
||||
command line.
|
||||
|
||||
## SPIRE 与 OCI 发布
|
||||
|
||||
runner Pod 使用专用 ServiceAccount `gitea-actions`,并由精确匹配 namespace、
|
||||
ServiceAccount 隐含的 Pod、以及 chart labels 的 `ClusterSPIFFEID` 获得:
|
||||
|
||||
```text
|
||||
spiffe://ddupan.top/ci/gitea-actions
|
||||
```
|
||||
|
||||
SPIFFE CSI socket 同时只读挂载到 runner 和 DinD。act 的 volume allowlist 只允许
|
||||
`/run/spire/agent-sockets`;workflow 仍必须在 job container 中显式请求该 bind
|
||||
mount。原因是 bind mount 由 DinD 内的 dockerd 解析,只挂 runner 容器无法让 job
|
||||
访问 Workload API。
|
||||
|
||||
该身份不是通用 registry 管理员。zot 只对明确列出的 CI 镜像仓库授予
|
||||
`read/create/update`,不授予 delete 或其他仓库写入。workflow 应获取
|
||||
`aud=zot` 的短期 JWT-SVID,并经 stdin 传给 registry client,不得把 JWT、X.509
|
||||
SVID 或 Docker auth 写入 workspace/artifact。
|
||||
|
||||
## Flux 接管状态
|
||||
|
||||
该 release 最初通过下述 review-first 流程手动 bootstrap。下一个 GitOps 阶段将
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
apiVersion: spire.spiffe.io/v1alpha1
|
||||
kind: ClusterSPIFFEID
|
||||
metadata:
|
||||
name: gitea-actions
|
||||
spec:
|
||||
className: spire-mgmt-spire
|
||||
spiffeIDTemplate: spiffe://{{ .TrustDomain }}/ci/gitea-actions
|
||||
namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: gitea-actions
|
||||
podSelector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/instance: gitea-actions
|
||||
app.kubernetes.io/name: actions-runner
|
||||
workloadSelectorTemplates:
|
||||
- k8s:ns:gitea-actions
|
||||
- k8s:sa:gitea-actions
|
||||
@@ -11,6 +11,8 @@ configMapGenerator:
|
||||
- values.yaml=values.yaml
|
||||
resources:
|
||||
- namespace.yaml
|
||||
- serviceaccount.yaml
|
||||
- clusterspiffeid.yaml
|
||||
- external-secret.yaml
|
||||
- helmrepository.yaml
|
||||
- helmrelease.yaml
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: gitea-actions
|
||||
namespace: gitea-actions
|
||||
automountServiceAccountToken: false
|
||||
@@ -7,6 +7,12 @@ existingSecretKey: token
|
||||
statefulset:
|
||||
replicas: 1
|
||||
timezone: Etc/UTC
|
||||
serviceAccountName: gitea-actions
|
||||
extraVolumes:
|
||||
- name: spiffe-workload-api
|
||||
csi:
|
||||
driver: csi.spiffe.io
|
||||
readOnly: true
|
||||
securityContext:
|
||||
fsGroup: 1000
|
||||
# Chart 0.1.1 applies this block to both runner and DinD containers.
|
||||
@@ -27,6 +33,10 @@ statefulset:
|
||||
repository: gitea/runner
|
||||
tag: 2.3.0
|
||||
pullPolicy: IfNotPresent
|
||||
extraVolumeMounts:
|
||||
- name: spiffe-workload-api
|
||||
mountPath: /run/spire/agent-sockets
|
||||
readOnly: true
|
||||
config: |
|
||||
log:
|
||||
level: info
|
||||
@@ -42,6 +52,10 @@ statefulset:
|
||||
container:
|
||||
require_docker: true
|
||||
docker_timeout: 300s
|
||||
# Workflows must still request this exact bind mount explicitly. The
|
||||
# allowlist prevents arbitrary host paths from reaching job containers.
|
||||
valid_volumes:
|
||||
- /run/spire/agent-sockets
|
||||
|
||||
dind:
|
||||
# The node enforces AppArmor's unprivileged-userns restriction, which blocks
|
||||
@@ -51,6 +65,12 @@ statefulset:
|
||||
repository: docker
|
||||
tag: 29.7.1-dind
|
||||
pullPolicy: IfNotPresent
|
||||
# Bind mounts are resolved by dockerd, so the CSI socket must exist in the
|
||||
# DinD container as well as in the runner container.
|
||||
extraVolumeMounts:
|
||||
- name: spiffe-workload-api
|
||||
mountPath: /run/spire/agent-sockets
|
||||
readOnly: true
|
||||
# k3s uses a 1450-byte pod MTU. Without matching it here, nested Actions
|
||||
# networks advertise 1500 and GitHub TLS packets disappear on the outer
|
||||
# overlay path while direct pod traffic remains healthy.
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
# NATS
|
||||
|
||||
共享的轻量消息基础设施。首期为 Gitea microVM runner 提供 JetStream work queue,
|
||||
但 Account、subject 与部署位置均不与 CI controller 绑定,其他服务可按独立 Account
|
||||
复用。
|
||||
|
||||
## 当前拓扑
|
||||
|
||||
- 单节点 NATS;当前 homelab 没有资源运行有意义的三副本 JetStream quorum。
|
||||
- JetStream file store 使用 `localpv-zfs-ceph`,PVC 2 GiB。
|
||||
- 服务通过 k3s ServiceLB 在 `nats.ad.ddupan.top:4222` 暴露给内网;集群内客户端
|
||||
使用 `nats.nats.svc.cluster.local:4222`。访问控制由 TLS、Account 与用户权限负责,
|
||||
不额外维护易漂移的源 IP 白名单。
|
||||
- TLS 证书由 `bao-acme` 签发。PVE 节点已信任内部 CA。
|
||||
- `bao-server` PKI role 只接受 RSA CSR,因此 Certificate 使用 RSA 2048;不要改成
|
||||
ECDSA,ACME challenge 会成功但 finalize 会以 `role requires keys of type rsa` 失败。
|
||||
- `SYS` Account 用于管理;`CI` Account 启用 JetStream,存储上限 1 GiB。
|
||||
|
||||
Account 内的 JetStream 配额会原样进入 `nats.conf`,必须使用 NATS 的 `MB`/`GB`
|
||||
格式;PVC 等 Kubernetes resource quantity 才使用 `Mi`/`Gi`。
|
||||
|
||||
首期使用静态用户,密码只存在 OpenBao `kv/k8s/nats`:
|
||||
|
||||
```text
|
||||
sys_password
|
||||
ci_producer_password
|
||||
ci_worker_password
|
||||
```
|
||||
|
||||
`ci-producer` 只能发布 `ci.runner.>` 并调用必要的 JetStream API;`ci-worker`
|
||||
只能调用 JetStream pull/ACK API。二者都不能读取另一个 Account 的 subject。
|
||||
|
||||
后续 SPIRE/Auth Callout 动态认证见 homelab-infra issue #56。该迁移只替换连接
|
||||
凭据,不改变 Account、stream、subject 或 consumer。
|
||||
|
||||
## CI stream 约定
|
||||
|
||||
controller 首次启动时幂等创建 `CI_RUNNER` stream:`ci.runner.>`、
|
||||
`WorkQueuePolicy`、file storage、24h/10000 条/256 MiB 上限。每类 runner 使用独立
|
||||
subject 和 durable pull consumer;`ci.runner.<backend>.binding` 传递 runner 实际
|
||||
领取任务后的身份绑定。同类型的多个 worker 共享 durable consumer。ACK 后消息立即
|
||||
删除,不保存 CI 历史。
|
||||
|
||||
## 验证
|
||||
|
||||
```bash
|
||||
kubectl -n nats get helmrelease,pod,pvc,certificate,externalsecret
|
||||
kubectl -n nats logs statefulset/nats -c nats
|
||||
```
|
||||
@@ -0,0 +1,21 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: nats-ad-ddupan-top
|
||||
namespace: nats
|
||||
spec:
|
||||
secretName: nats-server-tls
|
||||
issuerRef:
|
||||
name: bao-acme
|
||||
kind: ClusterIssuer
|
||||
group: cert-manager.io
|
||||
commonName: nats.ad.ddupan.top
|
||||
dnsNames:
|
||||
- nats.ad.ddupan.top
|
||||
duration: 720h
|
||||
renewBefore: 168h
|
||||
privateKey:
|
||||
# OpenBao's bao-server role intentionally accepts RSA keys only.
|
||||
algorithm: RSA
|
||||
size: 2048
|
||||
rotationPolicy: Always
|
||||
@@ -0,0 +1,16 @@
|
||||
apiVersion: external-secrets.io/v1
|
||||
kind: ExternalSecret
|
||||
metadata:
|
||||
name: nats-auth
|
||||
namespace: nats
|
||||
spec:
|
||||
refreshInterval: 1h
|
||||
secretStoreRef:
|
||||
kind: ClusterSecretStore
|
||||
name: openbao
|
||||
target:
|
||||
creationPolicy: Owner
|
||||
name: nats-auth
|
||||
dataFrom:
|
||||
- extract:
|
||||
key: k8s/nats
|
||||
@@ -0,0 +1,31 @@
|
||||
apiVersion: helm.toolkit.fluxcd.io/v2
|
||||
kind: HelmRelease
|
||||
metadata:
|
||||
name: nats
|
||||
namespace: nats
|
||||
spec:
|
||||
chart:
|
||||
spec:
|
||||
chart: nats
|
||||
interval: 1h
|
||||
sourceRef:
|
||||
kind: HelmRepository
|
||||
name: nats
|
||||
version: 2.14.2
|
||||
driftDetection:
|
||||
mode: enabled
|
||||
install:
|
||||
strategy:
|
||||
name: RetryOnFailure
|
||||
retryInterval: 5m
|
||||
interval: 30m
|
||||
releaseName: nats
|
||||
targetNamespace: nats
|
||||
timeout: 10m
|
||||
upgrade:
|
||||
strategy:
|
||||
name: RetryOnFailure
|
||||
retryInterval: 5m
|
||||
valuesFrom:
|
||||
- kind: ConfigMap
|
||||
name: nats-values
|
||||
@@ -0,0 +1,8 @@
|
||||
apiVersion: source.toolkit.fluxcd.io/v1
|
||||
kind: HelmRepository
|
||||
metadata:
|
||||
name: nats
|
||||
namespace: nats
|
||||
spec:
|
||||
interval: 1h
|
||||
url: https://nats-io.github.io/k8s/helm/charts/
|
||||
@@ -0,0 +1,17 @@
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
generatorOptions:
|
||||
disableNameSuffixHash: true
|
||||
labels:
|
||||
reconcile.fluxcd.io/watch: Enabled
|
||||
configMapGenerator:
|
||||
- name: nats-values
|
||||
namespace: nats
|
||||
files:
|
||||
- values.yaml=values.yaml
|
||||
resources:
|
||||
- namespace.yaml
|
||||
- helmrepository.yaml
|
||||
- external-secret.yaml
|
||||
- certificate.yaml
|
||||
- helmrelease.yaml
|
||||
@@ -0,0 +1,4 @@
|
||||
apiVersion: v1
|
||||
kind: Namespace
|
||||
metadata:
|
||||
name: nats
|
||||
@@ -0,0 +1,76 @@
|
||||
config:
|
||||
jetstream:
|
||||
enabled: true
|
||||
fileStore:
|
||||
enabled: true
|
||||
maxSize: 2G
|
||||
pvc:
|
||||
enabled: true
|
||||
size: 2Gi
|
||||
storageClassName: localpv-zfs-ceph
|
||||
memoryStore:
|
||||
enabled: true
|
||||
maxSize: 64M
|
||||
nats:
|
||||
tls:
|
||||
enabled: true
|
||||
secretName: nats-server-tls
|
||||
merge:
|
||||
system_account: SYS
|
||||
accounts:
|
||||
SYS:
|
||||
users:
|
||||
- user: sys
|
||||
password: "<< $NATS_SYS_PASSWORD >>"
|
||||
CI:
|
||||
jetstream:
|
||||
# Account limits are JSON-encoded by config.merge. Use explicit byte
|
||||
# counts so NATS receives integers rather than quoted size strings.
|
||||
max_memory: 33554432
|
||||
max_file: 1073741824
|
||||
max_streams: 16
|
||||
max_consumers: 64
|
||||
max_bytes_required: true
|
||||
users:
|
||||
- user: ci-producer
|
||||
password: "<< $NATS_CI_PRODUCER_PASSWORD >>"
|
||||
permissions:
|
||||
publish:
|
||||
allow: [ci.runner.>, $JS.API.>]
|
||||
subscribe:
|
||||
allow: [_INBOX.>]
|
||||
- user: ci-worker
|
||||
password: "<< $NATS_CI_WORKER_PASSWORD >>"
|
||||
permissions:
|
||||
publish:
|
||||
allow: [$JS.API.>, $JS.ACK.>]
|
||||
subscribe:
|
||||
allow: [_INBOX.>]
|
||||
|
||||
container:
|
||||
env:
|
||||
NATS_SYS_PASSWORD:
|
||||
valueFrom:
|
||||
secretKeyRef: {name: nats-auth, key: sys_password}
|
||||
NATS_CI_PRODUCER_PASSWORD:
|
||||
valueFrom:
|
||||
secretKeyRef: {name: nats-auth, key: ci_producer_password}
|
||||
NATS_CI_WORKER_PASSWORD:
|
||||
valueFrom:
|
||||
secretKeyRef: {name: nats-auth, key: ci_worker_password}
|
||||
resources:
|
||||
requests: {cpu: 25m, memory: 64Mi}
|
||||
limits: {memory: 192Mi}
|
||||
|
||||
natsBox:
|
||||
enabled: false
|
||||
|
||||
promExporter:
|
||||
enabled: true
|
||||
podMonitor:
|
||||
enabled: true
|
||||
|
||||
service:
|
||||
merge:
|
||||
spec:
|
||||
type: LoadBalancer
|
||||
@@ -2,9 +2,25 @@
|
||||
|
||||
Metrics, logs, and traces for the cluster — one VictoriaMetrics-ecosystem stack,
|
||||
operator-driven, running in the `monitoring` namespace. Replaces the standalone
|
||||
Docker Compose stack in `../../apps/victoriametrics/`.
|
||||
Docker Compose monitoring stack. The retired stack and its three Docker data volumes
|
||||
were removed on 2026-09-16 with the maintainer's explicit approval.
|
||||
|
||||
## Status — DEPLOYED (2026-07-10)
|
||||
## Grafana 当前入口(2026-09-16)
|
||||
|
||||
统一使用 <https://grafana.ad.ddupan.top>,A 记录指向 `192.168.10.127`,由共享
|
||||
Envoy Gateway 的 `https` listener 与 `*.ad.ddupan.top` 证书提供 TLS。
|
||||
Grafana root_url 和 Authelia 回调均使用此域名;旧专用 Tailscale Ingress 已停用。
|
||||
远程客户端仍需有到 LAN 的路由及已配置的内网 DNS 转发。
|
||||
|
||||
内存看板:`/d/homelab-memory`;采集配置及 AppArmor 规则见
|
||||
[主机与进程内存采集](metrics/exporters/README.md)。
|
||||
|
||||
Grafana 由 Flux 管理;修改 values 后通过 Git 合并触发 HelmRelease,避免现场 Helm
|
||||
修改被漂移检测回滚。Authelia 当前不由 Flux 管理,需单独 Helm upgrade 并先结构比较
|
||||
live values。迁移时先添加 DNS 和回调,再切换 Grafana。Recreate 策略会短暂中断访问,
|
||||
但保留原 PVC、用户、看板和数据源。回滚需同时恢复 root_url、回调和 Ingress 配置。
|
||||
|
||||
## 初始部署记录(2026-07-10)
|
||||
|
||||
Live and verified in the `monitoring` namespace (operator chart 0.66.2):
|
||||
metrics (data queryable), logs (pods ingesting), traces (VTSingle CRD, verified via
|
||||
@@ -56,8 +72,10 @@ hosts and `logs/vlogs-ingress.yaml` to push their logs.
|
||||
| Manage | **victoria-metrics-operator** | VMSingle/VMAgent/VMAlert/VMAlertmanager/VMRule **and** VLSingle as CRDs |
|
||||
| Expose | **Tailscale ingress** (private) + **Authelia OIDC** | admin tool: private + SSO |
|
||||
|
||||
Grafana's Prometheus-operator converter is on, so any chart shipping a
|
||||
`ServiceMonitor`/`PodMonitor`/`PrometheusRule` is scraped automatically.
|
||||
VictoriaMetrics Operator 的 Prometheus converter 已启用;官方
|
||||
`prometheus-operator-crds` chart 由 `operator/` 一并管理。因此应用 chart 可以原生
|
||||
声明 `ServiceMonitor`、`PodMonitor` 或 `PrometheusRule`,再由 converter 转换为对应
|
||||
VM 资源,不需要每个应用额外维护一份 `VM*Scrape`。
|
||||
|
||||
## Architecture
|
||||
|
||||
@@ -77,7 +95,7 @@ observability/
|
||||
operator/ victoria-metrics-operator (Helm)
|
||||
metrics/
|
||||
vmsingle|vmagent|vmalert|vmalertmanager.yaml core CRs
|
||||
rules/ VMRules ported from ../../apps/victoriametrics/rules
|
||||
rules/ VMRules migrated from the retired Compose stack (source in Git history)
|
||||
exporters/ node-exporter (+VMNodeScrape), kube-state-metrics (Helm)
|
||||
scrapes/ kubelet + cAdvisor VMNodeScrapes
|
||||
logs/ VLSingle CR + victoria-logs-collector (Helm)
|
||||
@@ -88,8 +106,8 @@ observability/
|
||||
|
||||
## Docker hosts
|
||||
|
||||
The Compose services on Docker hosts (`../../apps/vlmcsd`, `../../apps/ps3netsrv`, `../../apps/netboot`, ...
|
||||
and the legacy `../../apps/victoriametrics` stack) are monitored too:
|
||||
The Compose services on Docker hosts (`../../apps/vlmcsd`, `../../apps/ps3netsrv`, `../../apps/netboot`, ...)
|
||||
are monitored too:
|
||||
|
||||
- **Metrics — pull, nothing exposed.** Each host runs `docker-hosts/compose.yaml`
|
||||
(cAdvisor `:8080` + node-exporter `:9100`). The cluster's vmagent scrapes their LAN
|
||||
@@ -160,7 +178,9 @@ cd grafana && ./helm.sh && cd ..
|
||||
vmalert (14950), node-exporter (1860), cAdvisor now that those exporters exist.
|
||||
- Wire real Alertmanager receivers in `metrics/vmalertmanager.yaml` (email via
|
||||
`../../apps/smtp-relay/`), replacing the ported `blackhole`.
|
||||
- Once parity is confirmed, decommission the Compose stack: `../../apps/victoriametrics/`
|
||||
(`docker compose down`), and retire that folder.
|
||||
- 旧 Compose 监控栈已于 2026-09-16 按维护者要求清理:原先已无容器,本次移除
|
||||
旧配置及 `victoriametrics_vmdata`、`victoriametrics_vmagentdata`、
|
||||
`victoriametrics_grafanadata` 三个无引用 Docker 卷。没有创建备份或迁移旧数据,
|
||||
Kubernetes 监控资源未修改。旧配置仍可从 Git 历史查阅,旧卷中的数据已删除。
|
||||
- Add app instrumentation: point `OTEL_EXPORTER_OTLP_ENDPOINT` at
|
||||
`otel-collector.monitoring.svc:4317`.
|
||||
|
||||
@@ -0,0 +1,502 @@
|
||||
{
|
||||
"uid": "homelab-memory",
|
||||
"title": "Homelab 内存与 Swap",
|
||||
"schemaVersion": 39,
|
||||
"version": 1,
|
||||
"tags": [
|
||||
"homelab",
|
||||
"memory"
|
||||
],
|
||||
"timezone": "utc",
|
||||
"refresh": "1m",
|
||||
"time": {
|
||||
"from": "now-6h",
|
||||
"to": "now"
|
||||
},
|
||||
"templating": {
|
||||
"list": [
|
||||
{
|
||||
"name": "DS_VM",
|
||||
"label": "数据源",
|
||||
"type": "datasource",
|
||||
"query": "prometheus",
|
||||
"current": {
|
||||
"text": "VictoriaMetrics",
|
||||
"value": "VictoriaMetrics"
|
||||
}
|
||||
}
|
||||
]
|
||||
},
|
||||
"panels": [
|
||||
{
|
||||
"id": 1,
|
||||
"type": "timeseries",
|
||||
"title": "主机物理内存",
|
||||
"description": "ZFS ARC 是内存的一部分;不能与此面板或 Pod/PSS 再叠加。",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 0,
|
||||
"y": 0,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "node_memory_MemTotal_bytes{job=\"node-exporter\"} - node_memory_MemAvailable_bytes{job=\"node-exporter\"}",
|
||||
"legendFormat": "已用(total - available)"
|
||||
},
|
||||
{
|
||||
"refId": "B",
|
||||
"expr": "node_memory_MemAvailable_bytes{job=\"node-exporter\"}",
|
||||
"legendFormat": "可用"
|
||||
},
|
||||
{
|
||||
"refId": "C",
|
||||
"expr": "node_memory_MemTotal_bytes{job=\"node-exporter\"}",
|
||||
"legendFormat": "总量"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "bytes",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "timeseries",
|
||||
"title": "ZFS ARC",
|
||||
"description": "缓存可回收性取决于实际压力,ARC 不等同于 free 的 buff/cache。",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 12,
|
||||
"y": 0,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "node_zfs_arc_size{job=\"node-exporter\"}",
|
||||
"legendFormat": "当前 ARC"
|
||||
},
|
||||
{
|
||||
"refId": "B",
|
||||
"expr": "node_zfs_arc_c{job=\"node-exporter\"}",
|
||||
"legendFormat": "目标"
|
||||
},
|
||||
{
|
||||
"refId": "C",
|
||||
"expr": "node_zfs_arc_c_max{job=\"node-exporter\"}",
|
||||
"legendFormat": "上限"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "bytes",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"type": "timeseries",
|
||||
"title": "Swap 使用量",
|
||||
"description": "",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 0,
|
||||
"y": 8,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "node_memory_SwapTotal_bytes{job=\"node-exporter\"} - node_memory_SwapFree_bytes{job=\"node-exporter\"}",
|
||||
"legendFormat": "已用"
|
||||
},
|
||||
{
|
||||
"refId": "B",
|
||||
"expr": "node_memory_SwapTotal_bytes{job=\"node-exporter\"}",
|
||||
"legendFormat": "总量"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "bytes",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "timeseries",
|
||||
"title": "Swap 换页速率",
|
||||
"description": "单位为页/秒,不假定页大小。",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 12,
|
||||
"y": 8,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "rate(node_vmstat_pswpin{job=\"node-exporter\"}[5m])",
|
||||
"legendFormat": "读入 pages/s"
|
||||
},
|
||||
{
|
||||
"refId": "B",
|
||||
"expr": "rate(node_vmstat_pswpout{job=\"node-exporter\"}[5m])",
|
||||
"legendFormat": "写出 pages/s"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "ops",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "timeseries",
|
||||
"title": "程序与虚拟机 PSS:前 15",
|
||||
"description": "PSS 按共享页比例分摊。vm: 表示 QEMU 在宿主机的占用,不是来宾内部应用占用。与 Pod working set、ARC 不能相加。",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 0,
|
||||
"y": 16,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "topk(15, namedprocess_namegroup_memory_bytes{job=\"process-exporter\",memtype=\"proportionalResident\"})",
|
||||
"legendFormat": "{{groupname}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "bytes",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 6,
|
||||
"type": "timeseries",
|
||||
"title": "程序与虚拟机 SwapPss:前 15",
|
||||
"description": "通过 smaps 对共享换出页按比例分摊。",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 12,
|
||||
"y": 16,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "topk(15, namedprocess_namegroup_memory_bytes{job=\"process-exporter\",memtype=\"proportionalSwapped\"})",
|
||||
"legendFormat": "{{groupname}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "bytes",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 7,
|
||||
"type": "timeseries",
|
||||
"title": "Pod working set:前 15",
|
||||
"description": "cgroup working set 与 PSS 口径不同,不相加。",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 0,
|
||||
"y": 24,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "topk(15, sum by (namespace,pod) (container_memory_working_set_bytes{job=\"cadvisor\",container!=\"\",container!=\"POD\",pod!=\"\"}))",
|
||||
"legendFormat": "{{namespace}}/{{pod}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "bytes",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 8,
|
||||
"type": "timeseries",
|
||||
"title": "内存压力 PSI",
|
||||
"description": "",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 12,
|
||||
"y": 24,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "rate(node_pressure_memory_waiting_seconds_total{job=\"node-exporter\"}[5m])",
|
||||
"legendFormat": "some"
|
||||
},
|
||||
{
|
||||
"refId": "B",
|
||||
"expr": "rate(node_pressure_memory_stalled_seconds_total{job=\"node-exporter\"}[5m])",
|
||||
"legendFormat": "full"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "percentunit",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 9,
|
||||
"type": "timeseries",
|
||||
"title": "Samba RPC worker 数量",
|
||||
"description": "仅在 exporter 健康时将无 worker 解释为 0;见采集健康面板。",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 0,
|
||||
"y": 32,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "namedprocess_namegroup_num_procs{job=\"process-exporter\",groupname=\"rpcd_lsad\"} or on() (0 * max(up{job=\"process-exporter\"} == 1))",
|
||||
"legendFormat": "rpcd_lsad"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "short",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 10,
|
||||
"type": "timeseries",
|
||||
"title": "进程采集健康",
|
||||
"description": "up 仅表示抓取成功;还需检查读取错误。进程退出等竞态可能造成偶发 partial errors,持续增长时检查 AppArmor 审计。",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_VM}"
|
||||
},
|
||||
"gridPos": {
|
||||
"x": 12,
|
||||
"y": 32,
|
||||
"w": 12,
|
||||
"h": 8
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "up{job=\"process-exporter\"}",
|
||||
"legendFormat": "up"
|
||||
},
|
||||
{
|
||||
"refId": "B",
|
||||
"expr": "scrape_duration_seconds{job=\"process-exporter\"}",
|
||||
"legendFormat": "scrape 秒"
|
||||
},
|
||||
{
|
||||
"refId": "C",
|
||||
"expr": "namedprocess_scrape_errors{job=\"process-exporter\"}",
|
||||
"legendFormat": "采集错误"
|
||||
},
|
||||
{
|
||||
"refId": "D",
|
||||
"expr": "rate(namedprocess_scrape_partial_errors{job=\"process-exporter\"}[5m])",
|
||||
"legendFormat": "部分字段读取失败/s"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "short",
|
||||
"min": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "table",
|
||||
"placement": "bottom",
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
]
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
# Grafana 自身使用 OIDC,不增加第二层 forward-auth。
|
||||
apiVersion: gateway.networking.k8s.io/v1
|
||||
kind: HTTPRoute
|
||||
metadata:
|
||||
name: grafana-lan
|
||||
namespace: monitoring
|
||||
spec:
|
||||
parentRefs:
|
||||
- name: eg
|
||||
namespace: envoy-gateway-system
|
||||
sectionName: https
|
||||
hostnames:
|
||||
- grafana.ad.ddupan.top
|
||||
rules:
|
||||
- backendRefs:
|
||||
- name: grafana
|
||||
port: 80
|
||||
@@ -17,5 +17,13 @@ configMapGenerator:
|
||||
options:
|
||||
labels:
|
||||
grafana_dashboard: "1"
|
||||
- name: grafana-dashboard-homelab-memory
|
||||
namespace: monitoring
|
||||
files:
|
||||
- homelab-memory.json=dashboards/homelab-memory.json
|
||||
options:
|
||||
labels:
|
||||
grafana_dashboard: "1"
|
||||
resources:
|
||||
- helmrelease.yaml
|
||||
- httproute.yaml
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Grafana — single pane over metrics (VictoriaMetrics), logs (VictoriaLogs) and
|
||||
# traces (VictoriaTraces via Jaeger API). Exposed privately on the tailnet
|
||||
# (grafana.tail7e769.ts.net) and authenticated via Authelia OIDC. Local admin is
|
||||
# traces (VictoriaTraces via Jaeger API). LAN: grafana.ad.ddupan.top,
|
||||
# authenticated via Authelia OIDC. Local admin is
|
||||
# break-glass only.
|
||||
#
|
||||
# Chart: grafana/grafana (repo: https://grafana.github.io/helm-charts)
|
||||
@@ -54,16 +54,10 @@ persistence:
|
||||
deploymentStrategy:
|
||||
type: Recreate
|
||||
|
||||
# --- Private exposure via the Tailscale ingress (like seaweedfs-admin) ---
|
||||
# The tailscale operator provisions grafana.<tailnet>.ts.net and a TLS cert.
|
||||
# 内网入口由 httproute.yaml 接入 Envoy,使用现有通配 TLS 证书。
|
||||
# 客户端通过既有内网 DNS 转发解析,无需独立 Tailscale Ingress。
|
||||
ingress:
|
||||
enabled: true
|
||||
ingressClassName: tailscale
|
||||
hosts:
|
||||
- grafana
|
||||
tls:
|
||||
- hosts:
|
||||
- grafana
|
||||
enabled: false
|
||||
|
||||
# --- OIDC via Authelia (AD groups -> Grafana roles) ---
|
||||
# client_secret is injected from the grafana-oidc Secret (see oidc-secret.yaml),
|
||||
@@ -76,7 +70,7 @@ envValueFrom:
|
||||
|
||||
grafana.ini:
|
||||
server:
|
||||
root_url: "https://grafana.tail7e769.ts.net" # must match the tailnet FQDN + Authelia redirect_uri
|
||||
root_url: "https://grafana.ad.ddupan.top" # 与 Authelia redirect_uri 一致
|
||||
auth:
|
||||
# Keep the local admin login available as break-glass; don't force OIDC-only.
|
||||
disable_login_form: false
|
||||
|
||||
@@ -3,6 +3,7 @@ kind: Kustomization
|
||||
resources:
|
||||
- namespace.yaml
|
||||
- helmrepository.yaml
|
||||
- prometheus-helmrepository.yaml
|
||||
- grafana-helmrepository.yaml
|
||||
- operator
|
||||
- metrics
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
# 主机与进程内存采集
|
||||
|
||||
2026-09-16 已增加 `process-exporter.yaml`,固定上游 0.8.7,以 DaemonSet 运行,
|
||||
只读挂载宿主机 `/proc`,每 60 秒由 VMPodScrape 抓取,不开宿主机端口。
|
||||
NetworkPolicy 仅放行同 namespace 的 vmagent。进程按名称分组,QEMU 按 guest 名,
|
||||
NetBox 和 VS Code 按路径归组;不把完整命令行或 PID 放进指标标签。
|
||||
|
||||
当前 live 使用 `gather-smaps=true`,导出 RSS、VmSwap、PSS、SwapPss 与进程数。
|
||||
`apparmor/homelab-process-exporter` 已安装到 `/etc/apparmor.d/homelab-process-exporter`
|
||||
并以 enforce 加载。用户明确批准了跨进程读取:SYS_PTRACE、DAC_READ_SEARCH 与
|
||||
`ptrace (read) peer=**`;文件权限限于指标需要的 proc 文件,不允许 ptrace trace,
|
||||
不开放 `/proc/<pid>/mem`、`environ`,不关闭 AppArmor,也未修改虚拟机或容器默认 profile。
|
||||
|
||||
当前只部署到 laptop。其他节点必须先安装同名 profile,再扩展 nodeSelector:
|
||||
|
||||
```bash
|
||||
sudo install -m 0644 apparmor/homelab-process-exporter /etc/apparmor.d/homelab-process-exporter
|
||||
sudo apparmor_parser -r /etc/apparmor.d/homelab-process-exporter
|
||||
```
|
||||
|
||||
修改 profile 可原位重载,无需重启业务。禁用时先删除 exporter DaemonSet,再卸载
|
||||
专用 profile;不要让引用 profile 的 Pod 在 profile 缺失时启动。
|
||||
|
||||
验证已读到三台虚拟机、NetBox、Codex 的非零 PSS/SwapPss。`namedprocess_scrape_errors`
|
||||
和 `namedprocess_scrape_procread_errors` 为零。上游会先读取再匹配进程,内核线程没有
|
||||
可读的 smaps_rollup 会计入 partial errors;现场有约 430 个这种线程。
|
||||
partial errors 不保证为零,应结合业务组 PSS 和 AppArmor 审计判断。
|
||||
RSS、PSS、cAdvisor working set、ARC 是不同口径,不能直接相加。
|
||||
|
||||
已有采集无需重复部署:
|
||||
|
||||
- node-exporter:主机内存、Swap、PSI、ZFS ARC。
|
||||
- ARC 实际指标名为 `node_zfs_arc_size`、`node_zfs_arc_c`、`node_zfs_arc_c_max`。
|
||||
- kubelet/cAdvisor:Pod/container working set、Swap 等。
|
||||
|
||||
Grafana 新增 `Homelab 内存与 Swap`(UID `homelab-memory`),由 sidecar ConfigMap 加载。
|
||||
仅使用 `job="node-exporter"` 查询主机指标,避免目前 `docker-hosts` 对同一 9100 端口
|
||||
的重复抓取。另有 `192.168.10.127:8080` Docker cAdvisor target 失效,本次未修改该旧配置。
|
||||
|
||||
所有新增 manifest 已纳入对应 Kustomization,并已单独应用到 live;尚未提交到远端,
|
||||
Flux source 尚未包含这些新增资源。修改 exporter ConfigMap 后需滚动重启 DaemonSet,
|
||||
程序不会自动重载进程匹配配置。
|
||||
@@ -0,0 +1,27 @@
|
||||
#include <tunables/global>
|
||||
|
||||
# 限定为指标所需文件;跨 profile 读取由用户明确批准;不允许 trace 或读取进程 mem/environ。
|
||||
profile homelab-process-exporter flags=(attach_disconnected,mediate_deleted) {
|
||||
#include <abstractions/base>
|
||||
network inet stream,
|
||||
network inet6 stream,
|
||||
capability sys_ptrace,
|
||||
capability dac_read_search,
|
||||
ptrace (read) peer=**,
|
||||
signal (receive) peer=unconfined,
|
||||
signal (receive) peer=cri-containerd.apparmor.d,
|
||||
signal (receive) peer=runc,
|
||||
/bin/process-exporter mr,
|
||||
/process-exporter mr,
|
||||
/config/** r,
|
||||
/etc/{passwd,group,nsswitch.conf} r,
|
||||
/{host/,}proc/ r,
|
||||
/{host/,}proc/{stat,meminfo,cpuinfo,uptime,version,sys/kernel/random/boot_id} r,
|
||||
/{host/,}proc/[0-9]*/ r,
|
||||
/{host/,}proc/[0-9]*/{stat,status,cmdline,smaps,smaps_rollup,io,limits,cgroup,wchan} r,
|
||||
/{host/,}proc/[0-9]*/fd/ r,
|
||||
/{host/,}proc/[0-9]*/task/ r,
|
||||
/{host/,}proc/[0-9]*/task/[0-9]*/{stat,status,io,cmdline,wchan,cgroup,limits,smaps,smaps_rollup} r,
|
||||
/proc/sys/net/core/somaxconn r,
|
||||
/proc/self/{stat,status,smaps,smaps_rollup,limits,cgroup,mountinfo} r,
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: process-exporter-config
|
||||
namespace: monitoring
|
||||
data:
|
||||
config.yml: |
|
||||
process_names:
|
||||
# 只暴露虚拟机名,不把完整命令行、PID 或凭据写入标签。
|
||||
- name: 'vm:{{.Matches.VM}}'
|
||||
comm: [qemu-system-x86]
|
||||
cmdline: ['-name\s+guest=(?P<VM>[^,\s]+)']
|
||||
- name: netbox
|
||||
cmdline: ['/opt/netbox/']
|
||||
- name: vscode
|
||||
cmdline: ['\.vscode-server/']
|
||||
- name: '{{.Comm}}'
|
||||
cmdline: ['.+']
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: DaemonSet
|
||||
metadata:
|
||||
name: process-exporter
|
||||
namespace: monitoring
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: process-exporter
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/name: process-exporter
|
||||
spec:
|
||||
nodeSelector:
|
||||
kubernetes.io/hostname: laptop
|
||||
hostPID: true
|
||||
automountServiceAccountToken: false
|
||||
tolerations:
|
||||
- operator: Exists
|
||||
containers:
|
||||
- name: process-exporter
|
||||
image: ncabatoff/process-exporter:0.8.7
|
||||
args:
|
||||
- -procfs=/host/proc
|
||||
- -config.path=/config/config.yml
|
||||
- -gather-smaps=true
|
||||
- -threads=false
|
||||
- -children=false
|
||||
securityContext:
|
||||
runAsUser: 0
|
||||
readOnlyRootFilesystem: true
|
||||
allowPrivilegeEscalation: false
|
||||
appArmorProfile:
|
||||
type: Localhost
|
||||
localhostProfile: homelab-process-exporter
|
||||
capabilities:
|
||||
drop: [ALL]
|
||||
add: [SYS_PTRACE, DAC_READ_SEARCH]
|
||||
ports:
|
||||
- name: metrics
|
||||
containerPort: 9256
|
||||
readinessProbe:
|
||||
tcpSocket:
|
||||
port: metrics
|
||||
resources:
|
||||
requests:
|
||||
cpu: 25m
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: 500m
|
||||
memory: 128Mi
|
||||
volumeMounts:
|
||||
- name: proc
|
||||
mountPath: /host/proc
|
||||
readOnly: true
|
||||
- name: config
|
||||
mountPath: /config
|
||||
readOnly: true
|
||||
volumes:
|
||||
- name: proc
|
||||
hostPath:
|
||||
path: /proc
|
||||
type: Directory
|
||||
- name: config
|
||||
configMap:
|
||||
name: process-exporter-config
|
||||
---
|
||||
apiVersion: operator.victoriametrics.com/v1beta1
|
||||
kind: VMPodScrape
|
||||
metadata:
|
||||
name: process-exporter
|
||||
namespace: monitoring
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: process-exporter
|
||||
podMetricsEndpoints:
|
||||
- port: metrics
|
||||
interval: 60s
|
||||
scrapeTimeout: 30s
|
||||
relabelConfigs:
|
||||
- targetLabel: job
|
||||
replacement: process-exporter
|
||||
- sourceLabels: [__meta_kubernetes_pod_node_name]
|
||||
targetLabel: node
|
||||
---
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
metadata:
|
||||
name: process-exporter
|
||||
namespace: monitoring
|
||||
spec:
|
||||
podSelector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: process-exporter
|
||||
policyTypes: [Ingress]
|
||||
ingress:
|
||||
- from:
|
||||
- podSelector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: vmagent
|
||||
ports:
|
||||
- protocol: TCP
|
||||
port: 9256
|
||||
@@ -14,3 +14,4 @@ resources:
|
||||
- scrapes/docker-hosts.yaml
|
||||
- scrapes/kubelet.yaml
|
||||
- exporters/node-exporter.yaml
|
||||
- exporters/process-exporter.yaml
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# Ported from ../../../../apps/victoriametrics/rules/alerts-health.yml
|
||||
# 告警规则从已退役的 Compose 栈迁入,原始版本保留在 Git 历史。
|
||||
apiVersion: operator.victoriametrics.com/v1beta1
|
||||
kind: VMRule
|
||||
metadata:
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# Ported from ../../../../apps/victoriametrics/rules/alerts-vmagent.yml
|
||||
# 告警规则从已退役的 Compose 栈迁入,原始版本保留在 Git 历史。
|
||||
apiVersion: operator.victoriametrics.com/v1beta1
|
||||
kind: VMRule
|
||||
metadata:
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# Ported from ../../../../apps/victoriametrics/rules/alerts-vmalert.yml
|
||||
# 告警规则从已退役的 Compose 栈迁入,原始版本保留在 Git 历史。
|
||||
apiVersion: operator.victoriametrics.com/v1beta1
|
||||
kind: VMRule
|
||||
metadata:
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# Ported from ../../../../apps/victoriametrics/rules/alerts.yml
|
||||
# 告警规则从已退役的 Compose 栈迁入,原始版本保留在 Git 历史。
|
||||
apiVersion: operator.victoriametrics.com/v1beta1
|
||||
kind: VMRule
|
||||
metadata:
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# Alert router/notifier. Ported 1:1 from ../../../apps/victoriametrics/alertmanager.yaml,
|
||||
# Alert router/notifier. Migrated from the retired Compose stack (see Git history),
|
||||
# which currently blackholes everything. Wire real receivers here (email via the
|
||||
# in-cluster smtp-relay, or a webhook) when you want notifications.
|
||||
apiVersion: operator.victoriametrics.com/v1beta1
|
||||
|
||||
@@ -10,4 +10,5 @@ configMapGenerator:
|
||||
files:
|
||||
- values.yaml=values.yaml
|
||||
resources:
|
||||
- prometheus-crds-helmrelease.yaml
|
||||
- helmrelease.yaml
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
apiVersion: helm.toolkit.fluxcd.io/v2
|
||||
kind: HelmRelease
|
||||
metadata:
|
||||
name: prometheus-operator-crds
|
||||
namespace: monitoring
|
||||
spec:
|
||||
chart:
|
||||
spec:
|
||||
chart: prometheus-operator-crds
|
||||
interval: 1h
|
||||
sourceRef:
|
||||
kind: HelmRepository
|
||||
name: prometheus-community
|
||||
namespace: monitoring
|
||||
version: 32.0.0
|
||||
install:
|
||||
crds: CreateReplace
|
||||
strategy:
|
||||
name: RetryOnFailure
|
||||
retryInterval: 5m
|
||||
interval: 30m
|
||||
releaseName: prometheus-operator-crds
|
||||
targetNamespace: monitoring
|
||||
timeout: 10m
|
||||
upgrade:
|
||||
crds: CreateReplace
|
||||
strategy:
|
||||
name: RetryOnFailure
|
||||
retryInterval: 5m
|
||||
@@ -0,0 +1,8 @@
|
||||
apiVersion: source.toolkit.fluxcd.io/v1
|
||||
kind: HelmRepository
|
||||
metadata:
|
||||
name: prometheus-community
|
||||
namespace: monitoring
|
||||
spec:
|
||||
interval: 1h
|
||||
url: https://prometheus-community.github.io/helm-charts
|
||||
@@ -47,6 +47,37 @@ PostgreSQL保存 registration state;SPIRE Server 的 disk KeyManager 仍使用
|
||||
`ClusterSPIFFEID`,并以 namespace、ServiceAccount、Pod label 等 selector 收窄。
|
||||
不得仅因 Pod 能挂载 CSI socket 就给它签发身份。
|
||||
|
||||
## 宿主机本地开发身份
|
||||
|
||||
Kubernetes 节点 Agent 同时通过 hostPath 在宿主机发布 Workload API socket:
|
||||
|
||||
```text
|
||||
/run/spire/agent-sockets/spire-agent.sock
|
||||
```
|
||||
|
||||
Agent 已启用 Unix workload attestor,并为本机用户 `panxiao81`(UID `1000`)注册
|
||||
`spiffe://ddupan.top/dev/panxiao81`。本地开发程序应设置:
|
||||
|
||||
```bash
|
||||
export SPIFFE_ENDPOINT_SOCKET=unix:///run/spire/agent-sockets/spire-agent.sock
|
||||
```
|
||||
|
||||
该身份仅按 Unix UID 匹配,不是 SPIRE admin,也不会匹配 `sudo` 后以 root 运行的
|
||||
进程。`ClusterStaticEntry.spec.parentID` 绑定当前 `laptop` Kubernetes node UID;若
|
||||
节点被删除后重建,需从 `spire-server agent list` 取得新 Agent ID 并同步更新该字段。
|
||||
|
||||
下游授权仅绑定这个精确 SPIFFE ID:
|
||||
|
||||
- OpenBao `auth/jwt-spire/role/local-development` 接受 `aud=openbao`,签发 5 分钟
|
||||
token;允许读取和写入 KV v2 的 `kv/k8s/*` 与 `kv/infra/*` 子树、列出对应
|
||||
metadata、签发 `ai-agent` SSH
|
||||
短证书,以及查询、撤销自身 token;不允许删除/永久销毁 KV 数据或管理 auth;
|
||||
- zot 接受 `aud=zot`,允许本机开发身份对所有 repository 执行
|
||||
`read/create/update/delete`;其他 SPIFFE 身份仍保持全仓库只读。
|
||||
|
||||
当前没有其他服务直接消费 SPIFFE 身份;Gitea runner 与 dynamic runner 是独立身份
|
||||
使用方,NATS、SeaweedFS 等服务尚未通过 SPIFFE 做认证或授权。
|
||||
|
||||
稳定的 JWT issuer 预留为:
|
||||
|
||||
```text
|
||||
|
||||
@@ -15,3 +15,4 @@ resources:
|
||||
- helmrelease-crds.yaml
|
||||
- helmrelease.yaml
|
||||
- httproute.yaml
|
||||
- local-development-identity.yaml
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
apiVersion: spire.spiffe.io/v1alpha1
|
||||
kind: ClusterStaticEntry
|
||||
metadata:
|
||||
name: local-development-panxiao81
|
||||
labels:
|
||||
spire.spiffe.io/class-name: spire-mgmt-spire
|
||||
spec:
|
||||
className: spire-mgmt-spire
|
||||
parentID: spiffe://ddupan.top/spire/agent/k8s_psat/homelab/cd2d0233-c4ea-4031-8327-e7e359e766dd
|
||||
spiffeID: spiffe://ddupan.top/dev/panxiao81
|
||||
selectors:
|
||||
- unix:uid:1000
|
||||
@@ -72,7 +72,10 @@ spire-agent:
|
||||
k8s:
|
||||
enabled: true
|
||||
unix:
|
||||
enabled: false
|
||||
# The node Agent also exposes its Workload API socket on the host. Enable
|
||||
# Unix attestation so local development processes can receive an identity
|
||||
# through an explicitly scoped ClusterStaticEntry.
|
||||
enabled: true
|
||||
|
||||
spiffe-csi-driver:
|
||||
enabled: true
|
||||
|
||||
Reference in New Issue
Block a user