Author SHA1 Message Date
panxiao81 22bffe068c Merge pull request:使用字节数配置 NATS Account 配额
yaml / yaml (push) Successful in 10s
2026-09-16 12:49:59 +00:00
panxiao81 6078a06b99 使用字节数配置 NATS Account 配额
yaml / yaml (pull_request) Successful in 11s
2026-09-16 12:49:20 +00:00
panxiao81 47042d4df4 完善内存监控并迁移 Grafana 内网入口
yaml / yaml (push) Successful in 20s
ansible / collection-test (push) Successful in 1m10s
ansible / lint (push) Successful in 13m9s
Co-authored-by: panxiao81 <[email protected]>
2026-09-16 12:47:52 +00:00
panxiao81 fd6bd62a4f Merge pull request:修复 NATS 容量单位语法
yaml / yaml (push) Successful in 14s
2026-09-16 12:41:08 +00:00
panxiao81 3eb6f33dea 修复 NATS 容量单位语法
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:40:19 +00:00
panxiao81 818192f453 Merge pull request:修复 NATS Account 配额格式
yaml / yaml (push) Successful in 11s
2026-09-16 12:34:12 +00:00
panxiao81 387953c80a 修复 NATS Account 配额格式
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:33:16 +00:00
panxiao81 298db6a745 Merge pull request:修复 NATS Bao ACME 证书密钥类型
yaml / yaml (push) Successful in 10s
2026-09-16 12:26:19 +00:00
panxiao81 f36a1cbf11 修复 NATS Bao ACME 证书密钥类型
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:25:28 +00:00
panxiao81 b953db199e Merge pull request:由 monitoring 提供 Prometheus Operator CRD
yaml / yaml (push) Successful in 11s
2026-09-16 12:23:40 +00:00
panxiao81 b7b7975c9f 由 monitoring 提供 Prometheus Operator CRD
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:22:00 +00:00
panxiao81 9604ff1004 修复 NATS 监控 CRD 兼容性
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:18:34 +00:00
panxiao81 819b521039 Merge pull request:修复 cert-manager Gateway API ACME solver
yaml / yaml (push) Successful in 11s
2026-09-16 12:17:07 +00:00
panxiao81 40703782ea 修复 cert-manager Gateway API ACME solver
yaml / yaml (pull_request) Successful in 11s
2026-09-16 12:16:23 +00:00
panxiao81 aae19850cf Merge pull request:部署 NATS JetStream 消息基础设施
yaml / yaml (push) Successful in 16s
ansible / collection-test (push) Successful in 1m9s
ansible / lint (push) Successful in 18m16s
首期使用静态 Account 凭据;SPIRE Auth Callout 后续见 #56。YAML 与 collection tests 已通过,Ansible lint 卡在无关的 Galaxy 依赖下载。
2026-09-16 12:13:06 +00:00
panxiao81 c11e1d5e6f 部署 NATS JetStream 消息基础设施
yaml / yaml (pull_request) Successful in 51s
ansible / collection-test (pull_request) Successful in 2m0s
ansible / lint (pull_request) Successful in 19m43s
2026-09-16 12:00:24 +00:00
28 changed files with 1040 additions and 79 deletions
-63
View File
@@ -1,63 +0,0 @@
---
name: kind-on-kata-smoke
on:
push:
branches:
- poc/kind-on-kata
paths:
- .gitea/workflows/kind-on-kata-smoke.yml
workflow_dispatch:
jobs:
smoke:
runs-on: kata-poc
steps:
- name: Create nested kind cluster
shell: sh
env:
KIND_VERSION: v0.27.0
KIND_NODE_IMAGE: kindest/node:v1.32.2@sha256:f226345927d7e348497136874b6d207e0b32cc52154ad8323129352923a3142f
run: |
set -eu
apk add --no-cache ca-certificates curl docker-cli
curl --retry 5 --retry-all-errors --connect-timeout 15 -fsSLo /tmp/kind \
"https://kind.sigs.k8s.io/dl/${KIND_VERSION}/kind-linux-amd64"
curl --retry 5 --retry-all-errors --connect-timeout 15 -fsSLo /tmp/kind.sha256sum \
"https://kind.sigs.k8s.io/dl/${KIND_VERSION}/kind-linux-amd64.sha256sum"
expected="$(awk '{print $1}' /tmp/kind.sha256sum)"
printf '%s %s\n' "$expected" /tmp/kind | sha256sum -c -
install -m 0755 /tmp/kind /usr/local/bin/kind
docker info --format 'kernel={{.KernelVersion}} driver={{.Driver}}'
test "$(docker info --format '{{.Driver}}')" = overlay2
cleanup() {
kind delete cluster --name nested >/dev/null 2>&1 || true
}
trap cleanup EXIT
cat >/tmp/kind-config.yaml <<'EOF'
kind: Cluster
apiVersion: kind.x-k8s.io/v1alpha4
name: nested
nodes:
- role: control-plane
extraMounts:
- hostPath: /dev/kmsg
containerPath: /dev/kmsg
EOF
kind create cluster -v 9 --retain --config /tmp/kind-config.yaml --image "$KIND_NODE_IMAGE" --wait 5m
docker exec nested-control-plane kubectl \
--kubeconfig=/etc/kubernetes/admin.conf wait \
--for=condition=Ready node/nested-control-plane --timeout=2m
docker exec nested-control-plane kubectl \
--kubeconfig=/etc/kubernetes/admin.conf run smoke \
--image=docker.io/library/busybox:1.37 --restart=Never \
--command -- sh -c 'echo kind-on-kata-ok'
docker exec nested-control-plane kubectl \
--kubeconfig=/etc/kubernetes/admin.conf wait \
--for=jsonpath='{.status.phase}'=Succeeded pod/smoke --timeout=2m
test "$(docker exec nested-control-plane kubectl \
--kubeconfig=/etc/kubernetes/admin.conf logs smoke)" = kind-on-kata-ok
kind delete cluster --name nested
trap - EXIT
+1 -1
View File
@@ -229,7 +229,7 @@ configMap:
require_pkce: false require_pkce: false
token_endpoint_auth_method: 'client_secret_basic' token_endpoint_auth_method: 'client_secret_basic'
redirect_uris: redirect_uris:
- 'https://grafana.tail7e769.ts.net/login/generic_oauth' - 'https://grafana.ad.ddupan.top/login/generic_oauth'
scopes: scopes:
- 'openid' - 'openid'
- 'profile' - 'profile'
+18
View File
@@ -0,0 +1,18 @@
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: nats
namespace: flux-system
spec:
dependsOn:
- name: cert-manager
- name: external-secrets
- name: openebs
interval: 10m
path: ./platform/nats
prune: false
sourceRef:
kind: GitRepository
name: flux-system
timeout: 10m
wait: true
+1
View File
@@ -10,6 +10,7 @@ resources:
- apps/gitea-actions.yaml - apps/gitea-actions.yaml
- apps/http-echo.yaml - apps/http-echo.yaml
- apps/openebs.yaml - apps/openebs.yaml
- apps/nats.yaml
- apps/spire.yaml - apps/spire.yaml
- apps/observability.yaml - apps/observability.yaml
- apps/zot.yaml - apps/zot.yaml
+2
View File
@@ -11,7 +11,9 @@ homelab_dns:
- { zone: ad.ddupan.top, name: pve2, type: A, values: [192.168.10.7] } - { zone: ad.ddupan.top, name: pve2, type: A, values: [192.168.10.7] }
- { zone: ad.ddupan.top, name: pve3, type: A, values: [192.168.10.9] } - { zone: ad.ddupan.top, name: pve3, type: A, values: [192.168.10.9] }
- { zone: ad.ddupan.top, name: retrolab, type: A, values: [10.60.0.10] } - { zone: ad.ddupan.top, name: retrolab, type: A, values: [10.60.0.10] }
- { zone: ad.ddupan.top, name: grafana, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: netbox, type: A, values: [192.168.10.127] } - { zone: ad.ddupan.top, name: netbox, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: nats, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: s3, type: A, values: [192.168.10.127] } - { zone: ad.ddupan.top, name: s3, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: spire-oidc, type: A, values: [192.168.10.127] } - { zone: ad.ddupan.top, name: spire-oidc, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: zot, type: A, values: [192.168.10.127] } - { zone: ad.ddupan.top, name: zot, type: A, values: [192.168.10.127] }
+4
View File
@@ -119,6 +119,10 @@ issuerRef:
kind: ClusterIssuer kind: ClusterIssuer
``` ```
`values.yaml` 必须保持 `config.gatewayAPI.enabled: true`。`bao-acme` 的 HTTP-01
solver 通过共享 Gateway 创建临时 HTTPRoute;关闭该项不会让 ClusterIssuer 变为
NotReady,而是会让每个 Challenge 卡在 `gateway api is not enabled`。
Issuance is capped by `default_directory_policy = role:bao-server` Issuance is capped by `default_directory_policy = role:bao-server`
(`../../infrastructure/openbao/terraform/pki.tf`), which permits `ad.ddupan.top` subdomains only. Clients (`../../infrastructure/openbao/terraform/pki.tf`), which permits `ad.ddupan.top` subdomains only. Clients
need the internal CA in their trust store — already true for the PVE nodes, the DC and need the internal CA in their trust store — already true for the PVE nodes, the DC and
+7
View File
@@ -44,6 +44,13 @@ cainjector:
limits: limits:
memory: 256Mi memory: 256Mi
# bao-acme solves HTTP-01 through the shared Gateway. The ClusterIssuer can be
# accepted while this is disabled, but every Challenge then stays pending with
# "gateway api is not enabled". Gateway API CRDs are installed by Envoy Gateway.
config:
gatewayAPI:
enabled: true
# ⚠ DNS-01 self-check: cert-manager polls authoritative NS for the _acme-challenge # ⚠ DNS-01 self-check: cert-manager polls authoritative NS for the _acme-challenge
# TXT record before telling the CA to validate. By default it asks the cluster's # TXT record before telling the CA to validate. By default it asks the cluster's
# resolver, which for ad.ddupan.top is CoreDNS -> the Samba AD DC (k3s/coredns-custom.yaml). # resolver, which for ad.ddupan.top is CoreDNS -> the Samba AD DC (k3s/coredns-custom.yaml).
+48
View File
@@ -0,0 +1,48 @@
# NATS
共享的轻量消息基础设施。首期为 Gitea microVM runner 提供 JetStream work queue,
但 Account、subject 与部署位置均不与 CI controller 绑定,其他服务可按独立 Account
复用。
## 当前拓扑
- 单节点 NATS;当前 homelab 没有资源运行有意义的三副本 JetStream quorum。
- JetStream file store 使用 `localpv-zfs-ceph`,PVC 2 GiB。
- 服务通过 k3s ServiceLB 在 `nats.ad.ddupan.top:4222` 暴露给内网;集群内客户端
使用 `nats.nats.svc.cluster.local:4222`。访问控制由 TLS、Account 与用户权限负责,
不额外维护易漂移的源 IP 白名单。
- TLS 证书由 `bao-acme` 签发。PVE 节点已信任内部 CA。
- `bao-server` PKI role 只接受 RSA CSR,因此 Certificate 使用 RSA 2048;不要改成
ECDSA,ACME challenge 会成功但 finalize 会以 `role requires keys of type rsa` 失败。
- `SYS` Account 用于管理;`CI` Account 启用 JetStream,存储上限 1 GiB。
Account 内的 JetStream 配额会原样进入 `nats.conf`,必须使用 NATS 的 `MB`/`GB`
格式;PVC 等 Kubernetes resource quantity 才使用 `Mi`/`Gi`。
首期使用静态用户,密码只存在 OpenBao `kv/k8s/nats`:
```text
sys_password
ci_producer_password
ci_worker_password
```
`ci-producer` 只能发布 `ci.runner.>` 并调用必要的 JetStream API;`ci-worker`
只能调用 JetStream pull/ACK API。二者都不能读取另一个 Account 的 subject。
后续 SPIRE/Auth Callout 动态认证见 homelab-infra issue #56。该迁移只替换连接
凭据,不改变 Account、stream、subject 或 consumer。
## CI stream 约定
controller 首次启动时幂等创建 `CI_RUNNER` stream:`ci.runner.*`、
`WorkQueuePolicy`、file storage、24h/10000 条/256 MiB 上限。每类 runner 使用独立
subject 和 durable pull consumer;同类型的多个 worker 共享 durable consumer。
ACK 后消息立即删除,不保存 CI 历史。
## 验证
```bash
kubectl -n nats get helmrelease,pod,pvc,certificate,externalsecret
kubectl -n nats logs statefulset/nats -c nats
```
+21
View File
@@ -0,0 +1,21 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: nats-ad-ddupan-top
namespace: nats
spec:
secretName: nats-server-tls
issuerRef:
name: bao-acme
kind: ClusterIssuer
group: cert-manager.io
commonName: nats.ad.ddupan.top
dnsNames:
- nats.ad.ddupan.top
duration: 720h
renewBefore: 168h
privateKey:
# OpenBao's bao-server role intentionally accepts RSA keys only.
algorithm: RSA
size: 2048
rotationPolicy: Always
+16
View File
@@ -0,0 +1,16 @@
apiVersion: external-secrets.io/v1
kind: ExternalSecret
metadata:
name: nats-auth
namespace: nats
spec:
refreshInterval: 1h
secretStoreRef:
kind: ClusterSecretStore
name: openbao
target:
creationPolicy: Owner
name: nats-auth
dataFrom:
- extract:
key: k8s/nats
+31
View File
@@ -0,0 +1,31 @@
apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
metadata:
name: nats
namespace: nats
spec:
chart:
spec:
chart: nats
interval: 1h
sourceRef:
kind: HelmRepository
name: nats
version: 2.14.2
driftDetection:
mode: enabled
install:
strategy:
name: RetryOnFailure
retryInterval: 5m
interval: 30m
releaseName: nats
targetNamespace: nats
timeout: 10m
upgrade:
strategy:
name: RetryOnFailure
retryInterval: 5m
valuesFrom:
- kind: ConfigMap
name: nats-values
+8
View File
@@ -0,0 +1,8 @@
apiVersion: source.toolkit.fluxcd.io/v1
kind: HelmRepository
metadata:
name: nats
namespace: nats
spec:
interval: 1h
url: https://nats-io.github.io/k8s/helm/charts/
+17
View File
@@ -0,0 +1,17 @@
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
generatorOptions:
disableNameSuffixHash: true
labels:
reconcile.fluxcd.io/watch: Enabled
configMapGenerator:
- name: nats-values
namespace: nats
files:
- values.yaml=values.yaml
resources:
- namespace.yaml
- helmrepository.yaml
- external-secret.yaml
- certificate.yaml
- helmrelease.yaml
+4
View File
@@ -0,0 +1,4 @@
apiVersion: v1
kind: Namespace
metadata:
name: nats
+76
View File
@@ -0,0 +1,76 @@
config:
jetstream:
enabled: true
fileStore:
enabled: true
maxSize: 2G
pvc:
enabled: true
size: 2Gi
storageClassName: localpv-zfs-ceph
memoryStore:
enabled: true
maxSize: 64M
nats:
tls:
enabled: true
secretName: nats-server-tls
merge:
system_account: SYS
accounts:
SYS:
users:
- user: sys
password: "<< $NATS_SYS_PASSWORD >>"
CI:
jetstream:
# Account limits are JSON-encoded by config.merge. Use explicit byte
# counts so NATS receives integers rather than quoted size strings.
max_memory: 33554432
max_file: 1073741824
max_streams: 16
max_consumers: 64
max_bytes_required: true
users:
- user: ci-producer
password: "<< $NATS_CI_PRODUCER_PASSWORD >>"
permissions:
publish:
allow: [ci.runner.>, $JS.API.>]
subscribe:
allow: [_INBOX.>]
- user: ci-worker
password: "<< $NATS_CI_WORKER_PASSWORD >>"
permissions:
publish:
allow: [$JS.API.>, $JS.ACK.>]
subscribe:
allow: [_INBOX.>]
container:
env:
NATS_SYS_PASSWORD:
valueFrom:
secretKeyRef: {name: nats-auth, key: sys_password}
NATS_CI_PRODUCER_PASSWORD:
valueFrom:
secretKeyRef: {name: nats-auth, key: ci_producer_password}
NATS_CI_WORKER_PASSWORD:
valueFrom:
secretKeyRef: {name: nats-auth, key: ci_worker_password}
resources:
requests: {cpu: 25m, memory: 64Mi}
limits: {memory: 192Mi}
natsBox:
enabled: false
promExporter:
enabled: true
podMonitor:
enabled: true
service:
merge:
spec:
type: LoadBalancer
+20 -3
View File
@@ -4,7 +4,22 @@ Metrics, logs, and traces for the cluster — one VictoriaMetrics-ecosystem stac
operator-driven, running in the `monitoring` namespace. Replaces the standalone operator-driven, running in the `monitoring` namespace. Replaces the standalone
Docker Compose stack in `../../apps/victoriametrics/`. Docker Compose stack in `../../apps/victoriametrics/`.
## Status — DEPLOYED (2026-07-10) ## Grafana 当前入口(2026-09-16)
统一使用 <https://grafana.ad.ddupan.top>,A 记录指向 `192.168.10.127`,由共享
Envoy Gateway 的 `https` listener 与 `*.ad.ddupan.top` 证书提供 TLS。
Grafana root_url 和 Authelia 回调均使用此域名;旧专用 Tailscale Ingress 已停用。
远程客户端仍需有到 LAN 的路由及已配置的内网 DNS 转发。
内存看板:`/d/homelab-memory`;采集配置及 AppArmor 规则见
[主机与进程内存采集](metrics/exporters/README.md)。
Grafana 由 Flux 管理;修改 values 后通过 Git 合并触发 HelmRelease,避免现场 Helm
修改被漂移检测回滚。Authelia 当前不由 Flux 管理,需单独 Helm upgrade 并先结构比较
live values。迁移时先添加 DNS 和回调,再切换 Grafana。Recreate 策略会短暂中断访问,
但保留原 PVC、用户、看板和数据源。回滚需同时恢复 root_url、回调和 Ingress 配置。
## 初始部署记录(2026-07-10)
Live and verified in the `monitoring` namespace (operator chart 0.66.2): Live and verified in the `monitoring` namespace (operator chart 0.66.2):
metrics (data queryable), logs (pods ingesting), traces (VTSingle CRD, verified via metrics (data queryable), logs (pods ingesting), traces (VTSingle CRD, verified via
@@ -56,8 +71,10 @@ hosts and `logs/vlogs-ingress.yaml` to push their logs.
| Manage | **victoria-metrics-operator** | VMSingle/VMAgent/VMAlert/VMAlertmanager/VMRule **and** VLSingle as CRDs | | Manage | **victoria-metrics-operator** | VMSingle/VMAgent/VMAlert/VMAlertmanager/VMRule **and** VLSingle as CRDs |
| Expose | **Tailscale ingress** (private) + **Authelia OIDC** | admin tool: private + SSO | | Expose | **Tailscale ingress** (private) + **Authelia OIDC** | admin tool: private + SSO |
Grafana's Prometheus-operator converter is on, so any chart shipping a VictoriaMetrics Operator 的 Prometheus converter 已启用;官方
`ServiceMonitor`/`PodMonitor`/`PrometheusRule` is scraped automatically. `prometheus-operator-crds` chart 由 `operator/` 一并管理。因此应用 chart 可以原生
声明 `ServiceMonitor`、`PodMonitor` 或 `PrometheusRule`,再由 converter 转换为对应
VM 资源,不需要每个应用额外维护一份 `VM*Scrape`。
## Architecture ## Architecture
@@ -0,0 +1,502 @@
{
"uid": "homelab-memory",
"title": "Homelab 内存与 Swap",
"schemaVersion": 39,
"version": 1,
"tags": [
"homelab",
"memory"
],
"timezone": "utc",
"refresh": "1m",
"time": {
"from": "now-6h",
"to": "now"
},
"templating": {
"list": [
{
"name": "DS_VM",
"label": "数据源",
"type": "datasource",
"query": "prometheus",
"current": {
"text": "VictoriaMetrics",
"value": "VictoriaMetrics"
}
}
]
},
"panels": [
{
"id": 1,
"type": "timeseries",
"title": "主机物理内存",
"description": "ZFS ARC 是内存的一部分;不能与此面板或 Pod/PSS 再叠加。",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 0,
"y": 0,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "node_memory_MemTotal_bytes{job=\"node-exporter\"} - node_memory_MemAvailable_bytes{job=\"node-exporter\"}",
"legendFormat": "已用(total - available)"
},
{
"refId": "B",
"expr": "node_memory_MemAvailable_bytes{job=\"node-exporter\"}",
"legendFormat": "可用"
},
{
"refId": "C",
"expr": "node_memory_MemTotal_bytes{job=\"node-exporter\"}",
"legendFormat": "总量"
}
],
"fieldConfig": {
"defaults": {
"unit": "bytes",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
},
{
"id": 2,
"type": "timeseries",
"title": "ZFS ARC",
"description": "缓存可回收性取决于实际压力,ARC 不等同于 free 的 buff/cache。",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 12,
"y": 0,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "node_zfs_arc_size{job=\"node-exporter\"}",
"legendFormat": "当前 ARC"
},
{
"refId": "B",
"expr": "node_zfs_arc_c{job=\"node-exporter\"}",
"legendFormat": "目标"
},
{
"refId": "C",
"expr": "node_zfs_arc_c_max{job=\"node-exporter\"}",
"legendFormat": "上限"
}
],
"fieldConfig": {
"defaults": {
"unit": "bytes",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
},
{
"id": 3,
"type": "timeseries",
"title": "Swap 使用量",
"description": "",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 0,
"y": 8,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "node_memory_SwapTotal_bytes{job=\"node-exporter\"} - node_memory_SwapFree_bytes{job=\"node-exporter\"}",
"legendFormat": "已用"
},
{
"refId": "B",
"expr": "node_memory_SwapTotal_bytes{job=\"node-exporter\"}",
"legendFormat": "总量"
}
],
"fieldConfig": {
"defaults": {
"unit": "bytes",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
},
{
"id": 4,
"type": "timeseries",
"title": "Swap 换页速率",
"description": "单位为页/秒,不假定页大小。",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 12,
"y": 8,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "rate(node_vmstat_pswpin{job=\"node-exporter\"}[5m])",
"legendFormat": "读入 pages/s"
},
{
"refId": "B",
"expr": "rate(node_vmstat_pswpout{job=\"node-exporter\"}[5m])",
"legendFormat": "写出 pages/s"
}
],
"fieldConfig": {
"defaults": {
"unit": "ops",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
},
{
"id": 5,
"type": "timeseries",
"title": "程序与虚拟机 PSS:前 15",
"description": "PSS 按共享页比例分摊。vm: 表示 QEMU 在宿主机的占用,不是来宾内部应用占用。与 Pod working set、ARC 不能相加。",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 0,
"y": 16,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "topk(15, namedprocess_namegroup_memory_bytes{job=\"process-exporter\",memtype=\"proportionalResident\"})",
"legendFormat": "{{groupname}}"
}
],
"fieldConfig": {
"defaults": {
"unit": "bytes",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
},
{
"id": 6,
"type": "timeseries",
"title": "程序与虚拟机 SwapPss:前 15",
"description": "通过 smaps 对共享换出页按比例分摊。",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 12,
"y": 16,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "topk(15, namedprocess_namegroup_memory_bytes{job=\"process-exporter\",memtype=\"proportionalSwapped\"})",
"legendFormat": "{{groupname}}"
}
],
"fieldConfig": {
"defaults": {
"unit": "bytes",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
},
{
"id": 7,
"type": "timeseries",
"title": "Pod working set:前 15",
"description": "cgroup working set 与 PSS 口径不同,不相加。",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 0,
"y": 24,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "topk(15, sum by (namespace,pod) (container_memory_working_set_bytes{job=\"cadvisor\",container!=\"\",container!=\"POD\",pod!=\"\"}))",
"legendFormat": "{{namespace}}/{{pod}}"
}
],
"fieldConfig": {
"defaults": {
"unit": "bytes",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
},
{
"id": 8,
"type": "timeseries",
"title": "内存压力 PSI",
"description": "",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 12,
"y": 24,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "rate(node_pressure_memory_waiting_seconds_total{job=\"node-exporter\"}[5m])",
"legendFormat": "some"
},
{
"refId": "B",
"expr": "rate(node_pressure_memory_stalled_seconds_total{job=\"node-exporter\"}[5m])",
"legendFormat": "full"
}
],
"fieldConfig": {
"defaults": {
"unit": "percentunit",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
},
{
"id": 9,
"type": "timeseries",
"title": "Samba RPC worker 数量",
"description": "仅在 exporter 健康时将无 worker 解释为 0;见采集健康面板。",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 0,
"y": 32,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "namedprocess_namegroup_num_procs{job=\"process-exporter\",groupname=\"rpcd_lsad\"} or on() (0 * max(up{job=\"process-exporter\"} == 1))",
"legendFormat": "rpcd_lsad"
}
],
"fieldConfig": {
"defaults": {
"unit": "short",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
},
{
"id": 10,
"type": "timeseries",
"title": "进程采集健康",
"description": "up 仅表示抓取成功;还需检查读取错误。进程退出等竞态可能造成偶发 partial errors,持续增长时检查 AppArmor 审计。",
"datasource": {
"type": "prometheus",
"uid": "${DS_VM}"
},
"gridPos": {
"x": 12,
"y": 32,
"w": 12,
"h": 8
},
"targets": [
{
"refId": "A",
"expr": "up{job=\"process-exporter\"}",
"legendFormat": "up"
},
{
"refId": "B",
"expr": "scrape_duration_seconds{job=\"process-exporter\"}",
"legendFormat": "scrape 秒"
},
{
"refId": "C",
"expr": "namedprocess_scrape_errors{job=\"process-exporter\"}",
"legendFormat": "采集错误"
},
{
"refId": "D",
"expr": "rate(namedprocess_scrape_partial_errors{job=\"process-exporter\"}[5m])",
"legendFormat": "部分字段读取失败/s"
}
],
"fieldConfig": {
"defaults": {
"unit": "short",
"min": 0
},
"overrides": []
},
"options": {
"legend": {
"displayMode": "table",
"placement": "bottom",
"calcs": [
"lastNotNull"
]
},
"tooltip": {
"mode": "multi"
}
}
}
]
}
@@ -0,0 +1,17 @@
# Grafana 自身使用 OIDC,不增加第二层 forward-auth。
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: grafana-lan
namespace: monitoring
spec:
parentRefs:
- name: eg
namespace: envoy-gateway-system
sectionName: https
hostnames:
- grafana.ad.ddupan.top
rules:
- backendRefs:
- name: grafana
port: 80
@@ -17,5 +17,13 @@ configMapGenerator:
options: options:
labels: labels:
grafana_dashboard: "1" grafana_dashboard: "1"
- name: grafana-dashboard-homelab-memory
namespace: monitoring
files:
- homelab-memory.json=dashboards/homelab-memory.json
options:
labels:
grafana_dashboard: "1"
resources: resources:
- helmrelease.yaml - helmrelease.yaml
- httproute.yaml
+6 -12
View File
@@ -1,6 +1,6 @@
# Grafana — single pane over metrics (VictoriaMetrics), logs (VictoriaLogs) and # Grafana — single pane over metrics (VictoriaMetrics), logs (VictoriaLogs) and
# traces (VictoriaTraces via Jaeger API). Exposed privately on the tailnet # traces (VictoriaTraces via Jaeger API). LAN: grafana.ad.ddupan.top,
# (grafana.tail7e769.ts.net) and authenticated via Authelia OIDC. Local admin is # authenticated via Authelia OIDC. Local admin is
# break-glass only. # break-glass only.
# #
# Chart: grafana/grafana (repo: https://grafana.github.io/helm-charts) # Chart: grafana/grafana (repo: https://grafana.github.io/helm-charts)
@@ -54,16 +54,10 @@ persistence:
deploymentStrategy: deploymentStrategy:
type: Recreate type: Recreate
# --- Private exposure via the Tailscale ingress (like seaweedfs-admin) --- # 内网入口由 httproute.yaml 接入 Envoy,使用现有通配 TLS 证书。
# The tailscale operator provisions grafana.<tailnet>.ts.net and a TLS cert. # 客户端通过既有内网 DNS 转发解析,无需独立 Tailscale Ingress。
ingress: ingress:
enabled: true enabled: false
ingressClassName: tailscale
hosts:
- grafana
tls:
- hosts:
- grafana
# --- OIDC via Authelia (AD groups -> Grafana roles) --- # --- OIDC via Authelia (AD groups -> Grafana roles) ---
# client_secret is injected from the grafana-oidc Secret (see oidc-secret.yaml), # client_secret is injected from the grafana-oidc Secret (see oidc-secret.yaml),
@@ -76,7 +70,7 @@ envValueFrom:
grafana.ini: grafana.ini:
server: server:
root_url: "https://grafana.tail7e769.ts.net" # must match the tailnet FQDN + Authelia redirect_uri root_url: "https://grafana.ad.ddupan.top" # 与 Authelia redirect_uri 一致
auth: auth:
# Keep the local admin login available as break-glass; don't force OIDC-only. # Keep the local admin login available as break-glass; don't force OIDC-only.
disable_login_form: false disable_login_form: false
@@ -3,6 +3,7 @@ kind: Kustomization
resources: resources:
- namespace.yaml - namespace.yaml
- helmrepository.yaml - helmrepository.yaml
- prometheus-helmrepository.yaml
- grafana-helmrepository.yaml - grafana-helmrepository.yaml
- operator - operator
- metrics - metrics
@@ -0,0 +1,42 @@
# 主机与进程内存采集
2026-09-16 已增加 `process-exporter.yaml`,固定上游 0.8.7,以 DaemonSet 运行,
只读挂载宿主机 `/proc`,每 60 秒由 VMPodScrape 抓取,不开宿主机端口。
NetworkPolicy 仅放行同 namespace 的 vmagent。进程按名称分组,QEMU 按 guest 名,
NetBox 和 VS Code 按路径归组;不把完整命令行或 PID 放进指标标签。
当前 live 使用 `gather-smaps=true`,导出 RSS、VmSwap、PSS、SwapPss 与进程数。
`apparmor/homelab-process-exporter` 已安装到 `/etc/apparmor.d/homelab-process-exporter`
并以 enforce 加载。用户明确批准了跨进程读取:SYS_PTRACE、DAC_READ_SEARCH 与
`ptrace (read) peer=**`;文件权限限于指标需要的 proc 文件,不允许 ptrace trace,
不开放 `/proc/<pid>/mem`、`environ`,不关闭 AppArmor,也未修改虚拟机或容器默认 profile。
当前只部署到 laptop。其他节点必须先安装同名 profile,再扩展 nodeSelector:
```bash
sudo install -m 0644 apparmor/homelab-process-exporter /etc/apparmor.d/homelab-process-exporter
sudo apparmor_parser -r /etc/apparmor.d/homelab-process-exporter
```
修改 profile 可原位重载,无需重启业务。禁用时先删除 exporter DaemonSet,再卸载
专用 profile;不要让引用 profile 的 Pod 在 profile 缺失时启动。
验证已读到三台虚拟机、NetBox、Codex 的非零 PSS/SwapPss。`namedprocess_scrape_errors`
和 `namedprocess_scrape_procread_errors` 为零。上游会先读取再匹配进程,内核线程没有
可读的 smaps_rollup 会计入 partial errors;现场有约 430 个这种线程。
partial errors 不保证为零,应结合业务组 PSS 和 AppArmor 审计判断。
RSS、PSS、cAdvisor working set、ARC 是不同口径,不能直接相加。
已有采集无需重复部署:
- node-exporter:主机内存、Swap、PSI、ZFS ARC。
- ARC 实际指标名为 `node_zfs_arc_size`、`node_zfs_arc_c`、`node_zfs_arc_c_max`。
- kubelet/cAdvisor:Pod/container working set、Swap 等。
Grafana 新增 `Homelab 内存与 Swap`(UID `homelab-memory`),由 sidecar ConfigMap 加载。
仅使用 `job="node-exporter"` 查询主机指标,避免目前 `docker-hosts` 对同一 9100 端口
的重复抓取。另有 `192.168.10.127:8080` Docker cAdvisor target 失效,本次未修改该旧配置。
所有新增 manifest 已纳入对应 Kustomization,并已单独应用到 live;尚未提交到远端,
Flux source 尚未包含这些新增资源。修改 exporter ConfigMap 后需滚动重启 DaemonSet,
程序不会自动重载进程匹配配置。
@@ -0,0 +1,27 @@
#include <tunables/global>
# 限定为指标所需文件;跨 profile 读取由用户明确批准;不允许 trace 或读取进程 mem/environ。
profile homelab-process-exporter flags=(attach_disconnected,mediate_deleted) {
#include <abstractions/base>
network inet stream,
network inet6 stream,
capability sys_ptrace,
capability dac_read_search,
ptrace (read) peer=**,
signal (receive) peer=unconfined,
signal (receive) peer=cri-containerd.apparmor.d,
signal (receive) peer=runc,
/bin/process-exporter mr,
/process-exporter mr,
/config/** r,
/etc/{passwd,group,nsswitch.conf} r,
/{host/,}proc/ r,
/{host/,}proc/{stat,meminfo,cpuinfo,uptime,version,sys/kernel/random/boot_id} r,
/{host/,}proc/[0-9]*/ r,
/{host/,}proc/[0-9]*/{stat,status,cmdline,smaps,smaps_rollup,io,limits,cgroup,wchan} r,
/{host/,}proc/[0-9]*/fd/ r,
/{host/,}proc/[0-9]*/task/ r,
/{host/,}proc/[0-9]*/task/[0-9]*/{stat,status,io,cmdline,wchan,cgroup,limits,smaps,smaps_rollup} r,
/proc/sys/net/core/somaxconn r,
/proc/self/{stat,status,smaps,smaps_rollup,limits,cgroup,mountinfo} r,
}
@@ -0,0 +1,124 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: process-exporter-config
namespace: monitoring
data:
config.yml: |
process_names:
# 只暴露虚拟机名,不把完整命令行、PID 或凭据写入标签。
- name: 'vm:{{.Matches.VM}}'
comm: [qemu-system-x86]
cmdline: ['-name\s+guest=(?P<VM>[^,\s]+)']
- name: netbox
cmdline: ['/opt/netbox/']
- name: vscode
cmdline: ['\.vscode-server/']
- name: '{{.Comm}}'
cmdline: ['.+']
---
apiVersion: apps/v1
kind: DaemonSet
metadata:
name: process-exporter
namespace: monitoring
spec:
selector:
matchLabels:
app.kubernetes.io/name: process-exporter
template:
metadata:
labels:
app.kubernetes.io/name: process-exporter
spec:
nodeSelector:
kubernetes.io/hostname: laptop
hostPID: true
automountServiceAccountToken: false
tolerations:
- operator: Exists
containers:
- name: process-exporter
image: ncabatoff/process-exporter:0.8.7
args:
- -procfs=/host/proc
- -config.path=/config/config.yml
- -gather-smaps=true
- -threads=false
- -children=false
securityContext:
runAsUser: 0
readOnlyRootFilesystem: true
allowPrivilegeEscalation: false
appArmorProfile:
type: Localhost
localhostProfile: homelab-process-exporter
capabilities:
drop: [ALL]
add: [SYS_PTRACE, DAC_READ_SEARCH]
ports:
- name: metrics
containerPort: 9256
readinessProbe:
tcpSocket:
port: metrics
resources:
requests:
cpu: 25m
memory: 32Mi
limits:
cpu: 500m
memory: 128Mi
volumeMounts:
- name: proc
mountPath: /host/proc
readOnly: true
- name: config
mountPath: /config
readOnly: true
volumes:
- name: proc
hostPath:
path: /proc
type: Directory
- name: config
configMap:
name: process-exporter-config
---
apiVersion: operator.victoriametrics.com/v1beta1
kind: VMPodScrape
metadata:
name: process-exporter
namespace: monitoring
spec:
selector:
matchLabels:
app.kubernetes.io/name: process-exporter
podMetricsEndpoints:
- port: metrics
interval: 60s
scrapeTimeout: 30s
relabelConfigs:
- targetLabel: job
replacement: process-exporter
- sourceLabels: [__meta_kubernetes_pod_node_name]
targetLabel: node
---
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: process-exporter
namespace: monitoring
spec:
podSelector:
matchLabels:
app.kubernetes.io/name: process-exporter
policyTypes: [Ingress]
ingress:
- from:
- podSelector:
matchLabels:
app.kubernetes.io/name: vmagent
ports:
- protocol: TCP
port: 9256
@@ -14,3 +14,4 @@ resources:
- scrapes/docker-hosts.yaml - scrapes/docker-hosts.yaml
- scrapes/kubelet.yaml - scrapes/kubelet.yaml
- exporters/node-exporter.yaml - exporters/node-exporter.yaml
- exporters/process-exporter.yaml
@@ -10,4 +10,5 @@ configMapGenerator:
files: files:
- values.yaml=values.yaml - values.yaml=values.yaml
resources: resources:
- prometheus-crds-helmrelease.yaml
- helmrelease.yaml - helmrelease.yaml
@@ -0,0 +1,29 @@
apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
metadata:
name: prometheus-operator-crds
namespace: monitoring
spec:
chart:
spec:
chart: prometheus-operator-crds
interval: 1h
sourceRef:
kind: HelmRepository
name: prometheus-community
namespace: monitoring
version: 32.0.0
install:
crds: CreateReplace
strategy:
name: RetryOnFailure
retryInterval: 5m
interval: 30m
releaseName: prometheus-operator-crds
targetNamespace: monitoring
timeout: 10m
upgrade:
crds: CreateReplace
strategy:
name: RetryOnFailure
retryInterval: 5m
@@ -0,0 +1,8 @@
apiVersion: source.toolkit.fluxcd.io/v1
kind: HelmRepository
metadata:
name: prometheus-community
namespace: monitoring
spec:
interval: 1h
url: https://prometheus-community.github.io/helm-charts