181 Commits
Author SHA1 Message Date
panxiao81 e325a703dd Merge 部署内置 SPIRE CLI 的 runner 镜像
yaml / yaml (push) Successful in 16s
2026-09-24 07:29:53 +00:00
panxiao81 130849040c chore(runner): 部署内置 SPIRE CLI 镜像
yaml / yaml (pull_request) Successful in 22s
2026-09-24 07:29:36 +00:00
panxiao81 bcd034355e Merge 部署 runner 工作目录修复镜像
yaml / yaml (push) Failing after 8s
2026-09-24 06:57:48 +00:00
panxiao81 506e0cb983 chore(runner): 部署工作目录修复镜像
yaml / yaml (pull_request) Failing after 8s
2026-09-24 06:57:30 +00:00
panxiao81 c07709e078 Merge 为 OpenSandbox runner 挂载独立工作目录
yaml / yaml (push) Failing after 8s
2026-09-24 06:29:37 +00:00
panxiao81 e1979ebbb8 fix(runner): 挂载独立工作目录
yaml / yaml (pull_request) Failing after 7s
2026-09-24 06:29:02 +00:00
panxiao81 b02b2aec96 Merge 滚动部署最终 Docker runner 镜像
yaml / yaml (push) Successful in 18s
2026-09-24 06:12:24 +00:00
panxiao81 ff92037a28 chore(runner): 滚动部署最终 Docker 运行环境镜像
yaml / yaml (pull_request) Successful in 19s
2026-09-24 06:11:49 +00:00
panxiao81 9360eb2ea0 Merge 自动 Docker runner 镜像 rollout
yaml / yaml (push) Successful in 19s
2026-09-21 17:43:19 +00:00
panxiao81 0c7eb371e5 chore: rollout job-local Docker runner image
yaml / yaml (pull_request) Successful in 28s
2026-09-21 17:40:12 +00:00
panxiao81 757c21fc0b Merge pull request 'feat: 启用生产 VM runner 标签' (#136) from rollout/vm-runner-production into main
yaml / yaml (push) Successful in 12s
Reviewed-on: #136
2026-09-21 15:23:31 +00:00
panxiao81 c8e163f01e feat: 启用生产 VM runner 标签
yaml / yaml (pull_request) Successful in 21s
2026-09-21 15:22:36 +00:00
panxiao81 59fe28faa5 Merge Kata kind executor 最低资源配置
yaml / yaml (push) Successful in 20s
2026-09-21 12:36:20 +00:00
panxiao81 69c90b9fdf fix: 为 Kata kind executor 配置最低资源
yaml / yaml (pull_request) Successful in 19s
2026-09-21 12:33:35 +00:00
panxiao81 a6a861cc7e Merge 统一 Pod 与 VM executor 运行模型
yaml / yaml (push) Successful in 19s
2026-09-21 12:06:02 +00:00
panxiao81 5c9a09edde refactor: 统一 Pod 与 VM executor 运行模型
yaml / yaml (pull_request) Successful in 31s
2026-09-21 12:01:55 +00:00
panxiao81 cfb9c2d150 Merge Kata DinD guest ext4 存储修复
yaml / yaml (push) Successful in 16s
2026-09-21 11:50:26 +00:00
panxiao81 1ddec34149 fix: 为 Kata DinD 使用 guest ext4 数据盘
yaml / yaml (pull_request) Successful in 21s
2026-09-21 11:48:57 +00:00
panxiao81 db63503ec9 Merge Kata kind 资源规格
yaml / yaml (push) Successful in 16s
2026-09-21 11:34:00 +00:00
panxiao81 9953fae93d fix: 为 Kata kind 任务分配明确资源
yaml / yaml (pull_request) Successful in 25s
2026-09-21 11:31:57 +00:00
panxiao81 56598d6c84 Merge Kata DinD cgroup nesting 修复
yaml / yaml (push) Successful in 16s
2026-09-21 11:16:48 +00:00
panxiao81 834fdcb771 fix: 恢复 Kata DinD cgroup 初始化
yaml / yaml (pull_request) Successful in 17s
2026-09-21 11:14:52 +00:00
panxiao81 47ff73cefb Merge pull request '彻底删除旧静态 Gitea Runner' (#130) from retire/static-gitea-runner-phase2 into main
yaml / yaml (push) Successful in 19s
2026-09-21 10:23:05 +00:00
panxiao81 bb24c76704 chore: 删除旧静态 Gitea Runner
yaml / yaml (pull_request) Successful in 18s
2026-09-21 10:21:33 +00:00
panxiao81 b0791e4095 Merge pull request '退役旧静态 Gitea Runner(第一阶段)' (#129) from retire/static-gitea-runner-phase1 into main
yaml / yaml (push) Successful in 19s
2026-09-21 10:18:44 +00:00
panxiao81 26179f700f chore: 开始退役旧静态 Gitea Runner
yaml / yaml (pull_request) Successful in 16s
2026-09-21 10:17:06 +00:00
panxiao81 1ac99f4476 Merge pull request '部署 Runner 结构化生命周期日志' (#128) from deploy/structured-runner-logs into main
yaml / yaml (push) Successful in 23s
2026-09-21 09:46:47 +00:00
panxiao81 891c6b90e7 deploy: 更新 Runner 结构化日志镜像
yaml / yaml (pull_request) Successful in 21s
2026-09-21 09:45:33 +00:00
panxiao81 c2314256cc Merge pull request '为 sandbox executor 暴露内网 Runner facade' (#127) from fix/expose-runner-facade-internal into main
yaml / yaml (push) Successful in 26s
2026-09-21 09:18:16 +00:00
panxiao81 8244cdad79 fix: 为 sandbox 暴露 Runner facade
yaml / yaml (pull_request) Successful in 16s
2026-09-21 09:11:29 +00:00
panxiao81 23841556d3 Merge VM 集成测试入口隔离
yaml / yaml (push) Successful in 24s
2026-09-21 09:00:23 +00:00
panxiao81 4daaa38495 deploy: 隔离 VM 集成测试入口
yaml / yaml (pull_request) Successful in 12s
2026-09-21 08:58:21 +00:00
panxiao81 da0630cd29 Merge OpenSandbox Runner 标记修复部署
yaml / yaml (push) Successful in 46s
2026-09-21 08:35:24 +00:00
panxiao81 d0fcfdeddc deploy: 补回 OpenSandbox Runner 标记
yaml / yaml (pull_request) Successful in 23s
2026-09-21 08:34:37 +00:00
panxiao81 ecf395cc8f Merge OpenSandbox metadata 修复部署
yaml / yaml (push) Successful in 58s
2026-09-21 08:07:33 +00:00
panxiao81 9f646d51e7 deploy: 修复 OpenSandbox metadata 编码
yaml / yaml (pull_request) Successful in 56s
2026-09-21 08:04:15 +00:00
panxiao81 5c7e7ff8db Merge Runner 后端独立容量池部署
yaml / yaml (push) Successful in 34s
2026-09-21 07:35:32 +00:00
panxiao81 d6b97355d6 deploy: 启用 Runner 后端容量池
yaml / yaml (pull_request) Successful in 48s
2026-09-21 07:33:31 +00:00
panxiao81 b82ba4d5a0 Merge Pod Runner 四并发恢复
yaml / yaml (push) Successful in 22s
2026-09-21 06:57:37 +00:00
panxiao81 9dc7aabc94 恢复 Pod Runner 四并发
yaml / yaml (pull_request) Successful in 21s
2026-09-21 06:57:09 +00:00
panxiao81 b7c95fd0b8 Merge Runner 无中断滚动与容量隔离
yaml / yaml (push) Successful in 24s
2026-09-21 06:51:39 +00:00
panxiao81 098ff4e27f 更新 Runner 容量隔离镜像
yaml / yaml (pull_request) Successful in 27s
2026-09-21 06:51:37 +00:00
panxiao81 dfc55e1d9e 启用 Runner 无中断滚动更新
yaml / yaml (pull_request) Successful in 29s
2026-09-21 06:39:11 +00:00
panxiao81 18d9f83413 Merge Runner claim 恢复镜像
yaml / yaml (push) Successful in 34s
2026-09-21 06:24:31 +00:00
panxiao81 e5e2ab1b89 更新 Runner claim 恢复镜像
yaml / yaml (pull_request) Successful in 15s
2026-09-21 06:21:20 +00:00
panxiao81 0b236c43d9 Merge OpenSandbox VM Runner canary
yaml / yaml (push) Successful in 14s
2026-09-21 06:12:35 +00:00
panxiao81 762ce6458d 启用 OpenSandbox VM Runner canary
yaml / yaml (pull_request) Successful in 17s
2026-09-21 06:10:59 +00:00
panxiao81 15ef512cfc Merge runner Gitea proxy Host fix
yaml / yaml (push) Failing after 31s
更新 Runner Gitea 反代 Host 修复镜像
2026-09-21 05:42:33 +00:00
panxiao81 a13a5118cf 更新 Runner Gitea 反代 Host 修复镜像
yaml / yaml (pull_request) Failing after 33s
2026-09-21 05:41:50 +00:00
panxiao81 25bc7ee7d6 Merge runner cleanup and checkout proxy fix
yaml / yaml (push) Failing after 41s
更新 Runner 自动清理与 checkout 修复镜像
2026-09-21 05:31:26 +00:00
panxiao81 3d41e0e29b 更新 Runner 自动清理与 checkout 修复镜像
yaml / yaml (pull_request) Failing after 37s
2026-09-21 05:30:09 +00:00
panxiao81 3169fcabd3 Merge pull request '记录 CI 身份 Action 使用入口' (#116) from docs/ci-actions into main 2026-09-21 03:31:03 +00:00
panxiao81 0a42a3ad04 记录 CI 身份 Action 使用入口 2026-09-21 03:30:31 +00:00
panxiao81 9ec034c947 Merge runner facade readiness fix
yaml / yaml (push) Failing after 5m30s
更新 Runner facade 就绪等待镜像
2026-09-20 21:37:42 +00:00
panxiao81 ad9b305aba 更新 Runner facade 就绪等待镜像
yaml / yaml (pull_request) Failing after 5m26s
2026-09-20 21:37:19 +00:00
panxiao81 9696df1f51 Merge pull request #114
yaml / yaml (push) Failing after 10m47s
修正动态 Runner 的 SPIRE agent 绑定
2026-09-20 21:31:11 +00:00
panxiao81 a335bdbc38 修正动态 Runner 的 SPIRE agent 绑定
yaml / yaml (pull_request) Failing after 11m40s
2026-09-20 21:30:48 +00:00
panxiao81 b6623069ad Merge pull request #113
yaml / yaml (push) Failing after 11m36s
更新动态 Runner SPIFFE socket 修复镜像
2026-09-20 21:15:15 +00:00
panxiao81 4e3ef3c33f 更新动态 Runner SPIFFE socket 修复镜像
yaml / yaml (pull_request) Failing after 9m48s
2026-09-20 21:14:08 +00:00
panxiao81 8eac82c272 合并无重叠 scheduler 滚动策略
yaml / yaml (push) Failing after 1m1s
修复 Flux SSA strategy 冲突。
2026-09-20 21:05:02 +00:00
panxiao81 2b232c4cd2 使用无重叠滚动策略部署 scheduler
yaml / yaml (pull_request) Successful in 14s
2026-09-20 21:04:45 +00:00
panxiao81 b51dc3751b 合并 Recreate 策略修复
yaml / yaml (push) Successful in 15s
修复 Flux dry-run 阻塞,生产资源此前尚未切换。
2026-09-20 21:03:06 +00:00
panxiao81 616c87bafd 清理 Recreate 策略遗留字段
yaml / yaml (pull_request) Successful in 14s
2026-09-20 21:02:49 +00:00
panxiao81 0663642583 合并 Go 动态 Runner Pod canary
yaml / yaml (push) Successful in 20s
已完成 yamllint、kustomize 与生产 API server-side dry-run;按单副本 single-flight canary 部署。
2026-09-20 21:01:36 +00:00
panxiao81 68191e0ff4 部署 Go 动态 Runner Pod canary
yaml / yaml (pull_request) Successful in 18s
2026-09-20 21:00:28 +00:00
panxiao81 c368373e74 Merge pull request '纳管并验收 Nexus OCI 仓库' (#109) from feat/nexus-oci-poc into main
terraform / validate (push) Successful in 54s
2026-09-20 20:47:03 +00:00
panxiao81 7bdd152db9 纳管并验收 Nexus OCI 仓库
terraform / validate (pull_request) Successful in 1m21s
2026-09-20 20:46:26 +00:00
panxiao81 ab985c7979 Merge pull request '完成 Nexus Ansible 与 Go 缓存验收' (#108) from feat/nexus-poc-verification into main
terraform / validate (push) Successful in 1m48s
2026-09-20 20:32:45 +00:00
panxiao81 cb137f2044 完成 Nexus Ansible 与 Go 缓存验收
terraform / validate (pull_request) Successful in 49s
2026-09-20 20:28:23 +00:00
panxiao81 83e4ac2f01 Merge pull request '新增 Nexus 统一制品仓库 POC' (#103) from feat/nexus-poc into main
ansible / collection-test (push) Successful in 4m14s
ansible / lint (push) Successful in 5m15s
yaml / yaml (push) Successful in 35s
terraform / validate (push) Successful in 1m8s
Reviewed-on: #103
2026-09-20 19:49:49 +00:00
panxiao81 1df682bc6d Merge pull request '限制 OpenSandbox 仅调度 VM' (#107) from fix/enable-native-pod-worker into main
yaml / yaml (push) Successful in 20s
2026-09-20 17:19:36 +00:00
panxiao81 eb52197eee 限制 OpenSandbox 仅调度 VM
yaml / yaml (pull_request) Successful in 19s
2026-09-20 17:19:12 +00:00
panxiao81 9b9f44333e Merge pull request '在 homelab 启用原生 Pod Worker' (#106) from fix/enable-native-pod-worker into main
yaml / yaml (push) Successful in 24s
2026-09-20 17:16:28 +00:00
panxiao81 8896d58b89 在 homelab 启用原生 Pod Worker
yaml / yaml (pull_request) Successful in 24s
2026-09-20 17:16:26 +00:00
panxiao81 0fd7f87d7e Merge pull request '停止 OpenSandbox 消费 Pod 任务' (#105) from fix/separate-pod-opensandbox into main
yaml / yaml (push) Failing after 0s
2026-09-20 17:13:43 +00:00
panxiao81 89bd7864e6 停止 OpenSandbox 消费 Pod 任务
yaml / yaml (pull_request) Failing after 0s
2026-09-20 17:13:21 +00:00
panxiao81 93f803a0d9 合并 Runner completed ACK 镜像修复
yaml / yaml (push) Failing after 0s
2026-09-18 19:46:33 +00:00
panxiao81 f7d9579221 部署 Runner completed ACK 修复
yaml / yaml (pull_request) Failing after 0s
2026-09-18 19:46:30 +00:00
panxiao81 9fe910a4c9 合并 OpenSandbox Runner 部署修复
ansible / lint (push) Failing after 0s
ansible / collection-test (push) Failing after 0s
yaml / yaml (push) Failing after 0s
2026-09-18 19:34:25 +00:00
panxiao81 28c20fd3e5 恢复持久事件驱动 Runner 调度
ansible / lint (pull_request) Failing after 0s
ansible / collection-test (pull_request) Failing after 0s
yaml / yaml (pull_request) Failing after 0s
2026-09-18 19:27:25 +00:00
panxiao81 110dbcc2c3 新增 Nexus 统一制品仓库 POC
ansible / lint (pull_request) Failing after 0s
ansible / collection-test (pull_request) Failing after 0s
yaml / yaml (pull_request) Failing after 0s
terraform / validate (pull_request) Failing after 0s
2026-09-18 19:11:30 +00:00
panxiao81 28f1de35cb 启用 sandbox 静态身份登记
ansible / lint (pull_request) Failing after 2s
ansible / collection-test (pull_request) Failing after 1s
yaml / yaml (pull_request) Failing after 1s
2026-09-18 19:06:13 +00:00
panxiao81 ad4ed1f0d6 重新触发 OpenSandbox 集成验证
ansible / lint (pull_request) Failing after 0s
ansible / collection-test (pull_request) Failing after 0s
yaml / yaml (pull_request) Failing after 2s
2026-09-18 19:02:44 +00:00
panxiao81 611a5a3d3a 修复 Pool task-executor 数据目录与 guest 身份镜像
ansible / lint (pull_request) Failing after 0s
ansible / collection-test (pull_request) Failing after 0s
yaml / yaml (pull_request) Failing after 0s
2026-09-18 18:59:02 +00:00
panxiao81 7582990655 修复 OpenSandbox 冷启动超时与旧 worker 清理
ansible / lint (pull_request) Failing after 0s
ansible / collection-test (pull_request) Failing after 0s
yaml / yaml (pull_request) Failing after 0s
2026-09-18 18:53:26 +00:00
panxiao81 ec3ce94e12 Merge pull request '接入 OpenSandbox 动态 Runner' (#101) from feat/opensandbox-runner-integration into main
yaml / yaml (push) Successful in 23s
ansible / collection-test (push) Successful in 1m35s
ansible / lint (push) Successful in 2m55s
2026-09-18 18:45:24 +00:00
panxiao81 6eb4875db6 接入 OpenSandbox 动态 Runner
ansible / collection-test (pull_request) Successful in 2m58s
ansible / lint (pull_request) Successful in 4m33s
yaml / yaml (pull_request) Successful in 29s
2026-09-18 18:33:00 +00:00
panxiao81 518dba6036 Merge pull request 确保 OpenSandbox values 立即生效
yaml / yaml (push) Successful in 17s
ansible / collection-test (push) Successful in 1m15s
ansible / lint (push) Successful in 2m7s
2026-09-18 17:52:51 +00:00
panxiao81 4c823f8181 确保 OpenSandbox values 立即生效
yaml / yaml (pull_request) Successful in 27s
ansible / collection-test (pull_request) Successful in 1m16s
ansible / lint (pull_request) Successful in 2m33s
2026-09-18 17:49:26 +00:00
panxiao81 8290082fb4 Merge pull request 为 sandbox 接入 OpenBao External Secrets
yaml / yaml (push) Successful in 15s
terraform / validate (push) Successful in 42s
2026-09-18 17:41:51 +00:00
panxiao81 159a75b710 记录 sandbox ESO 当前状态
yaml / yaml (pull_request) Successful in 16s
terraform / validate (pull_request) Successful in 39s
2026-09-18 17:40:24 +00:00
panxiao81 e0e629794b 为 sandbox 接入 OpenBao External Secrets
yaml / yaml (pull_request) Successful in 17s
terraform / validate (pull_request) Successful in 49s
2026-09-18 17:34:53 +00:00
panxiao81 90ba945d85 Merge pull request 部署 OpenSandbox 控制面
yaml / yaml (push) Successful in 18s
2026-09-18 16:38:13 +00:00
panxiao81 d526fd75d3 部署 OpenSandbox 控制面
yaml / yaml (pull_request) Successful in 17s
2026-09-18 16:36:21 +00:00
panxiao81 18cb2858b9 Merge pull request '延长 sandbox Flux 根同步超时' (#93) from fix/sandbox-flux-root-timeout into main
ansible / collection-test (push) Successful in 1m13s
ansible / lint (push) Successful in 2m23s
Reviewed-on: #93
2026-09-18 16:18:46 +00:00
panxiao81 99d1ec1d6f Merge pull request '记录 Kata guest 内 SPIFFE 身份方案' (#95) from docs/kata-inner-spire into main
Reviewed-on: #95
2026-09-18 16:18:29 +00:00
panxiao81 c6ec310b0b Merge pull request 切换 SPIRE 到内部 chart fork
yaml / yaml (push) Successful in 19s
合并内部 chart fork 与 sandbox-kata Pod UID PSAT profile。
2026-09-18 00:24:54 +00:00
panxiao81 3b77af8da1 feat: 切换 SPIRE 到内部 chart fork
yaml / yaml (pull_request) Successful in 14s
2026-09-18 00:24:21 +00:00
panxiao81 583dab526a docs: 明确 SPIRE chart 字段缺口 2026-09-17 18:54:00 +00:00
panxiao81 3dbd4c5f31 docs: 记录 Kata guest SPIFFE 身份方案 2026-09-17 18:47:01 +00:00
panxiao81 e67bce5121 fix: 延长 sandbox Flux 根同步超时
ansible / collection-test (pull_request) Successful in 1m8s
ansible / lint (pull_request) Successful in 2m56s
2026-09-17 18:20:33 +00:00
panxiao81 8af511ecc8 Merge pull request '部署 sandbox Kata Containers' (#92) from feat/sandbox-kata into main
yaml / yaml (push) Successful in 22s
Reviewed-on: #92
2026-09-17 18:06:18 +00:00
panxiao81 94684d0722 feat: 部署 sandbox Kata Containers
yaml / yaml (pull_request) Successful in 17s
2026-09-17 18:02:46 +00:00
panxiao81 9515cde49b Merge pull request '补齐 sandbox PSAT reviewer 权限' (#90) from fix/sandbox-spire-psat-rbac into main
yaml / yaml (push) Successful in 16s
Reviewed-on: #90
2026-09-17 17:47:27 +00:00
panxiao81 accf2d8210 Merge pull request '修复 SPIRE Agent 连接 Server 的端口' (#91) from fix/spire-agent-server-port into main
yaml / yaml (push) Successful in 49s
Reviewed-on: #91
2026-09-17 17:44:50 +00:00
panxiao81 d1ccc99125 修复 SPIRE Agent Server 端口
yaml / yaml (pull_request) Failing after 1m1s
2026-09-17 17:43:39 +00:00
panxiao81 b619f6f681 fix: 补齐 sandbox PSAT reviewer 权限
yaml / yaml (pull_request) Successful in 13s
2026-09-17 17:30:54 +00:00
panxiao81 1806c678a4 Merge pull request '部署 sandbox SPIRE Agent 与 CSI' (#89) from feat/sandbox-spire-agents into main
yaml / yaml (push) Successful in 21s
ansible / collection-test (push) Successful in 1m16s
ansible / lint (push) Successful in 2m17s
Reviewed-on: #89
2026-09-17 17:27:14 +00:00
panxiao81 aeb8c49d0a feat: 部署 sandbox SPIRE Agent 与 CSI
yaml / yaml (pull_request) Successful in 20s
ansible / collection-test (pull_request) Successful in 1m9s
ansible / lint (pull_request) Successful in 2m15s
2026-09-17 17:23:25 +00:00
panxiao81 7b1a98280c Merge pull request '引导 sandbox SPIRE external controller' (#88) from feat/sandbox-spire-controller-bootstrap into main
yaml / yaml (push) Successful in 17s
ansible / collection-test (push) Successful in 1m12s
ansible / lint (push) Successful in 3m3s
Reviewed-on: #88
2026-09-17 17:14:50 +00:00
panxiao81 ea15841d4d feat: 引导 sandbox SPIRE external controller
yaml / yaml (pull_request) Successful in 17s
ansible / collection-test (pull_request) Successful in 1m5s
ansible / lint (pull_request) Successful in 4m4s
2026-09-17 17:00:44 +00:00
panxiao81 4b9aa9e164 Merge pull request '引导 sandbox 跨集群 SPIRE 认证' (#87) from feat/sandbox-spire-bootstrap into main
yaml / yaml (push) Successful in 20s
ansible / collection-test (push) Successful in 1m15s
ansible / lint (push) Successful in 2m12s
Reviewed-on: #87
2026-09-17 16:55:59 +00:00
panxiao81 b7b92b3465 feat: 引导 sandbox 跨集群 SPIRE 认证
yaml / yaml (pull_request) Successful in 19s
ansible / lint (pull_request) Successful in 2m13s
ansible / collection-test (pull_request) Successful in 1m11s
2026-09-17 16:50:02 +00:00
panxiao81 dc2b43f693 Merge pull request '接入 sandbox Kubernetes 监控' (#86) from feat/sandbox-monitoring into main
yaml / yaml (push) Successful in 17s
ansible / collection-test (push) Successful in 1m12s
ansible / lint (push) Successful in 2m11s
Reviewed-on: #86
2026-09-17 16:21:27 +00:00
panxiao81 60836c3360 feat: 接入 sandbox Kubernetes 监控
yaml / yaml (pull_request) Successful in 25s
ansible / collection-test (pull_request) Successful in 1m10s
ansible / lint (pull_request) Successful in 4m33s
2026-09-17 16:02:48 +00:00
panxiao81 2030751e6a Merge pull request '声明 sandbox 双节点 K3s 集群 IaC' (#85) from feat/sandbox-cluster-iac into main
yaml / yaml (push) Successful in 12s
ansible / collection-test (push) Successful in 1m6s
ansible / lint (push) Successful in 2m44s
Reviewed-on: #85
2026-09-17 15:42:41 +00:00
panxiao81 f810b1c674 docs: 明确 sandbox 监控边界
yaml / yaml (pull_request) Successful in 17s
ansible / collection-test (pull_request) Successful in 1m8s
ansible / lint (pull_request) Successful in 2m43s
2026-09-17 15:11:29 +00:00
panxiao81 6585d8c46a feat: 声明 sandbox 双节点 K3s 集群
yaml / yaml (pull_request) Successful in 18s
ansible / collection-test (pull_request) Successful in 1m16s
ansible / lint (pull_request) Successful in 2m18s
2026-09-17 14:57:48 +00:00
panxiao81 cadfee0aea Merge pull request '整理旧工作区并恢复未纳管 IaC' (#84) from chore/recover-old-workspace into main
yaml / yaml (push) Successful in 15s
terraform / validate (push) Successful in 45s
ansible / collection-test (push) Successful in 1m27s
ansible / lint (push) Successful in 2m31s
Reviewed-on: #84
2026-09-17 13:31:49 +00:00
panxiao81 412fa93018 修正网络配置 YAML 缩进
yaml / yaml (pull_request) Successful in 28s
terraform / validate (pull_request) Successful in 55s
ansible / collection-test (pull_request) Successful in 1m21s
ansible / lint (pull_request) Successful in 4m11s
2026-09-17 13:29:14 +00:00
panxiao81 2e05b1a96a 声明式管理 Proxmox API 对象与 Samba DNS 上游
yaml / yaml (pull_request) Failing after 25s
terraform / validate (pull_request) Successful in 51s
ansible / collection-test (pull_request) Successful in 1m27s
ansible / lint (pull_request) Successful in 3m45s
2026-09-17 13:22:50 +00:00
panxiao81 71eea7d8fc 恢复 OCI Terraform 与站点网络 IaC 2026-09-17 13:22:49 +00:00
panxiao81 a08c8a7303 统一 DN42 DNS 与宿主 Docker 地址池 2026-09-17 13:22:49 +00:00
panxiao81 76f94f4f31 Merge pull request '用统一 inventory 生成 DNS 后端配置' (#81) from feat/dns-config-generator into main
yaml / yaml (push) Successful in 26s
terraform / validate (push) Successful in 36s
ansible / collection-test (push) Successful in 1m35s
ansible / lint (push) Successful in 2m47s
Reviewed-on: #81
2026-09-16 18:43:02 +00:00
panxiao81 a2102e6708 Merge pull request '恢复本机对动态 Runner 镜像的发布权限' (#82) from fix/zot-local-publisher-exact-policy into main
yaml / yaml (push) Successful in 13s
Reviewed-on: #82
2026-09-16 18:39:51 +00:00
panxiao81 bb2b5119ab 恢复本机对动态 Runner 镜像的发布权限
yaml / yaml (pull_request) Successful in 19s
2026-09-16 18:38:30 +00:00
panxiao81 65a72ce7fa fix(ci): 为 Ansible lint 提供临时 Vault 占位文件
yaml / yaml (pull_request) Successful in 28s
terraform / validate (pull_request) Successful in 36s
ansible / collection-test (pull_request) Successful in 1m24s
ansible / lint (pull_request) Successful in 3m11s
2026-09-16 18:37:59 +00:00
panxiao81 5241eceb0c feat(dns): 用模板生成各后端配置
yaml / yaml (pull_request) Successful in 23s
terraform / validate (pull_request) Successful in 36s
ansible / collection-test (pull_request) Successful in 1m33s
ansible / lint (pull_request) Failing after 2m44s
2026-09-16 18:27:09 +00:00
panxiao81 5ba4411646 Merge pull request '授权动态 Runner CI 发布镜像' (#80) from feat/dynamic-runner-publisher-policy into main
yaml / yaml (push) Successful in 13s
Reviewed-on: #80
2026-09-16 18:12:01 +00:00
panxiao81 2fcb41adde 授权动态 Runner CI 发布镜像
yaml / yaml (pull_request) Successful in 10s
2026-09-16 18:11:11 +00:00
panxiao81 42d33e31da Merge pull request 'ci: 统一使用 pod runner' (#74) from ci/self-hosted-pod into main
yaml / yaml (push) Successful in 20s
terraform / validate (push) Successful in 27s
ansible / collection-test (push) Successful in 1m24s
ansible / lint (push) Failing after 2m1s
Reviewed-on: #74
2026-09-16 17:58:26 +00:00
panxiao81 b617b4bc23 Merge pull request '修复无 Vault 凭据时的 Ansible lint' (#79) from fix/ansible-lint-without-vault into main
ansible / collection-test (push) Failing after 1s
yaml / yaml (push) Successful in 10s
ansible / lint (push) Failing after 1m1s
Reviewed-on: #79
2026-09-16 17:58:00 +00:00
panxiao81 b6b4efe48f 修复无 Vault 凭据时的 Ansible lint
yaml / yaml (pull_request) Successful in 22s
ansible / collection-test (pull_request) Successful in 1m12s
ansible / lint (pull_request) Successful in 2m17s
2026-09-16 17:49:35 +00:00
panxiao81 d6518153c8 Merge pull request '使用 Zot 中的 Ubuntu Pod Runner 镜像' (#78) from fix/pod-runner-ubuntu-image into main
yaml / yaml (push) Successful in 13s
Reviewed-on: #78
2026-09-16 17:32:44 +00:00
panxiao81 e42bb12f8e ci: 完整验证固定 digest Runner
ansible / collection-test (pull_request) Successful in 1m8s
yaml / yaml (pull_request) Successful in 13s
terraform / validate (pull_request) Successful in 28s
ansible / lint (pull_request) Failing after 4m55s
2026-09-16 17:32:30 +00:00
panxiao81 f97505df2d ci: 在固定 digest 上重试集成测试
ansible / lint (pull_request) Failing after 6s
ansible / collection-test (pull_request) Failing after 10s
yaml / yaml (pull_request) Successful in 18s
terraform / validate (pull_request) Successful in 51s
2026-09-16 17:30:29 +00:00
panxiao81 8b4acc40f6 ci: 重试 Pod Runner 集成测试
yaml / yaml (pull_request) Failing after 9s
ansible / collection-test (pull_request) Failing after 9s
ansible / lint (pull_request) Failing after 9s
terraform / validate (pull_request) Failing after 11s
2026-09-16 17:29:45 +00:00
panxiao81 61df128e25 ci: 重新验证 Ubuntu Pod Runner
yaml / yaml (pull_request) Successful in 18s
terraform / validate (pull_request) Successful in 29s
ansible / collection-test (pull_request) Successful in 1m14s
ansible / lint (pull_request) Failing after 1m58s
2026-09-16 17:26:00 +00:00
panxiao81 5b81719675 使用 Zot 中的 Ubuntu Pod Runner 镜像
yaml / yaml (pull_request) Successful in 10s
2026-09-16 17:23:05 +00:00
panxiao81 62c0ce1fff Merge pull request '收窄本机 SPIFFE 身份的 KV 权限范围' (#77) from fix/spire-local-kv-scope into main
terraform / validate (push) Successful in 23s
Reviewed-on: #77
2026-09-16 17:17:11 +00:00
panxiao81 f3d8c38d28 收窄本机 SPIFFE 身份的 KV 权限范围
terraform / validate (pull_request) Successful in 28s
2026-09-16 17:15:07 +00:00
panxiao81 61f0f864aa Merge pull request '授权本机 SPIFFE 身份访问开发服务' (#76) from feat/spire-host-development into main
yaml / yaml (push) Successful in 13s
terraform / validate (push) Successful in 23s
Reviewed-on: #76
2026-09-16 17:07:50 +00:00
panxiao81 64b1acd6c3 授权本机 SPIFFE 身份访问开发服务
yaml / yaml (pull_request) Successful in 12s
terraform / validate (pull_request) Successful in 32s
2026-09-16 17:06:27 +00:00
panxiao81 bbf228de41 Merge pull request '为宿主机开发启用 SPIRE workload identity' (#75) from feat/spire-host-development into main
yaml / yaml (push) Successful in 9s
Reviewed-on: #75
2026-09-16 16:59:17 +00:00
panxiao81 7703b5d21e 为宿主机开发启用 SPIRE workload identity
yaml / yaml (pull_request) Failing after 2s
2026-09-16 16:57:33 +00:00
panxiao81 3871d34232 ci: 统一使用 pod runner
ansible / lint (pull_request) Failing after 4s
ansible / collection-test (pull_request) Failing after 4s
terraform / validate (pull_request) Failing after 4s
yaml / yaml (pull_request) Failing after 3s
2026-09-16 16:33:16 +00:00
panxiao81 9c64d31d3d Merge pull request '清理已退役的 VictoriaMetrics Compose 配置' (#73) from chore/remove-legacy-monitoring into main
yaml / yaml (push) Successful in 11s
Reviewed-on: #73
2026-09-16 16:05:32 +00:00
panxiao81 d9486b4c8c 清理已停用的 VictoriaMetrics Compose 配置并记录旧数据卷删除
yaml / yaml (pull_request) Successful in 15s
2026-09-16 15:53:00 +00:00
panxiao81 f36e8a1733 Merge pull request '允许 Gitea 投递动态 Runner 集群内 webhook' (#72) from fix/gitea-webhook-allowlist into main
yaml / yaml (push) Successful in 10s
Reviewed-on: #72
2026-09-16 15:37:35 +00:00
panxiao81 926a90508a 允许 Gitea 投递动态 Runner 集群内 webhook
yaml / yaml (pull_request) Successful in 11s
2026-09-16 15:35:38 +00:00
panxiao81 cb2a66ded5 合并 PR #70:更新 zot 运行状态与 GitOps 接管文档 2026-09-16 15:25:45 +00:00
panxiao81 f767cd9a5e Merge pull request:隔离动态 Runner 凭据挂载路径
yaml / yaml (push) Successful in 11s
2026-09-16 15:25:03 +00:00
panxiao81 13279256a5 避免动态 Runner 凭据覆盖服务账号令牌路径
yaml / yaml (pull_request) Successful in 11s
2026-09-16 15:24:16 +00:00
panxiao81 dc325f3a45 Merge pull request:固定 dynamic runner 初始化容器的数值用户
yaml / yaml (push) Successful in 10s
2026-09-16 15:14:08 +00:00
panxiao81 87bb556087 固定 curl 初始化容器的数值用户
yaml / yaml (pull_request) Successful in 11s
2026-09-16 15:13:09 +00:00
panxiao81 11f038794f docs(zot): 更新双入口运行状态与 GitOps 接管结果 2026-09-16 15:08:28 +00:00
panxiao81 654087ef46 Merge pull request:修正动态 Runner 的 Bao 凭据路径
yaml / yaml (push) Successful in 12s
2026-09-16 15:05:34 +00:00
panxiao81 14fb5a323a 修正动态 Runner 的 Bao 凭据路径
yaml / yaml (pull_request) Successful in 12s
2026-09-16 15:04:31 +00:00
panxiao81 eb68e9f0c5 Merge pull request '部署 Pod/VM 动态 Runner 控制器' (#67) from feat/dynamic-runner-controller into main
yaml / yaml (push) Successful in 11s
Reviewed-on: #67
2026-09-16 15:00:03 +00:00
panxiao81 c4f425e9ea 按执行职责拆分动态 Runner 组件
yaml / yaml (pull_request) Successful in 10s
2026-09-16 14:53:30 +00:00
panxiao81 d15733caac Merge pull request '修复 zot:同时提供匿名拉取与 SPIRE 鉴权推送' (#65) from fix/zot-anonymous-pull into main
ansible / collection-test (push) Successful in 1m9s
yaml / yaml (push) Successful in 14s
ansible / lint (push) Successful in 15m46s
Reviewed-on: #65
2026-09-16 14:50:13 +00:00
panxiao81 246b5023e7 部署动态 Pod Runner 控制器
yaml / yaml (pull_request) Successful in 12s
2026-09-16 14:34:47 +00:00
panxiao81 21b7b48cdf Merge pull request:为 Gitea runner 接入 SPIRE Workload API
yaml / yaml (push) Successful in 11s
2026-09-16 13:26:02 +00:00
panxiao81 94721c279b 为 Gitea runner 接入 SPIRE Workload API
yaml / yaml (pull_request) Successful in 12s
2026-09-16 13:25:01 +00:00
panxiao81 31d89817b2 fix(zot): 分离匿名拉取与 SPIRE 鉴权推送入口
yaml / yaml (pull_request) Successful in 22s
ansible / collection-test (pull_request) Successful in 1m10s
ansible / lint (pull_request) Successful in 17m42s
2026-09-16 13:06:35 +00:00
panxiao81 22bffe068c Merge pull request:使用字节数配置 NATS Account 配额
yaml / yaml (push) Successful in 10s
2026-09-16 12:49:59 +00:00
panxiao81 6078a06b99 使用字节数配置 NATS Account 配额
yaml / yaml (pull_request) Successful in 11s
2026-09-16 12:49:20 +00:00
panxiao81 47042d4df4 完善内存监控并迁移 Grafana 内网入口
yaml / yaml (push) Successful in 20s
ansible / collection-test (push) Successful in 1m10s
ansible / lint (push) Successful in 13m9s
Co-authored-by: panxiao81 <[email protected]>
2026-09-16 12:47:52 +00:00
panxiao81 fd6bd62a4f Merge pull request:修复 NATS 容量单位语法
yaml / yaml (push) Successful in 14s
2026-09-16 12:41:08 +00:00
panxiao81 3eb6f33dea 修复 NATS 容量单位语法
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:40:19 +00:00
panxiao81 818192f453 Merge pull request:修复 NATS Account 配额格式
yaml / yaml (push) Successful in 11s
2026-09-16 12:34:12 +00:00
panxiao81 387953c80a 修复 NATS Account 配额格式
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:33:16 +00:00
panxiao81 298db6a745 Merge pull request:修复 NATS Bao ACME 证书密钥类型
yaml / yaml (push) Successful in 10s
2026-09-16 12:26:19 +00:00
panxiao81 f36a1cbf11 修复 NATS Bao ACME 证书密钥类型
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:25:28 +00:00
panxiao81 b953db199e Merge pull request:由 monitoring 提供 Prometheus Operator CRD
yaml / yaml (push) Successful in 11s
2026-09-16 12:23:40 +00:00
panxiao81 b7b7975c9f 由 monitoring 提供 Prometheus Operator CRD
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:22:00 +00:00
panxiao81 9604ff1004 修复 NATS 监控 CRD 兼容性
yaml / yaml (pull_request) Successful in 10s
2026-09-16 12:18:34 +00:00
panxiao81 819b521039 Merge pull request:修复 cert-manager Gateway API ACME solver
yaml / yaml (push) Successful in 11s
2026-09-16 12:17:07 +00:00
panxiao81 40703782ea 修复 cert-manager Gateway API ACME solver
yaml / yaml (pull_request) Successful in 11s
2026-09-16 12:16:23 +00:00
panxiao81 aae19850cf Merge pull request:部署 NATS JetStream 消息基础设施
yaml / yaml (push) Successful in 16s
ansible / collection-test (push) Successful in 1m9s
ansible / lint (push) Successful in 18m16s
首期使用静态 Account 凭据;SPIRE Auth Callout 后续见 #56。YAML 与 collection tests 已通过,Ansible lint 卡在无关的 Galaxy 依赖下载。
2026-09-16 12:13:06 +00:00
panxiao81 c11e1d5e6f 部署 NATS JetStream 消息基础设施
yaml / yaml (pull_request) Successful in 51s
ansible / collection-test (pull_request) Successful in 2m0s
ansible / lint (pull_request) Successful in 19m43s
2026-09-16 12:00:24 +00:00
264 changed files with 8570 additions and 20920 deletions
View File
+12 -2
View File
@@ -21,7 +21,7 @@ env:
jobs: jobs:
lint: lint:
runs-on: self-hosted runs-on: [self-hosted, pod]
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
@@ -51,6 +51,16 @@ jobs:
run: | run: |
export PATH="$HOME/.local/bin:$PATH" export PATH="$HOME/.local/bin:$PATH"
export ANSIBLE_COLLECTIONS_PATH="$PWD/infrastructure/samba-ad/ansible/collections:/root/.ansible/collections" export ANSIBLE_COLLECTIONS_PATH="$PWD/infrastructure/samba-ad/ansible/collections:/root/.ansible/collections"
# 静态检查不应依赖生产 vault 凭据。一次性 checkout 可以去掉加密变量文件;
# syntax-check 只验证结构,不需要解析变量的运行时值。
rm -f \
infrastructure/openbao/ansible/group_vars/all/vault.yml \
infrastructure/samba-ad/ansible/group_vars/all/vault.yml
# ansible.cfg still declares vault_password_file. Even with encrypted
# vars removed, ansible-lint validates that the configured file exists
# before syntax-check starts. This throwaway value decrypts nothing.
export ANSIBLE_VAULT_PASSWORD_FILE="$RUNNER_TEMP/ansible-lint-vault-pass"
printf '%s\n' 'ci-placeholder-not-a-production-secret' > "$ANSIBLE_VAULT_PASSWORD_FILE"
rc=0 rc=0
for p in infrastructure/openbao infrastructure/samba-ad infrastructure/proxmox; do for p in infrastructure/openbao infrastructure/samba-ad infrastructure/proxmox; do
echo "::group::$p" echo "::group::$p"
@@ -60,7 +70,7 @@ jobs:
exit $rc exit $rc
collection-test: collection-test:
runs-on: self-hosted runs-on: [self-hosted, pod]
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
+10 -1
View File
@@ -13,18 +13,22 @@ on:
paths: paths:
- '**/*.yaml' - '**/*.yaml'
- '**/*.yml' - '**/*.yml'
- 'infrastructure/dns/**'
- 'infrastructure/cloudflared/terraform/dns.generated.tf'
- '.yamllint.yml' - '.yamllint.yml'
- '.gitea/workflows/lint.yml' - '.gitea/workflows/lint.yml'
pull_request: pull_request:
paths: paths:
- '**/*.yaml' - '**/*.yaml'
- '**/*.yml' - '**/*.yml'
- 'infrastructure/dns/**'
- 'infrastructure/cloudflared/terraform/dns.generated.tf'
- '.yamllint.yml' - '.yamllint.yml'
- '.gitea/workflows/lint.yml' - '.gitea/workflows/lint.yml'
jobs: jobs:
yaml: yaml:
runs-on: self-hosted runs-on: [self-hosted, pod]
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
@@ -52,3 +56,8 @@ jobs:
export PATH="$HOME/.local/bin:$PATH" export PATH="$HOME/.local/bin:$PATH"
files=$(git ls-files '*.yaml' '*.yml' | grep -vE '^apps/netboot/') files=$(git ls-files '*.yaml' '*.yml' | grep -vE '^apps/netboot/')
yamllint -c .yamllint.yml --no-warnings -f parsable $files yamllint -c .yamllint.yml --no-warnings -f parsable $files
- name: Verify generated DNS configuration
run: |
export PATH="$HOME/.local/bin:$PATH"
uv run infrastructure/dns/generate.py --check
+1 -1
View File
@@ -16,7 +16,7 @@ on:
jobs: jobs:
validate: validate:
runs-on: self-hosted runs-on: [self-hosted, pod]
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
- uses: hashicorp/setup-terraform@v3 - uses: hashicorp/setup-terraform@v3
+2
View File
@@ -49,6 +49,7 @@ authelia/secret.yaml
**/secret.yaml **/secret.yaml
**/credentials.yml **/credentials.yml
**/terraform.tfvars **/terraform.tfvars
**/credentials.auto.tfvars
**/tailscale/helm.sh **/tailscale/helm.sh
**/cloudflared/backup/ **/cloudflared/backup/
**/cloudflared/secret.yaml **/cloudflared/secret.yaml
@@ -118,3 +119,4 @@ apps/netboot/config/log/
# Blocky's per-day query logs. Bind-mounted into the container, one file per # Blocky's per-day query logs. Bind-mounted into the container, one file per
# day, and every DNS query the LAN makes ends up in them. # day, and every DNS query the LAN makes ends up in them.
apps/blocky/logs/ apps/blocky/logs/
.venv/
+1 -1
View File
@@ -229,7 +229,7 @@ configMap:
require_pkce: false require_pkce: false
token_endpoint_auth_method: 'client_secret_basic' token_endpoint_auth_method: 'client_secret_basic'
redirect_uris: redirect_uris:
- 'https://grafana.tail7e769.ts.net/login/generic_oauth' - 'https://grafana.ad.ddupan.top/login/generic_oauth'
scopes: scopes:
- 'openid' - 'openid'
- 'profile' - 'profile'
+8 -1
View File
@@ -1,4 +1,4 @@
# Blocky — LAN DNS. STAGED, NOT DEPLOYED. See README.md. # Blocky — 已部署的 LAN 主 DNS。见 README.md。
# #
# WHY compose on the laptop and NOT a k3s Deployment, given everything else here # WHY compose on the laptop and NOT a k3s Deployment, given everything else here
# is Kubernetes: # is Kubernetes:
@@ -52,3 +52,10 @@ services:
options: options:
max-size: "10m" max-size: "10m"
max-file: "3" max-file: "3"
# 避免与 DN42 的 172.20.0.0/14 重叠。
networks:
default:
ipam:
config:
- subnet: 172.28.0.0/24
+11 -4
View File
@@ -1,9 +1,7 @@
# Blocky — LAN resolver, ad-blocker and split-horizon DNS. # Blocky — LAN resolver, ad-blocker and split-horizon DNS.
# #
# DEPLOYED 2026-07-28 and verified, but NOT yet the LAN resolver — clients still # LAN 主 DNS 为 192.168.10.127,NEC IX 192.168.10.1 为备用。
# get the DC/router pair from DHCP. Making it the resolver needs a DHCP change on # DN42 条件转发经 VyOS,参见 README.md。
# the NEC IX; see README.md. Until then only clients that query 192.168.10.127
# explicitly are affected, so this is safely reversible.
ports: ports:
# These are the CONTAINER's listen addresses, so they must be unqualified — # These are the CONTAINER's listen addresses, so they must be unqualified —
@@ -35,6 +33,13 @@ conditional:
# Queries for the AD zone go straight to the DC, which is authoritative. This # Queries for the AD zone go straight to the DC, which is authoritative. This
# replaces the "DC first, router second" resolver ordering that clients use today. # replaces the "DC first, router second" resolver ordering that clients use today.
mapping: mapping:
# DN42 由 VyOS 使用注册地址转发,避免 LAN 私网源地址缺少回程。
dn42: 192.168.10.2
20.172.in-addr.arpa: 192.168.10.2
21.172.in-addr.arpa: 192.168.10.2
22.172.in-addr.arpa: 192.168.10.2
23.172.in-addr.arpa: 192.168.10.2
d.f.ip6.arpa: 192.168.10.2
ad.ddupan.top: 192.168.10.5 ad.ddupan.top: 192.168.10.5
# Reverse lookups for LAN hosts — the DC holds the reverse zone. # Reverse lookups for LAN hosts — the DC holds the reverse zone.
10.168.192.in-addr.arpa: 192.168.10.5 10.168.192.in-addr.arpa: 192.168.10.5
@@ -54,9 +59,11 @@ customDNS:
# laptop's only global IPv6 belongs to tun0, so a AAAA answer would send LAN # laptop's only global IPv6 belongs to tun0, so a AAAA answer would send LAN
# traffic into the VPN. See CLAUDE.md. # traffic into the VPN. See CLAUDE.md.
mapping: mapping:
# BEGIN GENERATED: homelab DNS (blocky)
git.ddupan.top: 192.168.10.127 git.ddupan.top: 192.168.10.127
auth.ddupan.top: 192.168.10.127 auth.ddupan.top: 192.168.10.127
obj.ddupan.top: 192.168.10.127 obj.ddupan.top: 192.168.10.127
# END GENERATED: homelab DNS (blocky)
blocking: blocking:
denylists: denylists:
+5
View File
@@ -48,6 +48,11 @@ gitea:
# github.com is reachable from this network (verified 2026-07-28) even when # github.com is reachable from this network (verified 2026-07-28) even when
# pypi.org/Fastly is not — see the flaky-WAN notes in the lint workflow. # pypi.org/Fastly is not — see the flaky-WAN notes in the lint workflow.
DEFAULT_ACTIONS_URL: github DEFAULT_ACTIONS_URL: github
webhook:
# Keep the default public-internet access for existing hooks while allowing
# only the dynamic Runner controller's exact in-cluster DNS name. Do not
# broaden this to the built-in `private` network group.
ALLOWED_HOST_LIST: external,dynamic-runner-controller.dynamic-runner.svc.cluster.local
mailer: mailer:
# Outbound mail via the in-cluster Postfix+OAuth relay (see ../smtp-relay/). # Outbound mail via the in-cluster Postfix+OAuth relay (see ../smtp-relay/).
# Plain SMTP on :25 — the relay does STARTTLS + OAuth to M365. From must be the # Plain SMTP on :25 — the relay does STARTTLS + OAuth to M365. From must be the
+12 -8
View File
@@ -52,11 +52,16 @@ prefix_roles:
# range. It makes the collision VISIBLE — the range shows 100% utilised and the # range. It makes the collision VISIBLE — the range shows 100% utilised and the
# address never appears as a suggestion — where plain YAML shows nothing at all. # address never appears as a suggestion — where plain YAML shows nothing at all.
ip_ranges: ip_ranges:
- start: 192.168.10.10/24 - start: 192.168.10.128/24
end: 192.168.10.250/24 end: 192.168.10.250/24
status: active status: active
mark_utilized: true mark_utilized: true
description: "NEC IX DHCP pool — do NOT statically allocate inside this." description: "NEC IX DHCP pool, updated 2026-09-14. Do NOT statically allocate inside this."
- start: 192.168.10.251/24
end: 192.168.10.254/24
status: reserved
mark_utilized: true
description: "用户确认预留,尚未分配;不可按扫描无响应视为空闲。"
vlan_group: vlan_group:
name: lab name: lab
@@ -144,8 +149,8 @@ devices:
role: hypervisor role: hypervisor
type: 10vgcto1ww type: 10vgcto1ww
serial: PC1AGX1Q serial: PC1AGX1Q
description: "Proxmox VE 9.2. LINSTOR satellite. The node that randomly froze." description: "Proxmox VE 9.2. LINSTOR satellite."
comments: "AMD Ryzen 5 PRO 2400GE w/ Vega, 8 threads, 7 GiB RAM. BIOS M1XKT45A. Raven Ridge idle bug fixed in BIOS: Power Supply Idle Control = Typical Current Idle." comments: "AMD Ryzen 5 PRO 2400GE w/ Vega, 8 threads, 7 GiB RAM. BIOS M1XKT45A."
interfaces: interfaces:
- { name: vmbr0, type: bridge, ip: 192.168.10.7/24, primary: true, mtu: 9000, dns_name: pve2.ad.ddupan.top } - { name: vmbr0, type: bridge, ip: 192.168.10.7/24, primary: true, mtu: 9000, dns_name: pve2.ad.ddupan.top }
@@ -154,7 +159,7 @@ devices:
type: 10vgcto1ww type: 10vgcto1ww
serial: PC1AGX1P serial: PC1AGX1P
description: "Proxmox VE 9.2. LINSTOR satellite." description: "Proxmox VE 9.2. LINSTOR satellite."
comments: "AMD Ryzen 5 PRO 2400GE w/ Vega, 8 threads, 7 GiB RAM. BIOS M1XKT55A. Same silicon as pve2, so susceptible to the same idle bug in principle." comments: "AMD Ryzen 5 PRO 2400GE w/ Vega, 8 threads, 7 GiB RAM. BIOS M1XKT55A."
interfaces: interfaces:
- { name: vmbr0, type: bridge, ip: 192.168.10.9/24, primary: true, mtu: 9000, dns_name: pve3.ad.ddupan.top } - { name: vmbr0, type: bridge, ip: 192.168.10.9/24, primary: true, mtu: 9000, dns_name: pve3.ad.ddupan.top }
@@ -181,9 +186,8 @@ devices:
# Wi-Fi. Runs as an AP/bridge, not a router — the NEC IX is the gateway, so this box's # Wi-Fi. Runs as an AP/bridge, not a router — the NEC IX is the gateway, so this box's
# routing, NAT and DHCP are not in play. Wireless clients land directly on the flat LAN. # routing, NAT and DHCP are not in play. Wireless clients land directly on the flat LAN.
# #
# ⚠ Its address .10 is the FIRST ADDRESS OF THE DHCP POOL above. Either it holds a lease # 2026-09-14: NEC IX 为此 MAC 固定分配 .10;动态池已迁到 .128–.250。
# (so the address can move) or it is a static that overlaps the pool. NetBox surfaces # 操作与回滚记录:infrastructure/samba-ad/router-dhcp-nec-ix.md。
# the overlap; the underlying config still needs a decision. See ../README.md.
# #
# Identified by MAC OUI d4:2c:46 = BUFFALO.INC plus the model string on its login page. # Identified by MAC OUI d4:2c:46 = BUFFALO.INC plus the model string on its login page.
- name: ap-buffalo - name: ap-buffalo
+157
View File
@@ -0,0 +1,157 @@
# Nexus Repository POC
本目录声明一个 Nexus Repository Community Edition POC,用来验证一次性 CI runner 通过
网络服务复用 Ansible Galaxy、Go Modules 与 OCI/BuildKit 缓存。Nexus 固定为 `3.96.1`,
镜像固定到官方 multi-arch index digest;LAN 入口为
`https://nexus.ad.ddupan.top`。
## POC 边界
- 单副本 Deployment,`Recreate` 更新,避免一个 RWO 卷被两个 Pod 同时挂载。
- `/nexus-data` 使用 `localpv-zfs-ceph` 上的 50 GiB RWO PVC。
- 资源预算为 250m/2 GiB request、2 CPU/4 GiB limit;JVM heap 上限 2 GiB。
- 使用容器默认的 embedded H2。它只用于 POC;正式接管 OCI 制品前必须迁移到外部
PostgreSQL,并验证备份恢复。
- 入口只在 LAN wildcard Gateway 上发布,不创建公网 DNS 或 Cloudflare route。
- 不套 Authelia forward-auth;它会破坏 Go、Ansible 与 OCI 非浏览器客户端。
- 现有 zot 保持不变。Nexus 完成 OCI、BuildKit cache 和恢复验收前不得迁移或删除 zot。
Terraform provider 创建 `ansible-public`、`go-public`、最小匿名权限与 OCI Bearer Token
Realm。Nexus 3.94 才加入的原生 OCI repository 已有 REST API,但当前锁定的 community
provider 尚未暴露 OCI resource;`terraform/reconcile-oci.sh` 因此根据 3.96.1 实例 Swagger
固定的 JSON schema,幂等调和 `oci-hosted`、`oci-proxy` 与 `oci-public`。不得绕过该入口在
UI 中创建无人管理的长期 repository。
## 部署
Flux 从 `clusters/homelab/apps/nexus.yaml` 协调本目录,并依赖 Envoy Gateway 与 OpenEBS。
合并前只渲染配置,不直接 apply:
```bash
kubectl kustomize apps/nexus
```
合并并由 Flux 部署后检查:
```bash
kubectl -n flux-system get kustomization nexus
kubectl -n nexus get pod,pvc,service,httproute
kubectl -n nexus logs deployment/nexus --tail=100
```
启动可能需要数分钟,startup probe 允许最多十分钟。不要因初次启动较慢反复删除 Pod;
先确认 PVC 已 Bound、Pod 没有 OOM,以及日志仍在推进。
DNS 期望状态已加入 `infrastructure/dns/records.yml`,需从 Samba AD Ansible root 以
`--check --diff` 核对后再按其 README 应用 DNS tag。没有 DNS 时可先用 port-forward
验证应用,但不能据此宣称 Gateway 路径已通过。
## 首次初始化与 Terraform
初始管理员密码生成在 PVC 的 `/nexus-data/admin.password`。只在交互式终端中读取并立即
完成首次密码轮换;不得把密码复制进 shell tracing、工单、Git 或命令参数。随后将
Terraform 管理账号的凭据存入 OpenBao,由 CI 通过 Terraform input variable 注入以下
环境变量:
```text
TF_VAR_nexus_url=https://nexus.ad.ddupan.top
TF_VAR_nexus_username=admin
TF_VAR_nexus_password=<OpenBao kv/infra/nexus 的 admin_password 字段>
```
`terraform/` 使用 `sonatype-nexus-community/sonatyperepo` 1.17.0,当前声明:
- `ansible-galaxy-proxy` → `https://galaxy.ansible.com`
- `ansible-public` group
- `go-proxy` → `https://proxy.golang.org`
- `go-public` group
provider credential 不写入 HCL 或 tfvars。正式 apply 前还必须为这个独立 Terraform root
配置远端 backend;本地 state 只允许用于可丢弃的 POC,不提交。验证命令:
```bash
terraform -chdir=apps/nexus/terraform init -backend=false
terraform -chdir=apps/nexus/terraform validate
```
先以 `--check` 查看 OCI repository 漂移,再明确 apply;脚本只从上述环境变量取得凭据,
用临时 `0600` netrc 调用 REST API,退出时删除:
```bash
apps/nexus/terraform/reconcile-oci.sh --check
apps/nexus/terraform/reconcile-oci.sh --apply
```
Terraform 同时把内置 `anonymous` 用户从默认的全仓库 `nx-anonymous` 角色收窄到
`ansible-public`、其返回制品 URL 使用的 `ansible-galaxy-proxy`、`go-public`,以及
`oci-public`/`oci-proxy` 的 `browse/read` 权限。`oci-hosted` 不向匿名用户开放。首次接管
已有实例时先执行
`terraform import sonatyperepo_user.anonymous anonymous,default`,再 apply;不要先启用默认的
全仓库匿名读取。
2026-09-20 的 POC 现场验收已确认:Flux 与 Pod Ready、PVC Bound、HTTPRoute 通过 HTTPS
返回 Nexus 状态 200,Samba DNS 已幂等收敛。全新客户端目录通过匿名入口下载
`community.general:11.2.0` 时冷缓存为 8.49 秒、热缓存为 1.89 秒,两次 tarball SHA-256
一致;`golang.org/x/[email protected]` 为 2.92 秒与 1.51 秒。
## 客户端验收
先验证冷缓存,再原样重复命令验证热缓存;记录 Nexus 请求、上游流量和耗时,不只观察
命令成功。Ansible 配置的 URL 必须以 `/` 结尾:
```ini
[galaxy]
server_list = nexus
[galaxy_server.nexus]
url = https://nexus.ad.ddupan.top/repository/ansible-public/
```
```bash
ansible-galaxy collection install -r collections/requirements.yml \
-p .ansible/collections
```
Go POC 使用:
```bash
GOPROXY=https://nexus.ad.ddupan.top/repository/go-public/ go mod download
```
私有 `git.ddupan.top/*` module 的 `GOPRIVATE`、凭据与是否允许 `direct` fallback 在实际
workflow 中单独决定;不要让私有 module path 意外发往公共 proxy。
OCI 使用 path-based routing:匿名公共拉取地址形如
`nexus.ad.ddupan.top/oci-public/library/alpine:3.22`,认证写入地址形如
`nexus.ad.ddupan.top/oci-hosted/<namespace>/<image>:<tag>`。2026-09-20 现场验收结果:
- `oci-public` 匿名代理拉取 Alpine 冷缓存 4.75 秒、热缓存 0.80 秒,digest 一致;
- `oci-hosted` 认证 push/pull 成功,匿名 pull 返回 401;
- amd64/arm64 OCI image index push 成功,两个平台 manifest 可见;
- Helm chart push/pull digest 与本地 tarball SHA-256 一致;
- Cosign 3.1.3 使用一次性密钥签名并验证成功,OCI 1.1 referrers API 返回一个
`application/vnd.dev.sigstore.bundle.v0.3+json` artifact;
- BuildKit `registry` cache 以 `mode=max` 导出成功,销毁首个 builder 后由新 builder 导入,
两个 `RUN` step 均明确命中 `CACHED`。
本机安装的测试客户端包括 `/usr/local/bin/cosign` 3.1.3;安装时核对官方 Linux amd64
binary SHA-256 `4629c757b7618056f8ddd7e2625ae9fdd94c0372a65049520bc7d9df9efc7f71`。
上述结果仍不代表备份恢复、外部 PostgreSQL 或正式 publisher service account 已完成;
这些项目通过前不得迁移或删除 zot。
## 数据与恢复
POC 的数据库、配置、blob、初始管理员状态都位于 `nexus-data` PVC。删除 Deployment
不会删除 PVC;删除 PVC 会永久删除整个 POC。当前没有独立备份,不能将它用于唯一副本的
正式制品。
恢复验证至少包括:停止写入、取得一致备份、在独立 PVC/实例恢复、登录、列出 repository、
拉取已缓存的 Ansible/Go 制品,并核对 OCI digest/referrers。正式化时再把数据库迁移至
外部 PostgreSQL,并分别定义数据库与 blob 的备份、恢复顺序和 RPO。
参考:
- [Nexus OCI repositories](https://help.sonatype.com/en/oci-repositories.html)
- [Nexus Ansible repositories](https://help.sonatype.com/en/ansible-repositories.html)
- [Nexus Go repositories](https://help.sonatype.com/en/go-repositories.html)
- [官方容器镜像](https://hub.docker.com/r/sonatype/nexus3)
+82
View File
@@ -0,0 +1,82 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: nexus
namespace: nexus
labels:
app.kubernetes.io/name: nexus
spec:
replicas: 1
strategy:
type: Recreate
selector:
matchLabels:
app.kubernetes.io/name: nexus
template:
metadata:
labels:
app.kubernetes.io/name: nexus
spec:
automountServiceAccountToken: false
securityContext:
fsGroup: 200
fsGroupChangePolicy: OnRootMismatch
runAsGroup: 200
runAsNonRoot: true
runAsUser: 200
seccompProfile:
type: RuntimeDefault
terminationGracePeriodSeconds: 120
containers:
- name: nexus
image: docker.io/sonatype/nexus3:3.96.1@sha256:56142f13432cf072e017aebb2025f201e42ae36ff40bb82618c702504c61f7dd
imagePullPolicy: IfNotPresent
env:
- name: INSTALL4J_ADD_VM_PARAMS
value: >-
-Xms1024m -Xmx2048m -XX:MaxDirectMemorySize=1024m
-Djava.util.prefs.userRoot=/nexus-data/javaprefs
ports:
- name: http
containerPort: 8081
protocol: TCP
resources:
requests:
cpu: 250m
memory: 2Gi
limits:
cpu: "2"
memory: 4Gi
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop:
- ALL
startupProbe:
httpGet:
path: /service/rest/v1/status
port: http
failureThreshold: 60
periodSeconds: 10
timeoutSeconds: 5
readinessProbe:
httpGet:
path: /service/rest/v1/status
port: http
failureThreshold: 6
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
httpGet:
path: /service/rest/v1/status
port: http
failureThreshold: 6
periodSeconds: 30
timeoutSeconds: 5
volumeMounts:
- name: data
mountPath: /nexus-data
volumes:
- name: data
persistentVolumeClaim:
claimName: nexus-data
+16
View File
@@ -0,0 +1,16 @@
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: nexus
namespace: nexus
spec:
parentRefs:
- name: eg
namespace: envoy-gateway-system
sectionName: https
hostnames:
- nexus.ad.ddupan.top
rules:
- backendRefs:
- name: nexus
port: 8081
+9
View File
@@ -0,0 +1,9 @@
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- namespace.yaml
- pvc.yaml
- deployment.yaml
- service.yaml
- httproute.yaml
- networkpolicy.yaml
+8
View File
@@ -0,0 +1,8 @@
apiVersion: v1
kind: Namespace
metadata:
name: nexus
labels:
pod-security.kubernetes.io/enforce: restricted
pod-security.kubernetes.io/audit: restricted
pod-security.kubernetes.io/warn: restricted
+23
View File
@@ -0,0 +1,23 @@
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: nexus-ingress
namespace: nexus
spec:
podSelector:
matchLabels:
app.kubernetes.io/name: nexus
policyTypes:
- Ingress
ingress:
- from:
- namespaceSelector:
matchLabels:
kubernetes.io/metadata.name: envoy-gateway-system
podSelector:
matchLabels:
gateway.envoyproxy.io/owning-gateway-name: eg
gateway.envoyproxy.io/owning-gateway-namespace: envoy-gateway-system
ports:
- protocol: TCP
port: 8081
+12
View File
@@ -0,0 +1,12 @@
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: nexus-data
namespace: nexus
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 50Gi
storageClassName: localpv-zfs-ceph
+14
View File
@@ -0,0 +1,14 @@
apiVersion: v1
kind: Service
metadata:
name: nexus
namespace: nexus
spec:
type: ClusterIP
selector:
app.kubernetes.io/name: nexus
ports:
- name: http
port: 8081
protocol: TCP
targetPort: http
+6
View File
@@ -0,0 +1,6 @@
.terraform/
*.tfstate
*.tfstate.*
*.tfplan
crash.log
crash.*.log
+24
View File
@@ -0,0 +1,24 @@
# This file is maintained automatically by "terraform init".
# Manual edits may be lost in future updates.
provider "registry.terraform.io/sonatype-nexus-community/sonatyperepo" {
version = "1.17.0"
constraints = "1.17.0"
hashes = [
"h1:uKhvhhhI7B+HBsh0zq/ybqKt+EnOGyI6rjcRCtj79ZA=",
"zh:0dde99e7b343fa01f8eefc378171fb8621bedb20f59157d6cc8e3d46c738105f",
"zh:3315929df254a3a6ac27c8c846c2006f7d2a91fadc014351bc4d617f948e5bf9",
"zh:36be5a455af3ce4e187de26753de63e78c1ee9a32dba0135c6cf96a6c1fff25f",
"zh:3f73f7ff57b8c339a7c7ac37653e2dc0b2dd9dcc3f3a538788e7e3ac838337b2",
"zh:40286ecca4c22ab7ae90618ac6d2743f5055199dac81cf5204a4a397c784d439",
"zh:4d24e5c0195fb3155b1967583ee64cfeda402d7cc7f3c73369438f6c69f4245b",
"zh:828a9d7aceaac36af7f9c07af43ec8d20a89148780645d170ffb1c68b2da792d",
"zh:a5ab04de3fe626ec57c832618c6f990abd6610f81e132621651e0b180b970cff",
"zh:a959fa6090a8c0f53739879184e7346423494aee598003df0d1ab4a22b2eee91",
"zh:bdda26c2f03f918bbe59e75abea44868fafda019c3a543725331195df126350b",
"zh:d8048e149ee97ba62971e6a79355d59887bc6d10fcf72cc2feff3d0a2582670c",
"zh:dd36f9988af4e1ca5b1ca7b7bb6f658df9a220dfcda7fec7392fedfe9064f652",
"zh:dda2688d46c7e539fe97e8fe9d3ec81fb364170e018d9c6a681364c8955d4e9d",
"zh:e6b519afe2dea1c0434f766eb6bc9ba78cc5b6ef2c311c2ca3c65cb24744f31f",
]
}
+17
View File
@@ -0,0 +1,17 @@
{
"name": "oci-hosted",
"online": true,
"storage": {
"blobStoreName": "default",
"strictContentTypeValidation": true,
"writePolicy": "ALLOW",
"latestPolicy": false
},
"oci": {
"v1Enabled": false,
"forceBasicAuth": false,
"pathEnabled": true
},
"component": { "proprietaryComponents": false },
"cosign": { "enforcement": "NONE" }
}
+27
View File
@@ -0,0 +1,27 @@
{
"name": "oci-proxy",
"online": true,
"storage": {
"blobStoreName": "default",
"strictContentTypeValidation": true
},
"oci": {
"v1Enabled": false,
"forceBasicAuth": false,
"pathEnabled": true
},
"ociProxy": {
"indexType": "HUB",
"cacheForeignLayers": false,
"foreignLayerUrlWhitelist": []
},
"proxy": {
"remoteUrl": "https://registry-1.docker.io",
"contentMaxAge": 1440,
"metadataMaxAge": 60,
"preserveEncodedCharacters": false
},
"negativeCache": { "enabled": true, "timeToLive": 60 },
"httpClient": { "blocked": false, "autoBlock": true },
"cosign": { "enforcement": "NONE" }
}
+15
View File
@@ -0,0 +1,15 @@
{
"name": "oci-public",
"online": true,
"storage": {
"blobStoreName": "default",
"strictContentTypeValidation": true
},
"group": { "memberNames": ["oci-proxy"] },
"oci": {
"v1Enabled": false,
"forceBasicAuth": false,
"pathEnabled": true
},
"cosign": { "enforcement": "NONE" }
}
+80
View File
@@ -0,0 +1,80 @@
#!/usr/bin/env bash
set -euo pipefail
mode="${1:---check}"
case "$mode" in
--check | --apply) ;;
*) echo "usage: $0 [--check|--apply]" >&2; exit 2 ;;
esac
: "${TF_VAR_nexus_url:?set TF_VAR_nexus_url}"
: "${TF_VAR_nexus_username:?set TF_VAR_nexus_username}"
: "${TF_VAR_nexus_password:?set TF_VAR_nexus_password}"
script_dir="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
auth_file="$(mktemp /tmp/nexus-oci-auth.XXXXXX)"
trap 'rm -f -- "$auth_file"' EXIT
chmod 0600 "$auth_file"
printf 'machine %s\nlogin %s\npassword %s\n' \
"${TF_VAR_nexus_url#*://}" "$TF_VAR_nexus_username" \
"$TF_VAR_nexus_password" >"$auth_file"
drift=0
for entry in \
"hosted:$script_dir/oci/oci-hosted.json" \
"proxy:$script_dir/oci/oci-proxy.json" \
"group:$script_dir/oci/oci-public.json"; do
repository_type="${entry%%:*}"
desired_file="${entry#*:}"
repository_name="$(jq -er '.name' "$desired_file")"
endpoint="$TF_VAR_nexus_url/service/rest/v1/repositories/oci/$repository_type"
current_file="$(mktemp /tmp/nexus-oci-current.XXXXXX)"
status="$(curl --silent --show-error --netrc-file "$auth_file" \
--output "$current_file" --write-out '%{http_code}' \
"$endpoint/$repository_name")"
if [[ "$status" == 404 ]]; then
drift=1
if [[ "$mode" == --apply ]]; then
curl --fail --silent --show-error --netrc-file "$auth_file" \
--header 'Content-Type: application/json' \
--data-binary "@$desired_file" "$endpoint"
echo "created $repository_name"
else
echo "missing $repository_name" >&2
fi
elif [[ "$status" == 200 ]]; then
if jq -e --slurpfile desired "$desired_file" '
def subset($actual; $wanted):
if ($wanted | type) == "object" then
all($wanted | keys[];
($actual[.] != null) and subset($actual[.]; $wanted[.]))
else
$actual == $wanted
end;
subset(.; $desired[0])
' "$current_file" >/dev/null; then
echo "in sync $repository_name"
else
drift=1
if [[ "$mode" == --apply ]]; then
curl --fail --silent --show-error --netrc-file "$auth_file" \
--request PUT --header 'Content-Type: application/json' \
--data-binary "@$desired_file" "$endpoint/$repository_name"
echo "updated $repository_name"
else
echo "drifted $repository_name" >&2
fi
fi
else
cat "$current_file" >&2
echo "unexpected HTTP $status for $repository_name" >&2
exit 1
fi
rm -f -- "$current_file"
done
if [[ "$mode" == --check && "$drift" -ne 0 ]]; then
exit 1
fi
+64
View File
@@ -0,0 +1,64 @@
locals {
proxy_http_client = {
auto_block = true
blocked = false
}
proxy_negative_cache = {
enabled = true
time_to_live = 60
}
repository_storage = {
blob_store_name = "default"
strict_content_type_validation = true
}
}
resource "sonatyperepo_repository_ansiblegalaxy_proxy" "galaxy" {
name = "ansible-galaxy-proxy"
online = true
http_client = local.proxy_http_client
negative_cache = local.proxy_negative_cache
proxy = {
remote_url = "https://galaxy.ansible.com"
content_max_age = 1440
metadata_max_age = 60
}
storage = local.repository_storage
}
resource "sonatyperepo_repository_ansiblegalaxy_group" "public" {
name = "ansible-public"
online = true
group = {
member_names = [sonatyperepo_repository_ansiblegalaxy_proxy.galaxy.name]
}
storage = local.repository_storage
}
resource "sonatyperepo_repository_go_proxy" "public" {
name = "go-proxy"
online = true
http_client = local.proxy_http_client
negative_cache = local.proxy_negative_cache
proxy = {
remote_url = "https://proxy.golang.org"
content_max_age = 1440
metadata_max_age = 60
}
storage = local.repository_storage
}
resource "sonatyperepo_repository_go_group" "public" {
name = "go-public"
online = true
group = {
member_names = [sonatyperepo_repository_go_proxy.public.name]
}
storage = local.repository_storage
}
+75
View File
@@ -0,0 +1,75 @@
resource "sonatyperepo_privilege_repository_view" "anonymous_ansible" {
name = "ci-anonymous-ansible-read"
description = "Anonymous read access to the Ansible Galaxy group"
actions = ["BROWSE", "READ"]
format = "ansiblegalaxy"
repository = sonatyperepo_repository_ansiblegalaxy_group.public.name
}
resource "sonatyperepo_privilege_repository_view" "anonymous_ansible_proxy" {
name = "ci-anonymous-ansible-proxy-read"
description = "Anonymous artifact read access to the Ansible Galaxy proxy"
actions = ["BROWSE", "READ"]
format = "ansiblegalaxy"
repository = sonatyperepo_repository_ansiblegalaxy_proxy.galaxy.name
}
resource "sonatyperepo_privilege_repository_view" "anonymous_go" {
name = "ci-anonymous-go-read"
description = "Anonymous read access to the Go module group"
actions = ["BROWSE", "READ"]
format = "go"
repository = sonatyperepo_repository_go_group.public.name
}
resource "sonatyperepo_privilege_repository_view" "anonymous_oci_public" {
name = "ci-anonymous-oci-public-read"
description = "Anonymous read access to the public OCI group"
actions = ["BROWSE", "READ"]
format = "oci"
repository = "oci-public"
}
resource "sonatyperepo_privilege_repository_view" "anonymous_oci_proxy" {
name = "ci-anonymous-oci-proxy-read"
description = "Anonymous read access to the OCI proxy member"
actions = ["BROWSE", "READ"]
format = "oci"
repository = "oci-proxy"
}
resource "sonatyperepo_role" "anonymous_ci" {
id = "ci-anonymous-read"
name = "CI anonymous read"
description = "Read-only access to public CI dependency proxy groups"
privileges = [
sonatyperepo_privilege_repository_view.anonymous_ansible.name,
sonatyperepo_privilege_repository_view.anonymous_ansible_proxy.name,
sonatyperepo_privilege_repository_view.anonymous_go.name,
sonatyperepo_privilege_repository_view.anonymous_oci_public.name,
sonatyperepo_privilege_repository_view.anonymous_oci_proxy.name,
]
roles = []
}
resource "sonatyperepo_user" "anonymous" {
user_id = "anonymous"
first_name = "Anonymous"
last_name = "User"
email_address = "[email protected]"
status = "active"
roles = [sonatyperepo_role.anonymous_ci.id]
}
resource "sonatyperepo_system_anonymous_access" "ci" {
enabled = true
user_id = sonatyperepo_user.anonymous.user_id
realm_name = "NexusAuthorizingRealm"
}
resource "sonatyperepo_security_realms" "active" {
active = [
"NexusAuthenticatingRealm",
"OciBearerToken",
]
}
+33
View File
@@ -0,0 +1,33 @@
terraform {
required_version = ">= 1.11.0"
required_providers {
sonatyperepo = {
source = "sonatype-nexus-community/sonatyperepo"
version = "1.17.0"
}
}
}
provider "sonatyperepo" {
url = var.nexus_url
username = var.nexus_username
password = var.nexus_password
}
variable "nexus_url" {
description = "Nexus Repository base URL"
type = string
}
variable "nexus_username" {
description = "Nexus Terraform management username"
type = string
sensitive = true
}
variable "nexus_password" {
description = "Nexus Terraform management password"
type = string
sensitive = true
}
+7
View File
@@ -12,3 +12,10 @@ services:
- "38008:38008" - "38008:38008"
volumes: volumes:
- "/mnt/pool/games/ps3:/games:rw" - "/mnt/pool/games/ps3:/games:rw"
# 避免与 DN42 的 172.20.0.0/14 重叠。
networks:
default:
ipam:
config:
- subnet: 172.28.1.0/24
-5
View File
@@ -1,5 +0,0 @@
route:
receiver: blackhole
receivers:
- name: blackhole
-94
View File
@@ -1,94 +0,0 @@
services:
# Metrics collector.
# It scrapes targets defined in --promscrape.config
# And forward them to --remoteWrite.url
vmagent:
image: victoriametrics/vmagent:v1.132.0
depends_on:
- "victoriametrics"
ports:
- 8429:8429
volumes:
- vmagentdata:/vmagentdata
- ./prometheus.yaml:/etc/prometheus/prometheus.yml
command:
- "--promscrape.config=/etc/prometheus/prometheus.yml"
- "--remoteWrite.url=http://victoriametrics:8428/api/v1/write"
restart: always
# VictoriaMetrics instance, a single process responsible for
# storing metrics and serve read requests.
victoriametrics:
image: victoriametrics/victoria-metrics:v1.132.0
ports:
- 8428:8428
- 8089:8089
- 8089:8089/udp
- 2003:2003
- 2003:2003/udp
- 4242:4242
volumes:
- vmdata:/storage
command:
- "--storageDataPath=/storage"
- "--graphiteListenAddr=:2003"
- "--opentsdbListenAddr=:4242"
- "--httpListenAddr=:8428"
- "--influxListenAddr=:8089"
- "--vmalert.proxyURL=http://vmalert:8880"
restart: always
grafana:
image: grafana/grafana:12.2.0
depends_on:
- "victoriametrics"
ports:
- 3000:3000
volumes:
- grafanadata:/var/lib/grafana
- ./provisioning/datasources/prometheus-datasource/single.yml:/etc/grafana/provisioning/datasources/single.yml
- ./provisioning/dashboards:/etc/grafana/provisioning/dashboards
- ./provisioning/dashboards/victoriametrics.json:/var/lib/grafana/dashboards/vm.json
- ./provisioning/dashboards/vmagent.json:/var/lib/grafana/dashboards/vmagent.json
- ./provisioning/dashboards/vmalert.json:/var/lib/grafana/dashboards/vmalert.json
restart: always
# vmalert executes alerting and recording rules
vmalert:
image: victoriametrics/vmalert:v1.132.0
depends_on:
- "victoriametrics"
- "alertmanager"
ports:
- 8880:8880
volumes:
- ./rules/alerts.yml:/etc/alerts/alerts.yml
- ./rules/alerts-health.yml:/etc/alerts/alerts-health.yml
- ./rules/alerts-vmagent.yml:/etc/alerts/alerts-vmagent.yml
- ./rules/alerts-vmalert.yml:/etc/alerts/alerts-vmalert.yml
command:
- "--datasource.url=http://victoriametrics:8428/"
- "--remoteRead.url=http://victoriametrics:8428/"
- "--remoteWrite.url=http://vmagent:8429/"
- "--notifier.url=http://alertmanager:9093/"
- "--rule=/etc/alerts/*.yml"
# display source of alerts in grafana
- "--external.url=http://127.0.0.1:3000" #grafana outside container
- '--external.alert.source=explore?orgId=1&left={"datasource":"VictoriaMetrics","queries":[{"expr":{{.Expr|jsonEscape|queryEscape}},"refId":"A"}],"range":{"from":"{{ .ActiveAt.UnixMilli }}","to":"now"}}'
restart: always
# alertmanager receives alerting notifications from vmalert
# and distributes them according to --config.file.
alertmanager:
image: prom/alertmanager:v0.28.1
volumes:
- ./alertmanager.yaml:/config/alertmanager.yml
command:
- "--config.file=/config/alertmanager.yml"
ports:
- 9093:9093
restart: always
volumes:
vmagentdata: {}
vmdata: {}
grafanadata: {}
-16
View File
@@ -1,16 +0,0 @@
global:
scrape_interval: 10s
scrape_configs:
- job_name: vmagent
static_configs:
- targets:
- vmagent:8429
- job_name: vmalert
static_configs:
- targets:
- vmalert:8880
- job_name: victoriametrics
static_configs:
- targets:
- victoriametrics:8428
@@ -1,9 +0,0 @@
apiVersion: 1
providers:
- name: Prometheus
orgId: 1
folder: ''
type: file
options:
path: /var/lib/grafana/dashboards
@@ -1,11 +0,0 @@
apiVersion: 1
datasources:
- name: VictoriaMetrics
type: prometheus
access: proxy
url: http://victoriametrics:8428
isDefault: true
jsonData:
prometheusType: Prometheus
prometheusVersion: 2.24.0
File diff suppressed because it is too large Load Diff
@@ -1,11 +0,0 @@
apiVersion: 1
datasources:
- name: VictoriaMetrics
type: prometheus
access: proxy
url: http://victoriametrics:8428
isDefault: true
jsonData:
prometheusType: Prometheus
prometheusVersion: 2.24.0
@@ -1,149 +0,0 @@
# File contains default list of alerts for various VM components.
# The following alerts are recommended for use for any VM installation.
# The alerts below are just recommendations and may require some updates
# and threshold calibration according to every specific setup.
groups:
- name: vm-health
# note the `job` filter and update accordingly to your setup
rules:
- alert: TooManyRestarts
expr: changes(process_start_time_seconds{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"}[15m]) > 2
labels:
severity: critical
annotations:
summary: "{{ $labels.job }} too many restarts (instance {{ $labels.instance }})"
description: >
Job {{ $labels.job }} (instance {{ $labels.instance }}) has restarted more than twice in the last 15 minutes.
It might be crashlooping.
- alert: ServiceDown
expr: up{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"} == 0
for: 2m
labels:
severity: critical
annotations:
summary: "Service {{ $labels.job }} is down on {{ $labels.instance }}"
description: "{{ $labels.instance }} of job {{ $labels.job }} has been down for more than 2 minutes."
- alert: ProcessNearFDLimits
expr: (process_max_fds - process_open_fds) < 100
for: 5m
labels:
severity: critical
annotations:
summary: "Number of free file descriptors is less than 100 for \"{{ $labels.job }}\"(\"{{ $labels.instance }}\") for the last 5m"
description: |
Exhausting OS file descriptors limit can cause severe degradation of the process.
Consider to increase the limit as fast as possible.
- alert: TooHighMemoryUsage
expr: (min_over_time(process_resident_memory_anon_bytes[10m]) / vm_available_memory_bytes) > 0.8
for: 5m
labels:
severity: critical
annotations:
summary: "It is more than 80% of memory used by \"{{ $labels.job }}\"(\"{{ $labels.instance }}\")"
description: |
Too high memory usage may result into multiple issues such as OOMs or degraded performance.
Consider to either increase available memory or decrease the load on the process.
- alert: TooHighCPUUsage
expr: rate(process_cpu_seconds_total[5m]) / process_cpu_cores_available > 0.9
for: 5m
labels:
severity: critical
annotations:
summary: "More than 90% of CPU is used by \"{{ $labels.job }}\"(\"{{ $labels.instance }}\") during the last 5m"
description: >
Too high CPU usage may be a sign of insufficient resources and make process unstable.
Consider to either increase available CPU resources or decrease the load on the process.
- alert: TooHighGoroutineSchedulingLatency
expr: histogram_quantile(0.99, sum(rate(go_sched_latencies_seconds_bucket{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"}[5m])) by (le, job, instance)) > 0.1
for: 15m
labels:
severity: critical
annotations:
summary: "\"{{ $labels.job }}\"(\"{{ $labels.instance }}\") has insufficient CPU resources for >15m"
description: >
Go runtime is unable to schedule goroutines execution in acceptable time. This is usually a sign of
insufficient CPU resources or CPU throttling. Verify that service has enough CPU resources. Otherwise,
the service could work unreliably with delays in processing.
- alert: TooManyLogs
expr: sum(increase(vm_log_messages_total{level="error"}[5m])) without (app_version, location) > 0
for: 15m
labels:
severity: warning
annotations:
summary: "Too many logs printed for job \"{{ $labels.job }}\" ({{ $labels.instance }})"
description: >
Logging rate for job \"{{ $labels.job }}\" ({{ $labels.instance }}) is {{ $value }} for last 15m.
Worth to check logs for specific error messages.
- alert: TooManyTSIDMisses
expr: increase(vm_missing_tsids_for_metric_id_total[5m]) > 0
for: 15m
labels:
severity: critical
annotations:
summary: "Unexpected TSID misses for job \"{{ $labels.job }}\" ({{ $labels.instance }}) for the last 15 minutes"
description: |
Unexpected TSID misses for \"{{ $labels.job }}\" ({{ $labels.instance }}) for the last 15 minutes.
If this happens after unclean shutdown of VictoriaMetrics process (via \"kill -9\", OOM or power off),
then this is OK - the alert must go away in a few minutes after the restart.
Otherwise this may point to the corruption of index data.
- alert: ConcurrentInsertsHitTheLimit
expr: avg_over_time(vm_concurrent_insert_current[1m]) >= vm_concurrent_insert_capacity
for: 15m
labels:
severity: warning
annotations:
summary: "{{ $labels.job }} on instance {{ $labels.instance }} is constantly hitting concurrent inserts limit"
description: |
The limit of concurrent inserts on instance {{ $labels.instance }} depends on the number of CPUs.
Usually, when component constantly hits the limit it is likely the component is overloaded and requires more CPU.
In some cases for components like vmagent or vminsert the alert might trigger if there are too many clients
making write attempts. If vmagent's or vminsert's CPU usage and network saturation are at normal level, then
it might be worth adjusting `-maxConcurrentInserts` cmd-line flag.
- alert: IndexDBRecordsDrop
expr: increase(vm_indexdb_items_dropped_total[5m]) > 0
labels:
severity: critical
annotations:
summary: "IndexDB skipped registering items during data ingestion with reason={{ $labels.reason }}."
description: |
VictoriaMetrics could skip registering new timeseries during ingestion if they fail the validation process.
For example, `reason=too_long_item` means that time series cannot exceed 64KB. Please, reduce the number
of labels or label values for such series. Or enforce these limits via `-maxLabelsPerTimeseries` and
`-maxLabelValueLen` command-line flags.
- alert: RowsRejectedOnIngestion
expr: rate(vm_rows_ignored_total[5m]) > 0
for: 15m
labels:
severity: warning
annotations:
summary: "Some rows are rejected on \"{{ $labels.instance }}\" on ingestion attempt"
description: "Ingested rows on instance \"{{ $labels.instance }}\" are rejected due to the
following reason: \"{{ $labels.reason }}\""
- alert: TooHighQueryLoad
expr: increase(vm_concurrent_select_limit_timeout_total[5m]) > 0
for: 15m
labels:
severity: warning
annotations:
summary: "Read queries fail with timeout for {{ $labels.job }} on instance {{ $labels.instance }}"
description: |
Instance {{ $labels.instance }} ({{ $labels.job }}) is failing to serve read queries during last 15m.
Concurrency limit `-search.maxConcurrentRequests` was reached on this instance and extra queries were
put into the queue for `-search.maxQueueDuration` interval. But even after waiting in the queue these queries weren't served.
This happens if instance is overloaded with the current workload, or datasource is too slow to respond.
Possible solutions are the following:
* reduce the query load;
* increase compute resources or number of replicas;
* adjust limits `-search.maxConcurrentRequests` and `-search.maxQueueDuration`.
See more at https://docs.victoriametrics.com/victoriametrics/troubleshooting/#slow-queries
@@ -1,172 +0,0 @@
# File contains default list of alerts for vmagent service.
# The alerts below are just recommendations and may require some updates
# and threshold calibration according to every specific setup.
groups:
# Alerts group for vmagent assumes that Grafana dashboard
# https://grafana.com/grafana/dashboards/12683 is installed.
# Pls update the `dashboard` annotation according to your setup.
- name: vmagent
interval: 30s
concurrency: 2
rules:
- alert: PersistentQueueIsDroppingData
expr: sum(increase(vm_persistentqueue_bytes_dropped_total[5m])) without (path) > 0
for: 10m
labels:
severity: critical
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=49&var-instance={{ $labels.instance }}"
summary: "Instance {{ $labels.instance }} is dropping data from persistent queue"
description: "Vmagent dropped {{ $value | humanize1024 }} from persistent queue
on instance {{ $labels.instance }} for the last 10m."
- alert: RejectedRemoteWriteDataBlocksAreDropped
expr: sum(increase(vmagent_remotewrite_packets_dropped_total[5m])) without (url) > 0
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=79&var-instance={{ $labels.instance }}"
summary: "Vmagent is dropping data blocks that are rejected by remote storage"
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} drops the rejected by
remote-write server data blocks. Check the logs to find the reason for rejects."
- alert: TooManyScrapeErrors
expr: increase(vm_promscrape_scrapes_failed_total[5m]) > 0
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=31&var-instance={{ $labels.instance }}"
summary: "Vmagent fails to scrape one or more targets"
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} fails to scrape targets for last 15m"
- alert: ScrapePoolHasNoTargets
expr: sum(vm_promscrape_scrape_pool_targets) without (status, instance, pod) == 0
for: 30m
labels:
severity: warning
annotations:
summary: "Vmagent has scrape_pool with 0 configured/discovered targets"
description: "Vmagent \"{{ $labels.job }}\" has scrape_pool \"{{ $labels.scrape_job }}\"
with 0 discovered targets. It is likely a misconfiguration. Please follow https://docs.victoriametrics.com/victoriametrics/vmagent/#debugging-scrape-targets
to troubleshoot the scraping config."
- alert: TooManyWriteErrors
expr: |
(sum(increase(vm_ingestserver_request_errors_total[5m])) without (name,net,type)
+
sum(increase(vmagent_http_request_errors_total[5m])) without (path,protocol)) > 0
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=77&var-instance={{ $labels.instance }}"
summary: "Vmagent responds with too many errors on data ingestion protocols"
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} responds with errors to write requests for last 15m."
- alert: TooManyRemoteWriteErrors
expr: rate(vmagent_remotewrite_retries_count_total[5m]) > 0
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=61&var-instance={{ $labels.instance }}"
summary: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} fails to push to remote storage"
description: "Vmagent fails to push data via remote write protocol to destination \"{{ $labels.url }}\"\n
Ensure that destination is up and reachable."
- alert: RemoteWriteConnectionIsSaturated
expr: |
(
rate(vmagent_remotewrite_send_duration_seconds_total[5m])
/
vmagent_remotewrite_queues
) > 0.9
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=84&var-instance={{ $labels.instance }}"
summary: "Remote write connection from \"{{ $labels.job }}\" (instance {{ $labels.instance }}) to {{ $labels.url }} is saturated"
description: "The remote write connection between vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }}) and destination \"{{ $labels.url }}\"
is saturated by more than 90% and vmagent won't be able to keep up.\n
There could be the following reasons for this:\n
* vmagent can't send data fast enough through the existing network connections. Increase `-remoteWrite.queues` cmd-line flag value to establish more connections per destination.\n
* remote destination can't accept data fast enough. Check if remote destination has enough resources for processing."
- alert: PersistentQueueForWritesIsSaturated
expr: rate(vm_persistentqueue_write_duration_seconds_total[5m]) > 0.9
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=98&var-instance={{ $labels.instance }}"
summary: "Persistent queue writes for instance {{ $labels.instance }} are saturated"
description: "Persistent queue writes for vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }})
are saturated by more than 90% and vmagent won't be able to keep up with flushing data on disk.
In this case, consider to decrease load on the vmagent or improve the disk throughput."
- alert: PersistentQueueForReadsIsSaturated
expr: rate(vm_persistentqueue_read_duration_seconds_total[5m]) > 0.9
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=99&var-instance={{ $labels.instance }}"
summary: "Persistent queue reads for instance {{ $labels.instance }} are saturated"
description: "Persistent queue reads for vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }})
are saturated by more than 90% and vmagent won't be able to keep up with reading data from the disk.
In this case, consider to decrease load on the vmagent or improve the disk throughput."
- alert: SeriesLimitHourReached
expr: (vmagent_hourly_series_limit_current_series / vmagent_hourly_series_limit_max_series) > 0.9
labels:
severity: critical
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=88&var-instance={{ $labels.instance }}"
summary: "Instance {{ $labels.instance }} reached 90% of the limit"
description: "Max series limit set via -remoteWrite.maxHourlySeries flag is close to reaching the max value.
Then samples for new time series will be dropped instead of sending them to remote storage systems."
- alert: SeriesLimitDayReached
expr: (vmagent_daily_series_limit_current_series / vmagent_daily_series_limit_max_series) > 0.9
labels:
severity: critical
annotations:
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=90&var-instance={{ $labels.instance }}"
summary: "Instance {{ $labels.instance }} reached 90% of the limit"
description: "Max series limit set via -remoteWrite.maxDailySeries flag is close to reaching the max value.
Then samples for new time series will be dropped instead of sending them to remote storage systems."
- alert: ConfigurationReloadFailure
expr: |
vm_promscrape_config_last_reload_successful != 1
or
vmagent_relabel_config_last_reload_successful != 1
labels:
severity: warning
annotations:
summary: "Configuration reload failed for vmagent instance {{ $labels.instance }}"
description: "Configuration hot-reload failed for vmagent on instance {{ $labels.instance }}.
Check vmagent's logs for detailed error message."
- alert: StreamAggrFlushTimeout
expr: |
increase(vm_streamaggr_flush_timeouts_total[5m]) > 0
labels:
severity: warning
annotations:
summary: "Streaming aggregation at \"{{ $labels.job }}\" (instance {{ $labels.instance }}) can't be finished within the configured aggregation interval."
description: "Stream aggregation process can't keep up with the load and might produce incorrect aggregation results. Check logs for more details.
Possible solutions: increase aggregation interval; aggregate smaller number of series; reduce samples' ingestion rate to stream aggregation."
- alert: StreamAggrDedupFlushTimeout
expr: |
increase(vm_streamaggr_dedup_flush_timeouts_total[5m]) > 0
labels:
severity: warning
annotations:
summary: "Deduplication \"{{ $labels.job }}\" (instance {{ $labels.instance }}) can't be finished within configured deduplication interval."
description: "Deduplication process can't keep up with the load and might produce incorrect results. Check docs https://docs.victoriametrics.com/victoriametrics/stream-aggregation/#deduplication and logs for more details.
Possible solutions: increase deduplication interval; deduplicate smaller number of series; reduce samples' ingestion rate."
@@ -1,96 +0,0 @@
# File contains default list of alerts for vmalert service.
# The alerts below are just recommendations and may require some updates
# and threshold calibration according to every specific setup.
groups:
# Alerts group for vmalert assumes that Grafana dashboard
# https://grafana.com/grafana/dashboards/14950 is installed.
# Pls update the `dashboard` annotation according to your setup.
- name: vmalert
interval: 30s
rules:
- alert: ConfigurationReloadFailure
expr: vmalert_config_last_reload_successful != 1
labels:
severity: warning
annotations:
summary: "Configuration reload failed for vmalert instance {{ $labels.instance }}"
description: "Configuration hot-reload failed for vmalert on instance {{ $labels.instance }}.
Check vmalert's logs for detailed error message."
- alert: AlertingRulesError
expr: sum(increase(vmalert_alerting_rules_errors_total[5m])) without(id) > 0
for: 5m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=13&var-instance={{ $labels.instance }}&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
summary: "Alerting rules are failing for vmalert instance {{ $labels.instance }}"
description: "Alerting rules execution is failing for \"{{ $labels.alertname }}\" from group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
Check vmalert's logs for detailed error message."
- alert: RecordingRulesError
expr: sum(increase(vmalert_recording_rules_errors_total[5m])) without(id) > 0
for: 5m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=30&var-instance={{ $labels.instance }}&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
summary: "Recording rules are failing for vmalert instance {{ $labels.instance }}"
description: "Recording rules execution is failing for \"{{ $labels.recording }}\" from group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
Check vmalert's logs for detailed error message."
- alert: RecordingRulesNoData
expr: sum(vmalert_recording_rules_last_evaluation_samples) without(id) < 1
for: 30m
labels:
severity: info
annotations:
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=33&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
summary: "Recording rule {{ $labels.recording }} ({{ $labels.group }}) produces no data"
description: "Recording rule \"{{ $labels.recording }}\" from group \"{{ $labels.group }}\ in file \"{{ $labels.file }}\"
produces 0 samples over the last 30min. It might be caused by a misconfiguration
or incorrect query expression."
- alert: TooManyMissedIterations
expr: increase(vmalert_iteration_missed_total[5m]) > 0
for: 15m
labels:
severity: warning
annotations:
summary: "vmalert instance {{ $labels.instance }} is missing rules evaluations"
description: "vmalert instance {{ $labels.instance }} is missing rules evaluations for group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
The group evaluation time takes longer than the configured evaluation interval. This may result in missed
alerting notifications or recording rules samples. Try increasing evaluation interval or concurrency of
group \"{{ $labels.group }}\". See https://docs.victoriametrics.com/victoriametrics/vmalert/#groups.
If rule expressions are taking longer than expected, please see https://docs.victoriametrics.com/victoriametrics/troubleshooting/#slow-queries."
- alert: RemoteWriteErrors
expr: increase(vmalert_remotewrite_errors_total[5m]) > 0
for: 15m
labels:
severity: warning
annotations:
summary: "vmalert instance {{ $labels.instance }} is failing to push metrics to remote write URL"
description: "vmalert instance {{ $labels.instance }} is failing to push metrics generated via alerting
or recording rules to the configured remote write URL. Check vmalert's logs for detailed error message."
- alert: RemoteWriteDroppingData
expr: increase(vmalert_remotewrite_dropped_rows_total[5m]) > 0
for: 5m
labels:
severity: critical
annotations:
summary: "vmalert instance {{ $labels.instance }} is dropping data sent to remote write URL"
description: "vmalert instance {{ $labels.instance }} is failing to send results of alerting or recording rules
to the configured remote write URL. This may result into gaps in recording rules or alerts state.
Check vmalert's logs for detailed error message."
- alert: AlertmanagerErrors
expr: increase(vmalert_alerts_send_errors_total[5m]) > 0
for: 15m
labels:
severity: warning
annotations:
summary: "vmalert instance {{ $labels.instance }} is failing to send notifications to Alertmanager"
description: "vmalert instance {{ $labels.instance }} is failing to send alert notifications to \"{{ $labels.addr }}\".
Check vmalert's logs for detailed error message."
-138
View File
@@ -1,138 +0,0 @@
# File contains default list of alerts for VictoriaMetrics single server.
# The alerts below are just recommendations and may require some updates
# and threshold calibration according to every specific setup.
groups:
# Alerts group for VM single assumes that Grafana dashboard
# https://grafana.com/grafana/dashboards/10229 is installed.
# Pls update the `dashboard` annotation according to your setup.
- name: vmsingle
interval: 30s
concurrency: 2
rules:
- alert: DiskRunsOutOfSpaceIn3Days
expr: |
sum(vm_free_disk_space_bytes) without(path) /
(
(rate(vm_rows_added_to_storage_total[1d]) - sum(rate(vm_deduplicated_samples_total[1d])) without(type)) * (
sum(vm_data_size_bytes{type!~"indexdb.*"}) without(type) /
sum(vm_rows{type!~"indexdb.*"}) without(type)
)
+
rate(vm_new_timeseries_created_total[1d]) * (
sum(vm_data_size_bytes{type="indexdb/file"}) without(type)/
sum(vm_rows{type="indexdb/file"}) without(type)
)
) < 3 * 24 * 3600 > 0
for: 30m
labels:
severity: critical
annotations:
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=53&var-instance={{ $labels.instance }}"
summary: "Instance {{ $labels.instance }} will run out of disk space soon"
description: "Taking into account current ingestion rate, free disk space will be enough only
for {{ $value | humanizeDuration }} on instance {{ $labels.instance }}.\n
Consider to limit the ingestion rate, decrease retention or scale the disk space if possible."
- alert: NodeBecomesReadonlyIn3Days
expr: |
sum(vm_free_disk_space_bytes - vm_free_disk_space_limit_bytes) without(path) /
(
(rate(vm_rows_added_to_storage_total[1d]) - sum(rate(vm_deduplicated_samples_total[1d])) without(type)) * (
sum(vm_data_size_bytes{type!~"indexdb.*"}) without(type) /
sum(vm_rows{type!~"indexdb.*"}) without(type)
)
+
rate(vm_new_timeseries_created_total[1d]) * (
sum(vm_data_size_bytes{type="indexdb/file"}) without(type) /
sum(vm_rows{type="indexdb/file"}) without(type)
)
) < 3 * 24 * 3600 > 0
for: 30m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/oS7Bi_0Wz?viewPanel=53&var-instance={{ $labels.instance }}"
summary: "Instance {{ $labels.instance }} will become read-only in 3 days"
description: "Taking into account current ingestion rate and free disk space
instance {{ $labels.instance }} is writable for {{ $value | humanizeDuration }}.\n
Consider to limit the ingestion rate, decrease retention or scale the disk space up if possible."
- alert: DiskRunsOutOfSpace
expr: |
sum(vm_data_size_bytes) by(job, instance) /
(
sum(vm_free_disk_space_bytes) by(job, instance) +
sum(vm_data_size_bytes) by(job, instance)
) > 0.8
for: 30m
labels:
severity: critical
annotations:
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=53&var-instance={{ $labels.instance }}"
summary: "Instance {{ $labels.instance }} (job={{ $labels.job }}) will run out of disk space soon"
description: "Disk utilisation on instance {{ $labels.instance }} is more than 80%.\n
Having less than 20% of free disk space could cripple merge processes and overall performance.
Consider to limit the ingestion rate, decrease retention or scale the disk space if possible."
- alert: RequestErrorsToAPI
expr: increase(vm_http_request_errors_total[5m]) > 0
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=35&var-instance={{ $labels.instance }}"
summary: "Too many errors served for path {{ $labels.path }} (instance {{ $labels.instance }})"
description: "Requests to path {{ $labels.path }} are receiving errors.
Please verify if clients are sending correct requests."
- alert: TooHighChurnRate
expr: |
(
sum(rate(vm_new_timeseries_created_total[5m])) by(instance)
/
sum(rate(vm_rows_inserted_total[5m])) by(instance)
) > 0.1
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=66&var-instance={{ $labels.instance }}"
summary: "Churn rate is more than 10% on \"{{ $labels.instance }}\" for the last 15m"
description: "VM constantly creates new time series on \"{{ $labels.instance }}\".\n
This effect is known as Churn Rate.\n
High Churn Rate is tightly connected with database performance and may
result in unexpected OOM's or slow queries."
- alert: TooHighChurnRate24h
expr: |
sum(increase(vm_new_timeseries_created_total[24h])) by(instance)
>
(sum(vm_cache_entries{type="storage/hour_metric_ids"}) by(instance) * 3)
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=66&var-instance={{ $labels.instance }}"
summary: "Too high number of new series on \"{{ $labels.instance }}\" created over last 24h"
description: "The number of created new time series over last 24h is 3x times higher than
current number of active series on \"{{ $labels.instance }}\".\n
This effect is known as Churn Rate.\n
High Churn Rate is tightly connected with database performance and may
result in unexpected OOM's or slow queries."
- alert: TooHighSlowInsertsRate
expr: |
(
sum(rate(vm_slow_row_inserts_total[5m])) by(instance)
/
sum(rate(vm_rows_inserted_total[5m])) by(instance)
) > 0.05
for: 15m
labels:
severity: warning
annotations:
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=68&var-instance={{ $labels.instance }}"
summary: "Percentage of slow inserts is more than 5% on \"{{ $labels.instance }}\" for the last 15m"
description: "High rate of slow inserts on \"{{ $labels.instance }}\" may be a sign of resource exhaustion
for the current load. It is likely more RAM is needed for optimal handling of the current number of active time series.
See also https://github.com/VictoriaMetrics/VictoriaMetrics/issues/3976#issuecomment-1476883183"
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+80 -14
View File
@@ -1,15 +1,32 @@
# zot OCI Registry # zot OCI Registry
内网入口为 `https://zot.ad.ddupan.top`。使用官方 Helm chart `0.1.124`,运行 内网匿名拉取入口为 `https://zot.ad.ddupan.top`,SPIRE 鉴权推送入口为
`https://zot-push.ad.ddupan.top`。使用官方 Helm chart `0.1.124`,运行
zot `v2.1.21`,镜像固定到官方 linux/amd64 digest。 zot `v2.1.21`,镜像固定到官方 linux/amd64 digest。
## 当前工作状态(2026-09-16 核验)
| 项目 | 状态 |
|---|---|
| 匿名拉取 | `zot.ad.ddupan.top` 已上线;空 `DOCKER_CONFIG` 的 crane pull 通过 |
| SPIRE 鉴权入口 | `zot-push.ad.ddupan.top` 已上线;真实 JWT-SVID 推送后可匿名拉取同一 digest |
| GitOps | 双入口配置已合并;Flux `zot` Kustomization 已应用 `d15733c`,状态 Ready |
| 运行与凭据同步 | `zot`、`zot-reader` HelmRelease 均 Ready,Pod 均 1/1;ESO SecretSynced |
| 临时配置清理 | 两个 HelmRelease 均无 `spec.values` 临时覆盖;暂停回写标记、测试身份和临时写权限已清理 |
| 接管复验 | 匿名拉取成功;推送入口无凭据返回 401,token realm 指向推送域名;接管未触发 Pod 重启 |
后续工作是给实际 CI 的 SPIFFE ID 配置具体仓库的 `create`/`update` 权限。
SPIRE 认证链路已经验证,但当前没有常驻 publisher 或删除授权;认证成功本身不代表
可以推送。S3 侧仍使用 Bao 管理的静态 AK/SK,尚未接入 SPIRE/STS。
## 存储与凭据 ## 存储与凭据
制品、manifest 和 OCI layout 保存在现有 SeaweedFS 的 `zot` bucket,前缀为 制品、manifest 和 OCI layout 保存在现有 SeaweedFS 的 `zot` bucket,前缀为
`registry/`,S3 endpoint 为 `https://s3.ad.ddupan.top`。**不创建 PVC**;chart 的 `registry/`,S3 endpoint 为 `https://s3.ad.ddupan.top`。**不创建 PVC**;chart 的
`/var/lib/registry` 是 `emptyDir`,仅用于运行时本地工作数据。 `/var/lib/registry` 是 `emptyDir`,仅用于运行时本地工作数据。
首期单副本,关闭跨仓库 dedupe,不额外部署 Redis/DynamoDB 缓存。保留 zot GC, 两个单副本实例共用同一 bucket 和前缀:`zot` 负责鉴权写入,`zot-reader` 负责匿名
读取。关闭跨仓库 dedupe,不额外部署 Redis/DynamoDB 缓存。只有写入实例启用 GC,
暂不配置自动删除已发布版本的 retention policy。增加副本、启用 dedupe 或搜索等 暂不配置自动删除已发布版本的 retention policy。增加副本、启用 dedupe 或搜索等
扩展前,需要重新检查共享元数据与缓存的持久化要求。 扩展前,需要重新检查共享元数据与缓存的持久化要求。
@@ -20,7 +37,7 @@ OpenBao kv/k8s/seaweedfs-s3
→ 原有 S3 身份及基础配置 ─┐ → 原有 S3 身份及基础配置 ─┐
├→ ESO 模板 → seaweedfs/seaweedfs-s3-config ├→ ESO 模板 → seaweedfs/seaweedfs-s3-config
OpenBao kv/k8s/zot-s3 ────┘ → 完整 s3.config OpenBao kv/k8s/zot-s3 ────┘ → 完整 s3.config
└→ ESO → zot/zot-s3 → zot 的 AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY └→ ESO → zot/zot-s3 → zot 与 zot-reader 的 AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY
``` ```
@@ -37,7 +54,7 @@ Terraform 的 `tfstate` bucket。`kv/k8s/zot-s3` 的字段是 `access_key` 和
本次归一没有轮换密钥,生成的完整配置与归一前语义一致。当前模板只有一组 zot 本次归一没有轮换密钥,生成的完整配置与归一前语义一致。当前模板只有一组 zot
凭据,尚未实现新旧密钥重叠轮换。后续轮换只修改 `kv/k8s/zot-s3`,但仍需协调 凭据,尚未实现新旧密钥重叠轮换。后续轮换只修改 `kv/k8s/zot-s3`,但仍需协调
两个 ExternalSecret 同步:确认 SeaweedFS Secret volume 更新后向 filer 的 两个 ExternalSecret 同步:确认 SeaweedFS Secret volume 更新后向 filer 的
`weed` 进程发送 SIGHUP,再确认 zot Secret 更新并重启 zot(环境变量不会热更新)。 `weed` 进程发送 SIGHUP,再确认 zot Secret 更新并重启 `zot` 和 `zot-reader`(环境变量不会热更新)。
两端异步更新期间可能短暂认证失败;需要无中断轮换时先扩展模板支持新旧凭据重叠。 两端异步更新期间可能短暂认证失败;需要无中断轮换时先扩展模板支持新旧凭据重叠。
## SPIRE 认证和授权 ## SPIRE 认证和授权
@@ -47,24 +64,40 @@ Terraform 的 `tfstate` bucket。`kv/k8s/zot-s3` 的字段是 `access_key` 和
| issuer | `https://spire-oidc.ad.ddupan.top` | | issuer | `https://spire-oidc.ad.ddupan.top` |
| JWT audience | `zot` | | JWT audience | `zot` |
| subject | `spiffe://ddupan.top/` 下的 workload SPIFFE ID | | subject | `spiffe://ddupan.top/` 下的 workload SPIFFE ID |
| token endpoint | `https://zot.ad.ddupan.top/zot/auth/token` | | token endpoint | `https://zot-push.ad.ddupan.top/zot/auth/token` |
| 当前权限 | 受信身份可以读取所有仓库;没有常驻写入或删除授权 | | 拉取入口 | 内网匿名读取所有仓库,不要求 SPIRE 身份 |
| 推送入口当前权限 | 受信身份可以读取所有仓库;没有常驻写入或删除授权 |
zot 通过已配置的 issuer discovery/JWKS 验证 JWT-SVID,再以 `sub` 作为授权身份。 zot 通过已配置的 issuer discovery/JWKS 验证 JWT-SVID,再以 `sub` 作为授权身份。
不接受任意 issuer,不关闭 TLS/issuer 验证。新的 Kata CI 负责取得并更新自己的 不接受任意 issuer,不关闭 TLS/issuer 验证。新的 Kata CI 负责取得并更新自己的
JWT-SVID;确认其身份命名后,再添加针对具体 repository 的 `create`/`update` JWT-SVID;确认其身份命名后,再添加针对具体 repository 的 `create`/`update`
授权,不能把整个 trust domain 都授予写权限。 授权,不能把整个 trust domain 都授予写权限。
现阶段拉取也需要 JWT-SVID。原定内网匿名拉取尚未启用:zot `v2.1.21` 的 zot `v2.1.21` 的 OIDC Bearer middleware 会在授权阶段之前拒绝无 token 请求。
OIDC Bearer middleware 会在授权阶段之前拒绝无 token 请求,单独增加 因此使用两个官方 zot 实例与两个域名,避免修改上游镜像,也避免同域名下匿名
`anonymousPolicy` 无法解决。匿名读取与 SPIRE 写入共存需后续单独验证方案。 `/v2/` 返回 200 导致标准客户端跳过 token 交换的问题。
- `zot-reader` 叠加 `reader-values.yaml`,没有认证 middleware,只有
`anonymousPolicy: [read]`。入口只转发 `/v2/` 的 GET/HEAD,并移除客户端遗留的
Authorization/Cookie;直接访问 reader Service 也不能写入。
- `zot` 保留 SPIRE issuer/audience/subject 校验及仓库授权,`externalUrl`、
Bearer realm、service 与 HTTPRoute 均使用 `zot-push.ad.ddupan.top`。
- reader 关闭 GC,没有同步或扫描扩展;读取同一份 S3 制品,不复制 bucket,
不新增 PVC 或 S3 密钥。镜像、安全上下文、资源和 Secret 引用由共用 values 继承。
- 两个配置的 `storageDriver` 必须保持一致;修改 S3 endpoint/bucket/prefix 时
同时更新 `values.yaml` 与 `reader-values.yaml`。
推送客户端应登录 `zot-push.ad.ddupan.top`;拉取客户端无需登录。
已有 SPIRE 身份的进程可以通过 Workload API 获取 `aud=zot` 的 JWT-SVID,然后 已有 SPIRE 身份的进程可以通过 Workload API 获取 `aud=zot` 的 JWT-SVID,然后
通过 `docker login` 或 `crane auth login` 的 `--password-stdin` 交给 Registry。 通过 `docker login` 或 `crane auth login` 的 `--password-stdin` 交给 Registry。
用户名可以使用 `zot`,实际权限取自已验证 JWT 的身份。使用独立、权限为 `0700` 用户名可以使用 `zot`,实际权限取自已验证 JWT 的身份。使用独立、权限为 `0700`
的临时 `DOCKER_CONFIG`,结束后删除;不要开启 shell tracing,不要打印 token, 的临时 `DOCKER_CONFIG`,结束后删除;不要开启 shell tracing,不要打印 token,
不要把 token 放进命令参数。token 接口不会延长 SVID 有效期。 不要把 token 放进命令参数。token 接口不会延长 SVID 有效期。
同一仓库在两个入口使用相同路径和 tag/digest,例如 CI 推送到
`zot-push.ad.ddupan.top/team/image:tag`,部署时使用
`zot.ad.ddupan.top/team/image:tag`;无需在两个仓库间复制。
## 部署与网络 ## 部署与网络
- 官方 chart 管理 Deployment、Service、ConfigMap 和 HTTPRoute。 - 官方 chart 管理 Deployment、Service、ConfigMap 和 HTTPRoute。
@@ -72,13 +105,17 @@ OIDC Bearer middleware 会在授权阶段之前拒绝无 token 请求,单独
`https` listener 与内网通配符证书终止。 `https` listener 与内网通配符证书终止。
- 仅配置 Samba AD 内网 DNS;不创建公网 DNS 或 Cloudflare Tunnel route。 - 仅配置 Samba AD 内网 DNS;不创建公网 DNS 或 Cloudflare Tunnel route。
- NetworkPolicy 只允许现有 Envoy Gateway 数据面访问 zot 的 5000 端口。 - NetworkPolicy 只允许现有 Envoy Gateway 数据面访问 zot 的 5000 端口。
- HTTPRoute 只暴露 `/v2/` 和 `/zot/auth/token`,不暴露内部健康检查或管理端点。 - 拉取域名仅暴露 `/v2/` 的 GET/HEAD;推送域名暴露 `/v2/` 和
`/zot/auth/token`,均不暴露内部健康检查或管理端点。
- namespace 使用 restricted PodSecurity,容器非 root、只读根文件系统。 - namespace 使用 restricted PodSecurity,容器非 root、只读根文件系统。
首次已按用户授权从本地执行 `kubectl apply -k apps/zot`,由集群 Helm controller `clusters/homelab/apps/zot.yaml` 已将 `zot` 和 `zot-reader` 一并纳入 Flux 管理。
安装。`clusters/homelab/apps/zot.yaml` 是 GitOps composition;对应文件合并进入 两个 HelmRelease 通过共用 `zot-values` 继承基础配置,reader 再叠加
Flux 跟踪分支后,才由根 Kustomization 持续管理,不能把未提交的本地部署写成 `zot-reader-values`。当前由 main 分支持续管理,不依赖本地覆盖或暂停回写。
已完成 Git 接管。
后续若需临时验收,收尾时先确认 Git 管理的配置与目标运行配置一致,再移除
`spec.values` 临时覆盖及 `kustomize.toolkit.fluxcd.io/reconcile=disabled` 标记,
触发 zot Kustomization reconcile 并复验。临时测试身份和写权限不得留在持久配置中。
检查与渲染: 检查与渲染:
@@ -114,6 +151,35 @@ sha256:b8d3b977a1235022759470903dab4a46b7cf8107958624f1f76a323eabe37c5e
它是仅含验证文本的 OCI 测试镜像,没有可执行入口,不用于运行服务。 它是仅含验证文本的 OCI 测试镜像,没有可执行入口,不用于运行服务。
双域名验收还使用 `verification/anonymous-spire:smoke`:标准 crane 从
`zot-push.ad.ddupan.top` 登录、推送,再从 `zot.ad.ddupan.top` 使用空
`DOCKER_CONFIG` 拉取,两个入口的 digest 必须一致。验证匿名 blob HEAD、tags、
referrers,以及客户端保存旧凭据时的公共拉取。推送入口检查无凭据、错误签名、
错误 audience、过期 SVID、跨仓库写入和删除拒绝;公共入口拒绝所有写方法,
reader Service 直连也拒绝写入。测试完成后撤回临时单仓库写权限。
2026-09-16 上述双域名验收通过;SVID 过期后推送入口返回 401,匿名拉取不受
影响。临时写权限已撤销,两个 HelmRelease Ready;推送 DNS 第二次检查 changed=0。
匿名拉取示例:
```bash
crane pull zot.ad.ddupan.top/verification/anonymous-spire:smoke image.tar --format oci
```
鉴权推送示例(先通过 Workload API 将短期 JWT-SVID 保存到当前进程的 `ZOT_JWT`,
不要启用 shell tracing;示例中的仓库仍需提前给具体 SPIFFE ID 授权):
```bash
export DOCKER_CONFIG="$(mktemp -d)"
printf '%s' "$ZOT_JWT" | crane auth login zot-push.ad.ddupan.top \
--username zot --password-stdin
crane push image.tar zot-push.ad.ddupan.top/team/image:tag
rm -rf -- "$DOCKER_CONFIG"
unset DOCKER_CONFIG ZOT_JWT
```
Registry 恢复需要完整的 SeaweedFS bucket 数据、Bao 专用凭据和此目录配置。 Registry 恢复需要完整的 SeaweedFS bucket 数据、Bao 专用凭据和此目录配置。
zot 的临时目录不是制品备份。独立异机/离线备份尚未在本次部署中建立;不能把同一 zot 的临时目录不是制品备份。独立异机/离线备份尚未在本次部署中建立;不能把同一
SeaweedFS 内的数据副本当作独立灾备。重装 zot 不得删除 `zot` bucket。 SeaweedFS 内的数据副本当作独立灾备。重装 zot 不得删除 `zot` bucket。
+32
View File
@@ -0,0 +1,32 @@
apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
metadata:
name: zot-reader
namespace: zot
spec:
chart:
spec:
chart: zot
version: 0.1.124
interval: 1h
sourceRef:
kind: HelmRepository
name: zot
releaseName: zot-reader
interval: 30m
timeout: 5m
driftDetection:
mode: enabled
install:
strategy:
name: RetryOnFailure
retryInterval: 5m
upgrade:
strategy:
name: RetryOnFailure
retryInterval: 5m
valuesFrom:
- kind: ConfigMap
name: zot-values
- kind: ConfigMap
name: zot-reader-values
+5
View File
@@ -6,6 +6,7 @@ resources:
- external-secret.yaml - external-secret.yaml
- helmrepository.yaml - helmrepository.yaml
- helmrelease.yaml - helmrelease.yaml
- helmrelease-reader.yaml
- networkpolicy.yaml - networkpolicy.yaml
generatorOptions: generatorOptions:
disableNameSuffixHash: true disableNameSuffixHash: true
@@ -16,3 +17,7 @@ configMapGenerator:
namespace: zot namespace: zot
files: files:
- values.yaml=values.yaml - values.yaml=values.yaml
- name: zot-reader-values
namespace: zot
files:
- values.yaml=reader-values.yaml
+64
View File
@@ -0,0 +1,64 @@
# 叠加于共用 values.yaml;同一镜像、S3、Secret、安全设置,无制品副本。
# 无 Bearer middleware,仅 anonymousPolicy=read;关闭 GC 避免多个实例清理共享存储。
configFiles:
config.json: |
{
"distSpecVersion": "1.1.1",
"storage": {
"rootDirectory": "/var/lib/registry",
"dedupe": false,
"gc": false,
"storageDriver": {
"name": "s3",
"region": "us-east-1",
"regionendpoint": "https://s3.ad.ddupan.top",
"bucket": "zot",
"rootdirectory": "/registry",
"secure": true,
"skipverify": false,
"forcepathstyle": true
}
},
"http": {
"address": "0.0.0.0",
"port": "5000",
"externalUrl": "https://zot.ad.ddupan.top",
"compat": [
"docker2s2"
],
"accessControl": {
"repositories": {
"**": {
"anonymousPolicy": [
"read"
]
}
}
}
},
"log": {
"level": "info"
}
}
httproute:
hostnames:
- zot.ad.ddupan.top
rules:
- matches:
- path:
type: PathPrefix
value: /v2/
method: GET
- path:
type: PathPrefix
value: /v2/
method: HEAD
filters:
- type: RequestHeaderModifier
requestHeaderModifier:
remove:
- Cookie
- Authorization
timeouts:
request: 900s
backendRequest: 900s
+53 -4
View File
@@ -41,14 +41,14 @@ configFiles:
"http": { "http": {
"address": "0.0.0.0", "address": "0.0.0.0",
"port": "5000", "port": "5000",
"externalUrl": "https://zot.ad.ddupan.top", "externalUrl": "https://zot-push.ad.ddupan.top",
"compat": [ "compat": [
"docker2s2" "docker2s2"
], ],
"auth": { "auth": {
"bearer": { "bearer": {
"realm": "https://zot.ad.ddupan.top/zot/auth/token", "realm": "https://zot-push.ad.ddupan.top/zot/auth/token",
"service": "zot.ad.ddupan.top", "service": "zot-push.ad.ddupan.top",
"oidc": [ "oidc": [
{ {
"issuer": "https://spire-oidc.ad.ddupan.top", "issuer": "https://spire-oidc.ad.ddupan.top",
@@ -70,7 +70,56 @@ configFiles:
}, },
"accessControl": { "accessControl": {
"repositories": { "repositories": {
"panxiao81/gitea-dynamic-runner-controller": {
"policies": [
{
"users": [
"spiffe://ddupan.top/ci/panxiao81/gitea-dynamic-runner/publish-images",
"spiffe://ddupan.top/dev/panxiao81"
],
"actions": [
"read",
"create",
"update"
]
}
],
"defaultPolicy": [
"read"
]
},
"panxiao81/gitea-dynamic-runner-runner": {
"policies": [
{
"users": [
"spiffe://ddupan.top/ci/panxiao81/gitea-dynamic-runner/publish-images",
"spiffe://ddupan.top/dev/panxiao81"
],
"actions": [
"read",
"create",
"update"
]
}
],
"defaultPolicy": [
"read"
]
},
"**": { "**": {
"policies": [
{
"users": [
"spiffe://ddupan.top/dev/panxiao81"
],
"actions": [
"read",
"create",
"update",
"delete"
]
}
],
"defaultPolicy": [ "defaultPolicy": [
"read" "read"
] ]
@@ -133,7 +182,7 @@ httproute:
namespace: envoy-gateway-system namespace: envoy-gateway-system
sectionName: https sectionName: https
hostnames: hostnames:
- zot.ad.ddupan.top - zot-push.ad.ddupan.top
rules: rules:
- matches: - matches:
- path: - path:
+4 -1
View File
@@ -50,6 +50,9 @@ sudo k3s kubectl -n flux-system get gitrepositories,kustomizations
- VictoriaMetrics Operator 已固定现有 chart `0.66.2` 并完成分阶段 Flux HelmRelease - VictoriaMetrics Operator 已固定现有 chart `0.66.2` 并完成分阶段 Flux HelmRelease
接管;Metrics、Logs、Traces 与 Grafana 也已统一完成 Flux 接管; 接管;Metrics、Logs、Traces 与 Grafana 也已统一完成 Flux 接管;
- External Secrets Operator 已固定 chart `2.8.0` 并完成分阶段接管; - External Secrets Operator 已固定 chart `2.8.0` 并完成分阶段接管;
- SPIRE 已按官方 hardened chart `0.30.2`(SPIRE `1.15.3`)声明,使用共享 - SPIRE 已按 hardened chart 内部 fork `0.30.2-ddupan.1`(基于上游 `0.30.2`,SPIRE
`1.15.3`)声明,使用共享
PostgreSQL 与独立 signing-key PVC;首次上线和 OpenBao JWT-SVID PoC 尚待合并后验证; PostgreSQL 与独立 signing-key PVC;首次上线和 OpenBao JWT-SVID PoC 尚待合并后验证;
- Nexus Repository CE POC 已加入 GitOps 声明,计划验证 Ansible Galaxy、Go Modules 与
OCI/BuildKit 缓存;尚未部署或完成现场验收,现有 zot 保持不变;
- root Kustomization 与所有 brownfield 子 Kustomization 继续保持 `prune: false`。 - root Kustomization 与所有 brownfield 子 Kustomization 继续保持 `prune: false`。
+19
View File
@@ -0,0 +1,19 @@
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: dynamic-runner
namespace: flux-system
spec:
dependsOn:
- name: external-secrets
- name: spire
interval: 10m
path: ./platform/dynamic-runner
# The runner backends are replaceable. Prune is required when a retired
# worker is removed from the component; otherwise it keeps consuming work.
prune: true
sourceRef:
kind: GitRepository
name: flux-system
timeout: 5m
wait: true
+18
View File
@@ -0,0 +1,18 @@
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: nats
namespace: flux-system
spec:
dependsOn:
- name: cert-manager
- name: external-secrets
- name: openebs
interval: 10m
path: ./platform/nats
prune: false
sourceRef:
kind: GitRepository
name: flux-system
timeout: 10m
wait: true
+22
View File
@@ -0,0 +1,22 @@
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: nexus
namespace: flux-system
spec:
dependsOn:
- name: envoy-gateway
- name: openebs
healthChecks:
- apiVersion: apps/v1
kind: Deployment
name: nexus
namespace: nexus
interval: 10m
path: ./apps/nexus
prune: false
sourceRef:
kind: GitRepository
name: flux-system
timeout: 15m
wait: true
+3 -1
View File
@@ -7,9 +7,11 @@ resources:
- apps/envoy-gateway.yaml - apps/envoy-gateway.yaml
- apps/external-secrets.yaml - apps/external-secrets.yaml
- apps/gitea.yaml - apps/gitea.yaml
- apps/gitea-actions.yaml
- apps/http-echo.yaml - apps/http-echo.yaml
- apps/openebs.yaml - apps/openebs.yaml
- apps/nats.yaml
- apps/dynamic-runner.yaml
- apps/spire.yaml - apps/spire.yaml
- apps/observability.yaml - apps/observability.yaml
- apps/zot.yaml - apps/zot.yaml
- apps/nexus.yaml
+72
View File
@@ -0,0 +1,72 @@
# Sandbox 集群
这里是 OpenSandbox、CI 和 AI Agent workload 所在双节点 k3s 集群的 Flux
reconciliation 入口。LXC、PostgreSQL、K3s、固定版本的 Flux controllers 与 root
sync 由 `infrastructure/sandbox-cluster/` 中的 Ansible 管理;本目录只组合集群内
workload。
Flux 通过 `https://git.ddupan.top/panxiao81/homelab-infra.git` 读取公开仓库。
Ansible 将 homelab CA 注入 `GitRepository/flux-system` 引用的同名 Secret,不使用
长期 Git 凭据。root Kustomization 从 `./clusters/sandbox` 开始 reconciliation,
初始保持 `prune: false`。
Root bootstrap 已完成。后续按依赖顺序分别引入:
1. 监控 CRD、kube-state-metrics 以及 kubelet/cAdvisor 抓取配置;
2. SPIRE Agent、SPIFFE CSI Driver 与 workload registration;
3. Kata Containers、`block-plain` RuntimeClass;
4. 独立 External Secrets Operator 与 sandbox 专用 OpenBao auth backend;
5. OpenSandbox controller/server;CI Pool 与 runner 调度器随后独立接入。
每一阶段单独合并并等待对应 Flux Kustomization Ready,不在 bootstrap 时一次性部署。
第一阶段监控拆为 `monitoring-operator` 与依赖它的 `monitoring`,防止 VM CR 在
VictoriaMetrics Operator CRD Ready 前进入 reconciliation。
SPIRE 阶段先由 `spire-bootstrap` 安装 CRD,并声明按上游 k8s_psat Server plugin
要求收窄的 reviewer:它可以调用 TokenReview,并只读查询用于证明的 Pod 与 Node。
Agent ServiceAccount 留给后续 HelmRelease 创建,避免两个声明方争夺同一资源。随后运行
`infrastructure/sandbox-cluster/ansible/spire-bootstrap.yml`:playbook 从 sandbox
读取 reviewer token,在内存中组成受限 kubeconfig,再通过 stdin reconcile 到 central
集群的 `spire-server/spire-external-kubeconfigs` Secret。凭据不写入仓库、日志或控制机
文件;该 Secret 准备完成后,才能启用 central external PSAT/controller-manager 和
sandbox Agent/CSI。
External controller-manager 使用独立的 `spire-controller-manager` ServiceAccount;其
RBAC 与上游 controller-manager 所需权限一致,用于读取 workload selectors、维护
SPIFFE CR status/finalizer 和 leader election。它不复用只允许 TokenReview 的 Server
reviewer。Ansible 将两份 kubeconfig 写入同一个 central Secret 的不同 key,便于 central
chart 分别绑定 `sandbox` 与 `sandbox-controller`。
Central SPIRE Server 通过内网 `spire-server.ad.ddupan.top:8081` 接收 sandbox Agent
attestation。Server 使用 external bundle publisher 持续维护 sandbox
`spire-system/spire-bundle`,Agent 不固定或复制 trust bundle。Sandbox HelmRelease
显式关闭 Server 与 OIDC Provider,只部署 Agent DaemonSet 和 SPIFFE CSI Driver;因此
不会产生第二个 trust root。
`spire-smoke` namespace、ServiceAccount 和 `sandbox-spire-smoke` ClusterSPIFFEID 只用于
普通 Pod 的 CSI 回归夹具,稳定身份为 `spiffe://ddupan.top/sandbox/smoke`。Kata guest
不能复用 node Agent 暴露的 Unix socket;virtio-fs 只能呈现 socket 路径,不能把连接
跨过 VM 边界。Kata workload 必须使用 guest 内 Agent,具体约束见
`platform/sandbox-kata/README.md`。测试 Pod 临时创建并在验收后删除,普通 Pod 的身份
声明保留。
Kata 阶段使用官方 4.1.0 `kata-deploy` chart 的短生命周期 `job` 模式,逐节点安装并
重启 K3s。只启用 `kata-clh-runtime-rs`,不创建默认 `kata` 别名;该 handler 的
`emptyDir` 固定使用 `block-plain`,为 Docker/BuildKit overlay2 与 kind 提供 guest
内块设备文件系统。详细限制与上线验收见 `platform/sandbox-kata/README.md`。
Sandbox 的 ESO 通过独立 `auth/kubernetes-sandbox` 向 OpenBao 证明 ServiceAccount
身份,只能读取共享的 `kv/k8s/opensandbox-api`。它不保存 reviewer JWT 或长期 Bao token;相关
Terraform 与 Flux 边界见 `platform/sandbox-external-secrets/README.md`。
OpenSandbox 阶段固定官方源码 commit 与 umbrella chart `0.2.2`,只部署 controller、
ClusterIP server 和 CRD。API key 由 ESO 从 OpenBao 投影,明文不进入 Git。
## 监控边界
这里只管理 sandbox LXC 内的 Kubernetes 监控,不负责 PVE 宿主监控。LXC 与宿主共享
内核,即使 lxcfs 虚拟化了内存和 uptime,容器内 `/proc/stat` 仍是宿主 CPU 视图;
在 LXC 内运行 node_exporter 会生成混合语义并重复采集宿主指标,因此禁止部署。
Sandbox 节点与 workload 指标来自 kubelet/cAdvisor 和 kube-state-metrics;K3s 或 LXC
特有但上述接口未覆盖的指标,应使用目标明确的 collector,不以 node_exporter 补齐。
+18
View File
@@ -0,0 +1,18 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: ci-runners
namespace: flux-system
spec:
dependsOn:
- name: opensandbox
- name: spire-agents
interval: 10m
path: ./platform/sandbox-ci-runners
prune: true
sourceRef:
kind: GitRepository
name: flux-system
timeout: 20m
wait: true
@@ -0,0 +1,17 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: external-secrets-operator
namespace: flux-system
spec:
dependsOn:
- name: monitoring-operator
interval: 10m
path: ./platform/sandbox-external-secrets/operator
prune: true
sourceRef:
kind: GitRepository
name: flux-system
timeout: 10m
wait: true
@@ -0,0 +1,17 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: external-secrets
namespace: flux-system
spec:
dependsOn:
- name: external-secrets-operator
interval: 10m
path: ./platform/sandbox-external-secrets/config
prune: true
sourceRef:
kind: GitRepository
name: flux-system
timeout: 10m
wait: true
+18
View File
@@ -0,0 +1,18 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: kata
namespace: flux-system
spec:
dependsOn:
- name: monitoring-operator
- name: spire-agents
interval: 10m
path: ./platform/sandbox-kata
prune: true
sourceRef:
kind: GitRepository
name: flux-system
timeout: 35m
wait: true
@@ -0,0 +1,15 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: monitoring-operator
namespace: flux-system
spec:
interval: 10m
path: ./platform/sandbox-monitoring/operator
prune: true
sourceRef:
kind: GitRepository
name: flux-system
timeout: 10m
wait: true
+17
View File
@@ -0,0 +1,17 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: monitoring
namespace: flux-system
spec:
dependsOn:
- name: monitoring-operator
interval: 10m
path: ./platform/sandbox-monitoring/workloads
prune: true
sourceRef:
kind: GitRepository
name: flux-system
timeout: 10m
wait: true
+19
View File
@@ -0,0 +1,19 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: opensandbox
namespace: flux-system
spec:
dependsOn:
- name: external-secrets
- name: kata
- name: monitoring-operator
interval: 10m
path: ./platform/sandbox-opensandbox
prune: true
sourceRef:
kind: GitRepository
name: flux-system
timeout: 15m
wait: true
+18
View File
@@ -0,0 +1,18 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: spire-agents
namespace: flux-system
spec:
dependsOn:
- name: spire-bootstrap
- name: monitoring-operator
interval: 10m
path: ./platform/sandbox-spire/agents
prune: true
sourceRef:
kind: GitRepository
name: flux-system
timeout: 15m
wait: true
@@ -1,14 +1,15 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1 apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization kind: Kustomization
metadata: metadata:
name: gitea-actions name: spire-bootstrap
namespace: flux-system namespace: flux-system
spec: spec:
interval: 10m interval: 10m
path: ./platform/gitea-runner path: ./platform/sandbox-spire/bootstrap
prune: false prune: true
sourceRef: sourceRef:
kind: GitRepository kind: GitRepository
name: flux-system name: flux-system
timeout: 3m timeout: 10m
wait: false wait: true
+13
View File
@@ -0,0 +1,13 @@
---
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- apps/monitoring-operator.yaml
- apps/monitoring.yaml
- apps/spire-bootstrap.yaml
- apps/spire-agents.yaml
- apps/kata.yaml
- apps/external-secrets-operator.yaml
- apps/external-secrets.yaml
- apps/opensandbox.yaml
- apps/ci-runners.yaml
@@ -0,0 +1,9 @@
# Generated by infrastructure/dns/generate.py. Do not edit directly.
resource "cloudflare_dns_record" "auth" {
zone_id = var.zone_id
name = "auth.ddupan.top"
type = "CNAME"
content = "ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com"
proxied = true
ttl = 1
}
@@ -38,17 +38,6 @@ resource "cloudflare_zero_trust_tunnel_cloudflared_config" "main" {
} }
} }
# Public DNS: proxied CNAME -> the tunnel. (auth was bootstrapped with
# `cloudflared tunnel route dns`; import it into state — see README.)
resource "cloudflare_dns_record" "auth" {
zone_id = var.zone_id
name = "auth.ddupan.top"
type = "CNAME"
content = "${var.tunnel_id}.cfargotunnel.com"
proxied = true
ttl = 1 # 1 = automatic (required when proxied)
}
# DKIM for Microsoft 365 mail sent as *@ddupan.top (via the smtp-relay). CNAMEs point # DKIM for Microsoft 365 mail sent as *@ddupan.top (via the smtp-relay). CNAMEs point
# at the tenant's DKIM keys; must be DNS-only (unproxied). Enable signing in Exchange # at the tenant's DKIM keys; must be DNS-only (unproxied). Enable signing in Exchange
# after these resolve: smtp-relay/scripts/enable-dkim.ps1. # after these resolve: smtp-relay/scripts/enable-dkim.ps1.
+33 -7
View File
@@ -5,10 +5,32 @@
| 视图 | 权威或递归服务 | 配置方式 | | 视图 | 权威或递归服务 | 配置方式 |
|---|---|---| |---|---|---|
| 公网 `ddupan.top` | Cloudflare | Terraform;尚待完整导入已有记录 | | 公网 `ddupan.top` | Cloudflare | 由生成器输出 Terraform;尚待完整导入已有记录 |
| AD `ad.ddupan.top` | Samba internal DNS | `samba_dns_record` Ansible module | | AD `ad.ddupan.top` | Samba internal DNS | `samba_dns_record` Ansible module |
| LAN split horizon | Blocky | 尚待从 inventory 渲染或校验 | | LAN split horizon | Blocky | 由生成器维护 `customDNS.mapping` 标记块 |
| Kubernetes Pod split horizon | CoreDNS | 尚待从 inventory 渲染或校验 | | Kubernetes Pod split horizon | CoreDNS | 由生成器维护 `.server` 标记块 |
## 生成配置
安装了 `uv` 后,在仓库根目录运行:
```bash
uv run infrastructure/dns/generate.py
uv run infrastructure/dns/generate.py --check
```
脚本使用内嵌锁定版本的 PyYAML 和 Jinja2,从 `records.yml` 渲染三个目标:
- `apps/blocky/config.yml` 中带 marker 的 LAN split-horizon mapping;
- `platform/k3s/coredns-custom.yaml` 中带 marker 的 Pod split-horizon server blocks;
- `infrastructure/cloudflared/terraform/dns.generated.tf` 中已经完成 Terraform 接管的公网记录。
生成文件需要提交进 Git,以便 PR 直接审阅最终配置。CI 执行 `--check`,任何手工修改生成块、
漏跑生成器或非确定性输出都会失败。Jinja 使用 `[[ ... ]]` 作为变量定界符,避免与 CoreDNS
模板表达式 `{{ .Name }}` 冲突。
`backends` 和 `terraform.managed` 是分阶段接管开关,而不是第二份记录数据:只有已经完成
零变更接管的后端才会生成。把记录加入新的后端前,应先完成相应的 live/state 对账。
## 安全边界 ## 安全边界
@@ -21,9 +43,13 @@
## 分阶段接管 ## 分阶段接管
1. 用 Samba module 接管现有静态 A RRset,首次 check mode 应为零变更。 1. 用 Samba module 接管现有静态 A RRset,首次 check mode 应为零变更。
2. 将 Cloudflare 已有 tunnel DNS 记录导入 Terraform state。 2. 将 Cloudflare 已有 tunnel DNS 记录逐条导入 Terraform state,再启用 `terraform.managed`。
3. 让 Blocky 与 CoreDNS 从 `split_horizon.records` 生成配置或执行 CI 一致性检查。 3. Blocky 与 CoreDNS 已从 `split_horizon.records` 生成;通过 `backends` 分阶段扩展。
4. 验证公网、LAN、Pod、AD 四个视图后,再单独修改 DHCP。 4. 验证公网、LAN、Pod、AD 四个视图后,再单独修改 DHCP。
当前 inventory 已明确暴露一个既有差异:`obj.ddupan.top` 在 Blocky 中存在,但 CoreDNS 当前 inventory 明确保留一个既有差异:`obj.ddupan.top` 的 `backends` 只有 Blocky,CoreDNS
尚无对应覆盖。本阶段不会偷偷修复它;后续在两个 resolver 同时接管时统一修复。 尚无对应覆盖。本阶段不改变线上语义;后续验证 Pod 侧入口后再加入 `coredns`。
CoreDNS split-horizon 的原因是避免集群内请求经 Cloudflare 公网绕回同一个集群。尤其 Gitea
启动时会访问 Authelia discovery URL,公网路径故障曾令其启动失败;生成块仍返回相同 LAN A
记录,并对 AAAA 返回 NOERROR/no-data。
+146
View File
@@ -0,0 +1,146 @@
#!/usr/bin/env -S uv run --script
# /// script
# requires-python = ">=3.12"
# dependencies = ["Jinja2==3.1.6", "PyYAML==6.0.3"]
# ///
"""Render backend DNS configuration from records.yml."""
from __future__ import annotations
import argparse
import difflib
from pathlib import Path
import sys
import yaml
from jinja2 import Environment, FileSystemLoader, StrictUndefined
ROOT = Path(__file__).resolve().parents[2]
DNS_DIR = ROOT / "infrastructure/dns"
BEGIN = "# BEGIN GENERATED: homelab DNS ([[ target ]])"
END = "# END GENERATED: homelab DNS ([[ target ]])"
def load_inventory() -> dict:
data = yaml.safe_load((DNS_DIR / "records.yml").read_text())
try:
inventory = data["homelab_dns"]
split_records = inventory["split_horizon"]["records"]
public_records = inventory["public"]["records"]
except (KeyError, TypeError) as exc:
raise ValueError(f"invalid DNS inventory: missing {exc}") from exc
for record in split_records:
require_fields(record, "name", "type", "values", "backends")
if record["type"] != "A" or len(record["values"]) != 1:
raise ValueError(f"split record must be a single A value: {record!r}")
unknown = set(record["backends"]) - {"blocky", "coredns"}
if unknown:
raise ValueError(f"unknown split DNS backends {sorted(unknown)}")
for record in public_records:
require_fields(record, "name", "type", "values", "proxied", "terraform")
terraform = record["terraform"]
if terraform.get("managed") and not terraform.get("resource_name"):
raise ValueError(f"managed Terraform record needs resource_name: {record['name']}")
if len(record["values"]) != 1:
raise ValueError(f"Cloudflare Terraform supports one value per record: {record['name']}")
return inventory
def require_fields(record: dict, *fields: str) -> None:
missing = [field for field in fields if field not in record]
if missing:
raise ValueError(f"record missing {', '.join(missing)}: {record!r}")
def environment() -> Environment:
return Environment(
loader=FileSystemLoader(DNS_DIR / "templates"),
undefined=StrictUndefined,
autoescape=False,
keep_trailing_newline=True,
trim_blocks=True,
lstrip_blocks=True,
variable_start_string="[[",
variable_end_string="]]",
block_start_string="[%",
block_end_string="%]",
)
def marker(target: str, end: bool = False) -> str:
return (END if end else BEGIN).replace("[[ target ]]", target)
def replace_block(original: str, target: str, rendered: str) -> str:
begin = marker(target)
end = marker(target, end=True)
if original.count(begin) != 1 or original.count(end) != 1:
raise ValueError(f"expected exactly one generated block for {target}")
prefix, remainder = original.split(begin, 1)
_, suffix = remainder.split(end, 1)
indent = prefix.rsplit("\n", 1)[-1]
body = rendered.rstrip("\n")
return f"{prefix}{begin}\n{body}\n{indent}{end}{suffix}"
def outputs(inventory: dict) -> dict[Path, str]:
env = environment()
split_records = inventory["split_horizon"]["records"]
public_records = inventory["public"]["records"]
result = {}
blocky_path = ROOT / "apps/blocky/config.yml"
blocky = env.get_template("blocky.yml.j2").render(
records=[record for record in split_records if "blocky" in record["backends"]]
)
result[blocky_path] = replace_block(blocky_path.read_text(), "blocky", blocky)
coredns_path = ROOT / "platform/k3s/coredns-custom.yaml"
coredns = env.get_template("coredns.yaml.j2").render(
records=[record for record in split_records if "coredns" in record["backends"]]
)
result[coredns_path] = replace_block(coredns_path.read_text(), "coredns", coredns)
terraform_path = ROOT / "infrastructure/cloudflared/terraform/dns.generated.tf"
terraform = env.get_template("cloudflare.tf.j2").render(
records=[record for record in public_records if record["terraform"]["managed"]]
)
result[terraform_path] = terraform
return result
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--check", action="store_true", help="fail when generated files differ")
args = parser.parse_args()
try:
rendered_outputs = outputs(load_inventory())
except (OSError, ValueError, yaml.YAMLError) as exc:
print(f"dns generation failed: {exc}", file=sys.stderr)
return 2
changed = False
for path, expected in rendered_outputs.items():
actual = path.read_text() if path.exists() else ""
if actual == expected:
continue
changed = True
if args.check:
print("".join(difflib.unified_diff(
actual.splitlines(keepends=True),
expected.splitlines(keepends=True),
fromfile=str(path.relative_to(ROOT)),
tofile=f"{path.relative_to(ROOT)} (generated)",
)))
else:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(expected)
print(f"rendered {path.relative_to(ROOT)}")
return 1 if args.check and changed else 0
if __name__ == "__main__":
raise SystemExit(main())
+35 -5
View File
@@ -10,19 +10,28 @@ homelab_dns:
- { zone: ad.ddupan.top, name: pve1, type: A, values: [192.168.10.4] } - { zone: ad.ddupan.top, name: pve1, type: A, values: [192.168.10.4] }
- { zone: ad.ddupan.top, name: pve2, type: A, values: [192.168.10.7] } - { zone: ad.ddupan.top, name: pve2, type: A, values: [192.168.10.7] }
- { zone: ad.ddupan.top, name: pve3, type: A, values: [192.168.10.9] } - { zone: ad.ddupan.top, name: pve3, type: A, values: [192.168.10.9] }
- { zone: ad.ddupan.top, name: sandbox1, type: A, values: [10.60.0.11] }
- { zone: ad.ddupan.top, name: sandbox2, type: A, values: [10.60.0.12] }
- { zone: ad.ddupan.top, name: sandbox-k8s, type: A, values: [10.60.0.13] }
- { zone: ad.ddupan.top, name: retrolab, type: A, values: [10.60.0.10] } - { zone: ad.ddupan.top, name: retrolab, type: A, values: [10.60.0.10] }
- { zone: ad.ddupan.top, name: grafana, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: metrics-write, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: netbox, type: A, values: [192.168.10.127] } - { zone: ad.ddupan.top, name: netbox, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: nats, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: nexus, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: s3, type: A, values: [192.168.10.127] } - { zone: ad.ddupan.top, name: s3, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: spire-oidc, type: A, values: [192.168.10.127] } - { zone: ad.ddupan.top, name: spire-oidc, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: spire-server, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: zot, type: A, values: [192.168.10.127] } - { zone: ad.ddupan.top, name: zot, type: A, values: [192.168.10.127] }
- { zone: ad.ddupan.top, name: zot-push, type: A, values: [192.168.10.127] }
split_horizon: split_horizon:
# LAN and pod resolvers should eventually render the same set from here. # backends records the current adoption boundary. obj is deliberately not
# Adoption of Blocky/CoreDNS is deliberately a separate change. # emitted to CoreDNS yet, preserving the current pod resolver behaviour.
records: records:
- { name: git.ddupan.top, type: A, values: [192.168.10.127] } - { name: git.ddupan.top, type: A, values: [192.168.10.127], backends: [blocky, coredns] }
- { name: auth.ddupan.top, type: A, values: [192.168.10.127] } - { name: auth.ddupan.top, type: A, values: [192.168.10.127], backends: [blocky, coredns] }
- { name: obj.ddupan.top, type: A, values: [192.168.10.127] } - { name: obj.ddupan.top, type: A, values: [192.168.10.127], backends: [blocky] }
public: public:
# Names expected at Cloudflare. Terraform adoption is a separate change; # Names expected at Cloudflare. Terraform adoption is a separate change;
@@ -32,15 +41,36 @@ homelab_dns:
type: CNAME type: CNAME
values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com] values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com]
proxied: true proxied: true
terraform:
managed: true
resource_name: auth
- name: git.ddupan.top - name: git.ddupan.top
type: CNAME type: CNAME
values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com] values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com]
proxied: true proxied: true
terraform: { managed: false }
- name: obj.ddupan.top - name: obj.ddupan.top
type: CNAME type: CNAME
values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com] values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com]
proxied: true proxied: true
terraform: { managed: false }
- name: e5renew.ddupan.top - name: e5renew.ddupan.top
type: CNAME type: CNAME
values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com] values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com]
proxied: true proxied: true
terraform: { managed: false }
# OCI 主机直接解析公网 IP,SSH 不经过 Cloudflare 代理。
- name: oci-arm.ddupan.top
type: A
values:
- 129.225.138.179
proxied: false
ttl: 300
terraform: { managed: false }
- name: oci-amd.ddupan.top
type: A
values:
- 129.225.176.134
proxied: false
ttl: 300
terraform: { managed: false }
@@ -0,0 +1,3 @@
[% for record in records %]
[[ record.name ]]: [[ record['values'][0] ]]
[% endfor %]
@@ -0,0 +1,11 @@
# Generated by infrastructure/dns/generate.py. Do not edit directly.
[% for record in records %]
resource "cloudflare_dns_record" "[[ record.terraform.resource_name ]]" {
zone_id = var.zone_id
name = "[[ record.name ]]"
type = "[[ record.type ]]"
content = "[[ record['values'][0] ]]"
proxied = [[ record.proxied | lower ]]
ttl = [[ record.ttl | default(1) ]]
}
[% endfor %]
@@ -0,0 +1,12 @@
[% for record in records %]
[[ record.name | replace('.', '-') ]].server: |
[[ record.name ]]:53 {
errors
template IN A {
answer "{{ .Name }} 60 IN A [[ record['values'][0] ]]"
}
template IN AAAA {
rcode NOERROR
}
}
[% endfor %]
+27
View File
@@ -0,0 +1,27 @@
# Docker 地址池与 DN42
DN42 使用 `172.20.0.0/14`。laptop 的 Docker 默认地址池改为 `172.28.0.0/16`,
按 `/24` 分配新 bridge,避免本地直连路由与 DN42 前缀重叠。
`ansible/site.yml` 合并现有 daemon.json,保留 NVIDIA runtime;先热加载 live-restore,
再重启 daemon 使默认地址池生效,避免已有容器随 daemon 停止。
```bash
ANSIBLE_LOCAL_TEMP=/tmp/ansible-docker ansible-playbook -i localhost, infrastructure/docker/ansible/site.yml --check --diff
ANSIBLE_LOCAL_TEMP=/tmp/ansible-docker ansible-playbook -i localhost, infrastructure/docker/ansible/site.yml
```
已有网络不会自动换地址。本次单独迁移结果:
| 网络 | 原地址 | 当前地址/状态 |
|---|---|---|
| blocky_default | 172.20.0.0/16 | 172.28.0.0/24,Compose 明确声明 |
| ps3netsrv_default | 172.21.0.0/16 | 172.28.1.0/24,Compose 明确声明 |
| research-auto_default | 172.22.0.0/16 | 172.28.2.0/24,仓库外 research-auto Compose 明确声明 |
| netboot_default | 172.23.0.0/16 | 删除无端点的遗留网络;netboot 两个容器均使用 host 网络 |
Blocky 健康检查与 DNS 查询通过;ps3netsrv 运行,游戏数据挂载保留。
research-auto 的 postgres 容器仅 create、未启动,原命名卷 `research-auto_postgres_data` 保留。
旧运行容器曾引用仓库重组前的 `/home/panxiao81/services/<app>` 挂载路径;
本次 Blocky 已用 `apps/blocky` 路径重建,netboot 等未重建的容器仍需在重建时使用当前 Compose。
不要在未检查 bind mount 路径的情况下关闭 live-restore 并重启所有容器。
@@ -0,0 +1,2 @@
[defaults]
local_tmp = /tmp/ansible-docker
+61
View File
@@ -0,0 +1,61 @@
---
- name: 为 DN42 排除 Docker 地址池重叠
hosts: localhost
connection: local
become: true
gather_facts: false
vars:
ansible_python_interpreter: /usr/bin/python3
docker_address_pools:
- base: 172.28.0.0/16
size: 24
tasks:
- name: 读取现有 Docker 配置并保留 runtimes 等设置
ansible.builtin.slurp:
src: /etc/docker/daemon.json
register: docker_config
no_log: true
# 先让旧 daemon 知道 live-restore,随后重启才能保留运行容器。
- name: 启用 live-restore
ansible.builtin.copy:
content: "{{ (docker_config.content | b64decode | from_json | combine({'live-restore': true})) | to_nice_json }}\n"
dest: /etc/docker/daemon.json
owner: root
group: root
mode: '0644'
backup: true
validate: /usr/bin/dockerd --validate --config-file %s
register: live_restore_config
- name: 热重载 live-restore
ansible.builtin.systemd_service:
name: docker
state: reloaded
when: live_restore_config.changed and not ansible_check_mode
- name: 确认运行中的 daemon 已启用 live-restore
ansible.builtin.command: docker info --format '{{ '{{' }}.LiveRestoreEnabled{{ '}}' }}'
register: live_restore_status
changed_when: false
retries: 5
delay: 2
until: live_restore_status.stdout == 'true'
when: not ansible_check_mode
- name: 配置 DN42 范围之外的默认地址池
ansible.builtin.copy:
content: "{{ (docker_config.content | b64decode | from_json | combine({'live-restore': true, 'default-address-pools': docker_address_pools})) | to_nice_json }}\n"
dest: /etc/docker/daemon.json
owner: root
group: root
mode: '0644'
backup: true
validate: /usr/bin/dockerd --validate --config-file %s
register: docker_pool_config
- name: 保留运行容器并重启 daemon 使地址池生效
ansible.builtin.systemd_service:
name: docker
state: restarted
when: docker_pool_config.changed and not ansible_check_mode
+110 -10
View File
@@ -1,13 +1,113 @@
# OCI infrastructure recovery # OCI 云上基础设施
The original Terraform source is currently unavailable. The likely authoritative `terraform/` 是独立 Terraform 根模块,从 OCI Object Storage 中的现有 state 恢复。
state remains in OCI Object Storage. Reconstruct configuration here only after Terraform 管理云 API 资源;实例内的软件、Kubernetes 和操作系统配置不在该 state 中。
taking an encrypted independent state backup.
Safety requirements: ## 资源与来源
- preserve the existing state lineage and serial; - Region:`ap-osaka-1`,compartment 为 tenancy 根。
- reproduce the current VM and public-network design first; - State:namespace `axckv9ylwqxr`,bucket `oci-k8s-free-tier-tfstate`,key `terraform.tfstate`。
- reach a zero-change plan before any apply; - 恢复源:2026-08-15 12:26:24 UTC 对象,25,261 字节,serial `249`,
- protect the instance and boot volume from destruction; lineage `c945c6c4-ee01-d7b7-23f5-37207ea60609`,Terraform `1.15.8`。
- treat migration to a private subnet as a separate reviewed change. - 7 个受管资源保留原地址:`oci_core_instance.vm`、`oci_core_vcn.vcn`、
`oci_core_subnet.public`、`oci_core_internet_gateway.igw`、`oci_core_route_table.public`、
`oci_core_security_list.public`、`oci_limits_quota.free_tier_quota`。
- 2 个数据源:`oci_core_images.ubuntu`、`oci_identity_availability_domains.ads`;保留 4 个原输出。
- VM:`homelab-vm`,A1 Flex,2 OCPU / 12 GB RAM / 100 GB 启动盘,
私网 `10.0.0.124`,恢复时公网 `129.225.138.179`。
- VCN `10.0.0.0/16`,公共子网 `10.0.0.0/24`,默认路由经 Internet Gateway;
入站保留 TCP 22、UDP 41641、ICMP type 3/code 4,出站全部允许。
- 配额语句保留 A1 4 核 / 24 GB、10 个卷、200 GB 总存储限制;这些语句不是费用保证。
- Bucket 自身不在 state 内,不由此根模块管理。
## 恢复设计
原变量、模块意图、provider 精确版本和生命周期规则无法从 state 完整恢复。
本次选择并锁定 `oracle/oci 9.1.0`,提交 lockfile;这不是声称找回了原 provider 版本。
资源间的 VCN、路由表、安全列表、子网和 DHCP 引用已重建。
启动镜像固定为现有 image OCID,避免数据源选中更新镜像导致 VM 替换。
新增 `prevent_destroy` 保护现有 VM;没有用 `ignore_changes` 掩盖配置差异。
Provider 使用本机 `~/.oci/config` 的 `DEFAULT` profile,可用变量覆盖 profile 和 region。
metadata 经敏感变量传入,只保存在忽略的 `terraform.tfvars.json`,不进入版本库。
State、plan、metadata 变量与 `.terraform/` 都不得提交;plan JSON 同样可能含敏感数据。
保留原目录的恢复要求:保留 lineage/serial,先复现当前 VM 和公共网络设计,
实际基础设施变更前达到严格零变更,并保护实例及启动盘;迁移私有子网须单独评审。
已用既有 GPG 加密子密钥 `5A6A04D1B216C64E` 创建独立加密源备份
`.recovery/source.tfstate.gpg`;本地原始副本的 lineage/serial 保持不变。
## 日常维护
旧 Terraform/CI 已由维护者确认停用,当前仓库已接管原 OCI Object Storage state。
`versions.tf` 使用 OCI backend,直接连接原对象,未迁移本地恢复副本。
认证沿用本机 OCI CLI 的 `DEFAULT` profile;不要在配置中写密钥。
```bash
cd infrastructure/oci/terraform
terraform init
terraform validate
terraform plan -input=false -out=change.tfplan
# 核对计划后执行:
terraform apply change.tfplan
```
新 checkout 需从受限 state 副本提取 metadata 至被忽略的 `terraform.tfvars.json`。
`prepare-local-state.py` 可执行这一步,同时保留 `.recovery/terraform.tfstate` 备份;
当前 backend 使用远端对象,`.recovery/` 中的副本不再参与日常 plan/apply。
下载前设置 `umask 077`,并使用 GPG 加密源备份;不要把本地副本上传覆盖远端。
## 恢复验证记录
Terraform 1.15.8、OCI provider 9.1.0 验证通过。
恢复计划唯一更新是 VM metadata 的敏感标记,plan JSON 中 before/after 值相同。
经授权 apply 后,本地完整刷新 plan 达到 `No changes`。
旧 state 在 provider 刷新后补充 VM shape/VNIC、subnet IPv4 CIDR、route type 字段,
这些读回差异没有产生基础设施修改计划。
接管时重新连接原远端对象,因此新增实例计划也包含原 VM 的同一敏感标记归一化。
## AMD 实例
`oci_core_instance.amd` 配置为 `homelab-amd`,`VM.Standard.E2.1.Micro`,1 GB RAM,
Ubuntu 24.04 x86_64,50 GB / 10 VPU 启动盘。复用现有公共子网和 SSH 公钥,
没有复制 A1 实例的其他初始化内容,也不部署 Kubernetes。
镜像固定为 `Canonical-Ubuntu-24.04-2026.08.25-0`,实例具有 `prevent_destroy` 保护。
创建前 API 确认大阪为 home region、机型计费类型为 `ALWAYS_FREE`,AMD 配额剩余 2 台。
存储盘点只有 A1 的 100 GB 启动盘;已核对两个启动盘合计 150 GB,200 GB 免费额度内剩余 50 GB。
免费额度跨启动盘和块存储共享;后续新增资源仍需重新核对实际占用。
实例已创建并确认 `RUNNING`,私网 `10.0.0.158`,公网 `129.225.176.134`。
登录命令:`ssh [email protected]`(本次未验证 SSH 登录)。
远端 state 已保存;创建后完整刷新 plan 为 `No changes`,退出码 0。
参考:[Always Free 资源](https://docs.oracle.com/en-us/iaas/Content/FreeTier/freetier_topic-Always_Free_Resources.htm)、
[OCI provider 认证](https://docs.oracle.com/en-us/iaas/Content/dev/terraform/configuring.htm)、
[OCI backend 配置](https://developer.hashicorp.com/terraform/language/backend/oci)。
## DNS 登录入口
- ARM:`ssh [email protected]`
- AMD:`ssh [email protected]`
公网 A 记录声明位于 `../dns/records.yml`,在 Cloudflare 上关闭代理,TTL 300 秒。
当前通过 DNS API 管理,未加入 OCI Terraform state;公网 IP 变化时需同步记录。
## WireGuard/BGP 与 DN42
家中端点已从 laptop 迁移到 VyOS `192.168.10.2`,AMD 与 VyOS 同属 AS4242421811,
通过独立 WireGuard 接口建立双栈 iBGP;laptop 保留原有 NEC BGP 和 OSPF,按路由经 VyOS 转发。
Ansible 配置与运行方法见 [ansible/README.md](ansible/README.md)。
注册前缀 `172.21.111.160/27`、`fdd0:98df:15b0::/48` 已在内部路由中准备:
VyOS `.161` / `::1`,AMD `.162` / `::2`,使用 loopback /32、/128。
首个外部 DN42 peer 已接入 RoutedBits Osaka(AS4242420207),AMD 使用独立 `wg-dn42-1`
和单 IPv6 MP-BGP 会话承载双栈;详见 [Ansible runbook](ansible/README.md)。
外部明细留在 AMD,`172.20.0.0/14`、`fd00::/8` 汇总经 iBGP 下发 VyOS;IPv4 /14 再经 OSPF 下发 LAN。
LAN 的 DN42 IPv6 /64 地址由 VyOS SLAAC 下发,fd00::/8 通过 RA RIO 分发,不通告 IPv6 默认路由。内部家中/OCI 业务路由不得向外部 DN42 邻居通告。
Terraform 管理 AMD NSG、VNIC 转发与 VCN 回程;Ansible 管理路由器/主机及 ARM 的 Tailscale
回程例外。Docker 与 DN42 的 /14 地址重叠已迁出,见 [Docker runbook](../docker/README.md)。
VyOS 的 `wg42` 主 IPv4 已改为注册的 `172.21.111.161/32`,内部 BGP 改为
单 link-local IPv6 会话承载双栈。三个 LAN 私网到 DN42 /14 由 VyOS 定向 masquerade,
排除本 AS /27;OCI 业务保持原源地址。IPv6 不做 NAT,由 `ansible/dn42-ra.yml` 管理三个 LAN 的 SLAAC 与专用路由通告。
+232
View File
@@ -0,0 +1,232 @@
# VyOS ↔ OCI:WireGuard 与 DN42 内部 BGP
站点端已从 laptop 迁移到 VyOS `192.168.10.2`。VyOS 位于双层 NAT 后,主动连接
`oci-amd.ddupan.top:51820`,keepalive 25 秒;LAN 访问 DN42 时在 VyOS 做定向 masquerade。
主机配置由 Ansible 管理;OCI NSG、VNIC 转发和 VCN 回程路由由旁边的 Terraform 管理。
| 节点 | 接口 | 传输 IPv4 | 传输 IPv6 | ASN |
|---|---|---|---|---|
| VyOS | wg42 | 172.21.111.161/32 | fe80::1811:1/64(BGP) | 4242421811 |
| AMD | wg-oci | 10.255.254.1/30(历史传输地址) | fe80::1811:2/64(BGP) | 4242421811 |
AMD 显式设置 `fe80::1811:2/64`;VyOS BGP 使用固定 `fe80::1811:1/64`,双方 AllowedIPs 包含
`fe80::/64`。FRR 在 ULA 建邻时仍需接口具有 link-local 下一跳地址。
MTU 1380,Linux 使用 `Table = off`,WireGuard AllowedIPs 用于选 peer 与源地址校验,
站点路由由 iBGP 安装。内外部均使用单条 IPv6 link-local 会话承载双 AFI,并启用 extended-nexthop。
内部旧 IPv4/ULA BGP 邻居已退役;ULA 传输地址保留用于路由下一跳,不再用于建邻。
VyOS 10.2.4 通过 `OCI-MP-IN` 的 `ipv6-next-hop prefer-global` 优先采用通告中的 ULA 下一跳,
避免 link-local NHT 显示 overlay index unresolved、BGP 已建邻但路由未安装。
## 地址与通告边界
已注册 `172.21.111.160/27`、`fdd0:98df:15b0::/48`:
| 节点 | 路由器 IPv4 | 路由器 IPv6 |
|---|---|---|
| VyOS(IPv4 在 wg42,IPv6 在 lo) | 172.21.111.161/32 | fdd0:98df:15b0::1/128 |
| AMD loopback | 172.21.111.162/32 | fdd0:98df:15b0::2/128 |
VyOS 为注册的 /27、/48 建立 distance 254 的 blackhole 聚合路由,保证精确前缀存在,
避免未分配地址落入默认路由。已分配的本地地址及 AMD 的 /32、/128 优先于聚合。
内部通告严格过滤:
- VyOS → AMD:`192.168.10.0/24`、`10.60.0.0/24`、`10.61.0.0/24` 和注册 /27、/48。
- AMD → VyOS:`10.0.0.0/24`、AMD 的注册 /32、/128,以及 DN42 汇总 `172.20.0.0/14`、`fd00::/8`。
- VyOS 向 LAN OSPF 只重分发 OCI /24、DN42 /14 和注册 IPv4 /27,使用精确 route-map、E1 metric。
既有直连 LAN/SDN 的 OSPF area 声明保持不变,不使用泛化的 redistribute connected。
**首个外部 peer 为 RoutedBits Osaka(AS4242420207),由 `dn42.yml` 单独管理。** 外部邻居使用独立的 import/export
过滤,只对外通告注册 /27、/48;禁止把上述内部业务前缀的过滤器复用到外部邻居。
入口使用 DN42 指南的保留地址、互联网络与前缀长度规则(IPv6 /44–/64),
并优先拒收本 AS、LAN、OCI 前缀;尚未配置注册表 ROA 校验。外部明细留在 AMD,汇总通过 iBGP 下发 VyOS;IPv4 汇总再经 OSPF 下发 LAN。
每个外部 WireGuard peer 使用独立接口,BGP 可复用本机 loopback 地址。
单接口多个 WireGuard peer 要求可明确区分的 AllowedIPs;多家 peer 都提供同一 DN42 路由范围时,
使用独立接口让 BGP 决定出口,避免相同 AllowedIPs 抢占 peer。
## 执行
```bash
cd infrastructure/oci/ansible
export SSH_AUTH_SOCK="$(gpgconf --list-dirs agent-ssh-socket)"
ansible-playbook site.yml --check --diff
ansible-playbook site.yml
ansible-playbook site.yml # 复跑应 changed=0
```
`vyos.yml` 先在路由器本机生成密钥并保存,再交换公钥并增量应用 VyOS set 命令。
VyOS native config 含私钥,因此相关模块使用 `no_log`,不把配置备份到 Git 或打印出来。
AMD 私钥在 `/etc/wireguard/wg-oci.key`(0600),由本机 PostUp 加载,不返回控制机。
首次 check mode 无法生成私钥,因此会跳过依赖不存在公钥的 Linux 配置渲染。
`retire-laptop.yml` 是迁移收尾:先检查 VyOS 邻居,再停止 laptop 的 wg-oci,
删除试验邻居、三个新增 network 语句及专用防火墙链,保留 laptop 原 AS65001 ↔ NEC AS65000
会话、原有 VPN 路由和 OSPF。旧私钥保留在 laptop 受限文件中,隧道和防火墙单元已禁用。
Linux 端 FRR 通过 `vtysh -f` 应用独立配置片段,另行 `write memory` 持久化;
文件变更时重建受管邻居,未变更的重跑不重置会话。移除前缀时还须显式删除已退出管理的
`network` 语句,不能只追加 set 命令,也不能清空整份 BGP 配置。
## OCI 回程与 Tailscale
VCN 虚拟路由器不参与主机间 iBGP;三个家中业务前缀的静态回程指向 AMD Private IP OCID。
首次启用时必须先将 VNIC `skip_source_dest_check` 设为 true,OCI 才接受该私有 IP 为路由目标。
当前 Terraform 的 route table → subnet → instance 依赖使首次引导需要先设置该标志。
重建 AMD 后须重新查询并更新 `amd_router_private_ip_ocid`。
ARM 接受 Tailscale 家中子网路由,table 52 原本抢走回程。
Ansible 在 ARM 设置 priority 5101–5103、仅匹配这三个目的前缀的 `lookup main` 规则,
让这些流量使用 OCI 网关→AMD;其他 Tailscale 地址保持原路径。
AMD 的 Zebra route-map 仅为本机发起的 BGP 业务流量选择 `10.0.0.158` 源地址,不改写转发源地址。
## Docker 地址冲突
DN42 使用 `172.20.0.0/14`。原 laptop Docker 的四个 /16 已清除,默认池改为 `172.28.0.0/16`,
详见 [Docker runbook](../../docker/README.md)。最长前缀匹配决定路由,但不能解决两套网络实际
地址重叠;不能只检查本 AS 注册的 /27 而忽略其他 DN42 注册前缀。
## 验证与停用
```bash
# VyOS operational mode
show interfaces wireguard wg42 summary
show bgp summary
show ip route 172.21.111.162/32
show ipv6 route fdd0:98df:15b0::2/128
# AMD
sudo wg show wg-oci latest-handshakes
sudo vtysh -c 'show bgp summary'
ping 192.168.10.4
ping -I 172.21.111.162 172.21.111.161
ping -6 -I fdd0:98df:15b0::2 fdd0:98df:15b0::1
```
业务网段测试应使用业务源地址;DN42 loopback 测试使用注册地址。
LAN 已通过 `dn42-ra.yml` 部署 DN42 SLAAC 与专用 RIO,不发布 IPv6 默认路由;OCI VCN 保持原配置。
停用或回滚须同时处理 BGP、WireGuard、OSPF 重分发、VCN 回程和 ARM 例外规则,
不能仅停止隧道后留下静态回程指向不可达节点。
参考:[WireGuard](https://www.wireguard.com/quickstart/)、
[VyOS WireGuard](https://docs.vyos.io/en/1.5/configuration/interfaces/wireguard.html)、
[DN42 入门](https://www.dn42.dev/howto/Getting-Started)。
## 迁移验收
2026-09-14:
- VyOS IPv4 iBGP 收到 OCI /24、AMD /32,通告三个家中前缀与注册 /27;IPv6 会话双方各收到一条前缀。
- 两端 DN42 loopback IPv4、IPv6 互通;IPv6 本次采样约 3.1 ms。
- laptop 的 `10.0.0.0/24` 与 `172.21.111.160/27` 经 OSPF 指向 `192.168.10.2`,不再使用旧隧道。
- PVE1 可访问 ARM 私网和 AMD 的 DN42 IPv4;AMD 可访问 PVE1、`10.60.0.1`、`10.61.0.1`。
- laptop 旧 WireGuard 和专用防火墙 service 已停止/禁用,原 NEC BGP 会话保留。
LAN 网关可能返回 ICMP Redirect,提示客户端将 VyOS 作为同网段下一跳;这是既有 LAN
拓扑的正常结果,没有为此修改客户端或网关的 redirect 策略。
## AMD 外部 DN42 首个接口
2026-09-15 已通过 `ansible-playbook dn42.yml` 准备独立监听:
- 接口:`wg-dn42-1`,UDP endpoint:`oci-amd.ddupan.top:51821`(`129.225.176.134:51821`)。
- 公钥:`YQ/X3QmNocnr0u4aUm5qhcV328StSNtg+ULd9AKCdhQ=`。
- 私钥仅保存在 AMD `/etc/wireguard/wg-dn42-1.key`,由 root 受限目录保护,不返回控制机。
- `wg-quick@wg-dn42-1` 开机启动;主机入站规则随接口启停,OCI NSG 规则由 Terraform 管理。
- Link-local:`fe80::1811:2/64`,本机作用域为 `%wg-dn42-1`。首个 peer 使用单条 IPv6 BGP 会话承载双栈(MP-BGP + RFC 8950 extended next hop)。
- 对端:AS4242420207,`router.osa1.routedbits.com:51811`,link-local `fe80::207`。
- 对端公钥:`96PwUEGi/ijdmKO+IjuZ+J6DeykuTRukZD5atajfeH4=`。
- WireGuard 使用 `Table = off`;AllowedIPs 为 `fe80::/64, 172.20.0.0/14, 10.0.0.0/8, 172.31.0.0/16, fd00::/8`,keepalive 25 秒。
- 双 AFI 共用 IPv6 TCP 会话并启用 extended-nexthop;每 AFI maximum-prefix 10000,出口只允许注册 /27 和 /48。
- 本机发起 DN42 流量使用注册地址作为 preferred source,不做 NAT。
- Ubuntu 自带 FRR 8.4.4 不满足 [DN42 FRR 指南](https://dn42.dev/howto/frr) 的 link-local 版本要求。
`tasks/frr-dn42.yml` 使用官方 frr-10.7 软件源,固定 10.7.1;升级前配置仅在 AMD `/var/backups/frr-before-dn42` 备份。
原 `wg-oci:51820` 继续承载与 VyOS 的内部互联。
2026-09-15 接入验收:WireGuard 握手与 `fe80::207%wg-dn42-1` 连通;FRR 10.7.1
会话 Established,双方已协商 IPv4/IPv6 AFI 与 extended nexthop。采样接收 IPv4 1174、IPv6 1273 条,
对外仅通告 `172.21.111.160/27`、`fdd0:98df:15b0::/48`。AMD 无需显式指定源地址,
即可访问对端 `172.20.19.78`、`fdb1:e72a:343d::f`,各 3/3 回复,约 1 ms;
IPv4 内核路由下一跳为 `via inet6 fe80::207 dev wg-dn42-1`。内部 VyOS 双栈 BGP 会话已恢复。
## 向 LAN 分发 DN42 汇总
`site.yml` 在 AMD 生成 `172.20.0.0/14`、`fd00::/8` 的 BGP aggregate,并通过内部精确
prefix-list 通告 VyOS。不使用全局 `summary-only`,避免抑制对外通告的注册 /27、/48;
对外出口过滤仍只允许这两个注册前缀。AMD 保留外部明细与汇总丢弃路径,无匹配明细的流量
在 AMD 丢弃。汇总只要仍有覆盖的 BGP 明细(包括本 AS 注册前缀)就可能存在,不能作为外部
peer 在线状态指示。自己的 /27、/48 更具体,继续指向本地站点。
VyOS 将 IPv4 /14 通过既有 OSPF E1 分发到 LAN;IPv6 /8 通过 RA 的 RIO 分发至接受该选项的客户端。
LAN 的 DN42 IPv6 地址通过 SLAAC 自动分配。
当前 LAN IPv4 客户端访问 DN42 /14 由 VyOS 定向 masquerade 到 `172.21.111.161`;
IPv6 使用 SLAAC 分配的注册地址直接路由。
AMD 只放行注册前缀在内部 `wg-oci` 与外部 `wg-dn42-1` 之间转发。
2026-09-15 汇总验收:VyOS 双栈 BGP 分别收到 /14、/8,下一跳为 AMD;
laptop 的 `172.20.0.0/14` 为 OSPF 路由,经 `192.168.10.2 dev br0`。
使用 VyOS 注册的 /32、/128 作为源,经 AMD 访问 RoutedBits 的双栈地址各 3/3 回复,
约 4–5 ms。`site.yml` 执行成功,ARM 无变更;LAN 地址配置保持原状。
## DN42 DNS 转发
`ansible-playbook dn42.yml dn42-dns.yml` 管理入口路由和 VyOS DNS。LAN 的 Blocky
(`192.168.10.127`) 将 `.dn42`、172.20–23 的 IPv4 反向区和 `d.f.ip6.arpa`
转给 VyOS `192.168.10.2:53`。VyOS 仅接受三个内部 LAN 网段,使用注册地址
`172.21.111.161` / `fdd0:98df:15b0::1` 发起递归转发,不做 NAT。
| 上游 | IPv4 | IPv6 |
|---|---|---|
| a0.recursive-servers.dn42 | 172.20.0.53 | fd42:d42:d42:54::1 |
| a3.recursive-servers.dn42 | 172.23.0.53 | fd42:d42:d42:53::1 |
两个上游的双栈地址均配置,递归请求设置 RD;转发域配置 NTA,避免使用公网根信任链
验证 DN42 私有命名空间。本地转发器不声明已完成 DN42 DNSSEC 信任锚验证。
IPv4 anycast /32 需要允许四个 `172.2x.0.0/24` 中的 /28–/32,不能只保留 /14 的 /21–/29。
新增互联网络明细仅由 AMD 接收;LAN 汇总仍是既定 /14、通过 RIO 通告的 IPv6 /8。
NEC 备用 DNS 与 k3s CoreDNS 本次未修改。
DNS 验收(2026-09-15):从 VyOS 用注册地址直查 a0/a3 的四个双栈地址均获得回复。
LAN 查询 Blocky 可得到 a0 的 A、a3 的 AAAA;AD 与公网域名正常,Blocky healthy。
VyOS DNS playbook 复跑 changed=0;AMD 已应用完整入口规则。
## VyOS LAN 到 DN42 masquerade
规则 18100 排除本 AS `172.21.111.160/27`;18110、18120、18130 分别匹配三个 LAN
源网段,目的仅 `172.20.0.0/14` 且出口 `wg42`,translation 为 `masquerade`。
为使 masquerade 选中注册地址,`172.21.111.161/32` 从 lo 移到 wg42,并移除
`10.255.254.2/30`;不能在 wg42 仍以传输私网地址为主 IPv4 时直接启用 masquerade。
现有 OCI 业务互联继续保留原源地址,DNS 转发使用的 `172.21.111.161` 保持可用。
AMD 仍仅允许 DN42 注册源前缀进入外部隧道,不在 AMD 做第二次 NAT。
2026-09-15 masquerade 验收:laptop `192.168.10.127` 到 RoutedBits `172.20.19.78`
3/3 回复约 4.7 ms,VyOS NAT 表显示转换为 `172.21.111.161`;到 ARM `10.0.0.124`
3/3 回复约 4.2 ms,NAT 表确认保留 `192.168.10.127`。
内部 link-local 单会话双 AFI 已建立;外部 peer 保持独立接口和精确出口。
## LAN DN42 IPv6 RA(不发布默认路由)
执行 `ansible-playbook dn42-ra.yml`,为三个 LAN 启用 SLAAC:
| LAN | VyOS 接口 | 前缀 | 路由器地址 |
|---|---|---|---|
| 192.168.10.0/24 | eth0 | fdd0:98df:15b0:10::/64 | fdd0:98df:15b0:10::1 |
| 10.60.0.0/24 | eth1 | fdd0:98df:15b0:60::/64 | fdd0:98df:15b0:60::1 |
| 10.61.0.0/24 | eth2 | fdd0:98df:15b0:61::/64 | fdd0:98df:15b0:61::1 |
Router Lifetime 为 **0**,不发布 `::/0`;PIO 开启 on-link 与 autonomous 标志,
preferred lifetime 14400 秒、valid lifetime 86400 秒。RIO 只包含 `fd00::/8`,
有效期 180 秒;RA 周期 10–30 秒。不发布 RDNSS、DNSSL、DHCPv6 标志或链路 MTU,
保留客户端现有 DNS 和公网出口。IPv6 经 VyOS→AMD→DN42 直接路由,不做 NAT66。
现有 /48 对外通告及内部回程已覆盖这三个 /64,无须泄漏每个 LAN 的明细到外部。
客户端须支持并接受 RIO。当前 laptop 的 br0 `accept_ra=0`、`forwarding=1`,
不会自动配置;本次保留其网络设置。Linux 路由主机如需接受 RA,需要在自身网络管理
配置中显式启用,并允许至少 /8 的 RIO(`accept_ra_rt_info_max_plen`)。
关闭 RA 时注意 PIO 的有效期;不要只删路由器接口地址而留下仍有效的客户端地址。
2026-09-15 RA 抓包验收:主 LAN 收到不带标签的 `:10::/64`;VLAN 100/110
分别携带 `:60::/64`、`:61::/64`。三者 Router Lifetime 均为 0,PIO 为 onlink/auto,
RIO 为 fd00::/8、180 秒,没有 DNS 或默认路由通告。`dn42-ra.yml` 复跑 changed=0。
同一 LAN 上的临时 Linux 测试客户端自动获得地址与 RIO,无 IPv6 默认路由;测试后自动删除。
+8
View File
@@ -0,0 +1,8 @@
[defaults]
inventory = inventory/hosts.yml
roles_path = roles
local_tmp = /tmp/ansible-oci
host_key_checking = True
interpreter_python = auto_silent
[ssh_connection]
ssh_args = -o ControlMaster=auto -o ControlPersist=60s -o StrictHostKeyChecking=accept-new
+23
View File
@@ -0,0 +1,23 @@
---
- name: 配置 VyOS DN42 DNS 转发
hosts: site_routers
gather_facts: false
vars:
dn42_dns_zones:
- dn42
- 20.172.in-addr.arpa
- 21.172.in-addr.arpa
- 22.172.in-addr.arpa
- 23.172.in-addr.arpa
- d.f.ip6.arpa
dn42_dns_servers:
- 172.20.0.53
- 172.23.0.53
- fd42:d42:d42:54::1
- fd42:d42:d42:53::1
tasks:
- name: 配置受限监听、注册地址源与条件转发
vyos.vyos.vyos_config:
lines: "{{ lookup('template', 'templates/vyos-dn42-dns.conf.j2').splitlines() | reject('equalto', '') | list }}"
save: true
comment: Ansible DN42 DNS forwarding
+21
View File
@@ -0,0 +1,21 @@
---
- name: 在 LAN 通告 DN42 SLAAC 地址与专用路由
hosts: site_routers
gather_facts: false
vars:
dn42_ra_lans:
- interface: eth0
prefix: fdd0:98df:15b0:10::/64
address: fdd0:98df:15b0:10::1/64
- interface: eth1
prefix: fdd0:98df:15b0:60::/64
address: fdd0:98df:15b0:60::1/64
- interface: eth2
prefix: fdd0:98df:15b0:61::/64
address: fdd0:98df:15b0:61::1/64
tasks:
- name: 配置接口地址、SLAAC 与 RIO,不通告默认路由或 DNS
vyos.vyos.vyos_config:
lines: "{{ lookup('template', 'templates/vyos-dn42-ra.conf.j2').splitlines() | reject('equalto', '') | list }}"
save: true
comment: Ansible DN42 SLAAC without default route
+98
View File
@@ -0,0 +1,98 @@
---
# 首个外部 peer:RoutedBits Osaka,IPv6 link-local MP-BGP。
- name: 准备 AMD DN42 WireGuard 监听
hosts: oci_amd
become: true
vars:
dn42_interface: wg-dn42-1
dn42_port: 51821
dn42_linklocal: fe80::1811:2/64
dn42_peer_linklocal: fe80::207
dn42_peer_asn: 4242420207
dn42_peer_endpoint: router.osa1.routedbits.com:51811
dn42_peer_public_key: 96PwUEGi/ijdmKO+IjuZ+J6DeykuTRukZD5atajfeH4=
tasks:
- name: 准备 FRR link-local 支持
ansible.builtin.import_tasks: tasks/frr-dn42.yml
- name: 创建受限密钥目录
ansible.builtin.file:
path: /etc/wireguard
state: directory
owner: root
group: root
mode: '0700'
- name: 在 AMD 本机生成独立私钥
ansible.builtin.shell: umask 077; wg genkey > /etc/wireguard/{{ dn42_interface }}.key
args:
creates: /etc/wireguard/{{ dn42_interface }}.key
no_log: true
- name: 写入监听配置
ansible.builtin.copy:
dest: /etc/wireguard/{{ dn42_interface }}.conf
owner: root
group: root
mode: '0600'
content: |
# Ansible 管理;首个 peer 使用 IPv6 link-local MP-BGP + extended next hop。
[Interface]
Address = {{ dn42_linklocal }}
ListenPort = {{ dn42_port }}
MTU = 1380
Table = off
PostUp = wg set %i private-key /etc/wireguard/{{ dn42_interface }}.key
PostUp = iptables -w -C INPUT -p udp --dport {{ dn42_port }} -j ACCEPT 2>/dev/null || iptables -w -I INPUT 1 -p udp --dport {{ dn42_port }} -j ACCEPT
PostDown = iptables -w -D INPUT -p udp --dport {{ dn42_port }} -j ACCEPT
[Peer]
PublicKey = {{ dn42_peer_public_key }}
Endpoint = {{ dn42_peer_endpoint }}
AllowedIPs = fe80::/64, 172.20.0.0/14, 10.0.0.0/8, 172.31.0.0/16, fd00::/8
PersistentKeepalive = 25
notify: 重启 DN42 接口
- name: 启用 DN42 监听
ansible.builtin.systemd_service:
name: wg-quick@{{ dn42_interface }}
enabled: true
state: started
- name: 应用配置
ansible.builtin.meta: flush_handlers
- name: 写入外部 BGP 配置片段
ansible.builtin.template:
src: templates/dn42-bgp.conf.j2
dest: /etc/frr/dn42-routedbits.vtysh
owner: root
group: frr
mode: '0640'
register: dn42_bgp_config
changed_when: dn42_bgp_config.changed or ('(deleted)' in frr_running.stdout)
notify: 应用 DN42 BGP
- name: 应用 BGP 配置
ansible.builtin.meta: flush_handlers
- name: 读取公开信息
ansible.builtin.command: wg show {{ dn42_interface }} {{ item }}
loop:
- public-key
- listen-port
changed_when: false
register: dn42_public_info
- name: 显示公钥和端口
ansible.builtin.debug:
msg: '{{ dn42_public_info.results | map(attribute="stdout") | list }}'
handlers:
- name: 重启 FRR
ansible.builtin.systemd_service:
name: frr
state: restarted
when: not ansible_check_mode
- name: 重启 DN42 接口
ansible.builtin.systemd_service:
name: wg-quick@{{ dn42_interface }}
state: restarted
when: not ansible_check_mode
- name: 应用 DN42 BGP
ansible.builtin.command: vtysh -f /etc/frr/dn42-routedbits.vtysh
notify: 保存 FRR 配置
when: not ansible_check_mode
- name: 保存 FRR 配置
ansible.builtin.command: vtysh -c 'write memory'
when: not ansible_check_mode
@@ -0,0 +1,9 @@
---
wg_interface: wg-oci
wg_port: 51820
wg_mtu: 1380
wg_home_prefixes: [192.168.10.0/24, 10.60.0.0/24, 10.61.0.0/24]
wg_cloud_prefixes: [10.0.0.0/24]
dn42_asn: 4242421811
dn42_ipv4: 172.21.111.160/27
dn42_ipv6: fdd0:98df:15b0::/48
@@ -0,0 +1,12 @@
---
wg_address: 10.255.254.2/30
wg_peer_address: 10.255.254.1
wg_peer_host: oci_amd
wg_endpoint: oci-amd.ddupan.top:51820
wg_keepalive: 25
wg_lan_interface: br0
bgp_asn: 65001
bgp_peer_asn: 4242421811
bgp_router_id: 192.168.10.127
bgp_export: '{{ wg_home_prefixes }}'
bgp_import: '{{ wg_cloud_prefixes }}'
@@ -0,0 +1,34 @@
---
wg_address: 10.255.254.1/30
wg_peer_address: 172.21.111.161
wg_peer_host: vyos_rtr
wg_endpoint: ''
wg_keepalive: 0
wg_lan_interface: ens3
bgp_asn: 4242421811
bgp_peer_asn: 4242421811
bgp_router_id: 10.0.0.158
bgp_export:
- 10.0.0.0/24
- 172.21.111.162/32
bgp_import:
- 192.168.10.0/24
- 10.60.0.0/24
- 10.61.0.0/24
- 172.21.111.160/27
wg_ipv6_address: fdd0:98df:15b0:ffff::1/64
wg_peer_ipv6: fdd0:98df:15b0:ffff::2
bgp_export6:
- fdd0:98df:15b0::2/128
bgp_import6:
- fdd0:98df:15b0::/48
wg_linklocal_address: fe80::1811:2/64
# 仅向内部邻居通告汇总;外部 peer 保持精确出口过滤。
bgp_summary: [172.20.0.0/14]
bgp_summary6: [fd00::/8]
dn42_external_interface: wg-dn42-1
# 内部与外部统一使用 link-local 单会话双 AFI;节点地址仍在 loopback。
bgp_transport_peer: fe80::1811:1
bgp_retired_peers: [10.255.254.2, "fdd0:98df:15b0:ffff::2"]
@@ -0,0 +1,26 @@
---
all:
children:
wireguard_sites:
hosts:
oci_amd:
ansible_host: oci-amd.ddupan.top
ansible_user: ubuntu
oci_routed_hosts:
hosts:
oci_arm:
ansible_host: oci-arm.ddupan.top
ansible_user: ubuntu
retired_wireguard_sites:
hosts:
laptop:
ansible_connection: local
ansible_python_interpreter: /usr/bin/python3
site_routers:
hosts:
vyos_rtr:
ansible_host: 192.168.10.2
ansible_user: vyos
ansible_connection: ansible.netcommon.network_cli
ansible_network_os: vyos.vyos.vyos
ansible_ssh_private_key_file: ~/.ssh/id_ed25519
@@ -0,0 +1,92 @@
---
- name: 确认路由器内部 BGP 已建立
hosts: site_routers
gather_facts: false
tasks:
- name: 确认到 AMD 的邻居
vyos.vyos.vyos_command:
commands: show bgp neighbors 10.255.254.1
register: migration_bgp
changed_when: false
failed_when: "'BGP state = Established' not in migration_bgp.stdout[0]"
- name: 退役 laptop 的 WireGuard 试验端点
hosts: retired_wireguard_sites
become: true
tasks:
- name: 停止并禁用旧隧道
ansible.builtin.systemd_service:
name: wg-quick@wg-oci
state: stopped
enabled: false
- name: 检查原 BGP 试验邻居是否存在
ansible.builtin.command: vtysh -c 'show running-config'
register: laptop_frr
changed_when: false
- name: 只移除本次试验添加的 BGP 节点,保留 NEC 邻居和 OSPF
ansible.builtin.command:
argv:
- vtysh
- -c
- configure terminal
- -c
- router bgp 65001
- -c
- no neighbor 10.255.254.1
- -c
- address-family ipv4 unicast
- -c
- no network 192.168.10.0/24
- -c
- no network 10.60.0.0/24
- -c
- no network 10.61.0.0/24
- -c
- exit-address-family
- -c
- exit
- -c
- no ip protocol bgp route-map OCI-WG-SOURCE
- -c
- no route-map OCI-WG-SOURCE
- -c
- no ip prefix-list OCI-WG-IN
- -c
- no ip prefix-list OCI-WG-OUT
- -c
- end
- -c
- write memory
when: "'neighbor 10.255.254.1 remote-as' in laptop_frr.stdout"
- name: 停止并禁用旧防火墙启动单元
ansible.builtin.systemd_service:
name: oci-wg-firewall
state: stopped
enabled: false
- name: 只删除旧隧道专用防火墙链
ansible.builtin.shell: |
set -eu
changed=0
for pair in INPUT:OCI-WG-IN FORWARD:OCI-WG-FWD; do
parent=${pair%%:*}; chain=${pair#*:}
if iptables -w -nL "$chain" >/dev/null 2>&1; then
while iptables -w -C "$parent" -j "$chain" 2>/dev/null; do
iptables -w -D "$parent" -j "$chain"
done
iptables -w -F "$chain"
iptables -w -X "$chain"
changed=1
fi
done
echo "$changed"
register: retired_chains
changed_when: retired_chains.stdout == '1'
- name: 移除旧的 BGP 配置片段,避免误用
ansible.builtin.file:
path: /etc/frr/oci-wireguard.vtysh
state: absent
@@ -0,0 +1,20 @@
---
- name: 重载 FRR daemons
ansible.builtin.systemd_service:
name: frr
state: reloaded
when: not ansible_check_mode
- name: 刷新专用防火墙
ansible.builtin.systemd_service:
name: oci-wg-firewall
daemon_reload: true
state: restarted
when: not ansible_check_mode
- name: 重启 WireGuard
ansible.builtin.systemd_service:
name: wg-quick@{{ wg_interface }}
state: restarted
when: not ansible_check_mode
- name: 应用 BGP 增量配置
ansible.builtin.include_tasks: apply-bgp.yml
when: not ansible_check_mode
@@ -0,0 +1,8 @@
---
- name: 应用受管 FRR 配置片段
ansible.builtin.command: vtysh -f /etc/frr/oci-wireguard.vtysh
register: bgp_apply
failed_when: bgp_apply.rc != 0 or '% Unknown' in bgp_apply.stdout or '% Unknown' in bgp_apply.stderr
- name: 保存完整运行配置以便重启恢复
ansible.builtin.command: vtysh -c 'write memory'
@@ -0,0 +1,32 @@
---
- name: 安装 AMD DN42 loopback 地址启动单元
ansible.builtin.copy:
dest: /etc/systemd/system/dn42-loopback.service
mode: '0644'
content: |
[Unit]
Description=DN42 router loopback addresses
Before=frr.service
[Service]
Type=oneshot
ExecStart=/usr/sbin/ip address replace 172.21.111.162/32 dev lo
ExecStart=/usr/sbin/ip -6 address replace fdd0:98df:15b0::2/128 dev lo
RemainAfterExit=yes
[Install]
WantedBy=multi-user.target
register: dn42_loopback_unit
- name: 启用 AMD DN42 loopback
ansible.builtin.systemd_service:
name: dn42-loopback
daemon_reload: true
enabled: true
state: "{{ 'restarted' if dn42_loopback_unit.changed else 'started' }}"
when: not ansible_check_mode
- name: 开启路由器 IPv6 转发
ansible.posix.sysctl:
name: net.ipv6.conf.all.forwarding
value: '1'
sysctl_file: /etc/sysctl.d/90-dn42-forwarding.conf
sysctl_set: true
@@ -0,0 +1,163 @@
---
- name: 安装 WireGuard 与 FRR
ansible.builtin.apt:
name: [wireguard-tools, frr, iptables]
state: present
update_cache: true
cache_valid_time: 3600
register: packages
retries: 3
delay: 5
until: packages is succeeded
- name: 创建 WireGuard 受限目录
ansible.builtin.file:
path: /etc/wireguard
state: directory
owner: root
group: root
mode: '0700'
- name: 在站点本机生成私钥,永不复制到控制机
ansible.builtin.shell: 'umask 077; wg genkey > /etc/wireguard/{{ wg_interface }}.key'
args:
creates: '/etc/wireguard/{{ wg_interface }}.key'
no_log: true
- name: 检查现有私钥
ansible.builtin.stat:
path: '/etc/wireguard/{{ wg_interface }}.key'
register: wg_key_file
- name: 读取可交换的公钥
ansible.builtin.shell: 'wg pubkey < /etc/wireguard/{{ wg_interface }}.key'
register: wg_public_key
changed_when: false
check_mode: false
when: wg_key_file.stat.exists
- name: 启用 BGP daemon,保留其他协议
ansible.builtin.lineinfile:
path: /etc/frr/daemons
regexp: '^bgpd='
line: bgpd=yes
notify: 重载 FRR daemons
when: not ansible_check_mode or not packages.changed
- name: 确保 FRR 运行
ansible.builtin.systemd_service:
name: frr
enabled: true
state: started
when: not ansible_check_mode
- name: 配置 DN42 路由器地址
ansible.builtin.include_tasks: dn42-loopback.yml
when: bgp_export6 is defined
- name: 开启 IPv4 转发
ansible.posix.sysctl:
name: net.ipv4.ip_forward
value: '1'
sysctl_file: /etc/sysctl.d/90-oci-wireguard.conf
sysctl_set: true
- name: 安装专用防火墙规则脚本
ansible.builtin.template:
src: firewall.sh.j2
dest: /usr/local/sbin/oci-wg-firewall
owner: root
group: root
mode: '0755'
notify: 刷新专用防火墙
- name: 安装防火墙启动单元
ansible.builtin.copy:
dest: /etc/systemd/system/oci-wg-firewall.service
mode: '0644'
content: |
[Unit]
Description=OCI WireGuard scoped forwarding (no NAT)
After=network-pre.target
Before=wg-quick@{{ wg_interface }}.service
[Service]
Type=oneshot
ExecStart=/usr/local/sbin/oci-wg-firewall
RemainAfterExit=yes
[Install]
WantedBy=multi-user.target
notify: 刷新专用防火墙
- name: 启动防火墙规则
ansible.builtin.systemd_service:
name: oci-wg-firewall
daemon_reload: true
enabled: true
state: started
when: not ansible_check_mode
- name: 写入只含公钥的隧道配置
ansible.builtin.template:
src: wg.conf.j2
dest: '/etc/wireguard/{{ wg_interface }}.conf'
owner: root
group: root
mode: '0600'
when: wg_key_file.stat.exists and (hostvars[wg_peer_host].wg_key_file | default({"stat":{"exists":false}})).stat.exists
notify: 重启 WireGuard
- name: 启动 WireGuard
ansible.builtin.systemd_service:
name: 'wg-quick@{{ wg_interface }}'
enabled: true
state: started
when: not ansible_check_mode
- name: 写入受管 BGP 增量配置
ansible.builtin.template:
src: bgp.conf.j2
dest: /etc/frr/oci-wireguard.vtysh
owner: frr
group: frr
mode: '0640'
notify: 应用 BGP 增量配置
when: not ansible_check_mode or not packages.changed
- name: 先完成配置变更再验证
ansible.builtin.meta: flush_handlers
- name: 检查 WireGuard 对端握手
ansible.builtin.command: 'wg show {{ wg_interface }} latest-handshakes'
changed_when: false
register: wg_handshake
retries: 12
delay: 5
until: wg_handshake.stdout.split() | length == 2 and (wg_handshake.stdout.split()[-1] | int) > 0
when: not ansible_check_mode
- name: 验证隧道互通
ansible.builtin.command: "{{ ('ping -6 -I ' ~ wg_interface ~ ' -c 3 -W 3 ' ~ bgp_transport_peer) if bgp_transport_peer is defined else ('ping -I ' ~ wg_interface ~ ' -c 3 -W 3 ' ~ wg_peer_address) }}"
changed_when: false
when: not ansible_check_mode
- name: 验证 BGP 建邻并收到预期前缀数量
ansible.builtin.command: vtysh -c 'show bgp summary json'
changed_when: false
register: bgp_summary
retries: 12
delay: 5
until: >-
(bgp_summary.stdout | from_json).get('ipv4Unicast', {}).get('peers', {}).get(bgp_transport_peer | default(wg_peer_address), {}).get('state') == 'Established'
and (bgp_summary.stdout | from_json).get('ipv4Unicast', {}).get('peers', {}).get(bgp_transport_peer | default(wg_peer_address), {}).get('pfxRcd', 0) == bgp_import | length
when: not ansible_check_mode
- name: 验证 IPv6 iBGP 建邻与前缀
ansible.builtin.command: vtysh -c 'show bgp summary json'
changed_when: false
register: bgp_summary6
retries: 12
delay: 5
until: >-
(bgp_summary6.stdout | from_json).get('ipv6Unicast', {}).get('peers', {}).get(bgp_transport_peer | default(wg_peer_ipv6), {}).get('state') == 'Established'
and (bgp_summary6.stdout | from_json).get('ipv6Unicast', {}).get('peers', {}).get(bgp_transport_peer | default(wg_peer_ipv6), {}).get('pfxRcd', 0) == bgp_import6 | length
when: not ansible_check_mode and bgp_import6 is defined
@@ -0,0 +1,80 @@
{% set v4_peer = bgp_transport_peer | default(wg_peer_address) %}
{% set v6_peer = bgp_transport_peer | default(wg_peer_ipv6 | default("")) %}
! vtysh -f 使用配置模式,write memory 由独立 handler 执行。
! 只重建本角色拥有的邻居和过滤器,保留既有 BGP/OSPF。
router bgp {{ bgp_asn }}
no neighbor {{ v4_peer }}
{% for peer in bgp_retired_peers | default([]) %}
no neighbor {{ peer }}
{% endfor %}
exit
no ip prefix-list OCI-WG-IN
no ip prefix-list OCI-WG-OUT
{% for prefix in bgp_import %}
ip prefix-list OCI-WG-IN seq {{ loop.index * 10 }} permit {{ prefix }}
{% endfor %}
{% for prefix in bgp_export + (bgp_summary | default([])) %}
ip prefix-list OCI-WG-OUT seq {{ loop.index * 10 }} permit {{ prefix }}
{% endfor %}
router bgp {{ bgp_asn }}
bgp router-id {{ bgp_router_id }}
neighbor {{ v4_peer }} remote-as {{ bgp_peer_asn }}
neighbor {{ v4_peer }} description OCI-WireGuard
neighbor {{ v4_peer }} update-source {{ wg_interface }}
{% if bgp_transport_peer is defined %}
neighbor {{ v4_peer }} interface {{ wg_interface }}
neighbor {{ v4_peer }} capability extended-nexthop
{% endif %}
address-family ipv4 unicast
{% for prefix in bgp_summary | default([]) %}
aggregate-address {{ prefix }}
{% endfor %}
{% for prefix in bgp_export %}
network {{ prefix }}
{% endfor %}
neighbor {{ v4_peer }} activate
neighbor {{ v4_peer }} prefix-list OCI-WG-IN in
neighbor {{ v4_peer }} prefix-list OCI-WG-OUT out
neighbor {{ v4_peer }} maximum-prefix {{ bgp_import | length }}
exit-address-family
exit
! 仅为这些 BGP 路由选择本机业务 IP;转发报文源地址保持不变,不是 NAT。
route-map OCI-WG-SOURCE permit 10
match ip address prefix-list OCI-WG-IN
set src {{ bgp_router_id }}
exit
route-map OCI-WG-SOURCE permit 100
exit
ip protocol bgp route-map OCI-WG-SOURCE
{% if bgp_export6 is defined %}
no ipv6 prefix-list OCI-WG6-IN
no ipv6 prefix-list OCI-WG6-OUT
{% for prefix in bgp_import6 %}
ipv6 prefix-list OCI-WG6-IN seq {{ loop.index * 10 }} permit {{ prefix }}
{% endfor %}
{% for prefix in bgp_export6 + (bgp_summary6 | default([])) %}
ipv6 prefix-list OCI-WG6-OUT seq {{ loop.index * 10 }} permit {{ prefix }}
{% endfor %}
router bgp {{ bgp_asn }}
neighbor {{ v6_peer }} remote-as {{ bgp_peer_asn }}
neighbor {{ v6_peer }} update-source {{ wg_interface }}
{% if bgp_transport_peer is not defined %}
address-family ipv4 unicast
no neighbor {{ v6_peer }} activate
exit-address-family
{% endif %}
address-family ipv6 unicast
{% for prefix in bgp_summary6 | default([]) %}
aggregate-address {{ prefix }}
{% endfor %}
{% for prefix in bgp_export6 %}
network {{ prefix }}
{% endfor %}
neighbor {{ v6_peer }} activate
neighbor {{ v6_peer }} prefix-list OCI-WG6-IN in
neighbor {{ v6_peer }} prefix-list OCI-WG6-OUT out
neighbor {{ v6_peer }} maximum-prefix {{ bgp_import6 | length }}
exit-address-family
exit
{% endif %}
@@ -0,0 +1,43 @@
#!/bin/sh
set -eu
# 仅重建专用链,不保存/覆盖 Docker、k3s、Tailscale 或 OCI 的其他动态规则。
for chain in OCI-WG-IN OCI-WG-FWD; do
iptables -w -nL "$chain" >/dev/null 2>&1 || iptables -w -N "$chain"
iptables -w -F "$chain"
done
iptables -w -A OCI-WG-IN -p udp --dport {{ wg_port }} -j ACCEPT
{% for prefix in [wg_peer_address ~ '/32'] + bgp_import %}
iptables -w -A OCI-WG-IN -i {{ wg_interface }} -s {{ prefix }} -j ACCEPT
{% endfor %}
{% for local_prefix in bgp_export %}
{% for remote_prefix in bgp_import %}
iptables -w -A OCI-WG-FWD -i {{ wg_interface }} -o {{ wg_lan_interface }} -s {{ remote_prefix }} -d {{ local_prefix }} -j ACCEPT
iptables -w -A OCI-WG-FWD -i {{ wg_lan_interface }} -o {{ wg_interface }} -s {{ local_prefix }} -d {{ remote_prefix }} -j ACCEPT
{% endfor %}
{% endfor %}
{% if dn42_external_interface is defined %}
# 仅允许注册地址在内部隧道与外部 DN42 之间转发,保持无 NAT。
iptables -w -A OCI-WG-FWD -i {{ wg_interface }} -o {{ dn42_external_interface }} -s {{ dn42_ipv4 }} -d 172.20.0.0/14 -j ACCEPT
iptables -w -A OCI-WG-FWD -i {{ dn42_external_interface }} -o {{ wg_interface }} -s 172.20.0.0/14 -d {{ dn42_ipv4 }} -j ACCEPT
{% endif %}
iptables -w -A OCI-WG-FWD -i {{ wg_interface }} -j DROP
iptables -w -A OCI-WG-FWD -o {{ wg_interface }} -j DROP
iptables -w -C INPUT -j OCI-WG-IN 2>/dev/null || iptables -w -I INPUT 1 -j OCI-WG-IN
iptables -w -C FORWARD -j OCI-WG-FWD 2>/dev/null || iptables -w -I FORWARD 1 -j OCI-WG-FWD
{% if bgp_import6 is defined %}
ip6tables -w -nL OCI-WG6-IN >/dev/null 2>&1 || ip6tables -w -N OCI-WG6-IN
ip6tables -w -F OCI-WG6-IN
ip6tables -w -A OCI-WG6-IN -i {{ wg_interface }} -s {{ dn42_ipv6 }} -j ACCEPT
ip6tables -w -C INPUT -j OCI-WG6-IN 2>/dev/null || ip6tables -w -I INPUT 1 -j OCI-WG6-IN
{% endif %}
{% if dn42_external_interface is defined %}
ip6tables -w -nL OCI-DN42-FWD >/dev/null 2>&1 || ip6tables -w -N OCI-DN42-FWD
ip6tables -w -F OCI-DN42-FWD
ip6tables -w -A OCI-DN42-FWD -i {{ wg_interface }} -o {{ dn42_external_interface }} -s {{ dn42_ipv6 }} -d fd00::/8 -j ACCEPT
ip6tables -w -A OCI-DN42-FWD -i {{ dn42_external_interface }} -o {{ wg_interface }} -s fd00::/8 -d {{ dn42_ipv6 }} -j ACCEPT
ip6tables -w -A OCI-DN42-FWD -i {{ dn42_external_interface }} -j DROP
ip6tables -w -A OCI-DN42-FWD -o {{ dn42_external_interface }} -j DROP
ip6tables -w -C FORWARD -j OCI-DN42-FWD 2>/dev/null || ip6tables -w -I FORWARD 1 -j OCI-DN42-FWD
{% endif %}
@@ -0,0 +1,16 @@
[Interface]
Address = {{ wg_address }}{% if wg_ipv6_address is defined %}, {{ wg_ipv6_address }}, {{ wg_linklocal_address }}{% endif %}
ListenPort = {{ wg_port }}
MTU = {{ wg_mtu }}
Table = off
# 私钥在本机生成和读取;配置模板与控制机不保存私钥。
PostUp = wg set %i private-key /etc/wireguard/{{ wg_interface }}.key
[Peer]
PublicKey = {{ hostvars[wg_peer_host].wg_public_key.stdout }}
AllowedIPs = {{ ([wg_peer_address ~ '/32'] + bgp_import + (bgp_import6 | default([])) + (['fe80::/64'] if bgp_import6 is defined else [])) | join(', ') }}
{% if wg_endpoint %}
Endpoint = {{ wg_endpoint }}
{% endif %}
PersistentKeepalive = {{ wg_keepalive }}
+52
View File
@@ -0,0 +1,52 @@
---
- import_playbook: vyos.yml
- name: 配置无 NAT 的 WireGuard 与 BGP 站点互联
hosts: wireguard_sites
become: true
strategy: linear
roles:
- wireguard_bgp
- name: 为 OCI ARM 设置经 AMD 的站点回程,保留其余 Tailscale 路由
hosts: oci_routed_hosts
become: true
tasks:
- name: 安装三个目标网段的策略路由脚本
ansible.builtin.template:
src: templates/arm-routes.sh.j2
dest: /usr/local/sbin/oci-site-return-routes
mode: '0755'
notify: 更新 OCI ARM 回程
- name: 安装回程策略启动单元
ansible.builtin.copy:
dest: /etc/systemd/system/oci-site-return-routes.service
mode: '0644'
content: |
[Unit]
Description=Route homelab prefixes through OCI AMD rather than Tailscale
After=network-online.target
Wants=network-online.target
[Service]
Type=oneshot
ExecStart=/usr/local/sbin/oci-site-return-routes
RemainAfterExit=yes
[Install]
WantedBy=multi-user.target
notify: 更新 OCI ARM 回程
- name: 启用回程策略
ansible.builtin.systemd_service:
name: oci-site-return-routes
daemon_reload: true
enabled: true
state: started
when: not ansible_check_mode
handlers:
- name: 更新 OCI ARM 回程
ansible.builtin.systemd_service:
name: oci-site-return-routes
daemon_reload: true
state: restarted
when: not ansible_check_mode
@@ -0,0 +1,38 @@
---
- name: 备份升级前 FRR 配置(仅远端)
ansible.builtin.command: cp -a /etc/frr /var/backups/frr-before-dn42
args:
creates: /var/backups/frr-before-dn42
- name: 安装 FRR 官方仓库签名密钥
ansible.builtin.get_url:
url: https://deb.frrouting.org/frr/keys.gpg
dest: /usr/share/keyrings/frrouting.gpg
mode: '0644'
- name: 配置 FRR 10.7 官方仓库
ansible.builtin.copy:
dest: /etc/apt/sources.list.d/frr.list
mode: '0644'
content: |
deb [signed-by=/usr/share/keyrings/frrouting.gpg] https://deb.frrouting.org/frr noble frr-10.7
register: frr_repo
- name: 安装支持 link-local peering 的固定 FRR 版本
ansible.builtin.apt:
name:
- frr=10.7.1-0~ubuntu24.04.1
- frr-pythontools=10.7.1-0~ubuntu24.04.1
update_cache: true
cache_valid_time: '{{ 0 if frr_repo.changed else 3600 }}'
dpkg_options: force-confdef,force-confold
register: frr_package
notify: 重启 FRR
- name: 检查 FRR 后台进程是否仍引用旧二进制
ansible.builtin.shell: readlink /proc/$(cat /run/frr/bgpd.pid)/exe
changed_when: false
register: frr_running
- name: 调度切换已升级的 FRR 进程
ansible.builtin.debug:
msg: 检查进程是否需要重启以使用已安装的 10.7.1
changed_when: "'(deleted)' in frr_running.stdout"
notify: 重启 FRR
- name: 完成 FRR 升级
ansible.builtin.meta: flush_handlers
@@ -0,0 +1,9 @@
#!/bin/sh
set -eu
# 5101..5103 保留给此互联,优先于 Tailscale 的 52 表(规则 5270)。
# 主表仍使用 OCI 网关;VCN 路由表再将这三个前缀交给 AMD。
{% for prefix in wg_home_prefixes %}
if ! ip -4 rule show | grep -F 'to {{ prefix }} lookup main' >/dev/null; then
ip -4 rule add priority {{ 5100 + loop.index }} to {{ prefix }} lookup main
fi
{% endfor %}
@@ -0,0 +1,62 @@
! Ansible 管理:首个 DN42 peer,单 IPv6 会话双 AFI。
! 精确出口与地址范围入口;尚未接入注册表 ROA 校验。
no ip prefix-list DN42-RB-IN
ip prefix-list DN42-RB-IN seq 1 deny 172.22.166.0/24 le 32
ip prefix-list DN42-RB-IN seq 5 deny 172.21.111.160/27 le 32
ip prefix-list DN42-RB-IN seq 10 deny 10.0.0.0/24 le 32
ip prefix-list DN42-RB-IN seq 20 deny 10.60.0.0/24 le 32
ip prefix-list DN42-RB-IN seq 30 deny 10.61.0.0/24 le 32
ip prefix-list DN42-RB-IN seq 40 deny 192.168.10.0/24 le 32
ip prefix-list DN42-RB-IN seq 1001 permit 172.20.0.0/24 ge 28 le 32
ip prefix-list DN42-RB-IN seq 1002 permit 172.21.0.0/24 ge 28 le 32
ip prefix-list DN42-RB-IN seq 1003 permit 172.22.0.0/24 ge 28 le 32
ip prefix-list DN42-RB-IN seq 1004 permit 172.23.0.0/24 ge 28 le 32
ip prefix-list DN42-RB-IN seq 1100 permit 172.20.0.0/14 ge 21 le 29
ip prefix-list DN42-RB-IN seq 2001 permit 10.100.0.0/14 le 32
ip prefix-list DN42-RB-IN seq 2002 permit 10.127.0.0/16 le 32
ip prefix-list DN42-RB-IN seq 2003 permit 10.0.0.0/8 ge 15 le 24
ip prefix-list DN42-RB-IN seq 3001 permit 172.31.0.0/16 le 32
ip prefix-list DN42-RB-IN seq 9999 deny 0.0.0.0/0 le 32
ip prefix-list DN42-RB-OUT seq 10 permit 172.21.111.160/27
no ipv6 prefix-list DN42-RB6-IN
ipv6 prefix-list DN42-RB6-IN seq 5 deny fdd0:98df:15b0::/48 le 128
ipv6 prefix-list DN42-RB6-IN seq 1001 permit fd00::/8 ge 44 le 64
ipv6 prefix-list DN42-RB6-IN seq 9999 deny ::/0 le 128
ipv6 prefix-list DN42-RB6-OUT seq 10 permit fdd0:98df:15b0::/48
interface {{ dn42_interface }}
exit
router bgp 4242421811
no neighbor {{ dn42_peer_linklocal }}
neighbor {{ dn42_peer_linklocal }} remote-as {{ dn42_peer_asn }}
neighbor {{ dn42_peer_linklocal }} shutdown
neighbor {{ dn42_peer_linklocal }} interface {{ dn42_interface }}
neighbor {{ dn42_peer_linklocal }} update-source {{ dn42_interface }}
neighbor {{ dn42_peer_linklocal }} description RoutedBits-OSA1
neighbor {{ dn42_peer_linklocal }} capability extended-nexthop
neighbor {{ dn42_peer_linklocal }} timers connect 10
address-family ipv4 unicast
neighbor {{ dn42_peer_linklocal }} activate
neighbor {{ dn42_peer_linklocal }} prefix-list DN42-RB-IN in
neighbor {{ dn42_peer_linklocal }} prefix-list DN42-RB-OUT out
neighbor {{ dn42_peer_linklocal }} maximum-prefix 10000
exit-address-family
address-family ipv6 unicast
neighbor {{ dn42_peer_linklocal }} activate
neighbor {{ dn42_peer_linklocal }} prefix-list DN42-RB6-IN in
neighbor {{ dn42_peer_linklocal }} prefix-list DN42-RB6-OUT out
neighbor {{ dn42_peer_linklocal }} maximum-prefix 10000
exit-address-family
no neighbor {{ dn42_peer_linklocal }} shutdown
exit
! 保留既有业务源地址规则,在其后为 DN42 路由选择注册地址;不是 NAT。
route-map OCI-WG-SOURCE permit 20
match ip address prefix-list DN42-RB-IN
set src 172.21.111.162
exit
route-map DN42-SOURCE6 permit 10
match ipv6 address prefix-list DN42-RB6-IN
set src fdd0:98df:15b0::2
exit
route-map DN42-SOURCE6 permit 100
exit
ipv6 protocol bgp route-map DN42-SOURCE6
@@ -0,0 +1,14 @@
set service dns forwarding listen-address '192.168.10.2'
set service dns forwarding source-address '172.21.111.161'
set service dns forwarding source-address 'fdd0:98df:15b0::1'
set service dns forwarding no-serve-rfc1918
{% for prefix in wg_home_prefixes %}
set service dns forwarding allow-from '{{ prefix }}'
{% endfor %}
{% for zone in dn42_dns_zones %}
set service dns forwarding domain {{ zone }} recursion-desired
set service dns forwarding domain {{ zone }} addnta
{% for server in dn42_dns_servers %}
set service dns forwarding domain {{ zone }} name-server '{{ server }}'
{% endfor %}
{% endfor %}
@@ -0,0 +1,10 @@
{% for lan in dn42_ra_lans %}
set interfaces ethernet {{ lan.interface }} address '{{ lan.address }}'
set service router-advert interface {{ lan.interface }} default-lifetime '0'
set service router-advert interface {{ lan.interface }} interval min '10'
set service router-advert interface {{ lan.interface }} interval max '30'
set service router-advert interface {{ lan.interface }} prefix {{ lan.prefix }} preferred-lifetime '14400'
set service router-advert interface {{ lan.interface }} prefix {{ lan.prefix }} valid-lifetime '86400'
set service router-advert interface {{ lan.interface }} route fd00::/8 route-preference 'medium'
set service router-advert interface {{ lan.interface }} route fd00::/8 valid-lifetime '180'
{% endfor %}
@@ -0,0 +1,12 @@
source /opt/vyatta/etc/functions/script-template
if /bin/cli-shell-api existsActive interfaces wireguard wg42 private-key; then
echo KEY_EXISTS
builtin exit 0
fi
configure
set interfaces wireguard wg42 address '10.255.254.2/30'
set interfaces wireguard wg42 private-key "$(wg genkey)"
commit || builtin exit 1
save || builtin exit 1
exit
echo KEY_CREATED
@@ -0,0 +1,84 @@
set interfaces wireguard wg42 description 'OCI AMD / DN42 internal transport'
set interfaces wireguard wg42 address '172.21.111.161/32'
set interfaces wireguard wg42 address 'fdd0:98df:15b0:ffff::2/64'
set interfaces wireguard wg42 mtu '1380'
set interfaces wireguard wg42 port '51820'
set interfaces wireguard wg42 peer amd address '129.225.176.134'
set interfaces wireguard wg42 peer amd port '51820'
set interfaces wireguard wg42 peer amd public-key '{{ hostvars.oci_amd.wg_public_key.stdout | trim }}'
set interfaces wireguard wg42 peer amd persistent-keepalive '25'
set interfaces wireguard wg42 peer amd allowed-ips '10.255.254.1/32'
set interfaces wireguard wg42 peer amd allowed-ips '10.0.0.0/24'
set interfaces wireguard wg42 peer amd allowed-ips '172.21.111.162/32'
set interfaces wireguard wg42 peer amd allowed-ips 'fdd0:98df:15b0::/48'
set interfaces loopback lo address 'fdd0:98df:15b0::1/128'
set protocols static route 172.21.111.160/27 blackhole distance '254'
set protocols static route6 fdd0:98df:15b0::/48 blackhole distance '254'
set protocols bgp system-as '{{ dn42_asn }}'
set protocols bgp parameters router-id '172.21.111.161'
{% for p in ['10.0.0.0/24', '172.21.111.162/32', '172.20.0.0/14'] %}
set policy prefix-list OCI-SITE-IN rule {{ loop.index * 10 }} action 'permit'
set policy prefix-list OCI-SITE-IN rule {{ loop.index * 10 }} prefix '{{ p }}'
{% endfor %}
{% for p in wg_home_prefixes + [dn42_ipv4] %}
set protocols bgp address-family ipv4-unicast network '{{ p }}'
set policy prefix-list OCI-SITE-OUT rule {{ loop.index * 10 }} action 'permit'
set policy prefix-list OCI-SITE-OUT rule {{ loop.index * 10 }} prefix '{{ p }}'
{% endfor %}
set protocols bgp address-family ipv6-unicast network '{{ dn42_ipv6 }}'
set policy prefix-list6 OCI-SITE6-IN rule 10 action 'permit'
set policy prefix-list6 OCI-SITE6-IN rule 10 prefix 'fdd0:98df:15b0::2/128'
set policy prefix-list6 OCI-SITE6-OUT rule 10 action 'permit'
set policy prefix-list6 OCI-SITE6-OUT rule 10 prefix '{{ dn42_ipv6 }}'
set policy prefix-list OCI-TO-OSPF rule 10 action 'permit'
set policy prefix-list OCI-TO-OSPF rule 10 prefix '10.0.0.0/24'
set policy route-map OCI-TO-OSPF rule 10 action 'permit'
set policy route-map OCI-TO-OSPF rule 10 match ip address prefix-list 'OCI-TO-OSPF'
set protocols ospf redistribute bgp route-map 'OCI-TO-OSPF'
set protocols ospf redistribute bgp metric-type '1'
set policy prefix-list DN42-TO-OSPF rule 10 action 'permit'
set policy prefix-list DN42-TO-OSPF rule 10 prefix '{{ dn42_ipv4 }}'
set policy route-map DN42-TO-OSPF rule 10 action 'permit'
set policy route-map DN42-TO-OSPF rule 10 match ip address prefix-list 'DN42-TO-OSPF'
set protocols ospf redistribute static route-map 'DN42-TO-OSPF'
set protocols ospf redistribute static metric-type '1'
set interfaces wireguard wg42 peer amd allowed-ips 'fe80::/64'
set interfaces wireguard wg42 peer amd allowed-ips '172.20.0.0/14'
set interfaces wireguard wg42 peer amd allowed-ips 'fd00::/8'
set policy prefix-list6 OCI-SITE6-IN rule 20 action 'permit'
set policy prefix-list6 OCI-SITE6-IN rule 20 prefix 'fd00::/8'
set policy prefix-list OCI-TO-OSPF rule 20 action 'permit'
set policy prefix-list OCI-TO-OSPF rule 20 prefix '172.20.0.0/14'
set interfaces wireguard wg42 address 'fe80::1811:1/64'
set protocols bgp neighbor fe80::1811:2 remote-as '{{ dn42_asn }}'
set protocols bgp neighbor fe80::1811:2 description 'OCI AMD internal MP-BGP'
set protocols bgp neighbor fe80::1811:2 interface source-interface 'wg42'
set protocols bgp neighbor fe80::1811:2 update-source 'fe80::1811:1'
set protocols bgp neighbor fe80::1811:2 capability extended-nexthop
set protocols bgp neighbor fe80::1811:2 address-family ipv4-unicast prefix-list import 'OCI-SITE-IN'
set protocols bgp neighbor fe80::1811:2 address-family ipv4-unicast prefix-list export 'OCI-SITE-OUT'
set protocols bgp neighbor fe80::1811:2 address-family ipv6-unicast prefix-list import 'OCI-SITE6-IN'
set protocols bgp neighbor fe80::1811:2 address-family ipv6-unicast prefix-list export 'OCI-SITE6-OUT'
{% for path in vyos_retired_paths %}
{% if vyos_retired_present.results[loop.index0].rc == 0 %}
delete {{ path }}
{% endif %}
{% endfor %}
set nat source rule 18100 description 'Exclude local DN42 allocation'
set nat source rule 18100 outbound-interface name 'wg42'
set nat source rule 18100 destination address '172.21.111.160/27'
set nat source rule 18100 exclude
{% for prefix in wg_home_prefixes %}
set nat source rule {{ 18100 + loop.index * 10 }} description 'LAN to DN42 masquerade'
set nat source rule {{ 18100 + loop.index * 10 }} outbound-interface name 'wg42'
set nat source rule {{ 18100 + loop.index * 10 }} source address '{{ prefix }}'
set nat source rule {{ 18100 + loop.index * 10 }} destination address '172.20.0.0/14'
set nat source rule {{ 18100 + loop.index * 10 }} translation address 'masquerade'
{% endfor %}
set policy route-map OCI-MP-IN rule 10 action 'permit'
set policy route-map OCI-MP-IN rule 10 set ipv6-next-hop prefer-global
set protocols bgp neighbor fe80::1811:2 address-family ipv4-unicast route-map import 'OCI-MP-IN'
set protocols bgp neighbor fe80::1811:2 address-family ipv6-unicast route-map import 'OCI-MP-IN'
+84
View File
@@ -0,0 +1,84 @@
---
- name: 读取 AMD 的现有 WireGuard 公钥
hosts: oci_amd
become: true
gather_facts: false
tasks:
- name: 仅交换公钥
ansible.builtin.shell: wg pubkey < /etc/wireguard/wg-oci.key
register: wg_public_key
changed_when: false
check_mode: false
- name: 配置 VyOS 站点与 DN42 内部路由
hosts: site_routers
gather_facts: false
vars:
vyos_retired_paths:
- interfaces wireguard wg42 address 10.255.254.2/30
- interfaces loopback lo address 172.21.111.161/32
- protocols bgp neighbor 10.255.254.1
- protocols bgp neighbor fdd0:98df:15b0:ffff::1
tasks:
- name: 检查 VyOS 是否已有 WireGuard 私钥,仅返回存在状态
ansible.builtin.command:
argv: [ssh, -o, BatchMode=yes, -o, ConnectTimeout=10, '[email protected]', /bin/cli-shell-api, existsActive, interfaces, wireguard, wg42, private-key]
delegate_to: localhost
vars:
ansible_connection: local
ansible_python_interpreter: /usr/bin/python3
register: vyos_key_present
changed_when: false
failed_when: vyos_key_present.rc not in [0, 1]
check_mode: false
- name: 首次在 VyOS 本机生成并保存私钥,控制机不接收私钥
ansible.builtin.command:
argv: [ssh, -o, BatchMode=yes, -o, ConnectTimeout=10, '[email protected]', /bin/vbash, -s]
stdin: "{{ lookup('template', 'templates/vyos-key-bootstrap.vbash.j2') }}"
delegate_to: localhost
vars:
ansible_connection: local
ansible_python_interpreter: /usr/bin/python3
no_log: true
register: vyos_key_bootstrap
changed_when: true
when: not ansible_check_mode and vyos_key_present.rc == 1
- name: 读取 VyOS WireGuard 公钥
vyos.vyos.vyos_command:
commands: show interfaces wireguard wg42 public-key
register: vyos_wg_public
changed_when: false
when: not ansible_check_mode
- name: 准備 Linux 端需要的公钥事实
ansible.builtin.set_fact:
wg_public_key:
stdout: '{{ vyos_wg_public.stdout[0] | trim }}'
wg_key_file:
stat:
exists: true
when: not ansible_check_mode
- name: 检查待退役的旧地址和 BGP 邻居是否存在
ansible.builtin.command:
cmd: 'ssh -o BatchMode=yes [email protected] /bin/cli-shell-api existsActive {{ item }}'
loop: '{{ vyos_retired_paths }}'
delegate_to: localhost
vars:
ansible_connection: local
ansible_python_interpreter: /usr/bin/python3
register: vyos_retired_present
changed_when: false
failed_when: vyos_retired_present.rc not in [0, 1]
check_mode: false
- name: 应用 VyOS 双栈 WireGuard、iBGP 与精确路由过滤
vyos.vyos.vyos_config:
lines: "{{ lookup('template', 'templates/vyos-site.conf.j2').splitlines() | map('trim') | reject('equalto', '') | list }}"
save: true
comment: Ansible OCI WireGuard DN42 preparation
register: vyos_site_config
changed_when: (vyos_site_config.commands | default([]) | length) > 0
no_log: true

Some files were not shown because too many files have changed in this diff Show More