Compare commits
178
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0e3564bf72
|
||
|
|
c6ec310b0b | ||
|
|
3b77af8da1
|
||
|
|
8af511ecc8 | ||
|
|
94684d0722
|
||
|
|
9515cde49b | ||
|
|
accf2d8210 | ||
|
|
d1ccc99125
|
||
|
|
b619f6f681
|
||
|
|
1806c678a4 | ||
|
|
aeb8c49d0a
|
||
|
|
7b1a98280c | ||
|
|
ea15841d4d
|
||
|
|
4b9aa9e164 | ||
|
|
b7b92b3465
|
||
|
|
dc2b43f693 | ||
|
|
60836c3360
|
||
|
|
2030751e6a | ||
|
|
f810b1c674
|
||
|
|
6585d8c46a
|
||
|
|
cadfee0aea | ||
|
|
412fa93018
|
||
|
|
2e05b1a96a
|
||
|
|
71eea7d8fc
|
||
|
|
a08c8a7303
|
||
|
|
76f94f4f31 | ||
|
|
a2102e6708 | ||
|
|
bb2b5119ab
|
||
|
|
65a72ce7fa
|
||
|
|
5241eceb0c
|
||
|
|
5ba4411646 | ||
|
|
2fcb41adde
|
||
|
|
42d33e31da | ||
|
|
b617b4bc23 | ||
|
|
b6b4efe48f
|
||
|
|
d6518153c8 | ||
|
|
e42bb12f8e
|
||
|
|
f97505df2d
|
||
|
|
8b4acc40f6
|
||
|
|
61df128e25
|
||
|
|
5b81719675
|
||
|
|
62c0ce1fff | ||
|
|
f3d8c38d28
|
||
|
|
61f0f864aa | ||
|
|
64b1acd6c3
|
||
|
|
bbf228de41 | ||
|
|
7703b5d21e
|
||
|
|
3871d34232
|
||
|
|
9c64d31d3d | ||
|
|
d9486b4c8c
|
||
|
|
f36e8a1733 | ||
|
|
926a90508a
|
||
|
|
cb2a66ded5 | ||
|
|
f767cd9a5e | ||
|
|
13279256a5
|
||
|
|
dc325f3a45 | ||
|
|
87bb556087
|
||
|
|
11f038794f
|
||
|
|
654087ef46 | ||
|
|
14fb5a323a
|
||
|
|
eb68e9f0c5 | ||
|
|
c4f425e9ea
|
||
|
|
d15733caac | ||
|
|
246b5023e7
|
||
|
|
21b7b48cdf | ||
|
|
94721c279b
|
||
|
|
31d89817b2
|
||
|
|
22bffe068c | ||
|
|
6078a06b99
|
||
|
|
47042d4df4 | ||
|
|
fd6bd62a4f | ||
|
|
3eb6f33dea
|
||
|
|
818192f453 | ||
|
|
387953c80a
|
||
|
|
298db6a745 | ||
|
|
f36a1cbf11
|
||
|
|
b953db199e | ||
|
|
b7b7975c9f
|
||
|
|
9604ff1004
|
||
|
|
819b521039 | ||
|
|
40703782ea
|
||
|
|
aae19850cf | ||
|
|
c11e1d5e6f
|
||
|
|
e0e8213b8f | ||
|
|
5c2b575a4f
|
||
|
|
67dfe1ddcb | ||
|
|
ed86738c4b
|
||
|
|
8c27a7286b | ||
|
|
c9987122e6
|
||
|
|
9139f6f1f1 | ||
|
|
b6ad65d768
|
||
|
|
d256682d11 | ||
|
|
bdacc03e74 | ||
|
|
22192a3c68
|
||
|
|
e2be2ca045
|
||
|
|
b390565146
|
||
|
|
97021109ac
|
||
|
|
422a640c0f | ||
|
|
d09f62c063
|
||
|
|
20a60b77d0 | ||
|
|
15444b9eb3
|
||
|
|
065f96d432 | ||
|
|
92820fc631
|
||
|
|
6a54afede3 | ||
|
|
1ea4604dfc
|
||
|
|
47c8133034 | ||
|
|
b9645be107
|
||
|
|
ffe21da281 | ||
|
|
b631bae5b3
|
||
|
|
addd103b2c | ||
|
|
f000301864
|
||
|
|
abf3be0eae | ||
|
|
4070b75436
|
||
|
|
7d72e12d60 | ||
|
|
c6e3abbfc0
|
||
|
|
392df6b6e6 | ||
|
|
d8754100f1
|
||
|
|
8c18d9daf3 | ||
|
|
2a1d1e095e
|
||
|
|
039ed46c8e | ||
|
|
c47922855f
|
||
|
|
9a581a2097 | ||
|
|
c4b425b079
|
||
|
|
02fd9a932a | ||
|
|
8f6b08bd20
|
||
|
|
1b2e551392 | ||
|
|
4d6274bb8f
|
||
|
|
def6ba185d | ||
|
|
19cd0a938b
|
||
|
|
c4f7046e1c | ||
|
|
33627573c3
|
||
|
|
5539793e10 | ||
|
|
789872264e
|
||
|
|
48f6e52e61 | ||
|
|
7a1c83054d
|
||
|
|
01cca07f48
|
||
|
|
f9bcde53ab | ||
|
|
1538a8a3c8
|
||
|
|
6bddb744c9 | ||
|
|
bac85b6335
|
||
|
|
af23195b04
|
||
|
|
91c6defe49 | ||
|
|
b4c61c5e10
|
||
|
|
bcf4f5c0f1 | ||
|
|
37abad9a04
|
||
|
|
e83cf1932c | ||
|
|
5e2e28e275
|
||
|
|
a8de4d257b | ||
|
|
745d2cc6c2
|
||
|
|
a9f6663069 | ||
|
|
a5cbe89ae2
|
||
|
|
f257a2aa0a | ||
|
|
b7775fce94
|
||
|
|
f1bcb9017a | ||
|
|
304e18d219
|
||
|
|
342ba4f114 | ||
|
|
45e3effcbe
|
||
|
|
f1d1a1a906 | ||
|
|
531011c257
|
||
|
|
11049f99c7 | ||
|
|
5d0441144d
|
||
|
|
ee75afe2d2 | ||
|
|
d1be2b1a9f
|
||
|
|
aaa54a1289 | ||
|
|
b343cb95f0
|
||
|
|
96f44c7e6a
|
||
|
|
4bfeef464d | ||
|
|
6d5f507faa
|
||
|
|
102fbf2ae6 | ||
|
|
0984a2d0ef
|
||
|
|
10a21df890 | ||
|
|
eeea0dc092
|
||
|
|
4ee123965c | ||
|
|
d5d529dcd9
|
||
|
|
661f841837 | ||
|
|
3de2a078b9
|
||
|
|
0644c25567 | ||
|
|
e513739ba0
|
@@ -0,0 +1,89 @@
|
|||||||
|
---
|
||||||
|
name: ansible
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [main]
|
||||||
|
paths:
|
||||||
|
- 'infrastructure/**/ansible/**'
|
||||||
|
- 'infrastructure/dns/**'
|
||||||
|
- '.ansible-lint'
|
||||||
|
- '.gitea/workflows/ansible.yml'
|
||||||
|
pull_request:
|
||||||
|
paths:
|
||||||
|
- 'infrastructure/**/ansible/**'
|
||||||
|
- 'infrastructure/dns/**'
|
||||||
|
- '.ansible-lint'
|
||||||
|
- '.gitea/workflows/ansible.yml'
|
||||||
|
|
||||||
|
env:
|
||||||
|
ANSIBLE_COLLECTIONS_PATH: /root/.ansible/collections
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
lint:
|
||||||
|
runs-on: [self-hosted, pod]
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Bootstrap uv
|
||||||
|
run: |
|
||||||
|
python3 -m pip install --user --break-system-packages \
|
||||||
|
--index-url https://pypi.org/simple --quiet uv==0.11.7
|
||||||
|
echo "$HOME/.local/bin" >> "$GITHUB_PATH"
|
||||||
|
|
||||||
|
- name: Install ansible-lint and collections
|
||||||
|
run: |
|
||||||
|
for i in 1 2 3 4 5; do
|
||||||
|
uv tool install ansible-core --with paramiko --with pywinrm --quiet && break
|
||||||
|
echo "attempt $i failed"; sleep 10
|
||||||
|
done
|
||||||
|
for i in 1 2 3 4 5; do
|
||||||
|
uv tool install ansible-lint --quiet && break
|
||||||
|
echo "attempt $i failed"; sleep 10
|
||||||
|
done
|
||||||
|
export PATH="$HOME/.local/bin:$PATH"
|
||||||
|
for p in infrastructure/proxmox infrastructure/samba-ad infrastructure/openbao; do
|
||||||
|
ansible-galaxy collection install \
|
||||||
|
-r "$p/ansible/requirements.yml" -p "$ANSIBLE_COLLECTIONS_PATH"
|
||||||
|
done
|
||||||
|
|
||||||
|
- name: ansible-lint
|
||||||
|
run: |
|
||||||
|
export PATH="$HOME/.local/bin:$PATH"
|
||||||
|
export ANSIBLE_COLLECTIONS_PATH="$PWD/infrastructure/samba-ad/ansible/collections:/root/.ansible/collections"
|
||||||
|
# 静态检查不应依赖生产 vault 凭据。一次性 checkout 可以去掉加密变量文件;
|
||||||
|
# syntax-check 只验证结构,不需要解析变量的运行时值。
|
||||||
|
rm -f \
|
||||||
|
infrastructure/openbao/ansible/group_vars/all/vault.yml \
|
||||||
|
infrastructure/samba-ad/ansible/group_vars/all/vault.yml
|
||||||
|
# ansible.cfg still declares vault_password_file. Even with encrypted
|
||||||
|
# vars removed, ansible-lint validates that the configured file exists
|
||||||
|
# before syntax-check starts. This throwaway value decrypts nothing.
|
||||||
|
export ANSIBLE_VAULT_PASSWORD_FILE="$RUNNER_TEMP/ansible-lint-vault-pass"
|
||||||
|
printf '%s\n' 'ci-placeholder-not-a-production-secret' > "$ANSIBLE_VAULT_PASSWORD_FILE"
|
||||||
|
rc=0
|
||||||
|
for p in infrastructure/openbao infrastructure/samba-ad infrastructure/proxmox; do
|
||||||
|
echo "::group::$p"
|
||||||
|
(cd "$p/ansible" && ansible-lint -c ../../../.ansible-lint --nocolor -f pep8 .) || rc=1
|
||||||
|
echo "::endgroup::"
|
||||||
|
done
|
||||||
|
exit $rc
|
||||||
|
|
||||||
|
collection-test:
|
||||||
|
runs-on: [self-hosted, pod]
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Install ansible-core
|
||||||
|
run: |
|
||||||
|
python3 -m pip install --user --break-system-packages \
|
||||||
|
--index-url https://pypi.org/simple --quiet uv==0.11.7
|
||||||
|
export PATH="$HOME/.local/bin:$PATH"
|
||||||
|
uv tool install ansible-core --quiet
|
||||||
|
|
||||||
|
- name: Run ansible-test
|
||||||
|
working-directory: infrastructure/samba-ad/ansible/collections/ansible_collections/ddupan/homelab
|
||||||
|
run: |
|
||||||
|
export PATH="$HOME/.local/bin:$PATH"
|
||||||
|
ansible-test sanity --venv --requirements --python 3.12 --color no
|
||||||
|
ansible-test units --venv --requirements --python 3.12 --color no
|
||||||
+27
-71
@@ -1,35 +1,44 @@
|
|||||||
---
|
---
|
||||||
# Stage 1 of the infra pipeline: static checks only. No cluster access, no
|
# Stage 1 of the infra pipeline: static checks only. No cluster access, no
|
||||||
# credentials, no mutation — so this is safe to run on every push from day one.
|
# credentials or mutation. It runs only when YAML-related paths change.
|
||||||
#
|
#
|
||||||
# Stages 2 (kubectl --dry-run=server) and 3 (k3d / molecule) come later and DO
|
# Stages 2 (kubectl --dry-run=server) and 3 (k3d / molecule) come later and DO
|
||||||
# need cluster access; keep them in separate workflows so a credential problem
|
# need cluster access; keep them in separate workflows so a credential problem
|
||||||
# there can never block this one.
|
# there can never block this one.
|
||||||
name: lint
|
name: yaml
|
||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
|
branches: [main]
|
||||||
|
paths:
|
||||||
|
- '**/*.yaml'
|
||||||
|
- '**/*.yml'
|
||||||
|
- 'infrastructure/dns/**'
|
||||||
|
- 'infrastructure/cloudflared/terraform/dns.generated.tf'
|
||||||
|
- '.yamllint.yml'
|
||||||
|
- '.gitea/workflows/lint.yml'
|
||||||
pull_request:
|
pull_request:
|
||||||
|
paths:
|
||||||
env:
|
- '**/*.yaml'
|
||||||
# pypi.org is NOT reachable from this network — it resolves fine but TCP/443 to
|
- '**/*.yml'
|
||||||
# Fastly (151.101.x) times out, while github.com and cloudflare.com are fine.
|
- 'infrastructure/dns/**'
|
||||||
# This is not the usual flaky-WAN symptom and a plain `uv tool install` will
|
- 'infrastructure/cloudflared/terraform/dns.generated.tf'
|
||||||
# hang until timeout. Use a mirror; verified reachable 2026-07-28.
|
- '.yamllint.yml'
|
||||||
UV_DEFAULT_INDEX: https://pypi.tuna.tsinghua.edu.cn/simple
|
- '.gitea/workflows/lint.yml'
|
||||||
|
|
||||||
# ansible-lint and ansible-core install as SEPARATE uv tools, each with its own
|
|
||||||
# venv. Collections installed under the ansible-core tool are invisible to
|
|
||||||
# ansible-lint, which then reports every module as `syntax-check[unknown-module]`
|
|
||||||
# — a false failure that looks exactly like a real one. Pin both to a shared path.
|
|
||||||
ANSIBLE_COLLECTIONS_PATH: /root/.ansible/collections
|
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
yaml:
|
yaml:
|
||||||
runs-on: self-hosted
|
runs-on: [self-hosted, pod]
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Bootstrap uv
|
||||||
|
# Pin the tool for reproducibility; PyPI also avoids another setup action.
|
||||||
|
run: |
|
||||||
|
python3 -m pip install --user --break-system-packages \
|
||||||
|
--index-url https://pypi.org/simple --quiet uv==0.11.7
|
||||||
|
echo "$HOME/.local/bin" >> "$GITHUB_PATH"
|
||||||
|
|
||||||
- name: Install yamllint
|
- name: Install yamllint
|
||||||
# The WAN drops at random (see CLAUDE.md); retry rather than fail a run.
|
# The WAN drops at random (see CLAUDE.md); retry rather than fail a run.
|
||||||
run: |
|
run: |
|
||||||
@@ -48,60 +57,7 @@ jobs:
|
|||||||
files=$(git ls-files '*.yaml' '*.yml' | grep -vE '^apps/netboot/')
|
files=$(git ls-files '*.yaml' '*.yml' | grep -vE '^apps/netboot/')
|
||||||
yamllint -c .yamllint.yml --no-warnings -f parsable $files
|
yamllint -c .yamllint.yml --no-warnings -f parsable $files
|
||||||
|
|
||||||
ansible:
|
- name: Verify generated DNS configuration
|
||||||
runs-on: self-hosted
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
|
|
||||||
- name: Install ansible-lint and collections
|
|
||||||
# pywinrm is not optional — without it every ansible.windows.* task dies
|
|
||||||
# with "No module named 'winrm'" (CLAUDE.md documents this trap).
|
|
||||||
run: |
|
|
||||||
for i in 1 2 3 4 5; do
|
|
||||||
uv tool install ansible-core --with ansible --with paramiko --with pywinrm --quiet && break
|
|
||||||
echo "attempt $i failed"; sleep 10
|
|
||||||
done
|
|
||||||
for i in 1 2 3 4 5; do
|
|
||||||
uv tool install ansible-lint --quiet && break
|
|
||||||
echo "attempt $i failed"; sleep 10
|
|
||||||
done
|
|
||||||
export PATH="$HOME/.local/bin:$PATH"
|
|
||||||
for p in infrastructure/proxmox infrastructure/samba-ad infrastructure/openbao; do
|
|
||||||
ansible-galaxy collection install \
|
|
||||||
-r "$p/ansible/requirements.yml" -p "$ANSIBLE_COLLECTIONS_PATH"
|
|
||||||
done
|
|
||||||
|
|
||||||
- name: ansible-lint
|
|
||||||
# Each project has its own ansible.cfg and relative roles_path, so lint
|
|
||||||
# must run from inside each one — a single run at the repo root resolves
|
|
||||||
# roles_path incorrectly and reports spurious missing-role errors.
|
|
||||||
run: |
|
run: |
|
||||||
export PATH="$HOME/.local/bin:$PATH"
|
export PATH="$HOME/.local/bin:$PATH"
|
||||||
rc=0
|
uv run infrastructure/dns/generate.py --check
|
||||||
for p in infrastructure/openbao infrastructure/samba-ad infrastructure/proxmox; do
|
|
||||||
echo "::group::$p"
|
|
||||||
(cd "$p/ansible" && ansible-lint -c ../../../.ansible-lint --nocolor -f pep8 .) || rc=1
|
|
||||||
echo "::endgroup::"
|
|
||||||
done
|
|
||||||
exit $rc
|
|
||||||
|
|
||||||
terraform:
|
|
||||||
runs-on: self-hosted
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
|
|
||||||
- name: fmt and validate
|
|
||||||
# -backend=false so validate never touches real state or needs credentials.
|
|
||||||
# These roots deliberately use different providers AND different interactive
|
|
||||||
# auth (bao login -method=oidc, az login), which is exactly why they are not
|
|
||||||
# merged — so validate is as far as static checking can go here.
|
|
||||||
run: |
|
|
||||||
rc=0
|
|
||||||
for d in $(git ls-files '*.tf' | xargs -n1 dirname | sort -u); do
|
|
||||||
echo "::group::$d"
|
|
||||||
terraform -chdir="$d" fmt -check -diff || rc=1
|
|
||||||
terraform -chdir="$d" init -backend=false -input=false || rc=1
|
|
||||||
terraform -chdir="$d" validate || rc=1
|
|
||||||
echo "::endgroup::"
|
|
||||||
done
|
|
||||||
exit $rc
|
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
---
|
||||||
|
name: terraform
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [main]
|
||||||
|
paths:
|
||||||
|
- '**/*.tf'
|
||||||
|
- '**/.terraform.lock.hcl'
|
||||||
|
- '.gitea/workflows/terraform.yml'
|
||||||
|
pull_request:
|
||||||
|
paths:
|
||||||
|
- '**/*.tf'
|
||||||
|
- '**/.terraform.lock.hcl'
|
||||||
|
- '.gitea/workflows/terraform.yml'
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
validate:
|
||||||
|
runs-on: [self-hosted, pod]
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- uses: hashicorp/setup-terraform@v3
|
||||||
|
|
||||||
|
- name: fmt and validate
|
||||||
|
# -backend=false so validate never touches real state or needs credentials.
|
||||||
|
run: |
|
||||||
|
rc=0
|
||||||
|
for d in $(git ls-files '*.tf' | xargs -n1 dirname | sort -u); do
|
||||||
|
echo "::group::$d"
|
||||||
|
terraform -chdir="$d" fmt -check -diff || rc=1
|
||||||
|
terraform -chdir="$d" init -backend=false -input=false || rc=1
|
||||||
|
terraform -chdir="$d" validate || rc=1
|
||||||
|
echo "::endgroup::"
|
||||||
|
done
|
||||||
|
exit $rc
|
||||||
@@ -49,6 +49,7 @@ authelia/secret.yaml
|
|||||||
**/secret.yaml
|
**/secret.yaml
|
||||||
**/credentials.yml
|
**/credentials.yml
|
||||||
**/terraform.tfvars
|
**/terraform.tfvars
|
||||||
|
**/credentials.auto.tfvars
|
||||||
**/tailscale/helm.sh
|
**/tailscale/helm.sh
|
||||||
**/cloudflared/backup/
|
**/cloudflared/backup/
|
||||||
**/cloudflared/secret.yaml
|
**/cloudflared/secret.yaml
|
||||||
@@ -118,3 +119,4 @@ apps/netboot/config/log/
|
|||||||
# Blocky's per-day query logs. Bind-mounted into the container, one file per
|
# Blocky's per-day query logs. Bind-mounted into the container, one file per
|
||||||
# day, and every DNS query the LAN makes ends up in them.
|
# day, and every DNS query the LAN makes ends up in them.
|
||||||
apps/blocky/logs/
|
apps/blocky/logs/
|
||||||
|
.venv/
|
||||||
|
|||||||
@@ -4,3 +4,8 @@
|
|||||||
- `apps/http-echo/` and `archive/traefik/` contain Kubernetes/Gateway API manifests; inspect their parent Gateway references before applying archived or brownfield resources.
|
- `apps/http-echo/` and `archive/traefik/` contain Kubernetes/Gateway API manifests; inspect their parent Gateway references before applying archived or brownfield resources.
|
||||||
- `apps/tailscale/helm.sh` contains live Tailscale OAuth values; do not copy, print, or commit those values anywhere else.
|
- `apps/tailscale/helm.sh` contains live Tailscale OAuth values; do not copy, print, or commit those values anywhere else.
|
||||||
- Preserve the existing README intent in `apps/http-echo/` and `archive/traefik/` when updating manifests.
|
- Preserve the existing README intent in `apps/http-echo/` and `archive/traefik/` when updating manifests.
|
||||||
|
- `CHANGELOG.md` 是冻结的历史快照,不再更新。持久的服务状态与运维知识写入对应
|
||||||
|
README/runbook;单次变化由 commit 和 PR 记录,agent 陷阱写入 `CLAUDE.md`。
|
||||||
|
- 在 homelab 工作中,所有提交到 `git.ddupan.top` 的 commit message、PR、issue
|
||||||
|
和项目文档默认优先使用中文。代码标识符、命令、配置键、上游专有名称,以及
|
||||||
|
使用英文能避免歧义的技术字段可保留英文。
|
||||||
|
|||||||
+57
-4
@@ -15,6 +15,42 @@ What changed in this homelab, when, and why. Newest first.
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
## 2026-09-10
|
||||||
|
|
||||||
|
**Flux 的 deployment、drift repair 和 scoped prune 闭环验证完成。**
|
||||||
|
|
||||||
|
| area | change |
|
||||||
|
|---|---|
|
||||||
|
| GitOps | PR #18 合并后,Flux 自行发现 revision `f257a2a` 并删除已在 `prune: true` 下重新进入 inventory 的测试 ConfigMap;未发送 reconcile annotation,`http-echo` Deployment/Service 保持 Ready,root 继续 `prune: false` |
|
||||||
|
| Helm migration | 选定 `gitea-actions` 作为第一个 Flux HelmRelease adoption:它不承载 Git、入口、DNS、证书、数据库或 secrets controller。live StatefulSet 与 Git 都使用 regular DinD,但 Helm 保存的 release values/manifest 仍是失败的 rootless 配置;接管先固定 chart `0.1.1` 并验证 live Pod spec 不变,升级另开 PR |
|
||||||
|
| Helm adoption stage | 为 `gitea-actions` 加入固定 chart `0.1.1` 的 HelmRepository、values ConfigMap 和 `suspend: true` HelmRelease;第一阶段只让 Flux 登记对象,确认 source 与固定 chart render 后再解除 suspend,失败策略使用 `RetryOnFailure` 以避免回滚到 stored rootless manifest |
|
||||||
|
| Helm adoption activate | 第一阶段合并后 Flux source 与子 Kustomization 均 Ready,Helm release 仍为 revision 1,runner Pod 未 rollout;再次确认固定 chart 的完整 render 对 live 集群为零差异后,第二阶段移除 `suspend`,允许 Flux 修正 Helm 存储状态并开始 drift detection |
|
||||||
|
| Gitea adoption stage | 开始用 Flux 接管关键 `gitea` release:固定现有 chart `12.5.3`,以 `suspend: true` 登记 HelmRelease、source、values ConfigMap 和现有 HTTPRoute,子 Kustomization 保持 `prune: false`;现有 OIDC Secret 继续只引用不覆盖,其尚未进入 OpenBao/ESO 的缺口独立跟踪 |
|
||||||
|
| Gitea adoption activate | 第一阶段合并后 source、子 Kustomization 和 HTTPRoute 均 Ready,Helm release 仍为 revision 14,Gitea Pod 未重建或重启;再次确认固定 chart 对 live 业务资源零差异后移除 `suspend`,允许 Flux 修正 Helm 存储状态并启用 drift detection |
|
||||||
|
| Gitea upgrade plan | 规划两跳升级:chart `12.6.0` + 显式 Gitea `1.26.4`,再到 chart `12.7.0` + 显式 Gitea `1.27.3`;每个 minor 都先以 suspended desired state 合并、停机建立 CNPG/PVC 一致回滚点,再用独立 PR 激活。当前 CNPG 无连续备份、local-path PVC 无 snapshot class,因此禁止无备份直接触发数据库 migration |
|
||||||
|
| Gitea 1.26 preparation | 将第一跳目标写入 Git:chart 固定为 `12.6.0`、rootless 镜像显式固定为 `1.26.4`,同时重新设置 HelmRelease `suspend: true`;该准备 revision 合并后只更新 desired state,不触发 Pod replacement 或数据库 migration |
|
||||||
|
| Gitea 1.26 backup | 预拉取 `1.26.4-rootless` 后,在 HelmRelease suspended 状态将 Gitea scale 到 0;生成并校验 539355-byte CNPG custom dump、2609188-byte PVC tar 和 2848667-byte GPG encrypted bundle,随后恢复旧版 `1.25.5` 并验证内外 API 与 Flux source。按明确决定不上传 OCI,本阶段接受只有节点本地回滚点的风险 |
|
||||||
|
| Gitea 1.26 activation | 停机一致备份门槛完成后,激活变更只移除 HelmRelease 的 `suspend`;chart `12.6.0`、显式 `1.26.4-rootless` image、values、数据库与 PVC 均保持已 review 的准备状态 |
|
||||||
|
| Gitea 1.26 result | Flux 以 Helm revision 16 成功完成 chart `12.6.0` / Gitea `1.26.4-rootless` 的 Recreate upgrade 和 migration 323–330;Pod 内/统一域名 API、临时 branch push/delete、Flux source 及 main/smoke 的全部 CI jobs 均通过,Pod 在约 15 分钟采样中保持零重启,Authelia OIDC init 同步与浏览器交互式管理员登录也已确认成功 |
|
||||||
|
| Gitea 1.27 preparation | 预拉取 `1.27.3-rootless` 并将第二跳 desired state 原子设置为 chart `12.7.0`、显式 image `1.27.3` 和 `suspend: true`;合并只暂停并登记目标,不执行 migration,激活前必须从当前 1.26.4 数据建立新的配套回滚点 |
|
||||||
|
| Gitea 1.27 activation | 按明确决定跳过新的 1.26.4 数据库/PVC 备份,激活变更只移除 HelmRelease 的 `suspend`;接受 migration 失败后不能无损回退到 1.26.4 的风险,现有 1.25.5 本地备份仅能作为会丢失第一跳后状态的灾难恢复点 |
|
||||||
|
| CI runner network | 修复 Actions job 容器访问 GitHub 超时:k3s Pod MTU 为 1450,而 DinD 动态 bridge 默认为 1500;为 Docker daemon 固定 `--mtu=1450`。隔离测试证明相同 curl 镜像在默认 bridge 超时、在 MTU 1450 bridge 下访问 GitHub 与 API 均约 0.1 秒成功 |
|
||||||
|
|
||||||
|
### Incident: Gitea 备份后的恢复命令被 stdin 校验阻塞
|
||||||
|
|
||||||
|
最初把本地 custom-format dump 通过 `kubectl exec -i` 输送给 CNPG Pod 内的
|
||||||
|
`pg_restore --list`;远端 stdin 没有正常结束,组合脚本因此停在校验步骤,尚未执行
|
||||||
|
后面的 scale-up。Deployment 保持预期的 0,没有失败 Pod 或数据写入。发现后终止
|
||||||
|
会话、先恢复 Gitea,再把 dump 临时复制到 CNPG 可写数据卷完成校验并立即删除。
|
||||||
|
旧版 Gitea 恢复后内外 API 和 Flux source 均正常;后续 runbook 不再把 stdin 管道与
|
||||||
|
恢复命令放进同一个 shell transaction。
|
||||||
|
|
||||||
|
`Carried forward`: complete the two-stage zero-change `gitea` HelmRelease
|
||||||
|
adoption, migrate its remaining manual OIDC Secret to OpenBao/ESO, then upgrade
|
||||||
|
Gitea and add credential-free PR plan output before ordering the remaining Helm
|
||||||
|
migrations by dependency and blast radius. Root Flux prune remains disabled
|
||||||
|
until brownfield ownership is audited.
|
||||||
|
|
||||||
## 2026-09-09
|
## 2026-09-09
|
||||||
|
|
||||||
**Recorded the brownfield GitOps/IaC redesign before changing live infrastructure.**
|
**Recorded the brownfield GitOps/IaC redesign before changing live infrastructure.**
|
||||||
@@ -26,14 +62,31 @@ What changed in this homelab, when, and why. Newest first.
|
|||||||
| secrets | Recorded that ESO 2.8.0, five ExternalSecrets and the scoped OpenBao Kubernetes-auth path already exist; the next gate is live recovery testing and migration of any remaining manual Secrets |
|
| secrets | Recorded that ESO 2.8.0, five ExternalSecrets and the scoped OpenBao Kubernetes-auth path already exist; the next gate is live recovery testing and migration of any remaining manual Secrets |
|
||||||
| Terraform | Recorded Gitea 1.27 State Registry as the preferred candidate for local roots after version and recovery testing; the OCI recovery root remains in OCI Object Storage to avoid a home-control-plane dependency loop |
|
| Terraform | Recorded Gitea 1.27 State Registry as the preferred candidate for local roots after version and recovery testing; the OCI recovery root remains in OCI Object Storage to avoid a home-control-plane dependency loop |
|
||||||
| cleanup | Removed the retired NapCat tree, the Contour and Kanidm archive trees, and seven generated Terraform plan files before establishing the clean Git baseline; plans may embed complete state and remain globally ignored |
|
| cleanup | Removed the retired NapCat tree, the Contour and Kanidm archive trees, and seven generated Terraform plan files before establishing the clean Git baseline; plans may embed complete state and remain globally ignored |
|
||||||
| CI | Added a review-first Gitea Actions runner bootstrap: official actions chart 0.1.1, pinned runner 2.3.0, one persistent Kubernetes runner with capacity four and rootless DinD, plus an ESO reference to a repository-scoped registration token in OpenBao. It is not deployed until the PR is merged |
|
| CI | Added a review-first Gitea Actions runner bootstrap: official actions chart 0.1.1, pinned runner 2.3.0, one persistent instance-scoped Kubernetes runner with capacity four, plus an ESO reference to its registration token in OpenBao. The first deployment proved that rootlesskit is blocked by the node's AppArmor unprivileged-userns policy; because the chart requires privileged DinD in either mode, the reviewed fix uses regular DinD instead of weakening the host-wide policy. The runner image intentionally carries neither `uv` nor Terraform: Terraform uses its versioned setup action, while `uv` is pinned and installed from official PyPI because the nested job network reaches PyPI but times out against the GitHub API queried by `setup-uv`. Ansible installs only `ansible-core` in its tool venv and puts declared Galaxy collections in a shared path visible to ansible-lint; installing the `ansible` meta-package had made Galaxy falsely skip that shared installation |
|
||||||
| identity | Declared the Samba AD `gitea-admins` group with `panxiao81` as its initial member. Gitea already maps this OIDC group to site administrators; the local `gitea_admin` account remains as break-glass access |
|
| identity | Declared the Samba AD `gitea-admins` group with `panxiao81` as its initial member. Gitea already maps this OIDC group to site administrators; the local `gitea_admin` account remains as break-glass access |
|
||||||
|
| docs | Reconciled the redesign and CI status with reality: the Gitea remote, instance-scoped runner, OpenBao-projected registration token, green Stage 1 and Flux bootstrap are live; off-site mirroring and recovery verification remain pending |
|
||||||
|
| 协作规范 | 在 `AGENTS.md` 中明确:homelab 向 `git.ddupan.top` 提交的 commit message、PR、issue 与项目文档默认优先使用中文,同时保留必要的英文技术标识符 |
|
||||||
|
| k3s | 在本机逐级从 `v1.33.6+k3s1` 升级到 `v1.33.13+k3s2`、`v1.34.11+k3s1`、`v1.35.8+k3s1`,最终到 `v1.36.4+k3s1`;每一级均建立 SQLite/server 冷备份并验证节点、工作负载、PVC、Gateway、DNS 与 Gitea。k3s 每次重启都会覆盖 CoreDNS 的手工 `serve_stale`,已按 `platform/k3s/Corefile.desired` 恢复 |
|
||||||
|
| GitOps | Flux `v2.9.5` 的四个核心 controller 已上线;集群内只读 Gitea source 与 `prune: false` 的 root Kustomization 均在合并 revision `aaa54a1` 上 Ready,完成了首个 pull reconciliation 闭环 |
|
||||||
|
| cleanup | 已把 `bao-acme` HTTP-01 solver 改到 Envoy Gateway 的明文 listener,并删除不再承载流量的 Contour namespace、provisioner、RBAC、GatewayClass 和全部 `projectcontour.io` CRD;Envoy Gateway、证书、DNS 与 Gitea 复查正常 |
|
||||||
|
| GitOps canary | 加入由 Flux 部署到独立 `gitops-canary` namespace 的 `http-echo` Deployment 和 Service;历史 Contour HTTPRoute 明确排除在 Kustomization 之外,初始保持 `prune: false` |
|
||||||
|
| prune 验证 | 为 `http-echo` 加入无业务依赖的 `flux-prune-canary` ConfigMap;先在 `prune: false` 下确认 Flux inventory,后续通过独立 PR 删除并仅为 canary 开启 prune |
|
||||||
|
| prune 验证第二阶段 | 第一阶段已确认 `flux-prune-canary` 带 Flux ownership 标签并进入 `http-echo` inventory;从 Git 删除该测试对象,同时仅为 `http-echo` 开启 `prune: true`,root 继续保持 `prune: false` |
|
||||||
|
| prune 验证修正 | 第二阶段证明“同一 revision 开启 prune 并删除旧对象”不会回收该对象:Flux 已从 inventory 移除它,但 live ConfigMap 保留。将 ConfigMap 在已经生效的 `prune: true` 下重新纳管,下一 revision 只做删除 |
|
||||||
|
| prune 验证最终阶段 | 已确认 `prune: true` 生效且测试 ConfigMap 重新进入 Flux inventory;本次只从 Git 删除该对象,不修改 canary 或 root 的 prune 设置,用于完成精确垃圾回收验证 |
|
||||||
|
|
||||||
|
### Incident: prune 启用与对象删除放在同一 revision
|
||||||
|
|
||||||
|
测试把 `http-echo` 从 `prune: false` 改为 `true` 的同时从 Git 删除测试
|
||||||
|
ConfigMap。Flux 按新 revision 更新了 inventory,但没有删除按旧设置管理的 live
|
||||||
|
对象,导致 ConfigMap 成为 inventory 之外的残留。没有业务影响。修正方式是在
|
||||||
|
`prune: true` 已经生效后先重新纳管对象,再用下一 revision 单独删除。
|
||||||
|
|
||||||
`Carried forward`: re-verify OpenBao/ESO recovery and remaining Secret inventory;
|
`Carried forward`: re-verify OpenBao/ESO recovery and remaining Secret inventory;
|
||||||
configure a Git remote and off-site mirror; confirm the running Gitea version;
|
configure an off-site Git mirror; plan the Gitea upgrade beyond 1.25.5;
|
||||||
take an encrypted independent OCI state copy before enabling bucket versioning;
|
take an encrypted independent OCI state copy before enabling bucket versioning;
|
||||||
reconstruct the missing root to a zero-change plan; bootstrap Gitea Actions and Flux on a low-risk
|
reconstruct the missing root to a zero-change plan; add a low-risk Flux canary workload;
|
||||||
service; then move Tunnel origins to Envoy one hostname at a time.
|
then move Tunnel origins to Envoy one hostname at a time.
|
||||||
|
|
||||||
## 2026-08-15
|
## 2026-08-15
|
||||||
|
|
||||||
|
|||||||
@@ -132,9 +132,10 @@ recovered, so `.vault_pass.gpg` is the authoritative recovery path.
|
|||||||
|
|
||||||
## Working rules
|
## Working rules
|
||||||
|
|
||||||
- **Record changes in `CHANGELOG.md`.** One dated section per day, newest first; incidents
|
- **Do not update `CHANGELOG.md`.** It is a frozen historical snapshot; requiring every PR
|
||||||
get their own subsection. Traps and procedures belong *here* in CLAUDE.md, not there —
|
to append to one shared text file caused needless conflicts and duplicated Git/PR history.
|
||||||
the changelog is for humans reading what changed.
|
Put durable service state and operational knowledge in the component README or runbook,
|
||||||
|
agent-facing traps here, and let commits/PRs record individual changes.
|
||||||
- **Verify, don't assert.** Check the end state (`pvesm status`, `linstor node list`,
|
- **Verify, don't assert.** Check the end state (`pvesm status`, `linstor node list`,
|
||||||
`kubectl get pod`, `show ip route`) rather than trusting that a command "should have" worked.
|
`kubectl get pod`, `show ip route`) rather than trusting that a command "should have" worked.
|
||||||
Several confident diagnoses in this repo's history were wrong until measured.
|
Several confident diagnoses in this repo's history were wrong until measured.
|
||||||
@@ -149,6 +150,11 @@ recovered, so `.vault_pass.gpg` is the authoritative recovery path.
|
|||||||
all-clear. Use `git check-ignore --no-index` and `git rm --cached` to actually remove it.
|
all-clear. Use `git check-ignore --no-index` and `git rm --cached` to actually remove it.
|
||||||
- **A `.tfplan` is a zip containing a full `tfstate`.** It walks straight past `*.tfstate`
|
- **A `.tfplan` is a zip containing a full `tfstate`.** It walks straight past `*.tfstate`
|
||||||
ignore rules. Ignore `*.tfplan` everywhere.
|
ignore rules. Ignore `*.tfplan` everywhere.
|
||||||
|
- **SPIRE CLI JSON can be an array of response blocks.** `spire-agent api fetch jwt
|
||||||
|
-output json` in 1.15.3 returns blocks containing `svids` and `bundles`. Capture stdout
|
||||||
|
privately and type-check before extracting fields; `list(response)` prints full tokens
|
||||||
|
when the response is already an array. Never inspect credential payloads by printing
|
||||||
|
their containers, and never put fetched JWTs in command arguments or Pod logs.
|
||||||
- **Quoting does not survive two ssh hops.** `ssh pve1 "ssh pve3 'cmd | qm monitor 103'"`
|
- **Quoting does not survive two ssh hops.** `ssh pve1 "ssh pve3 'cmd | qm monitor 103'"`
|
||||||
loses the inner quotes — ssh re-joins argv with spaces, so the pipeline splits and the
|
loses the inner quotes — ssh re-joins argv with spaces, so the pipeline splits and the
|
||||||
tail runs on the **jump host**. It fails silently if you discard stderr: a `screendump`
|
tail runs on the **jump host**. It fails silently if you discard stderr: a `screendump`
|
||||||
|
|||||||
@@ -229,7 +229,7 @@ configMap:
|
|||||||
require_pkce: false
|
require_pkce: false
|
||||||
token_endpoint_auth_method: 'client_secret_basic'
|
token_endpoint_auth_method: 'client_secret_basic'
|
||||||
redirect_uris:
|
redirect_uris:
|
||||||
- 'https://grafana.tail7e769.ts.net/login/generic_oauth'
|
- 'https://grafana.ad.ddupan.top/login/generic_oauth'
|
||||||
scopes:
|
scopes:
|
||||||
- 'openid'
|
- 'openid'
|
||||||
- 'profile'
|
- 'profile'
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
# Blocky — LAN DNS. STAGED, NOT DEPLOYED. See README.md.
|
# Blocky — 已部署的 LAN 主 DNS。见 README.md。
|
||||||
#
|
#
|
||||||
# WHY compose on the laptop and NOT a k3s Deployment, given everything else here
|
# WHY compose on the laptop and NOT a k3s Deployment, given everything else here
|
||||||
# is Kubernetes:
|
# is Kubernetes:
|
||||||
@@ -52,3 +52,10 @@ services:
|
|||||||
options:
|
options:
|
||||||
max-size: "10m"
|
max-size: "10m"
|
||||||
max-file: "3"
|
max-file: "3"
|
||||||
|
|
||||||
|
# 避免与 DN42 的 172.20.0.0/14 重叠。
|
||||||
|
networks:
|
||||||
|
default:
|
||||||
|
ipam:
|
||||||
|
config:
|
||||||
|
- subnet: 172.28.0.0/24
|
||||||
|
|||||||
+11
-4
@@ -1,9 +1,7 @@
|
|||||||
# Blocky — LAN resolver, ad-blocker and split-horizon DNS.
|
# Blocky — LAN resolver, ad-blocker and split-horizon DNS.
|
||||||
#
|
#
|
||||||
# DEPLOYED 2026-07-28 and verified, but NOT yet the LAN resolver — clients still
|
# LAN 主 DNS 为 192.168.10.127,NEC IX 192.168.10.1 为备用。
|
||||||
# get the DC/router pair from DHCP. Making it the resolver needs a DHCP change on
|
# DN42 条件转发经 VyOS,参见 README.md。
|
||||||
# the NEC IX; see README.md. Until then only clients that query 192.168.10.127
|
|
||||||
# explicitly are affected, so this is safely reversible.
|
|
||||||
|
|
||||||
ports:
|
ports:
|
||||||
# These are the CONTAINER's listen addresses, so they must be unqualified —
|
# These are the CONTAINER's listen addresses, so they must be unqualified —
|
||||||
@@ -35,6 +33,13 @@ conditional:
|
|||||||
# Queries for the AD zone go straight to the DC, which is authoritative. This
|
# Queries for the AD zone go straight to the DC, which is authoritative. This
|
||||||
# replaces the "DC first, router second" resolver ordering that clients use today.
|
# replaces the "DC first, router second" resolver ordering that clients use today.
|
||||||
mapping:
|
mapping:
|
||||||
|
# DN42 由 VyOS 使用注册地址转发,避免 LAN 私网源地址缺少回程。
|
||||||
|
dn42: 192.168.10.2
|
||||||
|
20.172.in-addr.arpa: 192.168.10.2
|
||||||
|
21.172.in-addr.arpa: 192.168.10.2
|
||||||
|
22.172.in-addr.arpa: 192.168.10.2
|
||||||
|
23.172.in-addr.arpa: 192.168.10.2
|
||||||
|
d.f.ip6.arpa: 192.168.10.2
|
||||||
ad.ddupan.top: 192.168.10.5
|
ad.ddupan.top: 192.168.10.5
|
||||||
# Reverse lookups for LAN hosts — the DC holds the reverse zone.
|
# Reverse lookups for LAN hosts — the DC holds the reverse zone.
|
||||||
10.168.192.in-addr.arpa: 192.168.10.5
|
10.168.192.in-addr.arpa: 192.168.10.5
|
||||||
@@ -54,9 +59,11 @@ customDNS:
|
|||||||
# laptop's only global IPv6 belongs to tun0, so a AAAA answer would send LAN
|
# laptop's only global IPv6 belongs to tun0, so a AAAA answer would send LAN
|
||||||
# traffic into the VPN. See CLAUDE.md.
|
# traffic into the VPN. See CLAUDE.md.
|
||||||
mapping:
|
mapping:
|
||||||
|
# BEGIN GENERATED: homelab DNS (blocky)
|
||||||
git.ddupan.top: 192.168.10.127
|
git.ddupan.top: 192.168.10.127
|
||||||
auth.ddupan.top: 192.168.10.127
|
auth.ddupan.top: 192.168.10.127
|
||||||
obj.ddupan.top: 192.168.10.127
|
obj.ddupan.top: 192.168.10.127
|
||||||
|
# END GENERATED: homelab DNS (blocky)
|
||||||
|
|
||||||
blocking:
|
blocking:
|
||||||
denylists:
|
denylists:
|
||||||
|
|||||||
@@ -0,0 +1,46 @@
|
|||||||
|
# Gitea
|
||||||
|
|
||||||
|
Gitea 使用外部 CloudNativePG 数据库和现有 `gitea-shared-storage` RWO PVC,入口由
|
||||||
|
Envoy Gateway HTTPRoute 提供。Helm chart 自带的无 class Ingress 暂时保留以确保
|
||||||
|
首次接管零变化;清理该 Ingress 与升级 chart 必须使用后续独立 PR。
|
||||||
|
|
||||||
|
## Flux 接管
|
||||||
|
|
||||||
|
初始接管 release 是 `gitea-12.5.3`(Gitea `1.25.5`)。接管分为两个 PR:第一阶段创建
|
||||||
|
固定版本且 `suspend: true` 的 HelmRelease,只让 Flux 登记对象;确认 source Ready
|
||||||
|
并重新验证完整 chart render 后,第二阶段解除 suspend。第一阶段已经确认 source、
|
||||||
|
子 Kustomization 和 HTTPRoute 均 Ready,Helm revision 与 Gitea Pod 未变化;第二次
|
||||||
|
live diff 仍只有不常驻的 test hook Pod。子 Kustomization 和集群 root 均保持
|
||||||
|
`prune: false`。
|
||||||
|
|
||||||
|
数据库密码已经由 External Secrets Operator 从 OpenBao 投射到 `gitea-db`。OIDC
|
||||||
|
client secret 仍是历史手工 Secret `gitea-oidc-secret`,本次接管只引用、不覆盖它;
|
||||||
|
将剩余 Secret 迁移到 OpenBao 是独立的后续工作。
|
||||||
|
|
||||||
|
Gitea 是 Flux GitRepository 的上游。升级或重启期间 Git source 暂时不可用不会删除
|
||||||
|
已经应用的资源;Gitea 恢复后 Flux 会继续同步。任何会改变 Pod template、数据库迁移
|
||||||
|
或 PVC identity 的变更都不得与首次接管合并。
|
||||||
|
|
||||||
|
跨 minor 的执行顺序、停机一致备份和失败恢复步骤见
|
||||||
|
[`../../docs/gitea-upgrade-plan.md`](../../docs/gitea-upgrade-plan.md)。
|
||||||
|
|
||||||
|
当前 release 是 chart `12.7.0` / Gitea `1.27.3-rootless`。两次跨 minor migration、
|
||||||
|
API、OIDC、Git/Flux 和 runner 均已验证。第二跳按明确决定跳过了新的 1.26.4 停机
|
||||||
|
一致回滚点;现有 1.25.5 本地备份没有 OCI 或异机副本,因此只作为会丢失后续状态的
|
||||||
|
灾难恢复点。
|
||||||
|
|
||||||
|
## 后续升级默认策略
|
||||||
|
|
||||||
|
已连续验证两次 Flux 驱动的跨 minor Recreate upgrade,后续常规 patch/minor 升级
|
||||||
|
不再默认执行长时间观察、临时 branch push/delete、重复 CI、逐条 migration 日志审计
|
||||||
|
或每个 minor 的停机备份。默认只需要:
|
||||||
|
|
||||||
|
1. 固定 chart 和实际 `image.tag`,阅读与本配置相关的 breaking/security notes;
|
||||||
|
2. Helm/Kustomize render 通过 review;
|
||||||
|
3. 合并后确认 HelmRelease `UpgradeSucceeded`、Pod Ready 且没有 CrashLoop;
|
||||||
|
4. API 返回目标版本,并简单确认 OIDC 登录和一次正常 Git 操作。
|
||||||
|
|
||||||
|
只有变更数据库后端或存储布局、rootless 模式、PVC identity、部署策略、重大 chart
|
||||||
|
结构,或者 release notes 指出相关 breaking migration 时,才恢复停机一致备份、分阶段
|
||||||
|
suspend、详细日志审计和扩展验收。出现启动失败或 migration error 时也立即升级为完整
|
||||||
|
故障流程。
|
||||||
@@ -48,6 +48,11 @@ gitea:
|
|||||||
# github.com is reachable from this network (verified 2026-07-28) even when
|
# github.com is reachable from this network (verified 2026-07-28) even when
|
||||||
# pypi.org/Fastly is not — see the flaky-WAN notes in the lint workflow.
|
# pypi.org/Fastly is not — see the flaky-WAN notes in the lint workflow.
|
||||||
DEFAULT_ACTIONS_URL: github
|
DEFAULT_ACTIONS_URL: github
|
||||||
|
webhook:
|
||||||
|
# Keep the default public-internet access for existing hooks while allowing
|
||||||
|
# only the dynamic Runner controller's exact in-cluster DNS name. Do not
|
||||||
|
# broaden this to the built-in `private` network group.
|
||||||
|
ALLOWED_HOST_LIST: external,dynamic-runner-controller.dynamic-runner.svc.cluster.local
|
||||||
mailer:
|
mailer:
|
||||||
# Outbound mail via the in-cluster Postfix+OAuth relay (see ../smtp-relay/).
|
# Outbound mail via the in-cluster Postfix+OAuth relay (see ../smtp-relay/).
|
||||||
# Plain SMTP on :25 — the relay does STARTTLS + OAuth to M365. From must be the
|
# Plain SMTP on :25 — the relay does STARTTLS + OAuth to M365. From must be the
|
||||||
|
|||||||
@@ -0,0 +1,39 @@
|
|||||||
|
apiVersion: helm.toolkit.fluxcd.io/v2
|
||||||
|
kind: HelmRelease
|
||||||
|
metadata:
|
||||||
|
name: gitea
|
||||||
|
namespace: gitea
|
||||||
|
spec:
|
||||||
|
chart:
|
||||||
|
spec:
|
||||||
|
chart: gitea
|
||||||
|
interval: 1h
|
||||||
|
sourceRef:
|
||||||
|
kind: HelmRepository
|
||||||
|
name: gitea-charts
|
||||||
|
version: 12.7.0
|
||||||
|
driftDetection:
|
||||||
|
mode: enabled
|
||||||
|
install:
|
||||||
|
strategy:
|
||||||
|
name: RetryOnFailure
|
||||||
|
retryInterval: 5m
|
||||||
|
interval: 30m
|
||||||
|
releaseName: gitea
|
||||||
|
targetNamespace: gitea
|
||||||
|
timeout: 15m
|
||||||
|
upgrade:
|
||||||
|
strategy:
|
||||||
|
name: RetryOnFailure
|
||||||
|
retryInterval: 5m
|
||||||
|
# Keep the patch release explicit because chart 12.7.0 defaults to 1.27.0.
|
||||||
|
# This override is in the suspended HelmRelease itself so chart, image and
|
||||||
|
# suspension are applied atomically; changing the watched values ConfigMap in
|
||||||
|
# the same revision could otherwise trigger reconciliation first.
|
||||||
|
values:
|
||||||
|
image:
|
||||||
|
rootless: true
|
||||||
|
tag: "1.27.3"
|
||||||
|
valuesFrom:
|
||||||
|
- kind: ConfigMap
|
||||||
|
name: gitea-values
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
apiVersion: source.toolkit.fluxcd.io/v1
|
||||||
|
kind: HelmRepository
|
||||||
|
metadata:
|
||||||
|
name: gitea-charts
|
||||||
|
namespace: gitea
|
||||||
|
spec:
|
||||||
|
interval: 1h
|
||||||
|
url: https://dl.gitea.com/charts/
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||||
|
kind: Kustomization
|
||||||
|
|
||||||
|
generatorOptions:
|
||||||
|
disableNameSuffixHash: true
|
||||||
|
annotations:
|
||||||
|
reconcile.fluxcd.io/watch: Enabled
|
||||||
|
|
||||||
|
configMapGenerator:
|
||||||
|
- name: gitea-values
|
||||||
|
namespace: gitea
|
||||||
|
files:
|
||||||
|
- values.yaml=gitea-values.yaml
|
||||||
|
|
||||||
|
resources:
|
||||||
|
- helmrepository.yaml
|
||||||
|
- helmrelease.yaml
|
||||||
|
- httproute.yaml
|
||||||
+21
-16
@@ -1,22 +1,27 @@
|
|||||||
# HTTP Echo (Gateway API workload)
|
# HTTP Echo(Flux GitOps canary)
|
||||||
|
|
||||||
**Purpose**
|
这是 Flux 首个低风险工作负载,用于验证 PR 合并后的自动部署、健康检查和漂移修复。
|
||||||
- Preserve a tiny HTTP echo `Deployment + Service` as a future GitOps canary.
|
Flux 将它部署到独立的 `gitops-canary` namespace。
|
||||||
- The current `HTTPRoute` still references the retired `contour-gateway`; do not
|
|
||||||
apply this folder until it is migrated and reviewed against Envoy Gateway.
|
|
||||||
|
|
||||||
**Resources**
|
| 文件 | 说明 |
|
||||||
| File | Description |
|
|
||||||
| --- | --- |
|
| --- | --- |
|
||||||
| `deployment.yaml` | Two replicas of `hashicorp/http-echo` returning `hello from contour gateway`. |
|
| `deployment.yaml` | 两个 `hashicorp/http-echo` 副本。 |
|
||||||
| `service.yaml` | ClusterIP service on port 80 targeted by the route. |
|
| `service.yaml` | 只在集群内可达的 ClusterIP Service。 |
|
||||||
| `httproute.yaml` | Gateway API `HTTPRoute` that targets `contour-gateway` and the `http-echo` service. |
|
| `kustomization.yaml` | Flux 实际构建入口;明确排除历史 HTTPRoute。 |
|
||||||
|
| `httproute.yaml` | 保留的历史 Contour 示例,**不在 Kustomization 中,不会部署**。 |
|
||||||
|
|
||||||
**How to verify**
|
## 验证
|
||||||
|
|
||||||
Migration and verification are intentionally deferred. Update `parentRefs` to the
|
合并 canary PR 后不手动 apply。等待 Flux 自动创建资源:
|
||||||
reviewed Envoy Gateway and choose a hostname covered by its listener before apply.
|
|
||||||
|
|
||||||
**Notes**
|
```bash
|
||||||
- This README records the old test intent; the retired Contour manifests are
|
sudo k3s kubectl -n flux-system get kustomization http-echo
|
||||||
available only in legacy Git history.
|
sudo k3s kubectl -n gitops-canary get deployment,service,pod
|
||||||
|
```
|
||||||
|
|
||||||
|
Deployment 漂移修复已经验证:手动把 replicas 改成 1 后,Flux 能按 Git 恢复为
|
||||||
|
2。删除验证使用无业务依赖的 `flux-prune-canary` ConfigMap。首次尝试在同一个
|
||||||
|
revision 中同时开启 prune 并删除对象,Flux 更新了 inventory 但保留了 live 对象。
|
||||||
|
因此先在已经生效的 `prune: true` 下重新纳管 ConfigMap,再用下一 revision 只删除
|
||||||
|
对象。最终删除阶段不再修改 prune,确保可以准确验证垃圾回收。root Kustomization
|
||||||
|
始终保持 `prune: false`,brownfield 资源不会进入此次删除范围。
|
||||||
|
|||||||
@@ -19,6 +19,6 @@ spec:
|
|||||||
- name: http-echo
|
- name: http-echo
|
||||||
image: hashicorp/http-echo:0.2.3
|
image: hashicorp/http-echo:0.2.3
|
||||||
args:
|
args:
|
||||||
- '-text=hello from contour gateway'
|
- '-text=hello from flux gitops canary'
|
||||||
ports:
|
ports:
|
||||||
- containerPort: 5678
|
- containerPort: 5678
|
||||||
|
|||||||
@@ -0,0 +1,5 @@
|
|||||||
|
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||||
|
kind: Kustomization
|
||||||
|
resources:
|
||||||
|
- deployment.yaml
|
||||||
|
- service.yaml
|
||||||
@@ -52,11 +52,16 @@ prefix_roles:
|
|||||||
# range. It makes the collision VISIBLE — the range shows 100% utilised and the
|
# range. It makes the collision VISIBLE — the range shows 100% utilised and the
|
||||||
# address never appears as a suggestion — where plain YAML shows nothing at all.
|
# address never appears as a suggestion — where plain YAML shows nothing at all.
|
||||||
ip_ranges:
|
ip_ranges:
|
||||||
- start: 192.168.10.10/24
|
- start: 192.168.10.128/24
|
||||||
end: 192.168.10.250/24
|
end: 192.168.10.250/24
|
||||||
status: active
|
status: active
|
||||||
mark_utilized: true
|
mark_utilized: true
|
||||||
description: "NEC IX DHCP pool — do NOT statically allocate inside this."
|
description: "NEC IX DHCP pool, updated 2026-09-14. Do NOT statically allocate inside this."
|
||||||
|
- start: 192.168.10.251/24
|
||||||
|
end: 192.168.10.254/24
|
||||||
|
status: reserved
|
||||||
|
mark_utilized: true
|
||||||
|
description: "用户确认预留,尚未分配;不可按扫描无响应视为空闲。"
|
||||||
|
|
||||||
vlan_group:
|
vlan_group:
|
||||||
name: lab
|
name: lab
|
||||||
@@ -144,8 +149,8 @@ devices:
|
|||||||
role: hypervisor
|
role: hypervisor
|
||||||
type: 10vgcto1ww
|
type: 10vgcto1ww
|
||||||
serial: PC1AGX1Q
|
serial: PC1AGX1Q
|
||||||
description: "Proxmox VE 9.2. LINSTOR satellite. The node that randomly froze."
|
description: "Proxmox VE 9.2. LINSTOR satellite."
|
||||||
comments: "AMD Ryzen 5 PRO 2400GE w/ Vega, 8 threads, 7 GiB RAM. BIOS M1XKT45A. Raven Ridge idle bug fixed in BIOS: Power Supply Idle Control = Typical Current Idle."
|
comments: "AMD Ryzen 5 PRO 2400GE w/ Vega, 8 threads, 7 GiB RAM. BIOS M1XKT45A."
|
||||||
interfaces:
|
interfaces:
|
||||||
- { name: vmbr0, type: bridge, ip: 192.168.10.7/24, primary: true, mtu: 9000, dns_name: pve2.ad.ddupan.top }
|
- { name: vmbr0, type: bridge, ip: 192.168.10.7/24, primary: true, mtu: 9000, dns_name: pve2.ad.ddupan.top }
|
||||||
|
|
||||||
@@ -154,7 +159,7 @@ devices:
|
|||||||
type: 10vgcto1ww
|
type: 10vgcto1ww
|
||||||
serial: PC1AGX1P
|
serial: PC1AGX1P
|
||||||
description: "Proxmox VE 9.2. LINSTOR satellite."
|
description: "Proxmox VE 9.2. LINSTOR satellite."
|
||||||
comments: "AMD Ryzen 5 PRO 2400GE w/ Vega, 8 threads, 7 GiB RAM. BIOS M1XKT55A. Same silicon as pve2, so susceptible to the same idle bug in principle."
|
comments: "AMD Ryzen 5 PRO 2400GE w/ Vega, 8 threads, 7 GiB RAM. BIOS M1XKT55A."
|
||||||
interfaces:
|
interfaces:
|
||||||
- { name: vmbr0, type: bridge, ip: 192.168.10.9/24, primary: true, mtu: 9000, dns_name: pve3.ad.ddupan.top }
|
- { name: vmbr0, type: bridge, ip: 192.168.10.9/24, primary: true, mtu: 9000, dns_name: pve3.ad.ddupan.top }
|
||||||
|
|
||||||
@@ -181,9 +186,8 @@ devices:
|
|||||||
# Wi-Fi. Runs as an AP/bridge, not a router — the NEC IX is the gateway, so this box's
|
# Wi-Fi. Runs as an AP/bridge, not a router — the NEC IX is the gateway, so this box's
|
||||||
# routing, NAT and DHCP are not in play. Wireless clients land directly on the flat LAN.
|
# routing, NAT and DHCP are not in play. Wireless clients land directly on the flat LAN.
|
||||||
#
|
#
|
||||||
# ⚠ Its address .10 is the FIRST ADDRESS OF THE DHCP POOL above. Either it holds a lease
|
# 2026-09-14: NEC IX 为此 MAC 固定分配 .10;动态池已迁到 .128–.250。
|
||||||
# (so the address can move) or it is a static that overlaps the pool. NetBox surfaces
|
# 操作与回滚记录:infrastructure/samba-ad/router-dhcp-nec-ix.md。
|
||||||
# the overlap; the underlying config still needs a decision. See ../README.md.
|
|
||||||
#
|
#
|
||||||
# Identified by MAC OUI d4:2c:46 = BUFFALO.INC plus the model string on its login page.
|
# Identified by MAC OUI d4:2c:46 = BUFFALO.INC plus the model string on its login page.
|
||||||
- name: ap-buffalo
|
- name: ap-buffalo
|
||||||
|
|||||||
@@ -12,3 +12,10 @@ services:
|
|||||||
- "38008:38008"
|
- "38008:38008"
|
||||||
volumes:
|
volumes:
|
||||||
- "/mnt/pool/games/ps3:/games:rw"
|
- "/mnt/pool/games/ps3:/games:rw"
|
||||||
|
|
||||||
|
# 避免与 DN42 的 172.20.0.0/14 重叠。
|
||||||
|
networks:
|
||||||
|
default:
|
||||||
|
ipam:
|
||||||
|
config:
|
||||||
|
- subnet: 172.28.1.0/24
|
||||||
|
|||||||
@@ -11,7 +11,9 @@
|
|||||||
| `helm.sh` | Installs or upgrades the SeaweedFS release. |
|
| `helm.sh` | Installs or upgrades the SeaweedFS release. |
|
||||||
|
|
||||||
**Install**
|
**Install**
|
||||||
1. Set real S3 access and secret keys in `values.yaml`.
|
1. 在 OpenBao `kv/k8s/seaweedfs-s3` 维护基础 S3 配置;zot 凭据单独以
|
||||||
|
`kv/k8s/zot-s3` 为唯一来源。ESO 合成为 `seaweedfs-s3-config`,详见下文。
|
||||||
|
不要把真实 AK/SK 放进 `values.yaml`。
|
||||||
2. Apply the manifests:
|
2. Apply the manifests:
|
||||||
```bash
|
```bash
|
||||||
bash ~/services/apps/seaweedfs/helm.sh
|
bash ~/services/apps/seaweedfs/helm.sh
|
||||||
@@ -31,4 +33,22 @@
|
|||||||
|
|
||||||
**Notes**
|
**Notes**
|
||||||
- The chart manages master, volume, filer, S3, and admin components.
|
- The chart manages master, volume, filer, S3, and admin components.
|
||||||
- The chart-managed S3 secret uses the current AK/SK pair for the admin user.
|
- The filer uses the ESO-managed `seaweedfs-s3-config` Secret for static S3 identities.
|
||||||
|
|
||||||
|
## zot 制品存储
|
||||||
|
|
||||||
|
`zot` bucket 专用于 [zot Registry](../zot/README.md),OCI 数据位于 `registry/`
|
||||||
|
前缀。静态身份 `zot` 只有该 bucket 的 Read/Write/List/Tagging 权限,凭据唯一来源为
|
||||||
|
Bao `kv/k8s/zot-s3` 的 `access_key` / `secret_key`,同时供 zot consumer 和
|
||||||
|
SeaweedFS 服务端使用。
|
||||||
|
|
||||||
|
[ExternalSecret 模板](../../platform/external-secrets/externalsecrets.yaml) 保留
|
||||||
|
`kv/k8s/seaweedfs-s3` 的原有身份及其他配置,再追加 zot 身份与限定 bucket 的权限。
|
||||||
|
基础配置当前版本不保存 zot AK/SK;旧 KV 版本历史仍保留。新增其他身份时使用
|
||||||
|
KV compare-and-set 保留已有内容,不覆盖 Terraform 或其他应用的 AK/SK。
|
||||||
|
不要直接编辑生成的 Kubernetes Secret。该 ExternalSecret 已单独应用到集群,
|
||||||
|
目前仍未加入 ESO 的 Flux Kustomization,遵循该组件现有 ownership 边界。
|
||||||
|
|
||||||
|
运行版本 `4.22` 可在 Secret volume 更新后向 filer/内嵌 S3 的 `weed` 进程发送
|
||||||
|
SIGHUP,重新加载静态配置,无需重启共享 S3 服务。本次接入没有启用 SeaweedFS
|
||||||
|
OIDC/STS;SPIRE 认证发生在 zot 的客户端入口。
|
||||||
|
|||||||
@@ -1,5 +0,0 @@
|
|||||||
route:
|
|
||||||
receiver: blackhole
|
|
||||||
|
|
||||||
receivers:
|
|
||||||
- name: blackhole
|
|
||||||
@@ -1,94 +0,0 @@
|
|||||||
services:
|
|
||||||
# Metrics collector.
|
|
||||||
# It scrapes targets defined in --promscrape.config
|
|
||||||
# And forward them to --remoteWrite.url
|
|
||||||
vmagent:
|
|
||||||
image: victoriametrics/vmagent:v1.132.0
|
|
||||||
depends_on:
|
|
||||||
- "victoriametrics"
|
|
||||||
ports:
|
|
||||||
- 8429:8429
|
|
||||||
volumes:
|
|
||||||
- vmagentdata:/vmagentdata
|
|
||||||
- ./prometheus.yaml:/etc/prometheus/prometheus.yml
|
|
||||||
command:
|
|
||||||
- "--promscrape.config=/etc/prometheus/prometheus.yml"
|
|
||||||
- "--remoteWrite.url=http://victoriametrics:8428/api/v1/write"
|
|
||||||
restart: always
|
|
||||||
# VictoriaMetrics instance, a single process responsible for
|
|
||||||
# storing metrics and serve read requests.
|
|
||||||
victoriametrics:
|
|
||||||
image: victoriametrics/victoria-metrics:v1.132.0
|
|
||||||
ports:
|
|
||||||
- 8428:8428
|
|
||||||
- 8089:8089
|
|
||||||
- 8089:8089/udp
|
|
||||||
- 2003:2003
|
|
||||||
- 2003:2003/udp
|
|
||||||
- 4242:4242
|
|
||||||
volumes:
|
|
||||||
- vmdata:/storage
|
|
||||||
command:
|
|
||||||
- "--storageDataPath=/storage"
|
|
||||||
- "--graphiteListenAddr=:2003"
|
|
||||||
- "--opentsdbListenAddr=:4242"
|
|
||||||
- "--httpListenAddr=:8428"
|
|
||||||
- "--influxListenAddr=:8089"
|
|
||||||
- "--vmalert.proxyURL=http://vmalert:8880"
|
|
||||||
restart: always
|
|
||||||
|
|
||||||
grafana:
|
|
||||||
image: grafana/grafana:12.2.0
|
|
||||||
depends_on:
|
|
||||||
- "victoriametrics"
|
|
||||||
ports:
|
|
||||||
- 3000:3000
|
|
||||||
volumes:
|
|
||||||
- grafanadata:/var/lib/grafana
|
|
||||||
- ./provisioning/datasources/prometheus-datasource/single.yml:/etc/grafana/provisioning/datasources/single.yml
|
|
||||||
- ./provisioning/dashboards:/etc/grafana/provisioning/dashboards
|
|
||||||
- ./provisioning/dashboards/victoriametrics.json:/var/lib/grafana/dashboards/vm.json
|
|
||||||
- ./provisioning/dashboards/vmagent.json:/var/lib/grafana/dashboards/vmagent.json
|
|
||||||
- ./provisioning/dashboards/vmalert.json:/var/lib/grafana/dashboards/vmalert.json
|
|
||||||
restart: always
|
|
||||||
|
|
||||||
# vmalert executes alerting and recording rules
|
|
||||||
vmalert:
|
|
||||||
image: victoriametrics/vmalert:v1.132.0
|
|
||||||
depends_on:
|
|
||||||
- "victoriametrics"
|
|
||||||
- "alertmanager"
|
|
||||||
ports:
|
|
||||||
- 8880:8880
|
|
||||||
volumes:
|
|
||||||
- ./rules/alerts.yml:/etc/alerts/alerts.yml
|
|
||||||
- ./rules/alerts-health.yml:/etc/alerts/alerts-health.yml
|
|
||||||
- ./rules/alerts-vmagent.yml:/etc/alerts/alerts-vmagent.yml
|
|
||||||
- ./rules/alerts-vmalert.yml:/etc/alerts/alerts-vmalert.yml
|
|
||||||
command:
|
|
||||||
- "--datasource.url=http://victoriametrics:8428/"
|
|
||||||
- "--remoteRead.url=http://victoriametrics:8428/"
|
|
||||||
- "--remoteWrite.url=http://vmagent:8429/"
|
|
||||||
- "--notifier.url=http://alertmanager:9093/"
|
|
||||||
- "--rule=/etc/alerts/*.yml"
|
|
||||||
# display source of alerts in grafana
|
|
||||||
- "--external.url=http://127.0.0.1:3000" #grafana outside container
|
|
||||||
- '--external.alert.source=explore?orgId=1&left={"datasource":"VictoriaMetrics","queries":[{"expr":{{.Expr|jsonEscape|queryEscape}},"refId":"A"}],"range":{"from":"{{ .ActiveAt.UnixMilli }}","to":"now"}}'
|
|
||||||
restart: always
|
|
||||||
|
|
||||||
# alertmanager receives alerting notifications from vmalert
|
|
||||||
# and distributes them according to --config.file.
|
|
||||||
alertmanager:
|
|
||||||
image: prom/alertmanager:v0.28.1
|
|
||||||
volumes:
|
|
||||||
- ./alertmanager.yaml:/config/alertmanager.yml
|
|
||||||
command:
|
|
||||||
- "--config.file=/config/alertmanager.yml"
|
|
||||||
ports:
|
|
||||||
- 9093:9093
|
|
||||||
restart: always
|
|
||||||
|
|
||||||
volumes:
|
|
||||||
vmagentdata: {}
|
|
||||||
vmdata: {}
|
|
||||||
grafanadata: {}
|
|
||||||
@@ -1,16 +0,0 @@
|
|||||||
global:
|
|
||||||
scrape_interval: 10s
|
|
||||||
|
|
||||||
scrape_configs:
|
|
||||||
- job_name: vmagent
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- vmagent:8429
|
|
||||||
- job_name: vmalert
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- vmalert:8880
|
|
||||||
- job_name: victoriametrics
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- victoriametrics:8428
|
|
||||||
@@ -1,9 +0,0 @@
|
|||||||
apiVersion: 1
|
|
||||||
|
|
||||||
providers:
|
|
||||||
- name: Prometheus
|
|
||||||
orgId: 1
|
|
||||||
folder: ''
|
|
||||||
type: file
|
|
||||||
options:
|
|
||||||
path: /var/lib/grafana/dashboards
|
|
||||||
-11
@@ -1,11 +0,0 @@
|
|||||||
apiVersion: 1
|
|
||||||
|
|
||||||
datasources:
|
|
||||||
- name: VictoriaMetrics
|
|
||||||
type: prometheus
|
|
||||||
access: proxy
|
|
||||||
url: http://victoriametrics:8428
|
|
||||||
isDefault: true
|
|
||||||
jsonData:
|
|
||||||
prometheusType: Prometheus
|
|
||||||
prometheusVersion: 2.24.0
|
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -1,11 +0,0 @@
|
|||||||
apiVersion: 1
|
|
||||||
|
|
||||||
datasources:
|
|
||||||
- name: VictoriaMetrics
|
|
||||||
type: prometheus
|
|
||||||
access: proxy
|
|
||||||
url: http://victoriametrics:8428
|
|
||||||
isDefault: true
|
|
||||||
jsonData:
|
|
||||||
prometheusType: Prometheus
|
|
||||||
prometheusVersion: 2.24.0
|
|
||||||
@@ -1,149 +0,0 @@
|
|||||||
# File contains default list of alerts for various VM components.
|
|
||||||
# The following alerts are recommended for use for any VM installation.
|
|
||||||
# The alerts below are just recommendations and may require some updates
|
|
||||||
# and threshold calibration according to every specific setup.
|
|
||||||
groups:
|
|
||||||
- name: vm-health
|
|
||||||
# note the `job` filter and update accordingly to your setup
|
|
||||||
rules:
|
|
||||||
- alert: TooManyRestarts
|
|
||||||
expr: changes(process_start_time_seconds{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"}[15m]) > 2
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "{{ $labels.job }} too many restarts (instance {{ $labels.instance }})"
|
|
||||||
description: >
|
|
||||||
Job {{ $labels.job }} (instance {{ $labels.instance }}) has restarted more than twice in the last 15 minutes.
|
|
||||||
It might be crashlooping.
|
|
||||||
|
|
||||||
- alert: ServiceDown
|
|
||||||
expr: up{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"} == 0
|
|
||||||
for: 2m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "Service {{ $labels.job }} is down on {{ $labels.instance }}"
|
|
||||||
description: "{{ $labels.instance }} of job {{ $labels.job }} has been down for more than 2 minutes."
|
|
||||||
|
|
||||||
- alert: ProcessNearFDLimits
|
|
||||||
expr: (process_max_fds - process_open_fds) < 100
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "Number of free file descriptors is less than 100 for \"{{ $labels.job }}\"(\"{{ $labels.instance }}\") for the last 5m"
|
|
||||||
description: |
|
|
||||||
Exhausting OS file descriptors limit can cause severe degradation of the process.
|
|
||||||
Consider to increase the limit as fast as possible.
|
|
||||||
|
|
||||||
- alert: TooHighMemoryUsage
|
|
||||||
expr: (min_over_time(process_resident_memory_anon_bytes[10m]) / vm_available_memory_bytes) > 0.8
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "It is more than 80% of memory used by \"{{ $labels.job }}\"(\"{{ $labels.instance }}\")"
|
|
||||||
description: |
|
|
||||||
Too high memory usage may result into multiple issues such as OOMs or degraded performance.
|
|
||||||
Consider to either increase available memory or decrease the load on the process.
|
|
||||||
|
|
||||||
- alert: TooHighCPUUsage
|
|
||||||
expr: rate(process_cpu_seconds_total[5m]) / process_cpu_cores_available > 0.9
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "More than 90% of CPU is used by \"{{ $labels.job }}\"(\"{{ $labels.instance }}\") during the last 5m"
|
|
||||||
description: >
|
|
||||||
Too high CPU usage may be a sign of insufficient resources and make process unstable.
|
|
||||||
Consider to either increase available CPU resources or decrease the load on the process.
|
|
||||||
|
|
||||||
- alert: TooHighGoroutineSchedulingLatency
|
|
||||||
expr: histogram_quantile(0.99, sum(rate(go_sched_latencies_seconds_bucket{job=~".*(victoriametrics|vmselect|vminsert|vmstorage|vmagent|vmalert|vmsingle|vmalertmanager|vmauth).*"}[5m])) by (le, job, instance)) > 0.1
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "\"{{ $labels.job }}\"(\"{{ $labels.instance }}\") has insufficient CPU resources for >15m"
|
|
||||||
description: >
|
|
||||||
Go runtime is unable to schedule goroutines execution in acceptable time. This is usually a sign of
|
|
||||||
insufficient CPU resources or CPU throttling. Verify that service has enough CPU resources. Otherwise,
|
|
||||||
the service could work unreliably with delays in processing.
|
|
||||||
|
|
||||||
- alert: TooManyLogs
|
|
||||||
expr: sum(increase(vm_log_messages_total{level="error"}[5m])) without (app_version, location) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Too many logs printed for job \"{{ $labels.job }}\" ({{ $labels.instance }})"
|
|
||||||
description: >
|
|
||||||
Logging rate for job \"{{ $labels.job }}\" ({{ $labels.instance }}) is {{ $value }} for last 15m.
|
|
||||||
Worth to check logs for specific error messages.
|
|
||||||
|
|
||||||
- alert: TooManyTSIDMisses
|
|
||||||
expr: increase(vm_missing_tsids_for_metric_id_total[5m]) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "Unexpected TSID misses for job \"{{ $labels.job }}\" ({{ $labels.instance }}) for the last 15 minutes"
|
|
||||||
description: |
|
|
||||||
Unexpected TSID misses for \"{{ $labels.job }}\" ({{ $labels.instance }}) for the last 15 minutes.
|
|
||||||
If this happens after unclean shutdown of VictoriaMetrics process (via \"kill -9\", OOM or power off),
|
|
||||||
then this is OK - the alert must go away in a few minutes after the restart.
|
|
||||||
Otherwise this may point to the corruption of index data.
|
|
||||||
|
|
||||||
- alert: ConcurrentInsertsHitTheLimit
|
|
||||||
expr: avg_over_time(vm_concurrent_insert_current[1m]) >= vm_concurrent_insert_capacity
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "{{ $labels.job }} on instance {{ $labels.instance }} is constantly hitting concurrent inserts limit"
|
|
||||||
description: |
|
|
||||||
The limit of concurrent inserts on instance {{ $labels.instance }} depends on the number of CPUs.
|
|
||||||
Usually, when component constantly hits the limit it is likely the component is overloaded and requires more CPU.
|
|
||||||
In some cases for components like vmagent or vminsert the alert might trigger if there are too many clients
|
|
||||||
making write attempts. If vmagent's or vminsert's CPU usage and network saturation are at normal level, then
|
|
||||||
it might be worth adjusting `-maxConcurrentInserts` cmd-line flag.
|
|
||||||
|
|
||||||
- alert: IndexDBRecordsDrop
|
|
||||||
expr: increase(vm_indexdb_items_dropped_total[5m]) > 0
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "IndexDB skipped registering items during data ingestion with reason={{ $labels.reason }}."
|
|
||||||
description: |
|
|
||||||
VictoriaMetrics could skip registering new timeseries during ingestion if they fail the validation process.
|
|
||||||
For example, `reason=too_long_item` means that time series cannot exceed 64KB. Please, reduce the number
|
|
||||||
of labels or label values for such series. Or enforce these limits via `-maxLabelsPerTimeseries` and
|
|
||||||
`-maxLabelValueLen` command-line flags.
|
|
||||||
|
|
||||||
- alert: RowsRejectedOnIngestion
|
|
||||||
expr: rate(vm_rows_ignored_total[5m]) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Some rows are rejected on \"{{ $labels.instance }}\" on ingestion attempt"
|
|
||||||
description: "Ingested rows on instance \"{{ $labels.instance }}\" are rejected due to the
|
|
||||||
following reason: \"{{ $labels.reason }}\""
|
|
||||||
|
|
||||||
- alert: TooHighQueryLoad
|
|
||||||
expr: increase(vm_concurrent_select_limit_timeout_total[5m]) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Read queries fail with timeout for {{ $labels.job }} on instance {{ $labels.instance }}"
|
|
||||||
description: |
|
|
||||||
Instance {{ $labels.instance }} ({{ $labels.job }}) is failing to serve read queries during last 15m.
|
|
||||||
Concurrency limit `-search.maxConcurrentRequests` was reached on this instance and extra queries were
|
|
||||||
put into the queue for `-search.maxQueueDuration` interval. But even after waiting in the queue these queries weren't served.
|
|
||||||
This happens if instance is overloaded with the current workload, or datasource is too slow to respond.
|
|
||||||
Possible solutions are the following:
|
|
||||||
* reduce the query load;
|
|
||||||
* increase compute resources or number of replicas;
|
|
||||||
* adjust limits `-search.maxConcurrentRequests` and `-search.maxQueueDuration`.
|
|
||||||
See more at https://docs.victoriametrics.com/victoriametrics/troubleshooting/#slow-queries
|
|
||||||
@@ -1,172 +0,0 @@
|
|||||||
# File contains default list of alerts for vmagent service.
|
|
||||||
# The alerts below are just recommendations and may require some updates
|
|
||||||
# and threshold calibration according to every specific setup.
|
|
||||||
groups:
|
|
||||||
# Alerts group for vmagent assumes that Grafana dashboard
|
|
||||||
# https://grafana.com/grafana/dashboards/12683 is installed.
|
|
||||||
# Pls update the `dashboard` annotation according to your setup.
|
|
||||||
- name: vmagent
|
|
||||||
interval: 30s
|
|
||||||
concurrency: 2
|
|
||||||
rules:
|
|
||||||
- alert: PersistentQueueIsDroppingData
|
|
||||||
expr: sum(increase(vm_persistentqueue_bytes_dropped_total[5m])) without (path) > 0
|
|
||||||
for: 10m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=49&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Instance {{ $labels.instance }} is dropping data from persistent queue"
|
|
||||||
description: "Vmagent dropped {{ $value | humanize1024 }} from persistent queue
|
|
||||||
on instance {{ $labels.instance }} for the last 10m."
|
|
||||||
|
|
||||||
- alert: RejectedRemoteWriteDataBlocksAreDropped
|
|
||||||
expr: sum(increase(vmagent_remotewrite_packets_dropped_total[5m])) without (url) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=79&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Vmagent is dropping data blocks that are rejected by remote storage"
|
|
||||||
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} drops the rejected by
|
|
||||||
remote-write server data blocks. Check the logs to find the reason for rejects."
|
|
||||||
|
|
||||||
- alert: TooManyScrapeErrors
|
|
||||||
expr: increase(vm_promscrape_scrapes_failed_total[5m]) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=31&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Vmagent fails to scrape one or more targets"
|
|
||||||
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} fails to scrape targets for last 15m"
|
|
||||||
|
|
||||||
- alert: ScrapePoolHasNoTargets
|
|
||||||
expr: sum(vm_promscrape_scrape_pool_targets) without (status, instance, pod) == 0
|
|
||||||
for: 30m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Vmagent has scrape_pool with 0 configured/discovered targets"
|
|
||||||
description: "Vmagent \"{{ $labels.job }}\" has scrape_pool \"{{ $labels.scrape_job }}\"
|
|
||||||
with 0 discovered targets. It is likely a misconfiguration. Please follow https://docs.victoriametrics.com/victoriametrics/vmagent/#debugging-scrape-targets
|
|
||||||
to troubleshoot the scraping config."
|
|
||||||
|
|
||||||
- alert: TooManyWriteErrors
|
|
||||||
expr: |
|
|
||||||
(sum(increase(vm_ingestserver_request_errors_total[5m])) without (name,net,type)
|
|
||||||
+
|
|
||||||
sum(increase(vmagent_http_request_errors_total[5m])) without (path,protocol)) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=77&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Vmagent responds with too many errors on data ingestion protocols"
|
|
||||||
description: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} responds with errors to write requests for last 15m."
|
|
||||||
|
|
||||||
- alert: TooManyRemoteWriteErrors
|
|
||||||
expr: rate(vmagent_remotewrite_retries_count_total[5m]) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=61&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Job \"{{ $labels.job }}\" on instance {{ $labels.instance }} fails to push to remote storage"
|
|
||||||
description: "Vmagent fails to push data via remote write protocol to destination \"{{ $labels.url }}\"\n
|
|
||||||
Ensure that destination is up and reachable."
|
|
||||||
|
|
||||||
- alert: RemoteWriteConnectionIsSaturated
|
|
||||||
expr: |
|
|
||||||
(
|
|
||||||
rate(vmagent_remotewrite_send_duration_seconds_total[5m])
|
|
||||||
/
|
|
||||||
vmagent_remotewrite_queues
|
|
||||||
) > 0.9
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=84&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Remote write connection from \"{{ $labels.job }}\" (instance {{ $labels.instance }}) to {{ $labels.url }} is saturated"
|
|
||||||
description: "The remote write connection between vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }}) and destination \"{{ $labels.url }}\"
|
|
||||||
is saturated by more than 90% and vmagent won't be able to keep up.\n
|
|
||||||
There could be the following reasons for this:\n
|
|
||||||
* vmagent can't send data fast enough through the existing network connections. Increase `-remoteWrite.queues` cmd-line flag value to establish more connections per destination.\n
|
|
||||||
* remote destination can't accept data fast enough. Check if remote destination has enough resources for processing."
|
|
||||||
|
|
||||||
- alert: PersistentQueueForWritesIsSaturated
|
|
||||||
expr: rate(vm_persistentqueue_write_duration_seconds_total[5m]) > 0.9
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=98&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Persistent queue writes for instance {{ $labels.instance }} are saturated"
|
|
||||||
description: "Persistent queue writes for vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }})
|
|
||||||
are saturated by more than 90% and vmagent won't be able to keep up with flushing data on disk.
|
|
||||||
In this case, consider to decrease load on the vmagent or improve the disk throughput."
|
|
||||||
|
|
||||||
- alert: PersistentQueueForReadsIsSaturated
|
|
||||||
expr: rate(vm_persistentqueue_read_duration_seconds_total[5m]) > 0.9
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=99&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Persistent queue reads for instance {{ $labels.instance }} are saturated"
|
|
||||||
description: "Persistent queue reads for vmagent \"{{ $labels.job }}\" (instance {{ $labels.instance }})
|
|
||||||
are saturated by more than 90% and vmagent won't be able to keep up with reading data from the disk.
|
|
||||||
In this case, consider to decrease load on the vmagent or improve the disk throughput."
|
|
||||||
|
|
||||||
- alert: SeriesLimitHourReached
|
|
||||||
expr: (vmagent_hourly_series_limit_current_series / vmagent_hourly_series_limit_max_series) > 0.9
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=88&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Instance {{ $labels.instance }} reached 90% of the limit"
|
|
||||||
description: "Max series limit set via -remoteWrite.maxHourlySeries flag is close to reaching the max value.
|
|
||||||
Then samples for new time series will be dropped instead of sending them to remote storage systems."
|
|
||||||
|
|
||||||
- alert: SeriesLimitDayReached
|
|
||||||
expr: (vmagent_daily_series_limit_current_series / vmagent_daily_series_limit_max_series) > 0.9
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/G7Z9GzMGz?viewPanel=90&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Instance {{ $labels.instance }} reached 90% of the limit"
|
|
||||||
description: "Max series limit set via -remoteWrite.maxDailySeries flag is close to reaching the max value.
|
|
||||||
Then samples for new time series will be dropped instead of sending them to remote storage systems."
|
|
||||||
|
|
||||||
- alert: ConfigurationReloadFailure
|
|
||||||
expr: |
|
|
||||||
vm_promscrape_config_last_reload_successful != 1
|
|
||||||
or
|
|
||||||
vmagent_relabel_config_last_reload_successful != 1
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Configuration reload failed for vmagent instance {{ $labels.instance }}"
|
|
||||||
description: "Configuration hot-reload failed for vmagent on instance {{ $labels.instance }}.
|
|
||||||
Check vmagent's logs for detailed error message."
|
|
||||||
|
|
||||||
- alert: StreamAggrFlushTimeout
|
|
||||||
expr: |
|
|
||||||
increase(vm_streamaggr_flush_timeouts_total[5m]) > 0
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Streaming aggregation at \"{{ $labels.job }}\" (instance {{ $labels.instance }}) can't be finished within the configured aggregation interval."
|
|
||||||
description: "Stream aggregation process can't keep up with the load and might produce incorrect aggregation results. Check logs for more details.
|
|
||||||
Possible solutions: increase aggregation interval; aggregate smaller number of series; reduce samples' ingestion rate to stream aggregation."
|
|
||||||
|
|
||||||
- alert: StreamAggrDedupFlushTimeout
|
|
||||||
expr: |
|
|
||||||
increase(vm_streamaggr_dedup_flush_timeouts_total[5m]) > 0
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Deduplication \"{{ $labels.job }}\" (instance {{ $labels.instance }}) can't be finished within configured deduplication interval."
|
|
||||||
description: "Deduplication process can't keep up with the load and might produce incorrect results. Check docs https://docs.victoriametrics.com/victoriametrics/stream-aggregation/#deduplication and logs for more details.
|
|
||||||
Possible solutions: increase deduplication interval; deduplicate smaller number of series; reduce samples' ingestion rate."
|
|
||||||
@@ -1,96 +0,0 @@
|
|||||||
# File contains default list of alerts for vmalert service.
|
|
||||||
# The alerts below are just recommendations and may require some updates
|
|
||||||
# and threshold calibration according to every specific setup.
|
|
||||||
groups:
|
|
||||||
# Alerts group for vmalert assumes that Grafana dashboard
|
|
||||||
# https://grafana.com/grafana/dashboards/14950 is installed.
|
|
||||||
# Pls update the `dashboard` annotation according to your setup.
|
|
||||||
- name: vmalert
|
|
||||||
interval: 30s
|
|
||||||
rules:
|
|
||||||
- alert: ConfigurationReloadFailure
|
|
||||||
expr: vmalert_config_last_reload_successful != 1
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "Configuration reload failed for vmalert instance {{ $labels.instance }}"
|
|
||||||
description: "Configuration hot-reload failed for vmalert on instance {{ $labels.instance }}.
|
|
||||||
Check vmalert's logs for detailed error message."
|
|
||||||
|
|
||||||
- alert: AlertingRulesError
|
|
||||||
expr: sum(increase(vmalert_alerting_rules_errors_total[5m])) without(id) > 0
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=13&var-instance={{ $labels.instance }}&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
|
|
||||||
summary: "Alerting rules are failing for vmalert instance {{ $labels.instance }}"
|
|
||||||
description: "Alerting rules execution is failing for \"{{ $labels.alertname }}\" from group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
|
|
||||||
Check vmalert's logs for detailed error message."
|
|
||||||
|
|
||||||
- alert: RecordingRulesError
|
|
||||||
expr: sum(increase(vmalert_recording_rules_errors_total[5m])) without(id) > 0
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=30&var-instance={{ $labels.instance }}&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
|
|
||||||
summary: "Recording rules are failing for vmalert instance {{ $labels.instance }}"
|
|
||||||
description: "Recording rules execution is failing for \"{{ $labels.recording }}\" from group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
|
|
||||||
Check vmalert's logs for detailed error message."
|
|
||||||
|
|
||||||
- alert: RecordingRulesNoData
|
|
||||||
expr: sum(vmalert_recording_rules_last_evaluation_samples) without(id) < 1
|
|
||||||
for: 30m
|
|
||||||
labels:
|
|
||||||
severity: info
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/LzldHAVnz?viewPanel=33&var-file={{ $labels.file }}&var-group={{ $labels.group }}"
|
|
||||||
summary: "Recording rule {{ $labels.recording }} ({{ $labels.group }}) produces no data"
|
|
||||||
description: "Recording rule \"{{ $labels.recording }}\" from group \"{{ $labels.group }}\ in file \"{{ $labels.file }}\"
|
|
||||||
produces 0 samples over the last 30min. It might be caused by a misconfiguration
|
|
||||||
or incorrect query expression."
|
|
||||||
|
|
||||||
- alert: TooManyMissedIterations
|
|
||||||
expr: increase(vmalert_iteration_missed_total[5m]) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "vmalert instance {{ $labels.instance }} is missing rules evaluations"
|
|
||||||
description: "vmalert instance {{ $labels.instance }} is missing rules evaluations for group \"{{ $labels.group }}\" in file \"{{ $labels.file }}\".
|
|
||||||
The group evaluation time takes longer than the configured evaluation interval. This may result in missed
|
|
||||||
alerting notifications or recording rules samples. Try increasing evaluation interval or concurrency of
|
|
||||||
group \"{{ $labels.group }}\". See https://docs.victoriametrics.com/victoriametrics/vmalert/#groups.
|
|
||||||
If rule expressions are taking longer than expected, please see https://docs.victoriametrics.com/victoriametrics/troubleshooting/#slow-queries."
|
|
||||||
|
|
||||||
- alert: RemoteWriteErrors
|
|
||||||
expr: increase(vmalert_remotewrite_errors_total[5m]) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "vmalert instance {{ $labels.instance }} is failing to push metrics to remote write URL"
|
|
||||||
description: "vmalert instance {{ $labels.instance }} is failing to push metrics generated via alerting
|
|
||||||
or recording rules to the configured remote write URL. Check vmalert's logs for detailed error message."
|
|
||||||
|
|
||||||
- alert: RemoteWriteDroppingData
|
|
||||||
expr: increase(vmalert_remotewrite_dropped_rows_total[5m]) > 0
|
|
||||||
for: 5m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
summary: "vmalert instance {{ $labels.instance }} is dropping data sent to remote write URL"
|
|
||||||
description: "vmalert instance {{ $labels.instance }} is failing to send results of alerting or recording rules
|
|
||||||
to the configured remote write URL. This may result into gaps in recording rules or alerts state.
|
|
||||||
Check vmalert's logs for detailed error message."
|
|
||||||
|
|
||||||
- alert: AlertmanagerErrors
|
|
||||||
expr: increase(vmalert_alerts_send_errors_total[5m]) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
summary: "vmalert instance {{ $labels.instance }} is failing to send notifications to Alertmanager"
|
|
||||||
description: "vmalert instance {{ $labels.instance }} is failing to send alert notifications to \"{{ $labels.addr }}\".
|
|
||||||
Check vmalert's logs for detailed error message."
|
|
||||||
@@ -1,138 +0,0 @@
|
|||||||
# File contains default list of alerts for VictoriaMetrics single server.
|
|
||||||
# The alerts below are just recommendations and may require some updates
|
|
||||||
# and threshold calibration according to every specific setup.
|
|
||||||
groups:
|
|
||||||
# Alerts group for VM single assumes that Grafana dashboard
|
|
||||||
# https://grafana.com/grafana/dashboards/10229 is installed.
|
|
||||||
# Pls update the `dashboard` annotation according to your setup.
|
|
||||||
- name: vmsingle
|
|
||||||
interval: 30s
|
|
||||||
concurrency: 2
|
|
||||||
rules:
|
|
||||||
- alert: DiskRunsOutOfSpaceIn3Days
|
|
||||||
expr: |
|
|
||||||
sum(vm_free_disk_space_bytes) without(path) /
|
|
||||||
(
|
|
||||||
(rate(vm_rows_added_to_storage_total[1d]) - sum(rate(vm_deduplicated_samples_total[1d])) without(type)) * (
|
|
||||||
sum(vm_data_size_bytes{type!~"indexdb.*"}) without(type) /
|
|
||||||
sum(vm_rows{type!~"indexdb.*"}) without(type)
|
|
||||||
)
|
|
||||||
+
|
|
||||||
rate(vm_new_timeseries_created_total[1d]) * (
|
|
||||||
sum(vm_data_size_bytes{type="indexdb/file"}) without(type)/
|
|
||||||
sum(vm_rows{type="indexdb/file"}) without(type)
|
|
||||||
)
|
|
||||||
) < 3 * 24 * 3600 > 0
|
|
||||||
for: 30m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=53&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Instance {{ $labels.instance }} will run out of disk space soon"
|
|
||||||
description: "Taking into account current ingestion rate, free disk space will be enough only
|
|
||||||
for {{ $value | humanizeDuration }} on instance {{ $labels.instance }}.\n
|
|
||||||
Consider to limit the ingestion rate, decrease retention or scale the disk space if possible."
|
|
||||||
|
|
||||||
- alert: NodeBecomesReadonlyIn3Days
|
|
||||||
expr: |
|
|
||||||
sum(vm_free_disk_space_bytes - vm_free_disk_space_limit_bytes) without(path) /
|
|
||||||
(
|
|
||||||
(rate(vm_rows_added_to_storage_total[1d]) - sum(rate(vm_deduplicated_samples_total[1d])) without(type)) * (
|
|
||||||
sum(vm_data_size_bytes{type!~"indexdb.*"}) without(type) /
|
|
||||||
sum(vm_rows{type!~"indexdb.*"}) without(type)
|
|
||||||
)
|
|
||||||
+
|
|
||||||
rate(vm_new_timeseries_created_total[1d]) * (
|
|
||||||
sum(vm_data_size_bytes{type="indexdb/file"}) without(type) /
|
|
||||||
sum(vm_rows{type="indexdb/file"}) without(type)
|
|
||||||
)
|
|
||||||
) < 3 * 24 * 3600 > 0
|
|
||||||
for: 30m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/oS7Bi_0Wz?viewPanel=53&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Instance {{ $labels.instance }} will become read-only in 3 days"
|
|
||||||
description: "Taking into account current ingestion rate and free disk space
|
|
||||||
instance {{ $labels.instance }} is writable for {{ $value | humanizeDuration }}.\n
|
|
||||||
Consider to limit the ingestion rate, decrease retention or scale the disk space up if possible."
|
|
||||||
|
|
||||||
- alert: DiskRunsOutOfSpace
|
|
||||||
expr: |
|
|
||||||
sum(vm_data_size_bytes) by(job, instance) /
|
|
||||||
(
|
|
||||||
sum(vm_free_disk_space_bytes) by(job, instance) +
|
|
||||||
sum(vm_data_size_bytes) by(job, instance)
|
|
||||||
) > 0.8
|
|
||||||
for: 30m
|
|
||||||
labels:
|
|
||||||
severity: critical
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=53&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Instance {{ $labels.instance }} (job={{ $labels.job }}) will run out of disk space soon"
|
|
||||||
description: "Disk utilisation on instance {{ $labels.instance }} is more than 80%.\n
|
|
||||||
Having less than 20% of free disk space could cripple merge processes and overall performance.
|
|
||||||
Consider to limit the ingestion rate, decrease retention or scale the disk space if possible."
|
|
||||||
|
|
||||||
- alert: RequestErrorsToAPI
|
|
||||||
expr: increase(vm_http_request_errors_total[5m]) > 0
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=35&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Too many errors served for path {{ $labels.path }} (instance {{ $labels.instance }})"
|
|
||||||
description: "Requests to path {{ $labels.path }} are receiving errors.
|
|
||||||
Please verify if clients are sending correct requests."
|
|
||||||
|
|
||||||
- alert: TooHighChurnRate
|
|
||||||
expr: |
|
|
||||||
(
|
|
||||||
sum(rate(vm_new_timeseries_created_total[5m])) by(instance)
|
|
||||||
/
|
|
||||||
sum(rate(vm_rows_inserted_total[5m])) by(instance)
|
|
||||||
) > 0.1
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=66&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Churn rate is more than 10% on \"{{ $labels.instance }}\" for the last 15m"
|
|
||||||
description: "VM constantly creates new time series on \"{{ $labels.instance }}\".\n
|
|
||||||
This effect is known as Churn Rate.\n
|
|
||||||
High Churn Rate is tightly connected with database performance and may
|
|
||||||
result in unexpected OOM's or slow queries."
|
|
||||||
|
|
||||||
- alert: TooHighChurnRate24h
|
|
||||||
expr: |
|
|
||||||
sum(increase(vm_new_timeseries_created_total[24h])) by(instance)
|
|
||||||
>
|
|
||||||
(sum(vm_cache_entries{type="storage/hour_metric_ids"}) by(instance) * 3)
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=66&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Too high number of new series on \"{{ $labels.instance }}\" created over last 24h"
|
|
||||||
description: "The number of created new time series over last 24h is 3x times higher than
|
|
||||||
current number of active series on \"{{ $labels.instance }}\".\n
|
|
||||||
This effect is known as Churn Rate.\n
|
|
||||||
High Churn Rate is tightly connected with database performance and may
|
|
||||||
result in unexpected OOM's or slow queries."
|
|
||||||
|
|
||||||
- alert: TooHighSlowInsertsRate
|
|
||||||
expr: |
|
|
||||||
(
|
|
||||||
sum(rate(vm_slow_row_inserts_total[5m])) by(instance)
|
|
||||||
/
|
|
||||||
sum(rate(vm_rows_inserted_total[5m])) by(instance)
|
|
||||||
) > 0.05
|
|
||||||
for: 15m
|
|
||||||
labels:
|
|
||||||
severity: warning
|
|
||||||
annotations:
|
|
||||||
dashboard: "http://localhost:3000/d/wNf0q_kZk?viewPanel=68&var-instance={{ $labels.instance }}"
|
|
||||||
summary: "Percentage of slow inserts is more than 5% on \"{{ $labels.instance }}\" for the last 15m"
|
|
||||||
description: "High rate of slow inserts on \"{{ $labels.instance }}\" may be a sign of resource exhaustion
|
|
||||||
for the current load. It is likely more RAM is needed for optimal handling of the current number of active time series.
|
|
||||||
See also https://github.com/VictoriaMetrics/VictoriaMetrics/issues/3976#issuecomment-1476883183"
|
|
||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,189 @@
|
|||||||
|
# zot OCI Registry
|
||||||
|
|
||||||
|
内网匿名拉取入口为 `https://zot.ad.ddupan.top`,SPIRE 鉴权推送入口为
|
||||||
|
`https://zot-push.ad.ddupan.top`。使用官方 Helm chart `0.1.124`,运行
|
||||||
|
zot `v2.1.21`,镜像固定到官方 linux/amd64 digest。
|
||||||
|
|
||||||
|
## 当前工作状态(2026-09-16 核验)
|
||||||
|
|
||||||
|
| 项目 | 状态 |
|
||||||
|
|---|---|
|
||||||
|
| 匿名拉取 | `zot.ad.ddupan.top` 已上线;空 `DOCKER_CONFIG` 的 crane pull 通过 |
|
||||||
|
| SPIRE 鉴权入口 | `zot-push.ad.ddupan.top` 已上线;真实 JWT-SVID 推送后可匿名拉取同一 digest |
|
||||||
|
| GitOps | 双入口配置已合并;Flux `zot` Kustomization 已应用 `d15733c`,状态 Ready |
|
||||||
|
| 运行与凭据同步 | `zot`、`zot-reader` HelmRelease 均 Ready,Pod 均 1/1;ESO SecretSynced |
|
||||||
|
| 临时配置清理 | 两个 HelmRelease 均无 `spec.values` 临时覆盖;暂停回写标记、测试身份和临时写权限已清理 |
|
||||||
|
| 接管复验 | 匿名拉取成功;推送入口无凭据返回 401,token realm 指向推送域名;接管未触发 Pod 重启 |
|
||||||
|
|
||||||
|
后续工作是给实际 CI 的 SPIFFE ID 配置具体仓库的 `create`/`update` 权限。
|
||||||
|
SPIRE 认证链路已经验证,但当前没有常驻 publisher 或删除授权;认证成功本身不代表
|
||||||
|
可以推送。S3 侧仍使用 Bao 管理的静态 AK/SK,尚未接入 SPIRE/STS。
|
||||||
|
|
||||||
|
## 存储与凭据
|
||||||
|
|
||||||
|
制品、manifest 和 OCI layout 保存在现有 SeaweedFS 的 `zot` bucket,前缀为
|
||||||
|
`registry/`,S3 endpoint 为 `https://s3.ad.ddupan.top`。**不创建 PVC**;chart 的
|
||||||
|
`/var/lib/registry` 是 `emptyDir`,仅用于运行时本地工作数据。
|
||||||
|
|
||||||
|
两个单副本实例共用同一 bucket 和前缀:`zot` 负责鉴权写入,`zot-reader` 负责匿名
|
||||||
|
读取。关闭跨仓库 dedupe,不额外部署 Redis/DynamoDB 缓存。只有写入实例启用 GC,
|
||||||
|
暂不配置自动删除已发布版本的 retention policy。增加副本、启用 dedupe 或搜索等
|
||||||
|
扩展前,需要重新检查共享元数据与缓存的持久化要求。
|
||||||
|
|
||||||
|
凭据链路:
|
||||||
|
|
||||||
|
```text
|
||||||
|
OpenBao kv/k8s/seaweedfs-s3
|
||||||
|
→ 原有 S3 身份及基础配置 ─┐
|
||||||
|
├→ ESO 模板 → seaweedfs/seaweedfs-s3-config
|
||||||
|
OpenBao kv/k8s/zot-s3 ────┘ → 完整 s3.config
|
||||||
|
└→ ESO → zot/zot-s3 → zot 与 zot-reader 的 AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
`kv/k8s/zot-s3` 是 zot AK/SK 的唯一维护来源。基础配置保留原有身份及其他字段,
|
||||||
|
不再保存 zot 凭据副本;[SeaweedFS ExternalSecret](../../platform/external-secrets/externalsecrets.yaml)
|
||||||
|
使用 ESO v2 模板追加 zot 身份。两个 Kubernetes Secret 都是自动生成的消费副本,
|
||||||
|
不手工编辑。Bao 的旧版本历史保留,回滚基础配置时模板也会替换其中的旧 zot 身份。
|
||||||
|
|
||||||
|
专用 S3 身份只有 `Read:zot`、`Write:zot`、`List:zot`、`Tagging:zot`,不能读取
|
||||||
|
Terraform 的 `tfstate` bucket。`kv/k8s/zot-s3` 的字段是 `access_key` 和
|
||||||
|
`secret_key`。AK/SK 不进入 Git、Helm values 或 CI;这里仍是静态 S3 凭据,尚未
|
||||||
|
接入 SPIRE/STS。
|
||||||
|
|
||||||
|
本次归一没有轮换密钥,生成的完整配置与归一前语义一致。当前模板只有一组 zot
|
||||||
|
凭据,尚未实现新旧密钥重叠轮换。后续轮换只修改 `kv/k8s/zot-s3`,但仍需协调
|
||||||
|
两个 ExternalSecret 同步:确认 SeaweedFS Secret volume 更新后向 filer 的
|
||||||
|
`weed` 进程发送 SIGHUP,再确认 zot Secret 更新并重启 `zot` 和 `zot-reader`(环境变量不会热更新)。
|
||||||
|
两端异步更新期间可能短暂认证失败;需要无中断轮换时先扩展模板支持新旧凭据重叠。
|
||||||
|
|
||||||
|
## SPIRE 认证和授权
|
||||||
|
|
||||||
|
| 参数 | 值 |
|
||||||
|
|---|---|
|
||||||
|
| issuer | `https://spire-oidc.ad.ddupan.top` |
|
||||||
|
| JWT audience | `zot` |
|
||||||
|
| subject | `spiffe://ddupan.top/` 下的 workload SPIFFE ID |
|
||||||
|
| token endpoint | `https://zot-push.ad.ddupan.top/zot/auth/token` |
|
||||||
|
| 拉取入口 | 内网匿名读取所有仓库,不要求 SPIRE 身份 |
|
||||||
|
| 推送入口当前权限 | 受信身份可以读取所有仓库;没有常驻写入或删除授权 |
|
||||||
|
|
||||||
|
zot 通过已配置的 issuer discovery/JWKS 验证 JWT-SVID,再以 `sub` 作为授权身份。
|
||||||
|
不接受任意 issuer,不关闭 TLS/issuer 验证。新的 Kata CI 负责取得并更新自己的
|
||||||
|
JWT-SVID;确认其身份命名后,再添加针对具体 repository 的 `create`/`update`
|
||||||
|
授权,不能把整个 trust domain 都授予写权限。
|
||||||
|
|
||||||
|
zot `v2.1.21` 的 OIDC Bearer middleware 会在授权阶段之前拒绝无 token 请求。
|
||||||
|
因此使用两个官方 zot 实例与两个域名,避免修改上游镜像,也避免同域名下匿名
|
||||||
|
`/v2/` 返回 200 导致标准客户端跳过 token 交换的问题。
|
||||||
|
|
||||||
|
- `zot-reader` 叠加 `reader-values.yaml`,没有认证 middleware,只有
|
||||||
|
`anonymousPolicy: [read]`。入口只转发 `/v2/` 的 GET/HEAD,并移除客户端遗留的
|
||||||
|
Authorization/Cookie;直接访问 reader Service 也不能写入。
|
||||||
|
- `zot` 保留 SPIRE issuer/audience/subject 校验及仓库授权,`externalUrl`、
|
||||||
|
Bearer realm、service 与 HTTPRoute 均使用 `zot-push.ad.ddupan.top`。
|
||||||
|
- reader 关闭 GC,没有同步或扫描扩展;读取同一份 S3 制品,不复制 bucket,
|
||||||
|
不新增 PVC 或 S3 密钥。镜像、安全上下文、资源和 Secret 引用由共用 values 继承。
|
||||||
|
- 两个配置的 `storageDriver` 必须保持一致;修改 S3 endpoint/bucket/prefix 时
|
||||||
|
同时更新 `values.yaml` 与 `reader-values.yaml`。
|
||||||
|
|
||||||
|
推送客户端应登录 `zot-push.ad.ddupan.top`;拉取客户端无需登录。
|
||||||
|
已有 SPIRE 身份的进程可以通过 Workload API 获取 `aud=zot` 的 JWT-SVID,然后
|
||||||
|
通过 `docker login` 或 `crane auth login` 的 `--password-stdin` 交给 Registry。
|
||||||
|
用户名可以使用 `zot`,实际权限取自已验证 JWT 的身份。使用独立、权限为 `0700`
|
||||||
|
的临时 `DOCKER_CONFIG`,结束后删除;不要开启 shell tracing,不要打印 token,
|
||||||
|
不要把 token 放进命令参数。token 接口不会延长 SVID 有效期。
|
||||||
|
|
||||||
|
同一仓库在两个入口使用相同路径和 tag/digest,例如 CI 推送到
|
||||||
|
`zot-push.ad.ddupan.top/team/image:tag`,部署时使用
|
||||||
|
`zot.ad.ddupan.top/team/image:tag`;无需在两个仓库间复制。
|
||||||
|
|
||||||
|
## 部署与网络
|
||||||
|
|
||||||
|
- 官方 chart 管理 Deployment、Service、ConfigMap 和 HTTPRoute。
|
||||||
|
- `persistence: false`,Service 为 ClusterIP,TLS 由已有 Envoy Gateway 的
|
||||||
|
`https` listener 与内网通配符证书终止。
|
||||||
|
- 仅配置 Samba AD 内网 DNS;不创建公网 DNS 或 Cloudflare Tunnel route。
|
||||||
|
- NetworkPolicy 只允许现有 Envoy Gateway 数据面访问 zot 的 5000 端口。
|
||||||
|
- 拉取域名仅暴露 `/v2/` 的 GET/HEAD;推送域名暴露 `/v2/` 和
|
||||||
|
`/zot/auth/token`,均不暴露内部健康检查或管理端点。
|
||||||
|
- namespace 使用 restricted PodSecurity,容器非 root、只读根文件系统。
|
||||||
|
|
||||||
|
`clusters/homelab/apps/zot.yaml` 已将 `zot` 和 `zot-reader` 一并纳入 Flux 管理。
|
||||||
|
两个 HelmRelease 通过共用 `zot-values` 继承基础配置,reader 再叠加
|
||||||
|
`zot-reader-values`。当前由 main 分支持续管理,不依赖本地覆盖或暂停回写。
|
||||||
|
|
||||||
|
后续若需临时验收,收尾时先确认 Git 管理的配置与目标运行配置一致,再移除
|
||||||
|
`spec.values` 临时覆盖及 `kustomize.toolkit.fluxcd.io/reconcile=disabled` 标记,
|
||||||
|
触发 zot Kustomization reconcile 并复验。临时测试身份和写权限不得留在持久配置中。
|
||||||
|
|
||||||
|
检查与渲染:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
helm template zot --repo https://zotregistry.dev/helm-charts \
|
||||||
|
--version 0.1.124 --namespace zot -f apps/zot/values.yaml --skip-tests
|
||||||
|
sudo k3s kubectl -n zot get helmrelease,pods,externalsecret,httproute
|
||||||
|
sudo k3s kubectl -n zot get pvc
|
||||||
|
```
|
||||||
|
|
||||||
|
上游 chart 的 Helm test Pod 不满足本 namespace 的 restricted 策略,也没有
|
||||||
|
SPIRE 凭据,因此不运行默认 `helm test`;使用下述真实身份验收。
|
||||||
|
|
||||||
|
## 验收与恢复
|
||||||
|
|
||||||
|
验收使用独立临时 Pod,通过 SPIFFE CSI socket 和真实 Workload API 取得 JWT-SVID,
|
||||||
|
没有修改现有 runner。仅在初始化 `verification/smoke:spire-s3` 测试镜像时临时
|
||||||
|
授予该测试身份针对该仓库的写权限;完成后必须撤回 HelmRelease override,并删除
|
||||||
|
临时 Pod、ServiceAccount 与 ClusterSPIFFEID。
|
||||||
|
|
||||||
|
验收项目:有效 SVID + crane pull、manifest digest 一致、错误 audience、错误
|
||||||
|
signature、过期 token、无凭据写入、跨仓库写入、只读身份写入和删除拒绝;另外检查
|
||||||
|
Pod 重建后镜像仍可拉取,以及 S3 身份不能访问 `tfstate`。
|
||||||
|
|
||||||
|
2026-09-14 已完成上述验收:HelmRelease Ready、HTTPRoute Accepted/ResolvedRefs,
|
||||||
|
DNS 第二次 Ansible check 为 `changed=0`;一分钟真实 JWT-SVID 到期后返回 401。
|
||||||
|
临时写权限已移除。测试镜像可供后续 CI 验证拉取:
|
||||||
|
|
||||||
|
```text
|
||||||
|
zot.ad.ddupan.top/verification/smoke:spire-s3
|
||||||
|
sha256:b8d3b977a1235022759470903dab4a46b7cf8107958624f1f76a323eabe37c5e
|
||||||
|
```
|
||||||
|
|
||||||
|
它是仅含验证文本的 OCI 测试镜像,没有可执行入口,不用于运行服务。
|
||||||
|
|
||||||
|
双域名验收还使用 `verification/anonymous-spire:smoke`:标准 crane 从
|
||||||
|
`zot-push.ad.ddupan.top` 登录、推送,再从 `zot.ad.ddupan.top` 使用空
|
||||||
|
`DOCKER_CONFIG` 拉取,两个入口的 digest 必须一致。验证匿名 blob HEAD、tags、
|
||||||
|
referrers,以及客户端保存旧凭据时的公共拉取。推送入口检查无凭据、错误签名、
|
||||||
|
错误 audience、过期 SVID、跨仓库写入和删除拒绝;公共入口拒绝所有写方法,
|
||||||
|
reader Service 直连也拒绝写入。测试完成后撤回临时单仓库写权限。
|
||||||
|
|
||||||
|
2026-09-16 上述双域名验收通过;SVID 过期后推送入口返回 401,匿名拉取不受
|
||||||
|
影响。临时写权限已撤销,两个 HelmRelease Ready;推送 DNS 第二次检查 changed=0。
|
||||||
|
|
||||||
|
匿名拉取示例:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
crane pull zot.ad.ddupan.top/verification/anonymous-spire:smoke image.tar --format oci
|
||||||
|
```
|
||||||
|
|
||||||
|
鉴权推送示例(先通过 Workload API 将短期 JWT-SVID 保存到当前进程的 `ZOT_JWT`,
|
||||||
|
不要启用 shell tracing;示例中的仓库仍需提前给具体 SPIFFE ID 授权):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export DOCKER_CONFIG="$(mktemp -d)"
|
||||||
|
printf '%s' "$ZOT_JWT" | crane auth login zot-push.ad.ddupan.top \
|
||||||
|
--username zot --password-stdin
|
||||||
|
crane push image.tar zot-push.ad.ddupan.top/team/image:tag
|
||||||
|
rm -rf -- "$DOCKER_CONFIG"
|
||||||
|
unset DOCKER_CONFIG ZOT_JWT
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
|
Registry 恢复需要完整的 SeaweedFS bucket 数据、Bao 专用凭据和此目录配置。
|
||||||
|
zot 的临时目录不是制品备份。独立异机/离线备份尚未在本次部署中建立;不能把同一
|
||||||
|
SeaweedFS 内的数据副本当作独立灾备。重装 zot 不得删除 `zot` bucket。
|
||||||
|
|
||||||
|
参考:[官方 Kubernetes 安装](https://zotregistry.dev/v2.1.21/install-guides/install-guide-k8s/)、
|
||||||
|
[S3 存储](https://zotregistry.dev/v2.1.21/articles/storage/)、
|
||||||
|
[OIDC workload identity](https://github.com/project-zot/zot/blob/v2.1.21/examples/README-OIDC-WORKLOAD-IDENTITY.md)。
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
apiVersion: external-secrets.io/v1
|
||||||
|
kind: ExternalSecret
|
||||||
|
metadata:
|
||||||
|
name: zot-s3
|
||||||
|
namespace: zot
|
||||||
|
spec:
|
||||||
|
refreshInterval: 1h
|
||||||
|
secretStoreRef:
|
||||||
|
kind: ClusterSecretStore
|
||||||
|
name: openbao
|
||||||
|
target:
|
||||||
|
name: zot-s3
|
||||||
|
creationPolicy: Owner
|
||||||
|
data:
|
||||||
|
- secretKey: access_key
|
||||||
|
remoteRef:
|
||||||
|
key: k8s/zot-s3
|
||||||
|
property: access_key
|
||||||
|
- secretKey: secret_key
|
||||||
|
remoteRef:
|
||||||
|
key: k8s/zot-s3
|
||||||
|
property: secret_key
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
apiVersion: helm.toolkit.fluxcd.io/v2
|
||||||
|
kind: HelmRelease
|
||||||
|
metadata:
|
||||||
|
name: zot-reader
|
||||||
|
namespace: zot
|
||||||
|
spec:
|
||||||
|
chart:
|
||||||
|
spec:
|
||||||
|
chart: zot
|
||||||
|
version: 0.1.124
|
||||||
|
interval: 1h
|
||||||
|
sourceRef:
|
||||||
|
kind: HelmRepository
|
||||||
|
name: zot
|
||||||
|
releaseName: zot-reader
|
||||||
|
interval: 30m
|
||||||
|
timeout: 5m
|
||||||
|
driftDetection:
|
||||||
|
mode: enabled
|
||||||
|
install:
|
||||||
|
strategy:
|
||||||
|
name: RetryOnFailure
|
||||||
|
retryInterval: 5m
|
||||||
|
upgrade:
|
||||||
|
strategy:
|
||||||
|
name: RetryOnFailure
|
||||||
|
retryInterval: 5m
|
||||||
|
valuesFrom:
|
||||||
|
- kind: ConfigMap
|
||||||
|
name: zot-values
|
||||||
|
- kind: ConfigMap
|
||||||
|
name: zot-reader-values
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
apiVersion: helm.toolkit.fluxcd.io/v2
|
||||||
|
kind: HelmRelease
|
||||||
|
metadata:
|
||||||
|
name: zot
|
||||||
|
namespace: zot
|
||||||
|
spec:
|
||||||
|
chart:
|
||||||
|
spec:
|
||||||
|
chart: zot
|
||||||
|
version: 0.1.124
|
||||||
|
interval: 1h
|
||||||
|
sourceRef:
|
||||||
|
kind: HelmRepository
|
||||||
|
name: zot
|
||||||
|
releaseName: zot
|
||||||
|
interval: 30m
|
||||||
|
timeout: 5m
|
||||||
|
driftDetection:
|
||||||
|
mode: enabled
|
||||||
|
install:
|
||||||
|
strategy:
|
||||||
|
name: RetryOnFailure
|
||||||
|
retryInterval: 5m
|
||||||
|
upgrade:
|
||||||
|
strategy:
|
||||||
|
name: RetryOnFailure
|
||||||
|
retryInterval: 5m
|
||||||
|
valuesFrom:
|
||||||
|
- kind: ConfigMap
|
||||||
|
name: zot-values
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
apiVersion: source.toolkit.fluxcd.io/v1
|
||||||
|
kind: HelmRepository
|
||||||
|
metadata:
|
||||||
|
name: zot
|
||||||
|
namespace: zot
|
||||||
|
spec:
|
||||||
|
interval: 1h
|
||||||
|
url: https://zotregistry.dev/helm-charts
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||||
|
kind: Kustomization
|
||||||
|
resources:
|
||||||
|
- namespace.yaml
|
||||||
|
- serviceaccount.yaml
|
||||||
|
- external-secret.yaml
|
||||||
|
- helmrepository.yaml
|
||||||
|
- helmrelease.yaml
|
||||||
|
- helmrelease-reader.yaml
|
||||||
|
- networkpolicy.yaml
|
||||||
|
generatorOptions:
|
||||||
|
disableNameSuffixHash: true
|
||||||
|
labels:
|
||||||
|
reconcile.fluxcd.io/watch: Enabled
|
||||||
|
configMapGenerator:
|
||||||
|
- name: zot-values
|
||||||
|
namespace: zot
|
||||||
|
files:
|
||||||
|
- values.yaml=values.yaml
|
||||||
|
- name: zot-reader-values
|
||||||
|
namespace: zot
|
||||||
|
files:
|
||||||
|
- values.yaml=reader-values.yaml
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
apiVersion: v1
|
||||||
|
kind: Namespace
|
||||||
|
metadata:
|
||||||
|
name: zot
|
||||||
|
labels:
|
||||||
|
pod-security.kubernetes.io/enforce: restricted
|
||||||
|
pod-security.kubernetes.io/enforce-version: v1.36
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
apiVersion: networking.k8s.io/v1
|
||||||
|
kind: NetworkPolicy
|
||||||
|
metadata:
|
||||||
|
name: zot-ingress
|
||||||
|
namespace: zot
|
||||||
|
spec:
|
||||||
|
podSelector:
|
||||||
|
matchLabels:
|
||||||
|
app.kubernetes.io/name: zot
|
||||||
|
policyTypes: [Ingress]
|
||||||
|
ingress:
|
||||||
|
- from:
|
||||||
|
- namespaceSelector:
|
||||||
|
matchLabels:
|
||||||
|
kubernetes.io/metadata.name: envoy-gateway-system
|
||||||
|
podSelector:
|
||||||
|
matchLabels:
|
||||||
|
gateway.envoyproxy.io/owning-gateway-name: eg
|
||||||
|
gateway.envoyproxy.io/owning-gateway-namespace: envoy-gateway-system
|
||||||
|
ports:
|
||||||
|
- protocol: TCP
|
||||||
|
port: 5000
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
# 叠加于共用 values.yaml;同一镜像、S3、Secret、安全设置,无制品副本。
|
||||||
|
# 无 Bearer middleware,仅 anonymousPolicy=read;关闭 GC 避免多个实例清理共享存储。
|
||||||
|
configFiles:
|
||||||
|
config.json: |
|
||||||
|
{
|
||||||
|
"distSpecVersion": "1.1.1",
|
||||||
|
"storage": {
|
||||||
|
"rootDirectory": "/var/lib/registry",
|
||||||
|
"dedupe": false,
|
||||||
|
"gc": false,
|
||||||
|
"storageDriver": {
|
||||||
|
"name": "s3",
|
||||||
|
"region": "us-east-1",
|
||||||
|
"regionendpoint": "https://s3.ad.ddupan.top",
|
||||||
|
"bucket": "zot",
|
||||||
|
"rootdirectory": "/registry",
|
||||||
|
"secure": true,
|
||||||
|
"skipverify": false,
|
||||||
|
"forcepathstyle": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"http": {
|
||||||
|
"address": "0.0.0.0",
|
||||||
|
"port": "5000",
|
||||||
|
"externalUrl": "https://zot.ad.ddupan.top",
|
||||||
|
"compat": [
|
||||||
|
"docker2s2"
|
||||||
|
],
|
||||||
|
"accessControl": {
|
||||||
|
"repositories": {
|
||||||
|
"**": {
|
||||||
|
"anonymousPolicy": [
|
||||||
|
"read"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"log": {
|
||||||
|
"level": "info"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
httproute:
|
||||||
|
hostnames:
|
||||||
|
- zot.ad.ddupan.top
|
||||||
|
rules:
|
||||||
|
- matches:
|
||||||
|
- path:
|
||||||
|
type: PathPrefix
|
||||||
|
value: /v2/
|
||||||
|
method: GET
|
||||||
|
- path:
|
||||||
|
type: PathPrefix
|
||||||
|
value: /v2/
|
||||||
|
method: HEAD
|
||||||
|
filters:
|
||||||
|
- type: RequestHeaderModifier
|
||||||
|
requestHeaderModifier:
|
||||||
|
remove:
|
||||||
|
- Cookie
|
||||||
|
- Authorization
|
||||||
|
timeouts:
|
||||||
|
request: 900s
|
||||||
|
backendRequest: 900s
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
apiVersion: v1
|
||||||
|
kind: ServiceAccount
|
||||||
|
metadata:
|
||||||
|
name: zot
|
||||||
|
namespace: zot
|
||||||
|
automountServiceAccountToken: false
|
||||||
@@ -0,0 +1,196 @@
|
|||||||
|
# 官方 chart 0.1.124 / zot v2.1.21;制品与 manifests 保存在 SeaweedFS S3。
|
||||||
|
# persistence=false 仅保留 chart 的 emptyDir,不创建 PVC。
|
||||||
|
# 首期关闭跨仓库 dedupe,不额外引入 Redis/DynamoDB 持久缓存。
|
||||||
|
replicaCount: 1
|
||||||
|
image:
|
||||||
|
repository: ghcr.io/project-zot/zot
|
||||||
|
tag: v2.1.21@sha256:8258443838e95989c13c891f78a02bc1c391b5a00591ffef24cb8c17cde28038
|
||||||
|
persistence: false
|
||||||
|
strategy:
|
||||||
|
type: Recreate
|
||||||
|
serviceAccount:
|
||||||
|
create: false
|
||||||
|
name: zot
|
||||||
|
service:
|
||||||
|
type: ClusterIP
|
||||||
|
port: 5000
|
||||||
|
mountConfig: true
|
||||||
|
mountSecret: false
|
||||||
|
secretFiles: {}
|
||||||
|
configFiles:
|
||||||
|
config.json: |
|
||||||
|
{
|
||||||
|
"distSpecVersion": "1.1.1",
|
||||||
|
"storage": {
|
||||||
|
"rootDirectory": "/var/lib/registry",
|
||||||
|
"dedupe": false,
|
||||||
|
"gc": true,
|
||||||
|
"gcDelay": "24h",
|
||||||
|
"gcInterval": "24h",
|
||||||
|
"storageDriver": {
|
||||||
|
"name": "s3",
|
||||||
|
"region": "us-east-1",
|
||||||
|
"regionendpoint": "https://s3.ad.ddupan.top",
|
||||||
|
"bucket": "zot",
|
||||||
|
"rootdirectory": "/registry",
|
||||||
|
"secure": true,
|
||||||
|
"skipverify": false,
|
||||||
|
"forcepathstyle": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"http": {
|
||||||
|
"address": "0.0.0.0",
|
||||||
|
"port": "5000",
|
||||||
|
"externalUrl": "https://zot-push.ad.ddupan.top",
|
||||||
|
"compat": [
|
||||||
|
"docker2s2"
|
||||||
|
],
|
||||||
|
"auth": {
|
||||||
|
"bearer": {
|
||||||
|
"realm": "https://zot-push.ad.ddupan.top/zot/auth/token",
|
||||||
|
"service": "zot-push.ad.ddupan.top",
|
||||||
|
"oidc": [
|
||||||
|
{
|
||||||
|
"issuer": "https://spire-oidc.ad.ddupan.top",
|
||||||
|
"audiences": [
|
||||||
|
"zot"
|
||||||
|
],
|
||||||
|
"claimMapping": {
|
||||||
|
"username": "claims.sub",
|
||||||
|
"validations": [
|
||||||
|
{
|
||||||
|
"expression": "claims.sub.startsWith('spiffe://ddupan.top/')",
|
||||||
|
"message": "SPIFFE trust domain mismatch"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"accessControl": {
|
||||||
|
"repositories": {
|
||||||
|
"panxiao81/gitea-dynamic-runner-controller": {
|
||||||
|
"policies": [
|
||||||
|
{
|
||||||
|
"users": [
|
||||||
|
"spiffe://ddupan.top/ci/panxiao81/gitea-dynamic-runner/publish-images",
|
||||||
|
"spiffe://ddupan.top/dev/panxiao81"
|
||||||
|
],
|
||||||
|
"actions": [
|
||||||
|
"read",
|
||||||
|
"create",
|
||||||
|
"update"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"defaultPolicy": [
|
||||||
|
"read"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"panxiao81/gitea-dynamic-runner-runner": {
|
||||||
|
"policies": [
|
||||||
|
{
|
||||||
|
"users": [
|
||||||
|
"spiffe://ddupan.top/ci/panxiao81/gitea-dynamic-runner/publish-images",
|
||||||
|
"spiffe://ddupan.top/dev/panxiao81"
|
||||||
|
],
|
||||||
|
"actions": [
|
||||||
|
"read",
|
||||||
|
"create",
|
||||||
|
"update"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"defaultPolicy": [
|
||||||
|
"read"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"**": {
|
||||||
|
"policies": [
|
||||||
|
{
|
||||||
|
"users": [
|
||||||
|
"spiffe://ddupan.top/dev/panxiao81"
|
||||||
|
],
|
||||||
|
"actions": [
|
||||||
|
"read",
|
||||||
|
"create",
|
||||||
|
"update",
|
||||||
|
"delete"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"defaultPolicy": [
|
||||||
|
"read"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"log": {
|
||||||
|
"level": "info"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
env:
|
||||||
|
- name: AWS_ACCESS_KEY_ID
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: zot-s3
|
||||||
|
key: access_key
|
||||||
|
- name: AWS_SECRET_ACCESS_KEY
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: zot-s3
|
||||||
|
key: secret_key
|
||||||
|
- name: AWS_EC2_METADATA_DISABLED
|
||||||
|
value: 'true'
|
||||||
|
podSecurityContext:
|
||||||
|
runAsNonRoot: true
|
||||||
|
runAsUser: 10001
|
||||||
|
runAsGroup: 10001
|
||||||
|
fsGroup: 10001
|
||||||
|
seccompProfile:
|
||||||
|
type: RuntimeDefault
|
||||||
|
securityContext:
|
||||||
|
allowPrivilegeEscalation: false
|
||||||
|
readOnlyRootFilesystem: true
|
||||||
|
capabilities:
|
||||||
|
drop:
|
||||||
|
- ALL
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
cpu: 100m
|
||||||
|
memory: 128Mi
|
||||||
|
limits:
|
||||||
|
cpu: '1'
|
||||||
|
memory: 512Mi
|
||||||
|
extraVolumes:
|
||||||
|
- name: tmp
|
||||||
|
emptyDir:
|
||||||
|
sizeLimit: 128Mi
|
||||||
|
extraVolumeMounts:
|
||||||
|
- name: tmp
|
||||||
|
mountPath: /tmp
|
||||||
|
startupProbe:
|
||||||
|
initialDelaySeconds: 5
|
||||||
|
periodSeconds: 5
|
||||||
|
failureThreshold: 60
|
||||||
|
httproute:
|
||||||
|
enabled: true
|
||||||
|
parentRefs:
|
||||||
|
- name: eg
|
||||||
|
namespace: envoy-gateway-system
|
||||||
|
sectionName: https
|
||||||
|
hostnames:
|
||||||
|
- zot-push.ad.ddupan.top
|
||||||
|
rules:
|
||||||
|
- matches:
|
||||||
|
- path:
|
||||||
|
type: PathPrefix
|
||||||
|
value: /v2/
|
||||||
|
- path:
|
||||||
|
type: Exact
|
||||||
|
value: /zot/auth/token
|
||||||
|
timeouts:
|
||||||
|
request: 900s
|
||||||
|
backendRequest: 900s
|
||||||
+52
-11
@@ -1,15 +1,56 @@
|
|||||||
# Homelab cluster
|
# Homelab 集群
|
||||||
|
|
||||||
This directory will become the Flux reconciliation entrypoint for the homelab
|
这里是单节点 k3s 集群的 Flux reconciliation 入口。集群当前运行 Kubernetes
|
||||||
k3s cluster. It is intentionally documentation-only until a Git remote, CI
|
`v1.36.4+k3s1`,Flux 固定为 `v2.9.5`。
|
||||||
checks and a low-risk bootstrap workload have been verified.
|
|
||||||
|
|
||||||
Planned reconciliation order:
|
## 首次 bootstrap
|
||||||
|
|
||||||
1. namespaces and CRDs;
|
仓库经过 PR 审查并合并后,在本机从合并后的 `main` 执行:
|
||||||
2. shared platform controllers;
|
|
||||||
3. secret references and storage;
|
|
||||||
4. applications.
|
|
||||||
|
|
||||||
Do not enable pruning for a path until its live resources and field ownership
|
```bash
|
||||||
have been audited.
|
sudo k3s kubectl apply -f clusters/homelab/flux-system/gotk-components.yaml
|
||||||
|
sudo k3s kubectl apply -f clusters/homelab/flux-system/gotk-sync.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
`GitRepository/flux-system` 通过集群内 Gitea Service 读取公开仓库,不需要长期
|
||||||
|
管理员 token,也不依赖 Cloudflare、公网 DNS 或 Envoy Gateway。Gitea 暂时不可用
|
||||||
|
时,已经应用的资源继续运行,Flux 在 Gitea 恢复后重新同步。
|
||||||
|
|
||||||
|
root Kustomization 从 `./clusters/homelab` 开始 reconciliation。初始设置
|
||||||
|
`prune: false`;在逐项审计现有资源和 field ownership 之前不得开启全局 prune。
|
||||||
|
|
||||||
|
计划中的 reconciliation 顺序:
|
||||||
|
|
||||||
|
1. namespaces 和 CRD;
|
||||||
|
2. platform controllers;
|
||||||
|
3. secret references 和 storage;
|
||||||
|
4. applications。
|
||||||
|
|
||||||
|
首次部署后的最低验证:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo k3s kubectl -n flux-system get pods
|
||||||
|
sudo k3s kubectl -n flux-system get gitrepositories,kustomizations
|
||||||
|
```
|
||||||
|
|
||||||
|
四个 controller、GitRepository 和 root Kustomization 都必须为 Ready,随后才能
|
||||||
|
通过单独 PR 引入低风险 canary workload。
|
||||||
|
|
||||||
|
## 当前状态
|
||||||
|
|
||||||
|
- Flux `v2.9.5`、GitRepository 和 root Kustomization 均为 Ready;
|
||||||
|
- `http-echo` canary 已验证 merge 后自动部署和 replicas 漂移修复;
|
||||||
|
- `http-echo` 的专用测试 ConfigMap 已在 `prune: true` 生效后重新纳管,并由下一
|
||||||
|
revision 自动删除;
|
||||||
|
- `http-echo` 保持 `prune: true`,root 保持 `prune: false`;
|
||||||
|
- `gitea-actions`、`gitea` 与 External Secrets 已由 Flux HelmRelease 接管,Gitea 已升级到 `1.27.3`;
|
||||||
|
- cert-manager 已固定现有 `v1.21.0` 并完成分阶段 Flux HelmRelease 接管;
|
||||||
|
- Envoy Gateway 已固定现有 `v1.5.6` 并完成分阶段 Flux HelmRelease 接管;
|
||||||
|
- OpenEBS 已固定现有 `4.4.0` 并完成分阶段 Flux HelmRelease 接管;
|
||||||
|
- VictoriaMetrics Operator 已固定现有 chart `0.66.2` 并完成分阶段 Flux HelmRelease
|
||||||
|
接管;Metrics、Logs、Traces 与 Grafana 也已统一完成 Flux 接管;
|
||||||
|
- External Secrets Operator 已固定 chart `2.8.0` 并完成分阶段接管;
|
||||||
|
- SPIRE 已按 hardened chart 内部 fork `0.30.2-ddupan.1`(基于上游 `0.30.2`,SPIRE
|
||||||
|
`1.15.3`)声明,使用共享
|
||||||
|
PostgreSQL 与独立 signing-key PVC;首次上线和 OpenBao JWT-SVID PoC 尚待合并后验证;
|
||||||
|
- root Kustomization 与所有 brownfield 子 Kustomization 继续保持 `prune: false`。
|
||||||
|
|||||||
@@ -0,0 +1,14 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: cert-manager
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/cert-manager
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 3m
|
||||||
|
wait: false
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: dynamic-runner
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
dependsOn:
|
||||||
|
- name: external-secrets
|
||||||
|
- name: nats
|
||||||
|
- name: spire
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/dynamic-runner
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 5m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: envoy-gateway
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/envoy-gateway
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 3m
|
||||||
|
wait: false
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: external-secrets
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/external-secrets
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 3m
|
||||||
|
wait: false
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: gitea-actions
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/gitea-runner
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 3m
|
||||||
|
wait: false
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: gitea
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./apps/gitea
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
wait: false
|
||||||
@@ -0,0 +1,20 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: http-echo
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
healthChecks:
|
||||||
|
- apiVersion: apps/v1
|
||||||
|
kind: Deployment
|
||||||
|
name: http-echo
|
||||||
|
namespace: gitops-canary
|
||||||
|
interval: 10m
|
||||||
|
path: ./apps/http-echo
|
||||||
|
prune: true
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
targetNamespace: gitops-canary
|
||||||
|
timeout: 3m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: nats
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
dependsOn:
|
||||||
|
- name: cert-manager
|
||||||
|
- name: external-secrets
|
||||||
|
- name: openebs
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/nats
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 10m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: observability
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/observability
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 3m
|
||||||
|
wait: false
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: openebs
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/openebs
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 3m
|
||||||
|
wait: false
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: spire
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/spire
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 15m
|
||||||
|
wait: false
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: zot
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
dependsOn:
|
||||||
|
- name: envoy-gateway
|
||||||
|
- name: external-secrets
|
||||||
|
- name: spire
|
||||||
|
interval: 10m
|
||||||
|
path: ./apps/zot
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 5m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
# Contour 退役清理
|
||||||
|
|
||||||
|
Envoy Gateway 已经承载全部现用 `Gateway` 和 `HTTPRoute`。Contour 在集群中仅剩
|
||||||
|
gateway provisioner、RBAC、旧 `GatewayClass` 和没有实例的 CRD,不再承载流量。
|
||||||
|
|
||||||
|
清理必须在本变更合并后进行,顺序如下:
|
||||||
|
|
||||||
|
1. 应用更新后的 `platform/cert-manager/clusterissuer-bao-acme.yaml`,把 OpenBao
|
||||||
|
HTTP-01 solver 改到 `envoy-gateway-system/eg` 的 `http` listener;
|
||||||
|
2. 确认 `Gateway/eg` 为 `Programmed=True`,所有现用 `HTTPRoute` 保持正常;
|
||||||
|
3. 删除 `GatewayClass/contour`;
|
||||||
|
4. 删除 `projectcontour` namespace;
|
||||||
|
5. 删除名称包含 `contour` 的遗留 ClusterRole/ClusterRoleBinding;
|
||||||
|
6. 在确认所有 Contour 自定义资源均为空后,删除 `projectcontour.io` 的五个 CRD;
|
||||||
|
7. 复查 Envoy Gateway、证书、DNS 和现用入口。
|
||||||
|
|
||||||
|
这些对象是 Flux 启用前留下的孤立资源,首次清理由本机 `kubectl` 完成。Flux 的
|
||||||
|
root Kustomization 初始保持 `prune: false`,不会借 bootstrap 顺带删除 brownfield
|
||||||
|
资源。
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,27 @@
|
|||||||
|
---
|
||||||
|
apiVersion: source.toolkit.fluxcd.io/v1
|
||||||
|
kind: GitRepository
|
||||||
|
metadata:
|
||||||
|
name: flux-system
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 1m
|
||||||
|
ref:
|
||||||
|
branch: main
|
||||||
|
timeout: 60s
|
||||||
|
url: http://gitea-http.gitea.svc.cluster.local:3000/panxiao81/homelab-infra.git
|
||||||
|
---
|
||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: flux-system
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./clusters/homelab
|
||||||
|
prune: false
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 3m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,5 @@
|
|||||||
|
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||||
|
kind: Kustomization
|
||||||
|
resources:
|
||||||
|
- gotk-components.yaml
|
||||||
|
- gotk-sync.yaml
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||||
|
kind: Kustomization
|
||||||
|
resources:
|
||||||
|
- flux-system
|
||||||
|
- namespaces/gitops-canary.yaml
|
||||||
|
- apps/cert-manager.yaml
|
||||||
|
- apps/envoy-gateway.yaml
|
||||||
|
- apps/external-secrets.yaml
|
||||||
|
- apps/gitea.yaml
|
||||||
|
- apps/gitea-actions.yaml
|
||||||
|
- apps/http-echo.yaml
|
||||||
|
- apps/openebs.yaml
|
||||||
|
- apps/nats.yaml
|
||||||
|
- apps/dynamic-runner.yaml
|
||||||
|
- apps/spire.yaml
|
||||||
|
- apps/observability.yaml
|
||||||
|
- apps/zot.yaml
|
||||||
@@ -0,0 +1,4 @@
|
|||||||
|
apiVersion: v1
|
||||||
|
kind: Namespace
|
||||||
|
metadata:
|
||||||
|
name: gitops-canary
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# Sandbox 集群
|
||||||
|
|
||||||
|
这里是 OpenSandbox、CI 和 AI Agent workload 所在双节点 k3s 集群的 Flux
|
||||||
|
reconciliation 入口。LXC、PostgreSQL、K3s、固定版本的 Flux controllers 与 root
|
||||||
|
sync 由 `infrastructure/sandbox-cluster/` 中的 Ansible 管理;本目录只组合集群内
|
||||||
|
workload。
|
||||||
|
|
||||||
|
Flux 通过 `https://git.ddupan.top/panxiao81/homelab-infra.git` 读取公开仓库。
|
||||||
|
Ansible 将 homelab CA 注入 `GitRepository/flux-system` 引用的同名 Secret,不使用
|
||||||
|
长期 Git 凭据。root Kustomization 从 `./clusters/sandbox` 开始 reconciliation,
|
||||||
|
初始保持 `prune: false`。
|
||||||
|
|
||||||
|
Root bootstrap 已完成。后续按依赖顺序分别引入:
|
||||||
|
|
||||||
|
1. 监控 CRD、kube-state-metrics 以及 kubelet/cAdvisor 抓取配置;
|
||||||
|
2. SPIRE Agent、SPIFFE CSI Driver 与 workload registration;
|
||||||
|
3. Kata Containers、`block-plain` RuntimeClass;
|
||||||
|
4. OpenSandbox operator/server 及 `ci-pod`、`ci-vm` Pools。
|
||||||
|
|
||||||
|
每一阶段单独合并并等待对应 Flux Kustomization Ready,不在 bootstrap 时一次性部署。
|
||||||
|
第一阶段监控拆为 `monitoring-operator` 与依赖它的 `monitoring`,防止 VM CR 在
|
||||||
|
VictoriaMetrics Operator CRD Ready 前进入 reconciliation。
|
||||||
|
|
||||||
|
SPIRE 阶段先由 `spire-bootstrap` 安装 CRD,并声明按上游 k8s_psat Server plugin
|
||||||
|
要求收窄的 reviewer:它可以调用 TokenReview,并只读查询用于证明的 Pod 与 Node。
|
||||||
|
Agent ServiceAccount 留给后续 HelmRelease 创建,避免两个声明方争夺同一资源。随后运行
|
||||||
|
`infrastructure/sandbox-cluster/ansible/spire-bootstrap.yml`:playbook 从 sandbox
|
||||||
|
读取 reviewer token,在内存中组成受限 kubeconfig,再通过 stdin reconcile 到 central
|
||||||
|
集群的 `spire-server/spire-external-kubeconfigs` Secret。凭据不写入仓库、日志或控制机
|
||||||
|
文件;该 Secret 准备完成后,才能启用 central external PSAT/controller-manager 和
|
||||||
|
sandbox Agent/CSI。
|
||||||
|
|
||||||
|
External controller-manager 使用独立的 `spire-controller-manager` ServiceAccount;其
|
||||||
|
RBAC 与上游 controller-manager 所需权限一致,用于读取 workload selectors、维护
|
||||||
|
SPIFFE CR status/finalizer 和 leader election。它不复用只允许 TokenReview 的 Server
|
||||||
|
reviewer。Ansible 将两份 kubeconfig 写入同一个 central Secret 的不同 key,便于 central
|
||||||
|
chart 分别绑定 `sandbox` 与 `sandbox-controller`。
|
||||||
|
|
||||||
|
Central SPIRE Server 通过内网 `spire-server.ad.ddupan.top:8081` 接收 sandbox Agent
|
||||||
|
attestation。Server 使用 external bundle publisher 持续维护 sandbox
|
||||||
|
`spire-system/spire-bundle`,Agent 不固定或复制 trust bundle。Sandbox HelmRelease
|
||||||
|
显式关闭 Server 与 OIDC Provider,只部署 Agent DaemonSet 和 SPIFFE CSI Driver;因此
|
||||||
|
不会产生第二个 trust root。
|
||||||
|
|
||||||
|
`spire-smoke` namespace、ServiceAccount 和 `sandbox-spire-smoke` ClusterSPIFFEID 是
|
||||||
|
普通 Pod 与后续 Kata guest 的回归夹具,稳定身份为
|
||||||
|
`spiffe://ddupan.top/sandbox/smoke`。测试 Pod 临时创建并在验收后删除,身份声明保留。
|
||||||
|
|
||||||
|
Kata 阶段使用官方 4.1.0 `kata-deploy` chart 的短生命周期 `job` 模式,逐节点安装并
|
||||||
|
重启 K3s。只启用 `kata-clh-runtime-rs`,不创建默认 `kata` 别名;该 handler 的
|
||||||
|
`emptyDir` 固定使用 `block-plain`,为 Docker/BuildKit overlay2 与 kind 提供 guest
|
||||||
|
内块设备文件系统。详细限制与上线验收见 `platform/sandbox-kata/README.md`。
|
||||||
|
|
||||||
|
## 监控边界
|
||||||
|
|
||||||
|
这里只管理 sandbox LXC 内的 Kubernetes 监控,不负责 PVE 宿主监控。LXC 与宿主共享
|
||||||
|
内核,即使 lxcfs 虚拟化了内存和 uptime,容器内 `/proc/stat` 仍是宿主 CPU 视图;
|
||||||
|
在 LXC 内运行 node_exporter 会生成混合语义并重复采集宿主指标,因此禁止部署。
|
||||||
|
|
||||||
|
Sandbox 节点与 workload 指标来自 kubelet/cAdvisor 和 kube-state-metrics;K3s 或 LXC
|
||||||
|
特有但上述接口未覆盖的指标,应使用目标明确的 collector,不以 node_exporter 补齐。
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
---
|
||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: kata
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
dependsOn:
|
||||||
|
- name: monitoring-operator
|
||||||
|
- name: spire-agents
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/sandbox-kata
|
||||||
|
prune: true
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 35m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
---
|
||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: monitoring-operator
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/sandbox-monitoring/operator
|
||||||
|
prune: true
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 10m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
---
|
||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: monitoring
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
dependsOn:
|
||||||
|
- name: monitoring-operator
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/sandbox-monitoring/workloads
|
||||||
|
prune: true
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 10m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
---
|
||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: opensandbox-pools
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
dependsOn:
|
||||||
|
- name: opensandbox
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/sandbox-opensandbox-pools
|
||||||
|
prune: true
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 20m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
---
|
||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: opensandbox
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
dependsOn:
|
||||||
|
- name: kata
|
||||||
|
- name: spire-agents
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/sandbox-opensandbox
|
||||||
|
prune: true
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 20m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
---
|
||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: spire-agents
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
dependsOn:
|
||||||
|
- name: spire-bootstrap
|
||||||
|
- name: monitoring-operator
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/sandbox-spire/agents
|
||||||
|
prune: true
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 15m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
---
|
||||||
|
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||||
|
kind: Kustomization
|
||||||
|
metadata:
|
||||||
|
name: spire-bootstrap
|
||||||
|
namespace: flux-system
|
||||||
|
spec:
|
||||||
|
interval: 10m
|
||||||
|
path: ./platform/sandbox-spire/bootstrap
|
||||||
|
prune: true
|
||||||
|
sourceRef:
|
||||||
|
kind: GitRepository
|
||||||
|
name: flux-system
|
||||||
|
timeout: 10m
|
||||||
|
wait: true
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
---
|
||||||
|
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||||
|
kind: Kustomization
|
||||||
|
resources:
|
||||||
|
- apps/monitoring-operator.yaml
|
||||||
|
- apps/monitoring.yaml
|
||||||
|
- apps/spire-bootstrap.yaml
|
||||||
|
- apps/spire-agents.yaml
|
||||||
|
- apps/kata.yaml
|
||||||
|
- apps/opensandbox.yaml
|
||||||
|
- apps/opensandbox-pools.yaml
|
||||||
+19
-14
@@ -1,6 +1,7 @@
|
|||||||
# CI/CD — what we are building
|
# CI/CD — what we are building
|
||||||
|
|
||||||
Status: **design, partly built.** Stage 1 is live. The substrate is undecided.
|
Status: **partly built.** Stage 1、Kubernetes runner 与 Flux 已上线;credentialed
|
||||||
|
stages 尚未实现。
|
||||||
Started 2026-07-28.
|
Started 2026-07-28.
|
||||||
|
|
||||||
## Goal
|
## Goal
|
||||||
@@ -22,19 +23,21 @@ to make drift between this repo and reality visible when it happens.
|
|||||||
|
|
||||||
| | |
|
| | |
|
||||||
|---|---|
|
|---|---|
|
||||||
| git | History since 2026-07-28. Four commits, **no remote yet** |
|
| git | `homelab-infra` is hosted on the local Gitea; an independent off-site mirror is still missing |
|
||||||
| stage 1 | Live and green — `yamllint`, `ansible-lint`, `terraform fmt`/`validate`. Configs tuned against a real run |
|
| stage 1 | Gitea runner 上的 `yamllint`、`ansible-lint`、Terraform fmt/validate 已上线;三个 workflow 按路径触发,feature push 不再与 PR 事件重复运行 |
|
||||||
| gitea | 1.25.5, Actions **enabled**, `DEFAULT_ACTIONS_URL=github`. **No runner deployed**, so nothing executes |
|
| gitea | 1.27.3,Actions 已启用;一个 instance-scoped Kubernetes runner 以 capacity 4 运行 |
|
||||||
| ansible | 33 roles across `infrastructure/proxmox/`, `infrastructure/samba-ad/`, `infrastructure/openbao/` |
|
| ansible | 33 roles across `infrastructure/proxmox/`, `infrastructure/samba-ad/`, `infrastructure/openbao/` |
|
||||||
| terraform | 4 roots, **local state**, each with **different interactive auth** (`bao login -method=oidc`, `az login`) |
|
| terraform | 4 roots, **local state**, each with **different interactive auth** (`bao login -method=oidc`, `az login`) |
|
||||||
| k8s | ~13 Helm releases, all deployed by hand |
|
| k8s | Flux 已接管 Gitea、Gitea Actions 与 http-echo canary;其余 brownfield release 逐项迁移 |
|
||||||
| secrets | 4 config files are gitignored because they embed live secrets, so their contents are **not** version controlled |
|
| secrets | 4 config files are gitignored because they embed live secrets, so their contents are **not** version controlled |
|
||||||
|
|
||||||
## The four stages
|
## The four stages
|
||||||
|
|
||||||
**Stage 1 — static. Built.**
|
**Stage 1 — static. Built.**
|
||||||
`yamllint`, `ansible-lint`, `terraform fmt -check` / `validate -backend=false`.
|
`yamllint`, `ansible-lint`, `terraform fmt -check` / `validate -backend=false`.
|
||||||
No cluster, no credentials, no mutation, so it is safe on every push. It already
|
No cluster, no credentials, no mutation. YAML、Ansible 和 Terraform 各自按相关路径
|
||||||
|
触发;feature branch 只由 `pull_request` 检查,合并后再由 `main` push 检查,避免
|
||||||
|
同一 revision 因 branch push 和 PR 各跑一遍。它已经
|
||||||
found a real defect: `infrastructure/proxmox/ansible/` had no `requirements.yml` at all, so a
|
found a real defect: `infrastructure/proxmox/ansible/` had no `requirements.yml` at all, so a
|
||||||
fresh checkout could not reproduce its collections.
|
fresh checkout could not reproduce its collections.
|
||||||
|
|
||||||
@@ -90,11 +93,13 @@ Gitea Actions is the CI control plane. It integrates directly with repository
|
|||||||
permissions and status checks and preserves GitHub Actions workflow syntax.
|
permissions and status checks and preserves GitHub Actions workflow syntax.
|
||||||
|
|
||||||
The bootstrap worker is the official Gitea Runner chart in Kubernetes: one
|
The bootstrap worker is the official Gitea Runner chart in Kubernetes: one
|
||||||
persistent StatefulSet Pod, rootless Docker-in-Docker and capacity four. Job
|
persistent StatefulSet Pod, Docker-in-Docker and capacity four. Job
|
||||||
containers are dynamic, while the runner and its Docker daemon remain resident.
|
containers are dynamic, while the runner and its Docker daemon remain resident.
|
||||||
Rootless DinD still needs a privileged Pod to establish its user namespace, so
|
The chart's DinD container is privileged in both modes. Rootless mode is blocked
|
||||||
the runner is repository-scoped and restricted to trusted workflows. See
|
by the node's AppArmor unprivileged-userns policy, so regular DinD avoids weakening
|
||||||
`platform/gitea-runner/`.
|
that host-wide policy without pretending the Pod has a stronger isolation boundary.
|
||||||
|
The runner is instance-scoped and restricted to trusted repositories and workflows.
|
||||||
|
See `platform/gitea-runner/`.
|
||||||
|
|
||||||
This is not native pod-per-job execution. If stronger isolation becomes useful,
|
This is not native pod-per-job execution. If stronger isolation becomes useful,
|
||||||
the runner's ephemeral registration and Gitea `workflow_job` webhook can later
|
the runner's ephemeral registration and Gitea `workflow_job` webhook can later
|
||||||
@@ -113,7 +118,7 @@ Proxmox provider exists, so ephemeral Proxmox VMs would mean writing one.
|
|||||||
Bootstrap dependencies and current status:
|
Bootstrap dependencies and current status:
|
||||||
|
|
||||||
1. **Secret delivery exists.** OpenBao and External Secrets Operator already
|
1. **Secret delivery exists.** OpenBao and External Secrets Operator already
|
||||||
synchronize five Secrets. Add the repository-scoped runner registration token
|
synchronize five Secrets. Add the instance-scoped runner registration token
|
||||||
at `kv/k8s/gitea-runner`; Git contains only its `ExternalSecret` reference.
|
at `kv/k8s/gitea-runner`; Git contains only its `ExternalSecret` reference.
|
||||||
2. **Git remote exists.** `homelab-infra` is hosted in Gitea. An off-cluster
|
2. **Git remote exists.** `homelab-infra` is hosted in Gitea. An off-cluster
|
||||||
read-only mirror remains required for disaster recovery.
|
read-only mirror remains required for disaster recovery.
|
||||||
@@ -169,10 +174,10 @@ the CI system itself.
|
|||||||
|
|
||||||
## Sequencing
|
## Sequencing
|
||||||
|
|
||||||
1. Externalise secrets — unblocks everything, valuable on its own
|
1. Externalise secrets — **partly complete**; ESO delivery works, recovery and the remaining inventory are pending
|
||||||
2. Capture `e5renew` and `rustfs` into the repo (`helm get values`)
|
2. Capture `e5renew` and `rustfs` into the repo (`helm get values`)
|
||||||
3. Git remote
|
3. Git remote — **complete locally**; off-site mirror pending
|
||||||
4. Pick the substrate; stand it up in an isolated namespace
|
4. Pick the substrate; stand it up in an isolated namespace — **complete**
|
||||||
5. Stage 2, then stage 3 on **one** container-friendly role first
|
5. Stage 2, then stage 3 on **one** container-friendly role first
|
||||||
6. Flux on one low-stakes namespace (`http-echo` or `marker`)
|
6. Flux on one low-stakes namespace (`http-echo` or `marker`)
|
||||||
7. Drift detection for Terraform and Ansible — scoped machine identities
|
7. Drift detection for Terraform and Ansible — scoped machine identities
|
||||||
|
|||||||
@@ -0,0 +1,174 @@
|
|||||||
|
# Gitea 1.25.5 → 1.27.3 升级计划
|
||||||
|
|
||||||
|
## 决策
|
||||||
|
|
||||||
|
现有 release 为 chart `12.5.3` / Gitea `1.25.5`,已经由 Flux HelmRelease 接管。
|
||||||
|
升级分两个独立阶段执行,不跨过中间 minor:
|
||||||
|
|
||||||
|
1. chart `12.6.0`,显式设置 `image.tag: 1.26.4`;
|
||||||
|
2. chart `12.7.0`,显式设置 `image.tag: 1.27.3`。
|
||||||
|
|
||||||
|
chart 当前默认 appVersion 分别只是 `1.26.1` 和 `1.27.0`,因此不能依赖默认镜像。
|
||||||
|
官方 `1.26.4` 修复了 1.26.3 的代码页回归并包含安全修复;`1.27.3` 又修复了
|
||||||
|
Actions fork PR 审批绕过等安全问题。两个 rootless 镜像标签都已经在官方 registry
|
||||||
|
验证存在。Gitea Runner `2.3.0` 已高于 1.27 推荐的 `2.0.0`,无需与服务端同时升级。
|
||||||
|
|
||||||
|
参考:
|
||||||
|
|
||||||
|
- [Gitea 升级说明](https://docs.gitea.com/installation/upgrade-from-gitea/)
|
||||||
|
- [Gitea 备份一致性说明](https://docs.gitea.com/1.26/administration/backup-and-restore/)
|
||||||
|
- [Gitea 1.26.4 发布说明](https://blog.gitea.com/release-of-1.26.3-and-1.26.4/)
|
||||||
|
- [Gitea 1.27.0 与 Runner 兼容说明](https://blog.gitea.com/release-of-1.27.0/)
|
||||||
|
- [Gitea 1.27.3 发布说明](https://blog.gitea.com/release-of-1.27.3/)
|
||||||
|
|
||||||
|
## 当前风险边界
|
||||||
|
|
||||||
|
- Gitea 使用外部单实例 CNPG;`gitea` 数据库约 19 MB,CNPG 没有配置连续备份,
|
||||||
|
`firstRecoverabilityPoint` 为空。
|
||||||
|
- `/data` 使用 `local-path` RWO PVC,实际数据约 6.2 MB;集群没有
|
||||||
|
VolumeSnapshotClass,因此不能把 CSI snapshot 当成回滚点。
|
||||||
|
- Gitea 是 Flux GitRepository 的上游。Gitea 停机时已应用的资源继续运行,但不能
|
||||||
|
依靠新的 Git commit 解救失败的 Gitea。
|
||||||
|
- 跨 minor 启动会执行数据库 migration。官方明确说明升级后的数据库不能由旧 minor
|
||||||
|
安全使用;回退必须同时恢复数据库和 `/data`。
|
||||||
|
- `gitea-oidc-secret` 仍是手工 Secret。本次只继续引用它,不与版本升级一起迁移。
|
||||||
|
|
||||||
|
## 已完成的静态检查
|
||||||
|
|
||||||
|
- 使用现有 `gitea-values.yaml` 渲染三个 chart,并从比较中排除 Secret 与 Helm test
|
||||||
|
hook;`12.5.3 → 12.6.0` 的业务资源变化只有 chart/app 标签和四处 Gitea 镜像。
|
||||||
|
- `12.6.0 → 12.7.0` 除同类版本变化外,只包含 Service template 文件重命名、字段
|
||||||
|
排版变化和 test hook 参数排版,没有新增或删除 live 业务对象。
|
||||||
|
- `docker.gitea.com/gitea:1.26.4-rootless` 与
|
||||||
|
`docker.gitea.com/gitea:1.27.3-rootless` 的 multi-arch manifests 均存在。
|
||||||
|
- 升级前没有 deprecation 或 error;启动会报告若干数据库 default 比较及内网明文 SMTP
|
||||||
|
warning,均为已有状态。HTTPRoute、内部 API、统一域名 API 与 Flux GitRepository
|
||||||
|
均 Ready。
|
||||||
|
|
||||||
|
## 每个 minor 的执行单元
|
||||||
|
|
||||||
|
两个阶段各自重复以下流程。1.26 验收完成前不得创建 1.27 的执行 PR。
|
||||||
|
|
||||||
|
### 1. 预拉取与 review
|
||||||
|
|
||||||
|
1. 在 HelmRelease 中显式设置目标 `image.tag`,同时把 chart 固定到该阶段目标版本,
|
||||||
|
并重新执行 Helm/Kustomize render。目标 chart、镜像与 `suspend: true` 必须在同一
|
||||||
|
HelmRelease 对象内原子提交;不得同时修改会触发 watch 的 values ConfigMap。
|
||||||
|
2. 在节点预拉取目标 rootless 镜像,避免维护窗口受 WAN 波动影响。
|
||||||
|
3. 创建 **保持 `spec.suspend: true`** 的准备 PR;合并并等 Flux 同步。此时 Git 只记录
|
||||||
|
目标版本,不执行 Helm action。
|
||||||
|
4. 从该准备 commit 创建只删除 `suspend` 的激活 PR,提前 push 到 Gitea,但暂不合并。
|
||||||
|
|
||||||
|
### 2. 建立一致回滚点
|
||||||
|
|
||||||
|
Gitea 官方要求停机备份才能保证数据库、仓库、LFS、附件和 metadata 一致。本环境
|
||||||
|
数据量很小,因此接受一个短维护窗口,不使用在线 `gitea dump` 冒充一致备份。
|
||||||
|
|
||||||
|
1. 确认 HelmRelease 已 suspended,记录 Deployment、ReplicaSet、Pod UID、Helm
|
||||||
|
revision、数据库大小和 PVC 路径。
|
||||||
|
2. 手工将 Gitea Deployment scale 到 0 并等待 Pod 消失。此时 runner 会暂时无法上报,
|
||||||
|
Flux source 会暂时 NotReady,但现有工作负载不受影响。
|
||||||
|
3. 对 CNPG 的 `gitea` 数据库执行 custom-format `pg_dump`;对 PVC host path 创建保留
|
||||||
|
owner/xattr 的 tar archive。两个文件写入节点上新建的、权限为 0700 的时间戳目录。
|
||||||
|
4. 生成 SHA-256,执行 `pg_restore --list` 和 `tar --list`,确认两份文件可读。
|
||||||
|
5. 用个人 GPG 公钥加密后复制一份到 OCI Object Storage(或另一台物理机);未验证
|
||||||
|
第二份副本前不得升级。
|
||||||
|
6. 将 Deployment scale 回 1,确认仍以旧版本启动,并验证 API、OIDC 登录、clone、
|
||||||
|
push 和 Actions runner online。
|
||||||
|
|
||||||
|
手工 scale 只用于取得一致备份;HelmRelease 此时 suspended,不会发生 drift repair。
|
||||||
|
备份结束后服务恢复,后续激活 PR 仍通过正常 Gitea review/merge,不需要在 Gitea 停机
|
||||||
|
时修改 Git 或 live HelmRelease。
|
||||||
|
|
||||||
|
### 3. 由 Flux 升级
|
||||||
|
|
||||||
|
1. review 并合并已经 push 的激活 PR;它唯一的行为变化应是删除 `suspend: true`。
|
||||||
|
2. 不手工 reconcile,观察 Flux 正常发现 revision、Helm upgrade 和 Gitea migration。
|
||||||
|
3. `Recreate` strategy 会先终止旧 Pod 再启动新 Pod,避免 RWO PVC 与 leveldb queue
|
||||||
|
lock 导致双 Pod deadlock。
|
||||||
|
4. 若 15 分钟内 HelmRelease 未 Ready,停止自动重试并进入回滚判断,不连续修改
|
||||||
|
values 猜测修复。
|
||||||
|
|
||||||
|
### 4. 每阶段验收
|
||||||
|
|
||||||
|
- HelmRelease `Ready=True`、`UpgradeSucceeded`,Helm revision 只增加一次;
|
||||||
|
- Deployment 和 Pod 使用目标 rootless 镜像,PVC UID/volume name 不变;
|
||||||
|
- Gitea `/api/v1/version` 从 Pod 内和 `https://git.ddupan.top` 返回目标版本;
|
||||||
|
- Authelia OIDC 管理员登录与 break-glass 本地管理员登录均有效;
|
||||||
|
- 对现有仓库完成 HTTPS clone、创建临时 branch、push、删除临时 branch;
|
||||||
|
- 当前仓库 Actions workflow 能入队并由 runner `2.3.0` 完成;
|
||||||
|
- Flux GitRepository 恢复 Ready 并能拉取升级 revision;
|
||||||
|
- package registry、Terraform package/state API(启用后)与附件/LFS 按实际使用情况抽查;
|
||||||
|
- 日志中没有 migration failure、panic、持续数据库错误或配置弃用警告;
|
||||||
|
- 稳定观察至少 15 分钟,再把该阶段记录为完成。
|
||||||
|
|
||||||
|
## 回滚
|
||||||
|
|
||||||
|
patch 版本原则上保持数据库结构兼容;本计划的两个步骤都是跨 minor,不能仅回退镜像。
|
||||||
|
若 migration 后目标版本无法健康启动:
|
||||||
|
|
||||||
|
1. 暂停 `Kustomization/gitea` 和 `HelmRelease/gitea`,停止 Gitea Deployment;
|
||||||
|
2. 保存失败现场的 Pod logs、Helm status 与 migration error;
|
||||||
|
3. 删除并从同一阶段备份恢复 `gitea` 数据库;
|
||||||
|
4. 清空并从配套 tar 恢复原 PVC 内容,保持 owner、mode 和 xattr;
|
||||||
|
5. 将 live HelmRelease 临时恢复到上一阶段 chart/image,并启动验证;
|
||||||
|
6. Gitea 恢复后,通过 PR 把 Git desired state 恢复到上一阶段,再恢复 Flux
|
||||||
|
Kustomization。不得让 Flux 在数据库尚未恢复时自动拉回新版本。
|
||||||
|
|
||||||
|
数据库 dump 与 PVC archive 是一个不可拆分的回滚点;禁止混用不同时间或不同阶段的
|
||||||
|
两份备份。
|
||||||
|
|
||||||
|
## 1.26.4 阶段执行记录
|
||||||
|
|
||||||
|
- 目标 `1.26.4-rootless` 镜像已经预拉取到 k3s 节点;
|
||||||
|
- Gitea 停写后生成了 custom-format PostgreSQL dump 与保留 owner/ACL/xattr 的 PVC
|
||||||
|
tar,两者分别通过 `pg_restore --list`、`tar --list` 和 SHA-256 校验;
|
||||||
|
- 本地回滚点位于
|
||||||
|
`/var/backups/homelab/gitea/20260910T060400Z-1.25.5/`,权限为 0700;其中另有使用
|
||||||
|
个人 GPG 公钥加密并通过 packet 检查的 bundle;
|
||||||
|
- 操作者明确选择不上传 OCI,因此本阶段接受只有节点本地副本的风险例外;
|
||||||
|
- 备份后旧版 `1.25.5` 已恢复,Pod 内与统一域名 API、Gitea API 和 Flux Git source
|
||||||
|
均验证正常。
|
||||||
|
|
||||||
|
激活后 Helm revision 16 以 chart `12.6.0` 成功部署
|
||||||
|
`docker.gitea.com/gitea:1.26.4-rootless`。Migration 323–330 完成;Pod 内和统一域名
|
||||||
|
API、临时 branch push/delete、Flux source,以及 main/smoke 的 YAML、Ansible、
|
||||||
|
Terraform CI 均通过。Pod 在约 15 分钟观察期内保持 Running、零重启;Authelia OIDC
|
||||||
|
配置 init 同步及浏览器交互式管理员登录也已确认成功。
|
||||||
|
|
||||||
|
## 1.27.3 阶段准备状态
|
||||||
|
|
||||||
|
- 目标 `1.27.3-rootless` 镜像已经预拉取到 k3s 节点;
|
||||||
|
- Git desired state 固定为 chart `12.7.0` 与显式 image `1.27.3`,并重新设置
|
||||||
|
`suspend: true`;
|
||||||
|
- 合并准备状态后必须从已经迁移的 1.26.4 数据重新建立一组数据库/PVC 回滚点,禁止
|
||||||
|
复用 1.25.5 备份作为 1.27 阶段的直接回滚点。
|
||||||
|
|
||||||
|
执行时操作者明确选择跳过上述 1.26.4 备份门槛并继续激活。该风险例外意味着 1.27
|
||||||
|
migration 后若失败,不能无损回到 1.26.4;现有本地 1.25.5 备份只可用于接受丢失
|
||||||
|
第一跳之后状态的灾难恢复。目标 1.27.3 镜像已预拉取,其余准备检查均已通过。
|
||||||
|
|
||||||
|
激活后 Helm revision 17 以 chart `12.7.0` 成功部署实际镜像 `1.27.3-rootless`,
|
||||||
|
migration 331–342 和全部 init containers 成功;Pod 内/统一域名 API、Git pull、临时
|
||||||
|
branch push/delete 与 Flux source 均通过,Pod Ready 且零重启。Chart metadata 显示
|
||||||
|
appVersion `1.27.0`,实际版本以固定 image 和 API 返回的 `1.27.3` 为准。
|
||||||
|
|
||||||
|
## 后续升级策略
|
||||||
|
|
||||||
|
本次连续两次跨 minor 升级证明现有 chart、外部 CNPG、rootless PVC 和 `Recreate`
|
||||||
|
组合工作稳定。后续常规 patch/minor 升级采用精简验收:固定 chart/image、审阅相关
|
||||||
|
release notes、render、确认 `UpgradeSucceeded`/Pod Ready/API 版本,再人工抽查 OIDC
|
||||||
|
与 Git。长时间观察、临时 branch、重复 CI、逐条 migration 日志和逐 minor 停机备份
|
||||||
|
不再是默认步骤。
|
||||||
|
|
||||||
|
以下任一条件出现时恢复本文的完整流程:数据库或存储变更、rootless/权限模型变化、
|
||||||
|
PVC identity 或 deployment strategy 变化、重大 chart 结构变化、相关 breaking migration,
|
||||||
|
以及任何启动失败、CrashLoop 或 migration error。
|
||||||
|
|
||||||
|
## 后续但不并入升级
|
||||||
|
|
||||||
|
- 将 `gitea-oidc-secret` 等剩余手工 Secret 迁入 OpenBao/ESO;
|
||||||
|
- Gitea 1.27 稳定后验证最小权限 Actions job token,再实现无集群凭据的 PR render
|
||||||
|
diff/summary;live `kubectl diff` 仍使用独立、受信任且限权的执行路径;
|
||||||
|
- 单独评估关闭 chart 生成但无人处理的 Ingress;
|
||||||
|
- 为 CNPG 和 Gitea `/data` 建立周期性、异机可恢复备份,替代升级前一次性备份。
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
# Homelab GitOps and IaC redesign
|
# Homelab GitOps and IaC redesign
|
||||||
|
|
||||||
Status: **design; no infrastructure changes have been applied.**
|
Status: **implementation in progress; CI、Flux bootstrap、漂移修复与受控 prune 已验证。**
|
||||||
|
|
||||||
Started 2026-09-09. This is the durable record of the redesign discussion. It
|
Started 2026-09-09. This is the durable record of the redesign discussion. It
|
||||||
separates observations, decisions and open work so an assumption cannot silently
|
separates observations, decisions and open work so an assumption cannot silently
|
||||||
@@ -35,13 +35,19 @@ or reconcile later CPU, memory, NIC or boot drift.
|
|||||||
|
|
||||||
### Kubernetes and delivery
|
### Kubernetes and delivery
|
||||||
|
|
||||||
- Kubernetes is a single-node k3s cluster.
|
- Kubernetes is a single-node k3s `v1.36.4+k3s1` cluster.
|
||||||
- Helm releases and manifests have historically been applied by hand.
|
- Helm releases and manifests have historically been applied by hand and are now
|
||||||
- No Flux or Argo CD installation was found during the initial audit.
|
being adopted by Flux one release at a time.
|
||||||
|
- Flux `v2.9.5` is live. Its internal Gitea source and root Kustomization are
|
||||||
|
Ready; `http-echo` proved automatic deployment, drift repair and scoped prune.
|
||||||
- Repository history records External Secrets Operator 2.8.0 as deployed. Five
|
- Repository history records External Secrets Operator 2.8.0 as deployed. Five
|
||||||
`ExternalSecret` resources cover Authelia, Gitea, Cloudflared and SeaweedFS.
|
`ExternalSecret` resources cover Authelia, Gitea, Cloudflared and SeaweedFS.
|
||||||
All five reported `SecretSynced=True` during a live check on 2026-09-09.
|
All five reported `SecretSynced=True` during a live check on 2026-09-09.
|
||||||
- The repository has no configured Git remote yet.
|
- Gitea Actions now has one instance-scoped runner in namespace `gitea-actions`.
|
||||||
|
Its Helm release is deployed, its Pod is `2/2 Running`, its identity PVC is
|
||||||
|
bound, and a sixth `ExternalSecret` delivers the registration token from OpenBao.
|
||||||
|
- The repository is hosted at `panxiao81/homelab-infra` on the local Gitea
|
||||||
|
instance. A one-way off-site mirror is still missing.
|
||||||
- Terraform roots remain per-service and must not be merged.
|
- Terraform roots remain per-service and must not be merged.
|
||||||
- The k3s node was `Ready` on 2026-09-09. The Snap-packaged `kubectl` could
|
- The k3s node was `Ready` on 2026-09-09. The Snap-packaged `kubectl` could
|
||||||
not start because the user systemd session was degraded; `k3s kubectl` with the
|
not start because the user systemd session was degraded; `k3s kubectl` with the
|
||||||
@@ -84,10 +90,12 @@ without becoming another deployment controller.
|
|||||||
### CI execution
|
### CI execution
|
||||||
|
|
||||||
Most CI uses Gitea Actions for its GitHub Actions compatibility. A persistent
|
Most CI uses Gitea Actions for its GitHub Actions compatibility. A persistent
|
||||||
Gitea Runner StatefulSet runs in Kubernetes with rootless Docker-in-Docker and
|
Gitea Runner StatefulSet runs in Kubernetes with Docker-in-Docker and capacity
|
||||||
capacity four; individual job containers are created dynamically. Rootless DinD
|
four; individual job containers are created dynamically. The chart requires a
|
||||||
still requires a privileged Pod, so the runner is repository-scoped and accepts
|
privileged DinD container in both modes, and rootlesskit is blocked by the node's
|
||||||
trusted workflows only.
|
AppArmor unprivileged-userns policy, so regular DinD is used instead of weakening
|
||||||
|
that host-wide policy. The runner is instance-scoped and accepts trusted
|
||||||
|
repositories and workflows only.
|
||||||
|
|
||||||
Only explicitly labelled jobs needing privilege, nested virtualization,
|
Only explicitly labelled jobs needing privilege, nested virtualization,
|
||||||
amd64-only software or isolation from k3s use an ephemeral Proxmox VM. IaC owns
|
amd64-only software or isolation from k3s use an ephemeral Proxmox VM. IaC owns
|
||||||
@@ -196,17 +204,23 @@ offline break-glass path. ESO-generated Secrets are projections, not backups.
|
|||||||
|
|
||||||
## Implementation phases
|
## Implementation phases
|
||||||
|
|
||||||
1. Preserve OCI state and sanitized libvirt/Helm evidence.
|
1. **In progress:** sanitized libvirt/Helm evidence is recorded; the independent
|
||||||
2. Re-verify the existing OpenBao/ESO delivery and recovery path, inventory
|
encrypted OCI state backup is still missing.
|
||||||
|
2. **In progress:** live OpenBao/ESO delivery is verified; verify recovery and inventory
|
||||||
remaining manually managed Secrets, then migrate them incrementally.
|
remaining manually managed Secrets, then migrate them incrementally.
|
||||||
3. Configure Gitea remote plus a one-way off-site mirror. Confirm whether the
|
3. **In progress:** the Gitea remote exists; add a one-way off-site mirror and
|
||||||
running Gitea supports the 1.27 Terraform State Registry.
|
revisit the Terraform State Registry after upgrading beyond Gitea 1.25.5.
|
||||||
4. Manually deploy the reviewed Gitea Runner bootstrap, then bootstrap Flux on
|
4. **Complete:** the reviewed Gitea Runner and Stage 1 CI are live. Flux deploys
|
||||||
`http-echo` or `marker` without enabling prune until live ownership is audited.
|
`http-echo`; automatic deployment, replica drift repair and scoped deletion
|
||||||
5. Move Tunnel origins to Envoy and consolidate split DNS through Blocky.
|
were verified. Root prune remains disabled for brownfield safety.
|
||||||
6. Deploy Backstage read-only with Catalog, Kubernetes, Flux and TechDocs.
|
5. **In progress:** Gitea Actions and Gitea are managed by Flux. Adopt External
|
||||||
7. Reconstruct the OCI root to a zero-change plan and add libvirt drift reports.
|
Secrets Operator next with its existing chart `2.8.0` and repository values;
|
||||||
8. Add dynamic PVE VM workers only when a real job requires one, with TTL cleanup
|
register the suspended release first, then activate it in a separate PR after
|
||||||
|
proving the fixed render matches Helm's stored manifest.
|
||||||
|
6. Move Tunnel origins to Envoy and consolidate split DNS through Blocky.
|
||||||
|
7. Deploy Backstage read-only with Catalog, Kubernetes, Flux and TechDocs.
|
||||||
|
8. Reconstruct the OCI root to a zero-change plan and add libvirt drift reports.
|
||||||
|
9. Add dynamic PVE VM workers only when a real job requires one, with TTL cleanup
|
||||||
and hard concurrency/resource limits first.
|
and hard concurrency/resource limits first.
|
||||||
|
|
||||||
## Open decisions
|
## Open decisions
|
||||||
@@ -216,7 +230,7 @@ offline break-glass path. ESO-generated Secrets are projections, not backups.
|
|||||||
- Whether OCI Object Storage passes the concurrent lockfile test.
|
- Whether OCI Object Storage passes the concurrent lockfile test.
|
||||||
- Whether the OCI VM should later move from its current public subnet.
|
- Whether the OCI VM should later move from its current public subnet.
|
||||||
- Schema/generator for the Git-owned service declaration.
|
- Schema/generator for the Git-owned service declaration.
|
||||||
- Which live Helm releases are absent from or differ from Git.
|
- Migration order for Helm releases after the `gitea-actions` adoption.
|
||||||
- Whether each local libvirt VM should autostart.
|
- Whether each local libvirt VM should autostart.
|
||||||
- Which first job genuinely requires a dynamic Proxmox VM.
|
- Which first job genuinely requires a dynamic Proxmox VM.
|
||||||
|
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ SUB-SKILL", checkbox task lists) that only ever suited one migration. Design
|
|||||||
documents now live directly in `docs/` — see `../cicd.md` — and the split that
|
documents now live directly in `docs/` — see `../cicd.md` — and the split that
|
||||||
matters is:
|
matters is:
|
||||||
|
|
||||||
- `CHANGELOG.md` — what changed, for humans
|
- `CHANGELOG.md` — frozen historical snapshot; Git commits and PRs now record changes
|
||||||
- `CLAUDE.md` — traps and procedures, for agents
|
- `CLAUDE.md` — traps and procedures, for agents
|
||||||
- `docs/*.md` — design docs for work not yet built
|
- `docs/*.md` — design docs for work not yet built
|
||||||
- `<service>/README.md` — how a service actually works
|
- `<service>/README.md` — how a service actually works
|
||||||
|
|||||||
@@ -0,0 +1,9 @@
|
|||||||
|
# Generated by infrastructure/dns/generate.py. Do not edit directly.
|
||||||
|
resource "cloudflare_dns_record" "auth" {
|
||||||
|
zone_id = var.zone_id
|
||||||
|
name = "auth.ddupan.top"
|
||||||
|
type = "CNAME"
|
||||||
|
content = "ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com"
|
||||||
|
proxied = true
|
||||||
|
ttl = 1
|
||||||
|
}
|
||||||
@@ -38,17 +38,6 @@ resource "cloudflare_zero_trust_tunnel_cloudflared_config" "main" {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
# Public DNS: proxied CNAME -> the tunnel. (auth was bootstrapped with
|
|
||||||
# `cloudflared tunnel route dns`; import it into state — see README.)
|
|
||||||
resource "cloudflare_dns_record" "auth" {
|
|
||||||
zone_id = var.zone_id
|
|
||||||
name = "auth.ddupan.top"
|
|
||||||
type = "CNAME"
|
|
||||||
content = "${var.tunnel_id}.cfargotunnel.com"
|
|
||||||
proxied = true
|
|
||||||
ttl = 1 # 1 = automatic (required when proxied)
|
|
||||||
}
|
|
||||||
|
|
||||||
# DKIM for Microsoft 365 mail sent as *@ddupan.top (via the smtp-relay). CNAMEs point
|
# DKIM for Microsoft 365 mail sent as *@ddupan.top (via the smtp-relay). CNAMEs point
|
||||||
# at the tenant's DKIM keys; must be DNS-only (unproxied). Enable signing in Exchange
|
# at the tenant's DKIM keys; must be DNS-only (unproxied). Enable signing in Exchange
|
||||||
# after these resolve: smtp-relay/scripts/enable-dkim.ps1.
|
# after these resolve: smtp-relay/scripts/enable-dkim.ps1.
|
||||||
|
|||||||
@@ -0,0 +1,55 @@
|
|||||||
|
# DNS 声明与权威边界
|
||||||
|
|
||||||
|
`records.yml` 是 homelab DNS 的唯一声明清单,但不是 DNS 服务本身。不同视图仍由最适合
|
||||||
|
它们的后端提供:
|
||||||
|
|
||||||
|
| 视图 | 权威或递归服务 | 配置方式 |
|
||||||
|
|---|---|---|
|
||||||
|
| 公网 `ddupan.top` | Cloudflare | 由生成器输出 Terraform;尚待完整导入已有记录 |
|
||||||
|
| AD `ad.ddupan.top` | Samba internal DNS | `samba_dns_record` Ansible module |
|
||||||
|
| LAN split horizon | Blocky | 由生成器维护 `customDNS.mapping` 标记块 |
|
||||||
|
| Kubernetes Pod split horizon | CoreDNS | 由生成器维护 `.server` 标记块 |
|
||||||
|
|
||||||
|
## 生成配置
|
||||||
|
|
||||||
|
安装了 `uv` 后,在仓库根目录运行:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
uv run infrastructure/dns/generate.py
|
||||||
|
uv run infrastructure/dns/generate.py --check
|
||||||
|
```
|
||||||
|
|
||||||
|
脚本使用内嵌锁定版本的 PyYAML 和 Jinja2,从 `records.yml` 渲染三个目标:
|
||||||
|
|
||||||
|
- `apps/blocky/config.yml` 中带 marker 的 LAN split-horizon mapping;
|
||||||
|
- `platform/k3s/coredns-custom.yaml` 中带 marker 的 Pod split-horizon server blocks;
|
||||||
|
- `infrastructure/cloudflared/terraform/dns.generated.tf` 中已经完成 Terraform 接管的公网记录。
|
||||||
|
|
||||||
|
生成文件需要提交进 Git,以便 PR 直接审阅最终配置。CI 执行 `--check`,任何手工修改生成块、
|
||||||
|
漏跑生成器或非确定性输出都会失败。Jinja 使用 `[[ ... ]]` 作为变量定界符,避免与 CoreDNS
|
||||||
|
模板表达式 `{{ .Name }}` 冲突。
|
||||||
|
|
||||||
|
`backends` 和 `terraform.managed` 是分阶段接管开关,而不是第二份记录数据:只有已经完成
|
||||||
|
零变更接管的后端才会生成。把记录加入新的后端前,应先完成相应的 live/state 对账。
|
||||||
|
|
||||||
|
## 安全边界
|
||||||
|
|
||||||
|
- Samba module 只管理 `homelab_dns.samba.records` 明确列出的 RRset,不遍历或清理 zone。
|
||||||
|
- `_ldap`、`_kerberos`、域控制器 locator 等由 Samba 自动维护的记录不进入 inventory。
|
||||||
|
- 一个受管 RRset 默认使用 `exact: true`:同名同类型的额外值会被删除,但其他名称和类型
|
||||||
|
不受影响。
|
||||||
|
- DHCP 切换不属于本阶段。Blocky 仍未成为 LAN 客户端的正式 resolver。
|
||||||
|
|
||||||
|
## 分阶段接管
|
||||||
|
|
||||||
|
1. 用 Samba module 接管现有静态 A RRset,首次 check mode 应为零变更。
|
||||||
|
2. 将 Cloudflare 已有 tunnel DNS 记录逐条导入 Terraform state,再启用 `terraform.managed`。
|
||||||
|
3. Blocky 与 CoreDNS 已从 `split_horizon.records` 生成;通过 `backends` 分阶段扩展。
|
||||||
|
4. 验证公网、LAN、Pod、AD 四个视图后,再单独修改 DHCP。
|
||||||
|
|
||||||
|
当前 inventory 明确保留一个既有差异:`obj.ddupan.top` 的 `backends` 只有 Blocky,CoreDNS
|
||||||
|
尚无对应覆盖。本阶段不改变线上语义;后续验证 Pod 侧入口后再加入 `coredns`。
|
||||||
|
|
||||||
|
CoreDNS split-horizon 的原因是避免集群内请求经 Cloudflare 公网绕回同一个集群。尤其 Gitea
|
||||||
|
启动时会访问 Authelia discovery URL,公网路径故障曾令其启动失败;生成块仍返回相同 LAN A
|
||||||
|
记录,并对 AAAA 返回 NOERROR/no-data。
|
||||||
@@ -0,0 +1,146 @@
|
|||||||
|
#!/usr/bin/env -S uv run --script
|
||||||
|
# /// script
|
||||||
|
# requires-python = ">=3.12"
|
||||||
|
# dependencies = ["Jinja2==3.1.6", "PyYAML==6.0.3"]
|
||||||
|
# ///
|
||||||
|
"""Render backend DNS configuration from records.yml."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import difflib
|
||||||
|
from pathlib import Path
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import yaml
|
||||||
|
from jinja2 import Environment, FileSystemLoader, StrictUndefined
|
||||||
|
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
DNS_DIR = ROOT / "infrastructure/dns"
|
||||||
|
BEGIN = "# BEGIN GENERATED: homelab DNS ([[ target ]])"
|
||||||
|
END = "# END GENERATED: homelab DNS ([[ target ]])"
|
||||||
|
|
||||||
|
|
||||||
|
def load_inventory() -> dict:
|
||||||
|
data = yaml.safe_load((DNS_DIR / "records.yml").read_text())
|
||||||
|
try:
|
||||||
|
inventory = data["homelab_dns"]
|
||||||
|
split_records = inventory["split_horizon"]["records"]
|
||||||
|
public_records = inventory["public"]["records"]
|
||||||
|
except (KeyError, TypeError) as exc:
|
||||||
|
raise ValueError(f"invalid DNS inventory: missing {exc}") from exc
|
||||||
|
|
||||||
|
for record in split_records:
|
||||||
|
require_fields(record, "name", "type", "values", "backends")
|
||||||
|
if record["type"] != "A" or len(record["values"]) != 1:
|
||||||
|
raise ValueError(f"split record must be a single A value: {record!r}")
|
||||||
|
unknown = set(record["backends"]) - {"blocky", "coredns"}
|
||||||
|
if unknown:
|
||||||
|
raise ValueError(f"unknown split DNS backends {sorted(unknown)}")
|
||||||
|
|
||||||
|
for record in public_records:
|
||||||
|
require_fields(record, "name", "type", "values", "proxied", "terraform")
|
||||||
|
terraform = record["terraform"]
|
||||||
|
if terraform.get("managed") and not terraform.get("resource_name"):
|
||||||
|
raise ValueError(f"managed Terraform record needs resource_name: {record['name']}")
|
||||||
|
if len(record["values"]) != 1:
|
||||||
|
raise ValueError(f"Cloudflare Terraform supports one value per record: {record['name']}")
|
||||||
|
return inventory
|
||||||
|
|
||||||
|
|
||||||
|
def require_fields(record: dict, *fields: str) -> None:
|
||||||
|
missing = [field for field in fields if field not in record]
|
||||||
|
if missing:
|
||||||
|
raise ValueError(f"record missing {', '.join(missing)}: {record!r}")
|
||||||
|
|
||||||
|
|
||||||
|
def environment() -> Environment:
|
||||||
|
return Environment(
|
||||||
|
loader=FileSystemLoader(DNS_DIR / "templates"),
|
||||||
|
undefined=StrictUndefined,
|
||||||
|
autoescape=False,
|
||||||
|
keep_trailing_newline=True,
|
||||||
|
trim_blocks=True,
|
||||||
|
lstrip_blocks=True,
|
||||||
|
variable_start_string="[[",
|
||||||
|
variable_end_string="]]",
|
||||||
|
block_start_string="[%",
|
||||||
|
block_end_string="%]",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def marker(target: str, end: bool = False) -> str:
|
||||||
|
return (END if end else BEGIN).replace("[[ target ]]", target)
|
||||||
|
|
||||||
|
|
||||||
|
def replace_block(original: str, target: str, rendered: str) -> str:
|
||||||
|
begin = marker(target)
|
||||||
|
end = marker(target, end=True)
|
||||||
|
if original.count(begin) != 1 or original.count(end) != 1:
|
||||||
|
raise ValueError(f"expected exactly one generated block for {target}")
|
||||||
|
prefix, remainder = original.split(begin, 1)
|
||||||
|
_, suffix = remainder.split(end, 1)
|
||||||
|
indent = prefix.rsplit("\n", 1)[-1]
|
||||||
|
body = rendered.rstrip("\n")
|
||||||
|
return f"{prefix}{begin}\n{body}\n{indent}{end}{suffix}"
|
||||||
|
|
||||||
|
|
||||||
|
def outputs(inventory: dict) -> dict[Path, str]:
|
||||||
|
env = environment()
|
||||||
|
split_records = inventory["split_horizon"]["records"]
|
||||||
|
public_records = inventory["public"]["records"]
|
||||||
|
result = {}
|
||||||
|
|
||||||
|
blocky_path = ROOT / "apps/blocky/config.yml"
|
||||||
|
blocky = env.get_template("blocky.yml.j2").render(
|
||||||
|
records=[record for record in split_records if "blocky" in record["backends"]]
|
||||||
|
)
|
||||||
|
result[blocky_path] = replace_block(blocky_path.read_text(), "blocky", blocky)
|
||||||
|
|
||||||
|
coredns_path = ROOT / "platform/k3s/coredns-custom.yaml"
|
||||||
|
coredns = env.get_template("coredns.yaml.j2").render(
|
||||||
|
records=[record for record in split_records if "coredns" in record["backends"]]
|
||||||
|
)
|
||||||
|
result[coredns_path] = replace_block(coredns_path.read_text(), "coredns", coredns)
|
||||||
|
|
||||||
|
terraform_path = ROOT / "infrastructure/cloudflared/terraform/dns.generated.tf"
|
||||||
|
terraform = env.get_template("cloudflare.tf.j2").render(
|
||||||
|
records=[record for record in public_records if record["terraform"]["managed"]]
|
||||||
|
)
|
||||||
|
result[terraform_path] = terraform
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--check", action="store_true", help="fail when generated files differ")
|
||||||
|
args = parser.parse_args()
|
||||||
|
try:
|
||||||
|
rendered_outputs = outputs(load_inventory())
|
||||||
|
except (OSError, ValueError, yaml.YAMLError) as exc:
|
||||||
|
print(f"dns generation failed: {exc}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
changed = False
|
||||||
|
for path, expected in rendered_outputs.items():
|
||||||
|
actual = path.read_text() if path.exists() else ""
|
||||||
|
if actual == expected:
|
||||||
|
continue
|
||||||
|
changed = True
|
||||||
|
if args.check:
|
||||||
|
print("".join(difflib.unified_diff(
|
||||||
|
actual.splitlines(keepends=True),
|
||||||
|
expected.splitlines(keepends=True),
|
||||||
|
fromfile=str(path.relative_to(ROOT)),
|
||||||
|
tofile=f"{path.relative_to(ROOT)} (generated)",
|
||||||
|
)))
|
||||||
|
else:
|
||||||
|
path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
path.write_text(expected)
|
||||||
|
print(f"rendered {path.relative_to(ROOT)}")
|
||||||
|
return 1 if args.check and changed else 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
---
|
||||||
|
# Homelab DNS desired state. This file is the canonical inventory; individual
|
||||||
|
# backends consume only the views they own.
|
||||||
|
homelab_dns:
|
||||||
|
samba:
|
||||||
|
# Samba remains authoritative for the AD zone. Only these explicitly listed
|
||||||
|
# RRsets are reconciled; Samba-generated AD/Kerberos records are untouched.
|
||||||
|
records:
|
||||||
|
- { zone: ad.ddupan.top, name: bao, type: A, values: [192.168.10.8] }
|
||||||
|
- { zone: ad.ddupan.top, name: pve1, type: A, values: [192.168.10.4] }
|
||||||
|
- { zone: ad.ddupan.top, name: pve2, type: A, values: [192.168.10.7] }
|
||||||
|
- { zone: ad.ddupan.top, name: pve3, type: A, values: [192.168.10.9] }
|
||||||
|
- { zone: ad.ddupan.top, name: sandbox1, type: A, values: [10.60.0.11] }
|
||||||
|
- { zone: ad.ddupan.top, name: sandbox2, type: A, values: [10.60.0.12] }
|
||||||
|
- { zone: ad.ddupan.top, name: sandbox-k8s, type: A, values: [10.60.0.13] }
|
||||||
|
- { zone: ad.ddupan.top, name: retrolab, type: A, values: [10.60.0.10] }
|
||||||
|
- { zone: ad.ddupan.top, name: grafana, type: A, values: [192.168.10.127] }
|
||||||
|
- { zone: ad.ddupan.top, name: metrics-write, type: A, values: [192.168.10.127] }
|
||||||
|
- { zone: ad.ddupan.top, name: netbox, type: A, values: [192.168.10.127] }
|
||||||
|
- { zone: ad.ddupan.top, name: nats, type: A, values: [192.168.10.127] }
|
||||||
|
- { zone: ad.ddupan.top, name: s3, type: A, values: [192.168.10.127] }
|
||||||
|
- { zone: ad.ddupan.top, name: spire-oidc, type: A, values: [192.168.10.127] }
|
||||||
|
- { zone: ad.ddupan.top, name: spire-server, type: A, values: [192.168.10.127] }
|
||||||
|
- { zone: ad.ddupan.top, name: zot, type: A, values: [192.168.10.127] }
|
||||||
|
- { zone: ad.ddupan.top, name: zot-push, type: A, values: [192.168.10.127] }
|
||||||
|
|
||||||
|
split_horizon:
|
||||||
|
# backends records the current adoption boundary. obj is deliberately not
|
||||||
|
# emitted to CoreDNS yet, preserving the current pod resolver behaviour.
|
||||||
|
records:
|
||||||
|
- { name: git.ddupan.top, type: A, values: [192.168.10.127], backends: [blocky, coredns] }
|
||||||
|
- { name: auth.ddupan.top, type: A, values: [192.168.10.127], backends: [blocky, coredns] }
|
||||||
|
- { name: obj.ddupan.top, type: A, values: [192.168.10.127], backends: [blocky] }
|
||||||
|
|
||||||
|
public:
|
||||||
|
# Names expected at Cloudflare. Terraform adoption is a separate change;
|
||||||
|
# complete RRsets here make the current ownership gap explicit.
|
||||||
|
records:
|
||||||
|
- name: auth.ddupan.top
|
||||||
|
type: CNAME
|
||||||
|
values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com]
|
||||||
|
proxied: true
|
||||||
|
terraform:
|
||||||
|
managed: true
|
||||||
|
resource_name: auth
|
||||||
|
- name: git.ddupan.top
|
||||||
|
type: CNAME
|
||||||
|
values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com]
|
||||||
|
proxied: true
|
||||||
|
terraform: { managed: false }
|
||||||
|
- name: obj.ddupan.top
|
||||||
|
type: CNAME
|
||||||
|
values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com]
|
||||||
|
proxied: true
|
||||||
|
terraform: { managed: false }
|
||||||
|
- name: e5renew.ddupan.top
|
||||||
|
type: CNAME
|
||||||
|
values: [ff392451-b0b1-45bb-964e-6d9372c3a9e3.cfargotunnel.com]
|
||||||
|
proxied: true
|
||||||
|
terraform: { managed: false }
|
||||||
|
# OCI 主机直接解析公网 IP,SSH 不经过 Cloudflare 代理。
|
||||||
|
- name: oci-arm.ddupan.top
|
||||||
|
type: A
|
||||||
|
values:
|
||||||
|
- 129.225.138.179
|
||||||
|
proxied: false
|
||||||
|
ttl: 300
|
||||||
|
terraform: { managed: false }
|
||||||
|
- name: oci-amd.ddupan.top
|
||||||
|
type: A
|
||||||
|
values:
|
||||||
|
- 129.225.176.134
|
||||||
|
proxied: false
|
||||||
|
ttl: 300
|
||||||
|
terraform: { managed: false }
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
[% for record in records %]
|
||||||
|
[[ record.name ]]: [[ record['values'][0] ]]
|
||||||
|
[% endfor %]
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
# Generated by infrastructure/dns/generate.py. Do not edit directly.
|
||||||
|
[% for record in records %]
|
||||||
|
resource "cloudflare_dns_record" "[[ record.terraform.resource_name ]]" {
|
||||||
|
zone_id = var.zone_id
|
||||||
|
name = "[[ record.name ]]"
|
||||||
|
type = "[[ record.type ]]"
|
||||||
|
content = "[[ record['values'][0] ]]"
|
||||||
|
proxied = [[ record.proxied | lower ]]
|
||||||
|
ttl = [[ record.ttl | default(1) ]]
|
||||||
|
}
|
||||||
|
[% endfor %]
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
[% for record in records %]
|
||||||
|
[[ record.name | replace('.', '-') ]].server: |
|
||||||
|
[[ record.name ]]:53 {
|
||||||
|
errors
|
||||||
|
template IN A {
|
||||||
|
answer "{{ .Name }} 60 IN A [[ record['values'][0] ]]"
|
||||||
|
}
|
||||||
|
template IN AAAA {
|
||||||
|
rcode NOERROR
|
||||||
|
}
|
||||||
|
}
|
||||||
|
[% endfor %]
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
# Docker 地址池与 DN42
|
||||||
|
|
||||||
|
DN42 使用 `172.20.0.0/14`。laptop 的 Docker 默认地址池改为 `172.28.0.0/16`,
|
||||||
|
按 `/24` 分配新 bridge,避免本地直连路由与 DN42 前缀重叠。
|
||||||
|
`ansible/site.yml` 合并现有 daemon.json,保留 NVIDIA runtime;先热加载 live-restore,
|
||||||
|
再重启 daemon 使默认地址池生效,避免已有容器随 daemon 停止。
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ANSIBLE_LOCAL_TEMP=/tmp/ansible-docker ansible-playbook -i localhost, infrastructure/docker/ansible/site.yml --check --diff
|
||||||
|
ANSIBLE_LOCAL_TEMP=/tmp/ansible-docker ansible-playbook -i localhost, infrastructure/docker/ansible/site.yml
|
||||||
|
```
|
||||||
|
|
||||||
|
已有网络不会自动换地址。本次单独迁移结果:
|
||||||
|
|
||||||
|
| 网络 | 原地址 | 当前地址/状态 |
|
||||||
|
|---|---|---|
|
||||||
|
| blocky_default | 172.20.0.0/16 | 172.28.0.0/24,Compose 明确声明 |
|
||||||
|
| ps3netsrv_default | 172.21.0.0/16 | 172.28.1.0/24,Compose 明确声明 |
|
||||||
|
| research-auto_default | 172.22.0.0/16 | 172.28.2.0/24,仓库外 research-auto Compose 明确声明 |
|
||||||
|
| netboot_default | 172.23.0.0/16 | 删除无端点的遗留网络;netboot 两个容器均使用 host 网络 |
|
||||||
|
|
||||||
|
Blocky 健康检查与 DNS 查询通过;ps3netsrv 运行,游戏数据挂载保留。
|
||||||
|
research-auto 的 postgres 容器仅 create、未启动,原命名卷 `research-auto_postgres_data` 保留。
|
||||||
|
|
||||||
|
旧运行容器曾引用仓库重组前的 `/home/panxiao81/services/<app>` 挂载路径;
|
||||||
|
本次 Blocky 已用 `apps/blocky` 路径重建,netboot 等未重建的容器仍需在重建时使用当前 Compose。
|
||||||
|
不要在未检查 bind mount 路径的情况下关闭 live-restore 并重启所有容器。
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
[defaults]
|
||||||
|
local_tmp = /tmp/ansible-docker
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
---
|
||||||
|
- name: 为 DN42 排除 Docker 地址池重叠
|
||||||
|
hosts: localhost
|
||||||
|
connection: local
|
||||||
|
become: true
|
||||||
|
gather_facts: false
|
||||||
|
vars:
|
||||||
|
ansible_python_interpreter: /usr/bin/python3
|
||||||
|
docker_address_pools:
|
||||||
|
- base: 172.28.0.0/16
|
||||||
|
size: 24
|
||||||
|
tasks:
|
||||||
|
- name: 读取现有 Docker 配置并保留 runtimes 等设置
|
||||||
|
ansible.builtin.slurp:
|
||||||
|
src: /etc/docker/daemon.json
|
||||||
|
register: docker_config
|
||||||
|
no_log: true
|
||||||
|
|
||||||
|
# 先让旧 daemon 知道 live-restore,随后重启才能保留运行容器。
|
||||||
|
- name: 启用 live-restore
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: "{{ (docker_config.content | b64decode | from_json | combine({'live-restore': true})) | to_nice_json }}\n"
|
||||||
|
dest: /etc/docker/daemon.json
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: '0644'
|
||||||
|
backup: true
|
||||||
|
validate: /usr/bin/dockerd --validate --config-file %s
|
||||||
|
register: live_restore_config
|
||||||
|
|
||||||
|
- name: 热重载 live-restore
|
||||||
|
ansible.builtin.systemd_service:
|
||||||
|
name: docker
|
||||||
|
state: reloaded
|
||||||
|
when: live_restore_config.changed and not ansible_check_mode
|
||||||
|
|
||||||
|
- name: 确认运行中的 daemon 已启用 live-restore
|
||||||
|
ansible.builtin.command: docker info --format '{{ '{{' }}.LiveRestoreEnabled{{ '}}' }}'
|
||||||
|
register: live_restore_status
|
||||||
|
changed_when: false
|
||||||
|
retries: 5
|
||||||
|
delay: 2
|
||||||
|
until: live_restore_status.stdout == 'true'
|
||||||
|
when: not ansible_check_mode
|
||||||
|
|
||||||
|
- name: 配置 DN42 范围之外的默认地址池
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: "{{ (docker_config.content | b64decode | from_json | combine({'live-restore': true, 'default-address-pools': docker_address_pools})) | to_nice_json }}\n"
|
||||||
|
dest: /etc/docker/daemon.json
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: '0644'
|
||||||
|
backup: true
|
||||||
|
validate: /usr/bin/dockerd --validate --config-file %s
|
||||||
|
register: docker_pool_config
|
||||||
|
|
||||||
|
- name: 保留运行容器并重启 daemon 使地址池生效
|
||||||
|
ansible.builtin.systemd_service:
|
||||||
|
name: docker
|
||||||
|
state: restarted
|
||||||
|
when: docker_pool_config.changed and not ansible_check_mode
|
||||||
+110
-10
@@ -1,13 +1,113 @@
|
|||||||
# OCI infrastructure recovery
|
# OCI 云上基础设施
|
||||||
|
|
||||||
The original Terraform source is currently unavailable. The likely authoritative
|
`terraform/` 是独立 Terraform 根模块,从 OCI Object Storage 中的现有 state 恢复。
|
||||||
state remains in OCI Object Storage. Reconstruct configuration here only after
|
Terraform 管理云 API 资源;实例内的软件、Kubernetes 和操作系统配置不在该 state 中。
|
||||||
taking an encrypted independent state backup.
|
|
||||||
|
|
||||||
Safety requirements:
|
## 资源与来源
|
||||||
|
|
||||||
- preserve the existing state lineage and serial;
|
- Region:`ap-osaka-1`,compartment 为 tenancy 根。
|
||||||
- reproduce the current VM and public-network design first;
|
- State:namespace `axckv9ylwqxr`,bucket `oci-k8s-free-tier-tfstate`,key `terraform.tfstate`。
|
||||||
- reach a zero-change plan before any apply;
|
- 恢复源:2026-08-15 12:26:24 UTC 对象,25,261 字节,serial `249`,
|
||||||
- protect the instance and boot volume from destruction;
|
lineage `c945c6c4-ee01-d7b7-23f5-37207ea60609`,Terraform `1.15.8`。
|
||||||
- treat migration to a private subnet as a separate reviewed change.
|
- 7 个受管资源保留原地址:`oci_core_instance.vm`、`oci_core_vcn.vcn`、
|
||||||
|
`oci_core_subnet.public`、`oci_core_internet_gateway.igw`、`oci_core_route_table.public`、
|
||||||
|
`oci_core_security_list.public`、`oci_limits_quota.free_tier_quota`。
|
||||||
|
- 2 个数据源:`oci_core_images.ubuntu`、`oci_identity_availability_domains.ads`;保留 4 个原输出。
|
||||||
|
- VM:`homelab-vm`,A1 Flex,2 OCPU / 12 GB RAM / 100 GB 启动盘,
|
||||||
|
私网 `10.0.0.124`,恢复时公网 `129.225.138.179`。
|
||||||
|
- VCN `10.0.0.0/16`,公共子网 `10.0.0.0/24`,默认路由经 Internet Gateway;
|
||||||
|
入站保留 TCP 22、UDP 41641、ICMP type 3/code 4,出站全部允许。
|
||||||
|
- 配额语句保留 A1 4 核 / 24 GB、10 个卷、200 GB 总存储限制;这些语句不是费用保证。
|
||||||
|
- Bucket 自身不在 state 内,不由此根模块管理。
|
||||||
|
|
||||||
|
## 恢复设计
|
||||||
|
|
||||||
|
原变量、模块意图、provider 精确版本和生命周期规则无法从 state 完整恢复。
|
||||||
|
本次选择并锁定 `oracle/oci 9.1.0`,提交 lockfile;这不是声称找回了原 provider 版本。
|
||||||
|
资源间的 VCN、路由表、安全列表、子网和 DHCP 引用已重建。
|
||||||
|
启动镜像固定为现有 image OCID,避免数据源选中更新镜像导致 VM 替换。
|
||||||
|
新增 `prevent_destroy` 保护现有 VM;没有用 `ignore_changes` 掩盖配置差异。
|
||||||
|
|
||||||
|
Provider 使用本机 `~/.oci/config` 的 `DEFAULT` profile,可用变量覆盖 profile 和 region。
|
||||||
|
metadata 经敏感变量传入,只保存在忽略的 `terraform.tfvars.json`,不进入版本库。
|
||||||
|
State、plan、metadata 变量与 `.terraform/` 都不得提交;plan JSON 同样可能含敏感数据。
|
||||||
|
保留原目录的恢复要求:保留 lineage/serial,先复现当前 VM 和公共网络设计,
|
||||||
|
实际基础设施变更前达到严格零变更,并保护实例及启动盘;迁移私有子网须单独评审。
|
||||||
|
已用既有 GPG 加密子密钥 `5A6A04D1B216C64E` 创建独立加密源备份
|
||||||
|
`.recovery/source.tfstate.gpg`;本地原始副本的 lineage/serial 保持不变。
|
||||||
|
|
||||||
|
## 日常维护
|
||||||
|
|
||||||
|
旧 Terraform/CI 已由维护者确认停用,当前仓库已接管原 OCI Object Storage state。
|
||||||
|
`versions.tf` 使用 OCI backend,直接连接原对象,未迁移本地恢复副本。
|
||||||
|
认证沿用本机 OCI CLI 的 `DEFAULT` profile;不要在配置中写密钥。
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd infrastructure/oci/terraform
|
||||||
|
terraform init
|
||||||
|
terraform validate
|
||||||
|
terraform plan -input=false -out=change.tfplan
|
||||||
|
# 核对计划后执行:
|
||||||
|
terraform apply change.tfplan
|
||||||
|
```
|
||||||
|
|
||||||
|
新 checkout 需从受限 state 副本提取 metadata 至被忽略的 `terraform.tfvars.json`。
|
||||||
|
`prepare-local-state.py` 可执行这一步,同时保留 `.recovery/terraform.tfstate` 备份;
|
||||||
|
当前 backend 使用远端对象,`.recovery/` 中的副本不再参与日常 plan/apply。
|
||||||
|
下载前设置 `umask 077`,并使用 GPG 加密源备份;不要把本地副本上传覆盖远端。
|
||||||
|
|
||||||
|
## 恢复验证记录
|
||||||
|
|
||||||
|
Terraform 1.15.8、OCI provider 9.1.0 验证通过。
|
||||||
|
恢复计划唯一更新是 VM metadata 的敏感标记,plan JSON 中 before/after 值相同。
|
||||||
|
经授权 apply 后,本地完整刷新 plan 达到 `No changes`。
|
||||||
|
旧 state 在 provider 刷新后补充 VM shape/VNIC、subnet IPv4 CIDR、route type 字段,
|
||||||
|
这些读回差异没有产生基础设施修改计划。
|
||||||
|
接管时重新连接原远端对象,因此新增实例计划也包含原 VM 的同一敏感标记归一化。
|
||||||
|
|
||||||
|
## AMD 实例
|
||||||
|
|
||||||
|
`oci_core_instance.amd` 配置为 `homelab-amd`,`VM.Standard.E2.1.Micro`,1 GB RAM,
|
||||||
|
Ubuntu 24.04 x86_64,50 GB / 10 VPU 启动盘。复用现有公共子网和 SSH 公钥,
|
||||||
|
没有复制 A1 实例的其他初始化内容,也不部署 Kubernetes。
|
||||||
|
镜像固定为 `Canonical-Ubuntu-24.04-2026.08.25-0`,实例具有 `prevent_destroy` 保护。
|
||||||
|
|
||||||
|
创建前 API 确认大阪为 home region、机型计费类型为 `ALWAYS_FREE`,AMD 配额剩余 2 台。
|
||||||
|
存储盘点只有 A1 的 100 GB 启动盘;已核对两个启动盘合计 150 GB,200 GB 免费额度内剩余 50 GB。
|
||||||
|
免费额度跨启动盘和块存储共享;后续新增资源仍需重新核对实际占用。
|
||||||
|
|
||||||
|
实例已创建并确认 `RUNNING`,私网 `10.0.0.158`,公网 `129.225.176.134`。
|
||||||
|
登录命令:`ssh [email protected]`(本次未验证 SSH 登录)。
|
||||||
|
远端 state 已保存;创建后完整刷新 plan 为 `No changes`,退出码 0。
|
||||||
|
|
||||||
|
参考:[Always Free 资源](https://docs.oracle.com/en-us/iaas/Content/FreeTier/freetier_topic-Always_Free_Resources.htm)、
|
||||||
|
[OCI provider 认证](https://docs.oracle.com/en-us/iaas/Content/dev/terraform/configuring.htm)、
|
||||||
|
[OCI backend 配置](https://developer.hashicorp.com/terraform/language/backend/oci)。
|
||||||
|
|
||||||
|
## DNS 登录入口
|
||||||
|
|
||||||
|
- ARM:`ssh [email protected]`
|
||||||
|
- AMD:`ssh [email protected]`
|
||||||
|
|
||||||
|
公网 A 记录声明位于 `../dns/records.yml`,在 Cloudflare 上关闭代理,TTL 300 秒。
|
||||||
|
当前通过 DNS API 管理,未加入 OCI Terraform state;公网 IP 变化时需同步记录。
|
||||||
|
|
||||||
|
## WireGuard/BGP 与 DN42
|
||||||
|
|
||||||
|
家中端点已从 laptop 迁移到 VyOS `192.168.10.2`,AMD 与 VyOS 同属 AS4242421811,
|
||||||
|
通过独立 WireGuard 接口建立双栈 iBGP;laptop 保留原有 NEC BGP 和 OSPF,按路由经 VyOS 转发。
|
||||||
|
Ansible 配置与运行方法见 [ansible/README.md](ansible/README.md)。
|
||||||
|
|
||||||
|
注册前缀 `172.21.111.160/27`、`fdd0:98df:15b0::/48` 已在内部路由中准备:
|
||||||
|
VyOS `.161` / `::1`,AMD `.162` / `::2`,使用 loopback /32、/128。
|
||||||
|
首个外部 DN42 peer 已接入 RoutedBits Osaka(AS4242420207),AMD 使用独立 `wg-dn42-1`
|
||||||
|
和单 IPv6 MP-BGP 会话承载双栈;详见 [Ansible runbook](ansible/README.md)。
|
||||||
|
外部明细留在 AMD,`172.20.0.0/14`、`fd00::/8` 汇总经 iBGP 下发 VyOS;IPv4 /14 再经 OSPF 下发 LAN。
|
||||||
|
LAN 的 DN42 IPv6 /64 地址由 VyOS SLAAC 下发,fd00::/8 通过 RA RIO 分发,不通告 IPv6 默认路由。内部家中/OCI 业务路由不得向外部 DN42 邻居通告。
|
||||||
|
|
||||||
|
Terraform 管理 AMD NSG、VNIC 转发与 VCN 回程;Ansible 管理路由器/主机及 ARM 的 Tailscale
|
||||||
|
回程例外。Docker 与 DN42 的 /14 地址重叠已迁出,见 [Docker runbook](../docker/README.md)。
|
||||||
|
|
||||||
|
VyOS 的 `wg42` 主 IPv4 已改为注册的 `172.21.111.161/32`,内部 BGP 改为
|
||||||
|
单 link-local IPv6 会话承载双栈。三个 LAN 私网到 DN42 /14 由 VyOS 定向 masquerade,
|
||||||
|
排除本 AS /27;OCI 业务保持原源地址。IPv6 不做 NAT,由 `ansible/dn42-ra.yml` 管理三个 LAN 的 SLAAC 与专用路由通告。
|
||||||
|
|||||||
@@ -0,0 +1,232 @@
|
|||||||
|
# VyOS ↔ OCI:WireGuard 与 DN42 内部 BGP
|
||||||
|
|
||||||
|
站点端已从 laptop 迁移到 VyOS `192.168.10.2`。VyOS 位于双层 NAT 后,主动连接
|
||||||
|
`oci-amd.ddupan.top:51820`,keepalive 25 秒;LAN 访问 DN42 时在 VyOS 做定向 masquerade。
|
||||||
|
主机配置由 Ansible 管理;OCI NSG、VNIC 转发和 VCN 回程路由由旁边的 Terraform 管理。
|
||||||
|
|
||||||
|
| 节点 | 接口 | 传输 IPv4 | 传输 IPv6 | ASN |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| VyOS | wg42 | 172.21.111.161/32 | fe80::1811:1/64(BGP) | 4242421811 |
|
||||||
|
| AMD | wg-oci | 10.255.254.1/30(历史传输地址) | fe80::1811:2/64(BGP) | 4242421811 |
|
||||||
|
|
||||||
|
AMD 显式设置 `fe80::1811:2/64`;VyOS BGP 使用固定 `fe80::1811:1/64`,双方 AllowedIPs 包含
|
||||||
|
`fe80::/64`。FRR 在 ULA 建邻时仍需接口具有 link-local 下一跳地址。
|
||||||
|
|
||||||
|
MTU 1380,Linux 使用 `Table = off`,WireGuard AllowedIPs 用于选 peer 与源地址校验,
|
||||||
|
站点路由由 iBGP 安装。内外部均使用单条 IPv6 link-local 会话承载双 AFI,并启用 extended-nexthop。
|
||||||
|
内部旧 IPv4/ULA BGP 邻居已退役;ULA 传输地址保留用于路由下一跳,不再用于建邻。
|
||||||
|
VyOS 10.2.4 通过 `OCI-MP-IN` 的 `ipv6-next-hop prefer-global` 优先采用通告中的 ULA 下一跳,
|
||||||
|
避免 link-local NHT 显示 overlay index unresolved、BGP 已建邻但路由未安装。
|
||||||
|
|
||||||
|
## 地址与通告边界
|
||||||
|
|
||||||
|
已注册 `172.21.111.160/27`、`fdd0:98df:15b0::/48`:
|
||||||
|
|
||||||
|
| 节点 | 路由器 IPv4 | 路由器 IPv6 |
|
||||||
|
|---|---|---|
|
||||||
|
| VyOS(IPv4 在 wg42,IPv6 在 lo) | 172.21.111.161/32 | fdd0:98df:15b0::1/128 |
|
||||||
|
| AMD loopback | 172.21.111.162/32 | fdd0:98df:15b0::2/128 |
|
||||||
|
|
||||||
|
VyOS 为注册的 /27、/48 建立 distance 254 的 blackhole 聚合路由,保证精确前缀存在,
|
||||||
|
避免未分配地址落入默认路由。已分配的本地地址及 AMD 的 /32、/128 优先于聚合。
|
||||||
|
|
||||||
|
内部通告严格过滤:
|
||||||
|
|
||||||
|
- VyOS → AMD:`192.168.10.0/24`、`10.60.0.0/24`、`10.61.0.0/24` 和注册 /27、/48。
|
||||||
|
- AMD → VyOS:`10.0.0.0/24`、AMD 的注册 /32、/128,以及 DN42 汇总 `172.20.0.0/14`、`fd00::/8`。
|
||||||
|
- VyOS 向 LAN OSPF 只重分发 OCI /24、DN42 /14 和注册 IPv4 /27,使用精确 route-map、E1 metric。
|
||||||
|
既有直连 LAN/SDN 的 OSPF area 声明保持不变,不使用泛化的 redistribute connected。
|
||||||
|
|
||||||
|
**首个外部 peer 为 RoutedBits Osaka(AS4242420207),由 `dn42.yml` 单独管理。** 外部邻居使用独立的 import/export
|
||||||
|
过滤,只对外通告注册 /27、/48;禁止把上述内部业务前缀的过滤器复用到外部邻居。
|
||||||
|
入口使用 DN42 指南的保留地址、互联网络与前缀长度规则(IPv6 /44–/64),
|
||||||
|
并优先拒收本 AS、LAN、OCI 前缀;尚未配置注册表 ROA 校验。外部明细留在 AMD,汇总通过 iBGP 下发 VyOS;IPv4 汇总再经 OSPF 下发 LAN。
|
||||||
|
|
||||||
|
每个外部 WireGuard peer 使用独立接口,BGP 可复用本机 loopback 地址。
|
||||||
|
单接口多个 WireGuard peer 要求可明确区分的 AllowedIPs;多家 peer 都提供同一 DN42 路由范围时,
|
||||||
|
使用独立接口让 BGP 决定出口,避免相同 AllowedIPs 抢占 peer。
|
||||||
|
|
||||||
|
## 执行
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd infrastructure/oci/ansible
|
||||||
|
export SSH_AUTH_SOCK="$(gpgconf --list-dirs agent-ssh-socket)"
|
||||||
|
ansible-playbook site.yml --check --diff
|
||||||
|
ansible-playbook site.yml
|
||||||
|
ansible-playbook site.yml # 复跑应 changed=0
|
||||||
|
```
|
||||||
|
|
||||||
|
`vyos.yml` 先在路由器本机生成密钥并保存,再交换公钥并增量应用 VyOS set 命令。
|
||||||
|
VyOS native config 含私钥,因此相关模块使用 `no_log`,不把配置备份到 Git 或打印出来。
|
||||||
|
AMD 私钥在 `/etc/wireguard/wg-oci.key`(0600),由本机 PostUp 加载,不返回控制机。
|
||||||
|
首次 check mode 无法生成私钥,因此会跳过依赖不存在公钥的 Linux 配置渲染。
|
||||||
|
|
||||||
|
`retire-laptop.yml` 是迁移收尾:先检查 VyOS 邻居,再停止 laptop 的 wg-oci,
|
||||||
|
删除试验邻居、三个新增 network 语句及专用防火墙链,保留 laptop 原 AS65001 ↔ NEC AS65000
|
||||||
|
会话、原有 VPN 路由和 OSPF。旧私钥保留在 laptop 受限文件中,隧道和防火墙单元已禁用。
|
||||||
|
|
||||||
|
Linux 端 FRR 通过 `vtysh -f` 应用独立配置片段,另行 `write memory` 持久化;
|
||||||
|
文件变更时重建受管邻居,未变更的重跑不重置会话。移除前缀时还须显式删除已退出管理的
|
||||||
|
`network` 语句,不能只追加 set 命令,也不能清空整份 BGP 配置。
|
||||||
|
|
||||||
|
## OCI 回程与 Tailscale
|
||||||
|
|
||||||
|
VCN 虚拟路由器不参与主机间 iBGP;三个家中业务前缀的静态回程指向 AMD Private IP OCID。
|
||||||
|
首次启用时必须先将 VNIC `skip_source_dest_check` 设为 true,OCI 才接受该私有 IP 为路由目标。
|
||||||
|
当前 Terraform 的 route table → subnet → instance 依赖使首次引导需要先设置该标志。
|
||||||
|
重建 AMD 后须重新查询并更新 `amd_router_private_ip_ocid`。
|
||||||
|
|
||||||
|
ARM 接受 Tailscale 家中子网路由,table 52 原本抢走回程。
|
||||||
|
Ansible 在 ARM 设置 priority 5101–5103、仅匹配这三个目的前缀的 `lookup main` 规则,
|
||||||
|
让这些流量使用 OCI 网关→AMD;其他 Tailscale 地址保持原路径。
|
||||||
|
AMD 的 Zebra route-map 仅为本机发起的 BGP 业务流量选择 `10.0.0.158` 源地址,不改写转发源地址。
|
||||||
|
|
||||||
|
## Docker 地址冲突
|
||||||
|
|
||||||
|
DN42 使用 `172.20.0.0/14`。原 laptop Docker 的四个 /16 已清除,默认池改为 `172.28.0.0/16`,
|
||||||
|
详见 [Docker runbook](../../docker/README.md)。最长前缀匹配决定路由,但不能解决两套网络实际
|
||||||
|
地址重叠;不能只检查本 AS 注册的 /27 而忽略其他 DN42 注册前缀。
|
||||||
|
|
||||||
|
## 验证与停用
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# VyOS operational mode
|
||||||
|
show interfaces wireguard wg42 summary
|
||||||
|
show bgp summary
|
||||||
|
show ip route 172.21.111.162/32
|
||||||
|
show ipv6 route fdd0:98df:15b0::2/128
|
||||||
|
# AMD
|
||||||
|
sudo wg show wg-oci latest-handshakes
|
||||||
|
sudo vtysh -c 'show bgp summary'
|
||||||
|
ping 192.168.10.4
|
||||||
|
ping -I 172.21.111.162 172.21.111.161
|
||||||
|
ping -6 -I fdd0:98df:15b0::2 fdd0:98df:15b0::1
|
||||||
|
```
|
||||||
|
|
||||||
|
业务网段测试应使用业务源地址;DN42 loopback 测试使用注册地址。
|
||||||
|
LAN 已通过 `dn42-ra.yml` 部署 DN42 SLAAC 与专用 RIO,不发布 IPv6 默认路由;OCI VCN 保持原配置。
|
||||||
|
停用或回滚须同时处理 BGP、WireGuard、OSPF 重分发、VCN 回程和 ARM 例外规则,
|
||||||
|
不能仅停止隧道后留下静态回程指向不可达节点。
|
||||||
|
|
||||||
|
参考:[WireGuard](https://www.wireguard.com/quickstart/)、
|
||||||
|
[VyOS WireGuard](https://docs.vyos.io/en/1.5/configuration/interfaces/wireguard.html)、
|
||||||
|
[DN42 入门](https://www.dn42.dev/howto/Getting-Started)。
|
||||||
|
|
||||||
|
## 迁移验收
|
||||||
|
|
||||||
|
2026-09-14:
|
||||||
|
|
||||||
|
- VyOS IPv4 iBGP 收到 OCI /24、AMD /32,通告三个家中前缀与注册 /27;IPv6 会话双方各收到一条前缀。
|
||||||
|
- 两端 DN42 loopback IPv4、IPv6 互通;IPv6 本次采样约 3.1 ms。
|
||||||
|
- laptop 的 `10.0.0.0/24` 与 `172.21.111.160/27` 经 OSPF 指向 `192.168.10.2`,不再使用旧隧道。
|
||||||
|
- PVE1 可访问 ARM 私网和 AMD 的 DN42 IPv4;AMD 可访问 PVE1、`10.60.0.1`、`10.61.0.1`。
|
||||||
|
- laptop 旧 WireGuard 和专用防火墙 service 已停止/禁用,原 NEC BGP 会话保留。
|
||||||
|
|
||||||
|
LAN 网关可能返回 ICMP Redirect,提示客户端将 VyOS 作为同网段下一跳;这是既有 LAN
|
||||||
|
拓扑的正常结果,没有为此修改客户端或网关的 redirect 策略。
|
||||||
|
|
||||||
|
## AMD 外部 DN42 首个接口
|
||||||
|
|
||||||
|
2026-09-15 已通过 `ansible-playbook dn42.yml` 准备独立监听:
|
||||||
|
|
||||||
|
- 接口:`wg-dn42-1`,UDP endpoint:`oci-amd.ddupan.top:51821`(`129.225.176.134:51821`)。
|
||||||
|
- 公钥:`YQ/X3QmNocnr0u4aUm5qhcV328StSNtg+ULd9AKCdhQ=`。
|
||||||
|
- 私钥仅保存在 AMD `/etc/wireguard/wg-dn42-1.key`,由 root 受限目录保护,不返回控制机。
|
||||||
|
- `wg-quick@wg-dn42-1` 开机启动;主机入站规则随接口启停,OCI NSG 规则由 Terraform 管理。
|
||||||
|
- Link-local:`fe80::1811:2/64`,本机作用域为 `%wg-dn42-1`。首个 peer 使用单条 IPv6 BGP 会话承载双栈(MP-BGP + RFC 8950 extended next hop)。
|
||||||
|
- 对端:AS4242420207,`router.osa1.routedbits.com:51811`,link-local `fe80::207`。
|
||||||
|
- 对端公钥:`96PwUEGi/ijdmKO+IjuZ+J6DeykuTRukZD5atajfeH4=`。
|
||||||
|
- WireGuard 使用 `Table = off`;AllowedIPs 为 `fe80::/64, 172.20.0.0/14, 10.0.0.0/8, 172.31.0.0/16, fd00::/8`,keepalive 25 秒。
|
||||||
|
- 双 AFI 共用 IPv6 TCP 会话并启用 extended-nexthop;每 AFI maximum-prefix 10000,出口只允许注册 /27 和 /48。
|
||||||
|
- 本机发起 DN42 流量使用注册地址作为 preferred source,不做 NAT。
|
||||||
|
- Ubuntu 自带 FRR 8.4.4 不满足 [DN42 FRR 指南](https://dn42.dev/howto/frr) 的 link-local 版本要求。
|
||||||
|
`tasks/frr-dn42.yml` 使用官方 frr-10.7 软件源,固定 10.7.1;升级前配置仅在 AMD `/var/backups/frr-before-dn42` 备份。
|
||||||
|
|
||||||
|
原 `wg-oci:51820` 继续承载与 VyOS 的内部互联。
|
||||||
|
|
||||||
|
2026-09-15 接入验收:WireGuard 握手与 `fe80::207%wg-dn42-1` 连通;FRR 10.7.1
|
||||||
|
会话 Established,双方已协商 IPv4/IPv6 AFI 与 extended nexthop。采样接收 IPv4 1174、IPv6 1273 条,
|
||||||
|
对外仅通告 `172.21.111.160/27`、`fdd0:98df:15b0::/48`。AMD 无需显式指定源地址,
|
||||||
|
即可访问对端 `172.20.19.78`、`fdb1:e72a:343d::f`,各 3/3 回复,约 1 ms;
|
||||||
|
IPv4 内核路由下一跳为 `via inet6 fe80::207 dev wg-dn42-1`。内部 VyOS 双栈 BGP 会话已恢复。
|
||||||
|
|
||||||
|
## 向 LAN 分发 DN42 汇总
|
||||||
|
|
||||||
|
`site.yml` 在 AMD 生成 `172.20.0.0/14`、`fd00::/8` 的 BGP aggregate,并通过内部精确
|
||||||
|
prefix-list 通告 VyOS。不使用全局 `summary-only`,避免抑制对外通告的注册 /27、/48;
|
||||||
|
对外出口过滤仍只允许这两个注册前缀。AMD 保留外部明细与汇总丢弃路径,无匹配明细的流量
|
||||||
|
在 AMD 丢弃。汇总只要仍有覆盖的 BGP 明细(包括本 AS 注册前缀)就可能存在,不能作为外部
|
||||||
|
peer 在线状态指示。自己的 /27、/48 更具体,继续指向本地站点。
|
||||||
|
|
||||||
|
VyOS 将 IPv4 /14 通过既有 OSPF E1 分发到 LAN;IPv6 /8 通过 RA 的 RIO 分发至接受该选项的客户端。
|
||||||
|
LAN 的 DN42 IPv6 地址通过 SLAAC 自动分配。
|
||||||
|
当前 LAN IPv4 客户端访问 DN42 /14 由 VyOS 定向 masquerade 到 `172.21.111.161`;
|
||||||
|
IPv6 使用 SLAAC 分配的注册地址直接路由。
|
||||||
|
AMD 只放行注册前缀在内部 `wg-oci` 与外部 `wg-dn42-1` 之间转发。
|
||||||
|
|
||||||
|
2026-09-15 汇总验收:VyOS 双栈 BGP 分别收到 /14、/8,下一跳为 AMD;
|
||||||
|
laptop 的 `172.20.0.0/14` 为 OSPF 路由,经 `192.168.10.2 dev br0`。
|
||||||
|
使用 VyOS 注册的 /32、/128 作为源,经 AMD 访问 RoutedBits 的双栈地址各 3/3 回复,
|
||||||
|
约 4–5 ms。`site.yml` 执行成功,ARM 无变更;LAN 地址配置保持原状。
|
||||||
|
|
||||||
|
## DN42 DNS 转发
|
||||||
|
|
||||||
|
`ansible-playbook dn42.yml dn42-dns.yml` 管理入口路由和 VyOS DNS。LAN 的 Blocky
|
||||||
|
(`192.168.10.127`) 将 `.dn42`、172.20–23 的 IPv4 反向区和 `d.f.ip6.arpa`
|
||||||
|
转给 VyOS `192.168.10.2:53`。VyOS 仅接受三个内部 LAN 网段,使用注册地址
|
||||||
|
`172.21.111.161` / `fdd0:98df:15b0::1` 发起递归转发,不做 NAT。
|
||||||
|
|
||||||
|
| 上游 | IPv4 | IPv6 |
|
||||||
|
|---|---|---|
|
||||||
|
| a0.recursive-servers.dn42 | 172.20.0.53 | fd42:d42:d42:54::1 |
|
||||||
|
| a3.recursive-servers.dn42 | 172.23.0.53 | fd42:d42:d42:53::1 |
|
||||||
|
|
||||||
|
两个上游的双栈地址均配置,递归请求设置 RD;转发域配置 NTA,避免使用公网根信任链
|
||||||
|
验证 DN42 私有命名空间。本地转发器不声明已完成 DN42 DNSSEC 信任锚验证。
|
||||||
|
IPv4 anycast /32 需要允许四个 `172.2x.0.0/24` 中的 /28–/32,不能只保留 /14 的 /21–/29。
|
||||||
|
新增互联网络明细仅由 AMD 接收;LAN 汇总仍是既定 /14、通过 RIO 通告的 IPv6 /8。
|
||||||
|
NEC 备用 DNS 与 k3s CoreDNS 本次未修改。
|
||||||
|
|
||||||
|
DNS 验收(2026-09-15):从 VyOS 用注册地址直查 a0/a3 的四个双栈地址均获得回复。
|
||||||
|
LAN 查询 Blocky 可得到 a0 的 A、a3 的 AAAA;AD 与公网域名正常,Blocky healthy。
|
||||||
|
VyOS DNS playbook 复跑 changed=0;AMD 已应用完整入口规则。
|
||||||
|
|
||||||
|
## VyOS LAN 到 DN42 masquerade
|
||||||
|
|
||||||
|
规则 18100 排除本 AS `172.21.111.160/27`;18110、18120、18130 分别匹配三个 LAN
|
||||||
|
源网段,目的仅 `172.20.0.0/14` 且出口 `wg42`,translation 为 `masquerade`。
|
||||||
|
为使 masquerade 选中注册地址,`172.21.111.161/32` 从 lo 移到 wg42,并移除
|
||||||
|
`10.255.254.2/30`;不能在 wg42 仍以传输私网地址为主 IPv4 时直接启用 masquerade。
|
||||||
|
现有 OCI 业务互联继续保留原源地址,DNS 转发使用的 `172.21.111.161` 保持可用。
|
||||||
|
AMD 仍仅允许 DN42 注册源前缀进入外部隧道,不在 AMD 做第二次 NAT。
|
||||||
|
|
||||||
|
2026-09-15 masquerade 验收:laptop `192.168.10.127` 到 RoutedBits `172.20.19.78`
|
||||||
|
3/3 回复约 4.7 ms,VyOS NAT 表显示转换为 `172.21.111.161`;到 ARM `10.0.0.124`
|
||||||
|
3/3 回复约 4.2 ms,NAT 表确认保留 `192.168.10.127`。
|
||||||
|
内部 link-local 单会话双 AFI 已建立;外部 peer 保持独立接口和精确出口。
|
||||||
|
|
||||||
|
## LAN DN42 IPv6 RA(不发布默认路由)
|
||||||
|
|
||||||
|
执行 `ansible-playbook dn42-ra.yml`,为三个 LAN 启用 SLAAC:
|
||||||
|
|
||||||
|
| LAN | VyOS 接口 | 前缀 | 路由器地址 |
|
||||||
|
|---|---|---|---|
|
||||||
|
| 192.168.10.0/24 | eth0 | fdd0:98df:15b0:10::/64 | fdd0:98df:15b0:10::1 |
|
||||||
|
| 10.60.0.0/24 | eth1 | fdd0:98df:15b0:60::/64 | fdd0:98df:15b0:60::1 |
|
||||||
|
| 10.61.0.0/24 | eth2 | fdd0:98df:15b0:61::/64 | fdd0:98df:15b0:61::1 |
|
||||||
|
|
||||||
|
Router Lifetime 为 **0**,不发布 `::/0`;PIO 开启 on-link 与 autonomous 标志,
|
||||||
|
preferred lifetime 14400 秒、valid lifetime 86400 秒。RIO 只包含 `fd00::/8`,
|
||||||
|
有效期 180 秒;RA 周期 10–30 秒。不发布 RDNSS、DNSSL、DHCPv6 标志或链路 MTU,
|
||||||
|
保留客户端现有 DNS 和公网出口。IPv6 经 VyOS→AMD→DN42 直接路由,不做 NAT66。
|
||||||
|
现有 /48 对外通告及内部回程已覆盖这三个 /64,无须泄漏每个 LAN 的明细到外部。
|
||||||
|
|
||||||
|
客户端须支持并接受 RIO。当前 laptop 的 br0 `accept_ra=0`、`forwarding=1`,
|
||||||
|
不会自动配置;本次保留其网络设置。Linux 路由主机如需接受 RA,需要在自身网络管理
|
||||||
|
配置中显式启用,并允许至少 /8 的 RIO(`accept_ra_rt_info_max_plen`)。
|
||||||
|
关闭 RA 时注意 PIO 的有效期;不要只删路由器接口地址而留下仍有效的客户端地址。
|
||||||
|
|
||||||
|
2026-09-15 RA 抓包验收:主 LAN 收到不带标签的 `:10::/64`;VLAN 100/110
|
||||||
|
分别携带 `:60::/64`、`:61::/64`。三者 Router Lifetime 均为 0,PIO 为 onlink/auto,
|
||||||
|
RIO 为 fd00::/8、180 秒,没有 DNS 或默认路由通告。`dn42-ra.yml` 复跑 changed=0。
|
||||||
|
同一 LAN 上的临时 Linux 测试客户端自动获得地址与 RIO,无 IPv6 默认路由;测试后自动删除。
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
[defaults]
|
||||||
|
inventory = inventory/hosts.yml
|
||||||
|
roles_path = roles
|
||||||
|
local_tmp = /tmp/ansible-oci
|
||||||
|
host_key_checking = True
|
||||||
|
interpreter_python = auto_silent
|
||||||
|
[ssh_connection]
|
||||||
|
ssh_args = -o ControlMaster=auto -o ControlPersist=60s -o StrictHostKeyChecking=accept-new
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
---
|
||||||
|
- name: 配置 VyOS DN42 DNS 转发
|
||||||
|
hosts: site_routers
|
||||||
|
gather_facts: false
|
||||||
|
vars:
|
||||||
|
dn42_dns_zones:
|
||||||
|
- dn42
|
||||||
|
- 20.172.in-addr.arpa
|
||||||
|
- 21.172.in-addr.arpa
|
||||||
|
- 22.172.in-addr.arpa
|
||||||
|
- 23.172.in-addr.arpa
|
||||||
|
- d.f.ip6.arpa
|
||||||
|
dn42_dns_servers:
|
||||||
|
- 172.20.0.53
|
||||||
|
- 172.23.0.53
|
||||||
|
- fd42:d42:d42:54::1
|
||||||
|
- fd42:d42:d42:53::1
|
||||||
|
tasks:
|
||||||
|
- name: 配置受限监听、注册地址源与条件转发
|
||||||
|
vyos.vyos.vyos_config:
|
||||||
|
lines: "{{ lookup('template', 'templates/vyos-dn42-dns.conf.j2').splitlines() | reject('equalto', '') | list }}"
|
||||||
|
save: true
|
||||||
|
comment: Ansible DN42 DNS forwarding
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
---
|
||||||
|
- name: 在 LAN 通告 DN42 SLAAC 地址与专用路由
|
||||||
|
hosts: site_routers
|
||||||
|
gather_facts: false
|
||||||
|
vars:
|
||||||
|
dn42_ra_lans:
|
||||||
|
- interface: eth0
|
||||||
|
prefix: fdd0:98df:15b0:10::/64
|
||||||
|
address: fdd0:98df:15b0:10::1/64
|
||||||
|
- interface: eth1
|
||||||
|
prefix: fdd0:98df:15b0:60::/64
|
||||||
|
address: fdd0:98df:15b0:60::1/64
|
||||||
|
- interface: eth2
|
||||||
|
prefix: fdd0:98df:15b0:61::/64
|
||||||
|
address: fdd0:98df:15b0:61::1/64
|
||||||
|
tasks:
|
||||||
|
- name: 配置接口地址、SLAAC 与 RIO,不通告默认路由或 DNS
|
||||||
|
vyos.vyos.vyos_config:
|
||||||
|
lines: "{{ lookup('template', 'templates/vyos-dn42-ra.conf.j2').splitlines() | reject('equalto', '') | list }}"
|
||||||
|
save: true
|
||||||
|
comment: Ansible DN42 SLAAC without default route
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
---
|
||||||
|
# 首个外部 peer:RoutedBits Osaka,IPv6 link-local MP-BGP。
|
||||||
|
- name: 准备 AMD DN42 WireGuard 监听
|
||||||
|
hosts: oci_amd
|
||||||
|
become: true
|
||||||
|
vars:
|
||||||
|
dn42_interface: wg-dn42-1
|
||||||
|
dn42_port: 51821
|
||||||
|
dn42_linklocal: fe80::1811:2/64
|
||||||
|
dn42_peer_linklocal: fe80::207
|
||||||
|
dn42_peer_asn: 4242420207
|
||||||
|
dn42_peer_endpoint: router.osa1.routedbits.com:51811
|
||||||
|
dn42_peer_public_key: 96PwUEGi/ijdmKO+IjuZ+J6DeykuTRukZD5atajfeH4=
|
||||||
|
tasks:
|
||||||
|
- name: 准备 FRR link-local 支持
|
||||||
|
ansible.builtin.import_tasks: tasks/frr-dn42.yml
|
||||||
|
- name: 创建受限密钥目录
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: /etc/wireguard
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: '0700'
|
||||||
|
- name: 在 AMD 本机生成独立私钥
|
||||||
|
ansible.builtin.shell: umask 077; wg genkey > /etc/wireguard/{{ dn42_interface }}.key
|
||||||
|
args:
|
||||||
|
creates: /etc/wireguard/{{ dn42_interface }}.key
|
||||||
|
no_log: true
|
||||||
|
- name: 写入监听配置
|
||||||
|
ansible.builtin.copy:
|
||||||
|
dest: /etc/wireguard/{{ dn42_interface }}.conf
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: '0600'
|
||||||
|
content: |
|
||||||
|
# Ansible 管理;首个 peer 使用 IPv6 link-local MP-BGP + extended next hop。
|
||||||
|
[Interface]
|
||||||
|
Address = {{ dn42_linklocal }}
|
||||||
|
ListenPort = {{ dn42_port }}
|
||||||
|
MTU = 1380
|
||||||
|
Table = off
|
||||||
|
PostUp = wg set %i private-key /etc/wireguard/{{ dn42_interface }}.key
|
||||||
|
PostUp = iptables -w -C INPUT -p udp --dport {{ dn42_port }} -j ACCEPT 2>/dev/null || iptables -w -I INPUT 1 -p udp --dport {{ dn42_port }} -j ACCEPT
|
||||||
|
PostDown = iptables -w -D INPUT -p udp --dport {{ dn42_port }} -j ACCEPT
|
||||||
|
|
||||||
|
[Peer]
|
||||||
|
PublicKey = {{ dn42_peer_public_key }}
|
||||||
|
Endpoint = {{ dn42_peer_endpoint }}
|
||||||
|
AllowedIPs = fe80::/64, 172.20.0.0/14, 10.0.0.0/8, 172.31.0.0/16, fd00::/8
|
||||||
|
PersistentKeepalive = 25
|
||||||
|
notify: 重启 DN42 接口
|
||||||
|
- name: 启用 DN42 监听
|
||||||
|
ansible.builtin.systemd_service:
|
||||||
|
name: wg-quick@{{ dn42_interface }}
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
- name: 应用配置
|
||||||
|
ansible.builtin.meta: flush_handlers
|
||||||
|
- name: 写入外部 BGP 配置片段
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: templates/dn42-bgp.conf.j2
|
||||||
|
dest: /etc/frr/dn42-routedbits.vtysh
|
||||||
|
owner: root
|
||||||
|
group: frr
|
||||||
|
mode: '0640'
|
||||||
|
register: dn42_bgp_config
|
||||||
|
changed_when: dn42_bgp_config.changed or ('(deleted)' in frr_running.stdout)
|
||||||
|
notify: 应用 DN42 BGP
|
||||||
|
- name: 应用 BGP 配置
|
||||||
|
ansible.builtin.meta: flush_handlers
|
||||||
|
- name: 读取公开信息
|
||||||
|
ansible.builtin.command: wg show {{ dn42_interface }} {{ item }}
|
||||||
|
loop:
|
||||||
|
- public-key
|
||||||
|
- listen-port
|
||||||
|
changed_when: false
|
||||||
|
register: dn42_public_info
|
||||||
|
- name: 显示公钥和端口
|
||||||
|
ansible.builtin.debug:
|
||||||
|
msg: '{{ dn42_public_info.results | map(attribute="stdout") | list }}'
|
||||||
|
handlers:
|
||||||
|
- name: 重启 FRR
|
||||||
|
ansible.builtin.systemd_service:
|
||||||
|
name: frr
|
||||||
|
state: restarted
|
||||||
|
when: not ansible_check_mode
|
||||||
|
- name: 重启 DN42 接口
|
||||||
|
ansible.builtin.systemd_service:
|
||||||
|
name: wg-quick@{{ dn42_interface }}
|
||||||
|
state: restarted
|
||||||
|
when: not ansible_check_mode
|
||||||
|
- name: 应用 DN42 BGP
|
||||||
|
ansible.builtin.command: vtysh -f /etc/frr/dn42-routedbits.vtysh
|
||||||
|
notify: 保存 FRR 配置
|
||||||
|
when: not ansible_check_mode
|
||||||
|
- name: 保存 FRR 配置
|
||||||
|
ansible.builtin.command: vtysh -c 'write memory'
|
||||||
|
when: not ansible_check_mode
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
---
|
||||||
|
wg_interface: wg-oci
|
||||||
|
wg_port: 51820
|
||||||
|
wg_mtu: 1380
|
||||||
|
wg_home_prefixes: [192.168.10.0/24, 10.60.0.0/24, 10.61.0.0/24]
|
||||||
|
wg_cloud_prefixes: [10.0.0.0/24]
|
||||||
|
dn42_asn: 4242421811
|
||||||
|
dn42_ipv4: 172.21.111.160/27
|
||||||
|
dn42_ipv6: fdd0:98df:15b0::/48
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
---
|
||||||
|
wg_address: 10.255.254.2/30
|
||||||
|
wg_peer_address: 10.255.254.1
|
||||||
|
wg_peer_host: oci_amd
|
||||||
|
wg_endpoint: oci-amd.ddupan.top:51820
|
||||||
|
wg_keepalive: 25
|
||||||
|
wg_lan_interface: br0
|
||||||
|
bgp_asn: 65001
|
||||||
|
bgp_peer_asn: 4242421811
|
||||||
|
bgp_router_id: 192.168.10.127
|
||||||
|
bgp_export: '{{ wg_home_prefixes }}'
|
||||||
|
bgp_import: '{{ wg_cloud_prefixes }}'
|
||||||
@@ -0,0 +1,34 @@
|
|||||||
|
---
|
||||||
|
wg_address: 10.255.254.1/30
|
||||||
|
wg_peer_address: 172.21.111.161
|
||||||
|
wg_peer_host: vyos_rtr
|
||||||
|
wg_endpoint: ''
|
||||||
|
wg_keepalive: 0
|
||||||
|
wg_lan_interface: ens3
|
||||||
|
bgp_asn: 4242421811
|
||||||
|
bgp_peer_asn: 4242421811
|
||||||
|
bgp_router_id: 10.0.0.158
|
||||||
|
bgp_export:
|
||||||
|
- 10.0.0.0/24
|
||||||
|
- 172.21.111.162/32
|
||||||
|
bgp_import:
|
||||||
|
- 192.168.10.0/24
|
||||||
|
- 10.60.0.0/24
|
||||||
|
- 10.61.0.0/24
|
||||||
|
- 172.21.111.160/27
|
||||||
|
wg_ipv6_address: fdd0:98df:15b0:ffff::1/64
|
||||||
|
wg_peer_ipv6: fdd0:98df:15b0:ffff::2
|
||||||
|
bgp_export6:
|
||||||
|
- fdd0:98df:15b0::2/128
|
||||||
|
bgp_import6:
|
||||||
|
- fdd0:98df:15b0::/48
|
||||||
|
wg_linklocal_address: fe80::1811:2/64
|
||||||
|
|
||||||
|
# 仅向内部邻居通告汇总;外部 peer 保持精确出口过滤。
|
||||||
|
bgp_summary: [172.20.0.0/14]
|
||||||
|
bgp_summary6: [fd00::/8]
|
||||||
|
dn42_external_interface: wg-dn42-1
|
||||||
|
|
||||||
|
# 内部与外部统一使用 link-local 单会话双 AFI;节点地址仍在 loopback。
|
||||||
|
bgp_transport_peer: fe80::1811:1
|
||||||
|
bgp_retired_peers: [10.255.254.2, "fdd0:98df:15b0:ffff::2"]
|
||||||
@@ -0,0 +1,26 @@
|
|||||||
|
---
|
||||||
|
all:
|
||||||
|
children:
|
||||||
|
wireguard_sites:
|
||||||
|
hosts:
|
||||||
|
oci_amd:
|
||||||
|
ansible_host: oci-amd.ddupan.top
|
||||||
|
ansible_user: ubuntu
|
||||||
|
oci_routed_hosts:
|
||||||
|
hosts:
|
||||||
|
oci_arm:
|
||||||
|
ansible_host: oci-arm.ddupan.top
|
||||||
|
ansible_user: ubuntu
|
||||||
|
retired_wireguard_sites:
|
||||||
|
hosts:
|
||||||
|
laptop:
|
||||||
|
ansible_connection: local
|
||||||
|
ansible_python_interpreter: /usr/bin/python3
|
||||||
|
site_routers:
|
||||||
|
hosts:
|
||||||
|
vyos_rtr:
|
||||||
|
ansible_host: 192.168.10.2
|
||||||
|
ansible_user: vyos
|
||||||
|
ansible_connection: ansible.netcommon.network_cli
|
||||||
|
ansible_network_os: vyos.vyos.vyos
|
||||||
|
ansible_ssh_private_key_file: ~/.ssh/id_ed25519
|
||||||
@@ -0,0 +1,92 @@
|
|||||||
|
---
|
||||||
|
- name: 确认路由器内部 BGP 已建立
|
||||||
|
hosts: site_routers
|
||||||
|
gather_facts: false
|
||||||
|
tasks:
|
||||||
|
- name: 确认到 AMD 的邻居
|
||||||
|
vyos.vyos.vyos_command:
|
||||||
|
commands: show bgp neighbors 10.255.254.1
|
||||||
|
register: migration_bgp
|
||||||
|
changed_when: false
|
||||||
|
failed_when: "'BGP state = Established' not in migration_bgp.stdout[0]"
|
||||||
|
|
||||||
|
- name: 退役 laptop 的 WireGuard 试验端点
|
||||||
|
hosts: retired_wireguard_sites
|
||||||
|
become: true
|
||||||
|
tasks:
|
||||||
|
- name: 停止并禁用旧隧道
|
||||||
|
ansible.builtin.systemd_service:
|
||||||
|
name: wg-quick@wg-oci
|
||||||
|
state: stopped
|
||||||
|
enabled: false
|
||||||
|
|
||||||
|
- name: 检查原 BGP 试验邻居是否存在
|
||||||
|
ansible.builtin.command: vtysh -c 'show running-config'
|
||||||
|
register: laptop_frr
|
||||||
|
changed_when: false
|
||||||
|
|
||||||
|
- name: 只移除本次试验添加的 BGP 节点,保留 NEC 邻居和 OSPF
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- vtysh
|
||||||
|
- -c
|
||||||
|
- configure terminal
|
||||||
|
- -c
|
||||||
|
- router bgp 65001
|
||||||
|
- -c
|
||||||
|
- no neighbor 10.255.254.1
|
||||||
|
- -c
|
||||||
|
- address-family ipv4 unicast
|
||||||
|
- -c
|
||||||
|
- no network 192.168.10.0/24
|
||||||
|
- -c
|
||||||
|
- no network 10.60.0.0/24
|
||||||
|
- -c
|
||||||
|
- no network 10.61.0.0/24
|
||||||
|
- -c
|
||||||
|
- exit-address-family
|
||||||
|
- -c
|
||||||
|
- exit
|
||||||
|
- -c
|
||||||
|
- no ip protocol bgp route-map OCI-WG-SOURCE
|
||||||
|
- -c
|
||||||
|
- no route-map OCI-WG-SOURCE
|
||||||
|
- -c
|
||||||
|
- no ip prefix-list OCI-WG-IN
|
||||||
|
- -c
|
||||||
|
- no ip prefix-list OCI-WG-OUT
|
||||||
|
- -c
|
||||||
|
- end
|
||||||
|
- -c
|
||||||
|
- write memory
|
||||||
|
when: "'neighbor 10.255.254.1 remote-as' in laptop_frr.stdout"
|
||||||
|
|
||||||
|
- name: 停止并禁用旧防火墙启动单元
|
||||||
|
ansible.builtin.systemd_service:
|
||||||
|
name: oci-wg-firewall
|
||||||
|
state: stopped
|
||||||
|
enabled: false
|
||||||
|
|
||||||
|
- name: 只删除旧隧道专用防火墙链
|
||||||
|
ansible.builtin.shell: |
|
||||||
|
set -eu
|
||||||
|
changed=0
|
||||||
|
for pair in INPUT:OCI-WG-IN FORWARD:OCI-WG-FWD; do
|
||||||
|
parent=${pair%%:*}; chain=${pair#*:}
|
||||||
|
if iptables -w -nL "$chain" >/dev/null 2>&1; then
|
||||||
|
while iptables -w -C "$parent" -j "$chain" 2>/dev/null; do
|
||||||
|
iptables -w -D "$parent" -j "$chain"
|
||||||
|
done
|
||||||
|
iptables -w -F "$chain"
|
||||||
|
iptables -w -X "$chain"
|
||||||
|
changed=1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
echo "$changed"
|
||||||
|
register: retired_chains
|
||||||
|
changed_when: retired_chains.stdout == '1'
|
||||||
|
|
||||||
|
- name: 移除旧的 BGP 配置片段,避免误用
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: /etc/frr/oci-wireguard.vtysh
|
||||||
|
state: absent
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user