diff --git a/public/banners/running-production-infrastructure-solo.png b/public/banners/running-production-infrastructure-solo.png new file mode 100644 index 0000000..489b30e Binary files /dev/null and b/public/banners/running-production-infrastructure-solo.png differ diff --git a/public/og/running-production-infrastructure-solo.png b/public/og/running-production-infrastructure-solo.png new file mode 100644 index 0000000..d95b219 Binary files /dev/null and b/public/og/running-production-infrastructure-solo.png differ diff --git a/scripts/banner-gen/generate.mjs b/scripts/banner-gen/generate.mjs index 1d6f3c6..1fad950 100644 --- a/scripts/banner-gen/generate.mjs +++ b/scripts/banner-gen/generate.mjs @@ -1091,6 +1091,27 @@ BANNERS['stop-writing-agent-prompts-for-deterministic-work'] = { ], }; +BANNERS['running-production-infrastructure-solo'] = { + titlebar: "root@hoe-lee — three hosts, one operator", + lines: [ + { t: 'prompt', text: "$" }, { t: 'cmd', text: "docker ps — three hosts, one operator" }, + { t: 'prompt', text: "DSM" }, { t: 'cmd', text: "86 containers · 44 stacks — storage, SSO, services" }, + { t: 'prompt', text: "unRaid" }, { t: 'cmd', text: "37 containers · 21 stacks — CI runner, monitoring" }, + { t: 'prompt', text: "VPS" }, { t: 'cmd', text: "39 containers · 13 stacks — mail, public sites" }, + { t: 'prompt', text: "WARN" }, { t: 'err', text: "cAdvisor count inflated · one volume counted three times" }, + { t: 'prompt', text: "$" }, { t: 'cmd', text: "verify_infra.py — re-measure, never remember" }, + { t: 'prompt', text: "" }, { t: 'ok', text: "162 containers · 78 compose stacks ✓" }, + { t: 'prompt', text: "" }, { t: 'dim', text: "a number you cannot re-check is not a number you can defend" }, + ], + flow: [ + { n: '1', label: "3 hosts" }, + { n: '2', label: "162 ctrs" }, + { n: '3', label: "monitor lied", err: true }, + { n: '4', label: "re-measure" }, + { n: '5', label: "honest ✓" }, + ], +}; + // ---------- read frontmatter ---------- const postPath = join(ROOT, 'src', 'content', 'posts', `${slug}.md`); let category = 'devops'; diff --git a/scripts/og-gen/generate.mjs b/scripts/og-gen/generate.mjs index d8dfd83..cc7936a 100644 --- a/scripts/og-gen/generate.mjs +++ b/scripts/og-gen/generate.mjs @@ -331,6 +331,12 @@ TERMINALS['stop-writing-agent-prompts-for-deterministic-work'] = `
 [drift_skip] × 29 runs — silent, and silence reads as "nothing to report"
$rewrite as a script · stdout IS the notification→ green since ✓
`; +TERMINALS['running-production-infrastructure-solo'] = ` +
$3 hosts · 162 containers · 78 compose stacks
+
grafana: cAdvisor 'containers' inflated by cgroup pseudo-entries
+
grafana: one NAS volume counted three times
+
$verify_infra.py — re-derive from the live daemons→ no remembered numbers ✓
`; + // ---------- read frontmatter ---------- const postPath = join(ROOT, 'src', 'content', 'posts', `${slug}.md`); if (!existsSync(postPath)) { diff --git a/src/content/posts/running-production-infrastructure-solo.md b/src/content/posts/running-production-infrastructure-solo.md index 750059b..dc5425c 100644 --- a/src/content/posts/running-production-infrastructure-solo.md +++ b/src/content/posts/running-production-infrastructure-solo.md @@ -3,8 +3,10 @@ title: "Running Production Infrastructure Solo: 162 Containers Across Three Host description: "One operator, three hosts, 162 containers in 78 compose stacks — Traefik, SSO, monitoring, mail, CI/CD and backups — and the two failures that changed how I verify my own numbers." pubDate: 2026-10-06 category: case-studies -tags: [docker, traefik, monitoring, self-hosting, infrastructure, devops] -draft: true +tags: [docker, traefik, monitoring, infrastructure, devops] +ogImage: /og/running-production-infrastructure-solo.png +banner: /banners/running-production-infrastructure-solo.png +draft: false --- ## Why this matters diff --git a/src/content/posts/zh/running-production-infrastructure-solo.md b/src/content/posts/zh/running-production-infrastructure-solo.md new file mode 100644 index 0000000..4a7e7d1 --- /dev/null +++ b/src/content/posts/zh/running-production-infrastructure-solo.md @@ -0,0 +1,82 @@ +--- +title: "一个人跑生产基础设施:三台主机、162 个容器" +description: "一个人、三台主机、78 个 compose stack 里的 162 个容器——Traefik、SSO、监控、邮件、CI/CD 与备份,以及两次改变了我「怎么核对数字」的故障。" +pubDate: 2026-10-06 +category: case-studies +tags: [docker, traefik, monitoring, infrastructure, devops] +ogImage: /og/running-production-infrastructure-solo.png +banner: /banners/running-production-infrastructure-solo.png +draft: false +--- + +## 为什么这件事重要 + +卖软件的公司其实同时需要两样东西:一个能干活的应用,以及底下那套让它随时可达、有备份、够安全的机器。大多数情况是这两件事被拆开——开发的人看不见服务器,托管公司又读不懂代码。 + +我两件都做。也就是说,写这个应用的人,也是在 TLS 自动续期挂掉时被叫醒的人,而且能直接修应用,而不是把问题推给平台。 + +这是我自己的生产环境:三台自管的 Docker 主机,162 个容器分布在 78 个 compose stack 上,跑着真正对外的服务——客户网站、邮件平台、CI/CD、工作流自动化、访问统计,以及这个博客。没有云平台把它粘在一起,也没有团队在运维。下面是这套架构,以及两次改变了我「怎么检查自己的工作」的故障。 + +## 它的样子 + +| 主机 | 承担什么 | 容器 | Compose stack | +|---|---|---|---| +| Synology DSM | 存储、SSO、大部分内部服务 | 86 | 44 | +| unRaid | CI runner、媒体流水线、监控 | 37 | 21 | +| Ubuntu 24.04 VPS | 公网入口、邮件平台、客户站点 | 39 | 13 | + +这个划分是刻意的,不是历史遗留。VPS 承担的是「家里网断了也必须可达、必须快」的东西,因为公网流量和客户邮件都落在那里。Synology 承担需要磁盘和稳定文件系统的部分,以及所有其他服务都信任的身份源。unRaid 跑 CI runner 和那些又重又突发的批处理任务——否则它们会和服务抢内存。 + +## 让三台主机像一套系统的东西 + +**统一的入口模式。** 每台主机都跑一个 Traefik 反向代理,配自动化的 ACME 证书,所以证书自己续期,也没有哪个服务需要自己保管证书文件。后面是一套统一的路由和跳转约定,公网域名前面再套 Cloudflare。 + +**统一的身份源。** authentik 用 OIDC 和 proxy provider 做 SSO,所以新增一个内部服务是「在一个地方加一个应用」,而不是「在表格里再加一个密码」。对外的服务保持匿名可访问;其余的都从这里认证。 + +**统一的监控。** Prometheus、Grafana、cAdvisor 和 node-exporter 铺在三台主机上,收进同一个 Grafana 而不是三块面板——因为真正有意思的故障,都是跨主机的那种。 + +**统一的交付路径。** 自建的 Gitea Actions runner 负责构建和部署。这个博客就是走这条流水线:提交、runner 接单、容器重建、Cloudflare 把新构建送出去。 + +**经过验证、而不是假设的备份。** NAS 上的快照备份,加上 Duplicati 和 restic 任务,并把恢复步骤写下来——因为没验证过的备份只是一种感觉,不是备份。 + +## 失败一:仪表盘在骗我 + +搭监控的时候我选择相信 Grafana 显示的东西,结果有三块面板是错的。 + +cAdvisor 报出来的「容器」其实是 cgroup 的伪条目,把容器数量虚报了。我的「总容量」面板把一个 NAS 卷算了三遍,因为三个看起来差不多的挂载点被当成独立文件系统抓取了——这个数字看起来完全合理,而「看起来合理」正是它危险的地方。还有一台虚拟机根本不导出 CPU 频率,于是那块面板一直是空的,而我把它读成「暂时没数据」,而不是「这个 exporter 配错了」。 + +这三件事都没有弄坏任何服务。但三件事都会让我向客户报出一个假数字,因为仪表盘上的错数字和真数字长得一模一样。整个过程我写在 [One Prometheus for unRaid, a Synology NAS, and a VPS](https://blog.hoelee.com/posts/one-prometheus-for-unraid-synology-and-a-vps/) 里了。 + +## 失败二:少了一个配置文件,三种故障模式 + +我的团队密码管理器 Passbolt 开始返回 504,它的定时任务看起来卡住了。最后发现是三种不同的故障模式、同一个根因:一个只存在于运行中的容器里、从来没进 compose 文件的配置值。 + +空的 GPG fingerprint 让邮件发不出去,而发邮件这一步被放在定时任务里,于是整个任务挂住,UI 跟着一起挂。修好之后确实好了——直到容器被重建,配置被抹掉,原来的症状又回来了,看起来像问题自己复发。接着我为了追这两个问题引入的 `ssl.force`,在反向代理后面造成了跳转循环。 + +三个症状、三个晚上、一个教训:没有进版本控制的配置,就不算配置。完整的过程写在 [Passbolt UI Kept Hanging — Three Failure Modes From One Missing Config File](https://blog.hoelee.com/posts/passbolt-hang-three-failure-modes/)。 + +## 我怎么让数字保持诚实 + +简历上的数字会悄悄过期。我的数字是可以重新算出来的:一个小脚本通过 Portainer API 读活的 Docker 端点,逐台主机统计容器和 compose stack 的数量,所以「162 个容器、78 个 compose stack」是按需重新测量,而不是记住它曾经为真的那一天。同一个脚本还会打印出一份「绝不能出现在任何对客户材料里」的服务清单。 + +背后的规则很简单:一个无法被重新核对的数字,就是一个没人能在面试里、在报价里、在客户问「你怎么算出来的」那一通电话里守住的说法。凡是不能重新测量的数字,就不写进文档。 + +## 我会做得不一样的地方 + +**第一天就把每一个配置值放进 compose 文件**,包括那些感觉「属于机器而不是属于服务」的。Passbolt 那次复发,只因为一个修复活在容器里。 + +**对「没有数据」告警,而不是只对「坏数据」告警。** 一块空白的面板和一台健康服务,在仪表盘上长得一样。如果某个 exporter 十分钟没送来任何东西,那就是该告警的事件。 + +**用专门用来衡量可用性的东西去衡量可用性。** 指标流水线告诉你系统现在表现如何,它并不统计「成功请求 ÷ 总请求」。如果一个百分比要出现在客户面前,它需要一个按计划探测端点并记录结果的探针;否则诚实的回答是「有监控」,而不是一个数字。这个区别比数字本身更重要。 + +## 结果 + +三台主机、162 个容器、78 个 compose stack、一个人——加上自动化证书、统一 SSO、跨主机监控、经过验证的备份和自建 CI/CD。它同时跑过客户网站、一个邮件平台、一个 Telegram 客服 bot、排程自动化和这个博客;也正是它让我能把项目从「这是需求」做到「已经上线、有监控、有备份」,而不用把后半段交给别人。 + +## 想给你的生意也做一套? + +如果你需要一个既能写应用、又能跑它脚下那套基础设施的人——Docker 和 Traefik、自动化 TLS、SSO、监控、邮件、备份、CI/CD——我做的就是这个,而且我更想聊你的环境,而不是发一份宣传单。 + +**WhatsApp:[+60 12-797 2969](https://wa.me/60127972969)** · **Email:[me@hoelee.com](mailto:me@hoelee.com?subject=Infrastructure%20setup%20enquiry)** · **[hoelee.com](https://hoelee.com)** + +网站设计开发是我的主业;自托管基础设施、监控与自动化是它的另一半。