diff --git a/public/banners/n8n-v1-to-v2-upgrade-gotchas.png b/public/banners/n8n-v1-to-v2-upgrade-gotchas.png
new file mode 100644
index 0000000..0aec44d
Binary files /dev/null and b/public/banners/n8n-v1-to-v2-upgrade-gotchas.png differ
diff --git a/public/banners/running-tts-as-a-service-with-token-sidecars.png b/public/banners/running-tts-as-a-service-with-token-sidecars.png
new file mode 100644
index 0000000..d3b602e
Binary files /dev/null and b/public/banners/running-tts-as-a-service-with-token-sidecars.png differ
diff --git a/public/banners/self-healing-digital-goods-entitlements.png b/public/banners/self-healing-digital-goods-entitlements.png
new file mode 100644
index 0000000..85874ab
Binary files /dev/null and b/public/banners/self-healing-digital-goods-entitlements.png differ
diff --git a/public/og/n8n-v1-to-v2-upgrade-gotchas.png b/public/og/n8n-v1-to-v2-upgrade-gotchas.png
new file mode 100644
index 0000000..d4fb835
Binary files /dev/null and b/public/og/n8n-v1-to-v2-upgrade-gotchas.png differ
diff --git a/public/og/running-tts-as-a-service-with-token-sidecars.png b/public/og/running-tts-as-a-service-with-token-sidecars.png
new file mode 100644
index 0000000..baea8b3
Binary files /dev/null and b/public/og/running-tts-as-a-service-with-token-sidecars.png differ
diff --git a/public/og/self-healing-digital-goods-entitlements.png b/public/og/self-healing-digital-goods-entitlements.png
new file mode 100644
index 0000000..097e988
Binary files /dev/null and b/public/og/self-healing-digital-goods-entitlements.png differ
diff --git a/scripts/banner-gen/generate.mjs b/scripts/banner-gen/generate.mjs
index dfabd31..011c790 100644
--- a/scripts/banner-gen/generate.mjs
+++ b/scripts/banner-gen/generate.mjs
@@ -445,6 +445,69 @@ const BANNERS = {
],
};
+BANNERS['n8n-v1-to-v2-upgrade-gotchas'] = {
+ titlebar: 'root@dsm — n8n v1 → v2 migration',
+ lines: [
+ { t: 'prompt', text: '$' }, { t: 'cmd', text: 'docker pull n8nio/n8n:2.40.1 · container up in 90s' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'v1.123.x → v2.40.1 · 17 workflows · 7 active' },
+ { t: 'prompt', text: 'WARN' }, { t: 'err', text: 'Telemetry failed schema validation: executions_data_save_on_error' },
+ { t: 'prompt', text: 'WARN' }, { t: 'err', text: 'Failed to start Python task runner — Python 3 missing' },
+ { t: 'prompt', text: 'WARN' }, { t: 'err', text: 'Sandbox: enabled=false (DB override; env was enabled=true)' },
+ { t: 'prompt', text: '$' }, { t: 'cmd', text: 'N8N_WEBHOOK_URL · pin TASK_TIMEOUT=300 · pin compression limits' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'storage rename flagged for v3 · migrate + remount together' },
+ { t: 'prompt', text: '' }, { t: 'ok', text: '→ 7 deprecations resolved · downstream workflows intact ✓' },
+ ],
+ flow: [
+ { n: '1', label: 'pull v2' },
+ { n: '2', label: 'read boot log' },
+ { n: '3', label: 'schema reject', err: true },
+ { n: '4', label: 'pin defaults' },
+ { n: '5', label: 'verified ✓' },
+ ],
+};
+
+BANNERS['self-healing-digital-goods-entitlements'] = {
+ titlebar: 'root@dsm — entitlement lifecycle (W1–W5)',
+ lines: [
+ { t: 'prompt', text: '$' }, { t: 'cmd', text: 'nocodb webhook → n8n compute → alist role scopes' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'desired = union(active purchases, direct grants) · MAX expiry wins' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'shared/effective-grants.js inlined at build · 4 workflows' },
+ { t: 'prompt', text: 'WARN' }, { t: 'err', text: 'public hostname: >60s → nginx 504 → 1s retries → pool → 503' },
+ { t: 'prompt', text: '$' }, { t: 'cmd', text: 'http://nocodb:10380 (30ms, no tunnel) → cascade gone' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'W3 sweeps every 5min · W4 repairs drift 03:00 + verifies' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'W5 public: CORS-locked, minimal fields, no HMAC theatre' },
+ { t: 'prompt', text: '' }, { t: 'ok', text: '→ repair verified · reconcile log empty when healthy ✓' },
+ ],
+ flow: [
+ { n: '1', label: 'grant' },
+ { n: '2', label: 'expire' },
+ { n: '3', label: 'drift', err: true },
+ { n: '4', label: 'repair' },
+ { n: '5', label: 'verify ✓' },
+ ],
+};
+
+BANNERS['running-tts-as-a-service-with-token-sidecars'] = {
+ titlebar: 'root@dsm — tts service · 1 year uptime',
+ lines: [
+ { t: 'prompt', text: '$' }, { t: 'cmd', text: 'reader → GET /webhook/{mtts,gtts}?pass=…&text=…&speed=…' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'azure speech F0 · google cloud cmn-CN-Wavenet-A' },
+ { t: 'prompt', text: 'WARN' }, { t: 'err', text: 'token stored in workflow → 401 after 10min (azure) / 1h (google)' },
+ { t: 'prompt', text: '$' }, { t: 'cmd', text: '2 cron sidecars → accesstoken.txt on shared volume' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'refresh 570s / 3500s · margin before expiry' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'map speed 5–50 → -20%…+150% · strip trailing newline' },
+ { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: '8 neural voices selected by integer index' },
+ { t: 'prompt', text: '' }, { t: 'ok', text: '→ secrets in zero workflows · still running ✓' },
+ ],
+ flow: [
+ { n: '1', label: 'webhook' },
+ { n: '2', label: 'read token' },
+ { n: '3', label: 'synthesize' },
+ { n: '4', label: 'audio/wav' },
+ { n: '5', label: '1yr ✓' },
+ ],
+};
+
// ---------- read frontmatter ----------
const postPath = join(ROOT, 'src', 'content', 'posts', `${slug}.md`);
let category = 'devops';
diff --git a/scripts/banner-gen/template.html b/scripts/banner-gen/template.html
index b61a43d..25eca10 100644
--- a/scripts/banner-gen/template.html
+++ b/scripts/banner-gen/template.html
@@ -64,7 +64,7 @@
.terminal .dot.g { background: #3fb950; }
.terminal .title-text { margin-left: 12px; color: #6b7c99; font-size: 16px; }
- .terminal .body { padding: 30px 36px; font-size: 22px; line-height: 1.8; flex: 1; }
+ .terminal .body { padding: 30px 36px; font-size: 22px; line-height: 1.8; flex: 1; display: flex; flex-direction: column; justify-content: center; }
.terminal .line { display: flex; gap: 12px; margin-bottom: 8px; }
.terminal .prompt { color: #7599ff; flex-shrink: 0; }
.terminal .cmd { color: #e6edf3; }
diff --git a/scripts/og-gen/generate.mjs b/scripts/og-gen/generate.mjs
index a25ee21..0970684 100644
--- a/scripts/og-gen/generate.mjs
+++ b/scripts/og-gen/generate.mjs
@@ -144,6 +144,21 @@ const TERMINALS = {
$wp plugin list · hre-ai-remix 0.0.1 → 0.0.22
404 …/wp-json/hre/v1admin/photos — rest_url() has no trailing slash
$fix = '/admin/…' · 58 commits · 4 were the AI→ shipped ✓
`,
+
+ 'n8n-v1-to-v2-upgrade-gotchas': `
+ $pull n8nio/n8n:2.40.1 · restart · upgrade took 90s
+ telemetry schema: executions_data_save_on_error rejected
+ $read the boot log · pin timeouts + limits→ 7 fixed ✓
`,
+
+ 'self-healing-digital-goods-entitlements': `
+ $nocodb → n8n W1–W5 → alist role scopes
+ public hostname: 60s latency → nginx 504 → retry storm → 503
+ $http://nocodb:10380 · 30ms · W4 repairs drift 03:00→ self-healing ✓
`,
+
+ 'running-tts-as-a-service-with-token-sidecars': `
+ $reading app → GET /webhook/mtts?pass=…&text=…
+ azure token expires in ~10min · google in ~1h
+ $2 cron sidecars write accesstoken.txt · 570s / 3500s→ 1 year uptime ✓
`,
};
const DEFAULT_TERMINAL = `
diff --git a/src/content/posts/n8n-v1-to-v2-upgrade-gotchas.md b/src/content/posts/n8n-v1-to-v2-upgrade-gotchas.md
new file mode 100644
index 0000000..4c95027
--- /dev/null
+++ b/src/content/posts/n8n-v1-to-v2-upgrade-gotchas.md
@@ -0,0 +1,270 @@
+---
+title: "Upgrading n8n v1 to v2: Seven Deprecations in One Log File"
+description: "My self-hosted n8n 1.x to 2.40.1 upgrade surfaced seven silent breakages at once — a telemetry schema rejection, deprecated webhook vars, and a DB override that quietly disabled the AI sandbox."
+pubDate: 2026-09-18
+category: devops
+tags: [n8n, docker, upgrade, self-hosting, debugging, automation]
+ogImage: /og/n8n-v1-to-v2-upgrade-gotchas.png
+banner: /banners/n8n-v1-to-v2-upgrade-gotchas.png
+---
+
+I run n8n as the automation backbone for my self-hosted stack — it handles
+file-delivery permissions, database backups, and a text-to-speech API that a
+reading app depends on. It had been sitting on the `1.123.x` line for the
+better part of a year, quietly doing its job.
+
+Then I pulled `n8nio/n8n:2.40.1` and restarted the container. The upgrade
+itself took about ninety seconds. Understanding what it *broke* took the rest
+of the evening — and almost all of it was already written down in a single log
+file that n8n prints on boot. I just hadn't read it carefully enough the first
+time.
+
+This is that log file, decoded, so you can plan your own v1 → v2 jump instead
+of discovering these at 11 PM.
+
+## Why it matters
+
+Major-version upgrades of an automation platform are different from upgrading
+a leaf service. n8n is *the thing that runs everything else*: if it comes up
+broken, your backups, your access-control syncs, and your internal APIs all
+stop with it. Worse, most of what breaks in v2 doesn't throw an error — it
+logs a deprecation notice once and then quietly does something different.
+
+The seven items below are the ones that actually applied to a real, messy,
+production-shaped install. Three of them changed behavior in my stack. One of
+them silently turned a feature *off*.
+
+## Start here: n8n tells you what's wrong at boot
+
+Before touching a single workflow, read the container log from the top. On a
+fresh v2 boot, n8n prints an explicit deprecation block:
+
+```text
+There are deprecations related to your n8n setup. Please take the recommended
+actions to update your configuration:
+ - WEBHOOK_URL -> Use N8N_WEBHOOK_URL instead, which sets the base URL for
+ both test and production webhooks.
+ - N8N_UNVERIFIED_PACKAGES_ENABLED -> The default for this variable will
+ change to `false` in a future version.
+ - N8N_RUNNERS_MODE -> Internal task runner mode is deprecated and will be
+ removed in a future version.
+ - N8N_RUNNERS_TASK_TIMEOUT -> The default for this variable will be reduced
+ from 300 (5 minutes) to 60 (1 minute) in a future version.
+ - N8N_COMPRESSION_NODE_MAX_DECOMPRESSED_SIZE_BYTES -> The default will be
+ reduced from 2 GiB to 256 MiB in a future version.
+ - N8N_COMPRESSION_NODE_MAX_ZIP_ENTRIES -> The default will be reduced from
+ 5000 to 1000 in a future version.
+```
+
+That block is your migration checklist. Six of my seven gotchas are in it.
+
+## 1. Your env values can now fail *schema validation*
+
+Here is the one that confused me most, because it looked like a nonsense error:
+
+```text
+Telemetry event "Instance started" failed schema validation:
+execution_variables.executions_data_save_on_error: Invalid option:
+expected one of "all"|"none"
+```
+
+I had `EXECUTIONS_DATA_SAVE_ON_ERROR=error` set — a value that was perfectly
+legal in v1, and that I'd chosen deliberately, because saving *only* failed
+executions is the sane default for a busy instance. In v2 that value is no
+longer in the allowed set, which is now `all` or `none`.
+
+The failure mode is the interesting part. It didn't crash. It didn't even warn
+in a way that reads like an error at a glance — it emitted a *telemetry schema
+validation* message, which sounds like an n8n-internal problem, not a
+configuration problem of mine. The setting was effectively ignored.
+
+The fix is to move that intent somewhere coherent: pick a legal value and
+control volume with pruning instead.
+
+```env
+EXECUTIONS_DATA_SAVE_ON_ERROR=all
+EXECUTIONS_DATA_SAVE_ON_SUCCESS=none
+EXECUTIONS_DATA_PRUNE=true
+EXECUTIONS_DATA_MAX_AGE=336
+EXECUTIONS_DATA_PRUNE_MAX_COUNT=10000
+```
+
+**Lesson:** in v2, treat your environment variables as a typed interface with
+a schema. An invalid value may be dropped silently rather than rejected loudly.
+
+## 2. `WEBHOOK_URL` is deprecated for `N8N_WEBHOOK_URL`
+
+If you publish webhooks behind a reverse proxy — which you almost certainly do,
+because that's how they become reachable — the base URL variable is load-bearing.
+It's what makes n8n report the *public* webhook path instead of
+`http://localhost:5678/...`.
+
+The old name still works today, so this one won't bite immediately. But note
+the wording: the new variable sets the base URL for **both test and production
+webhooks**. In my setup those had drifted apart in behavior, which is exactly
+the class of bug this consolidation is meant to eliminate.
+
+```env
+# before (still functional, deprecated)
+WEBHOOK_URL=https://auto.example.com/
+
+# after
+N8N_WEBHOOK_URL=https://auto.example.com/
+```
+
+## 3. Internal task runner mode is going away — and mine was already broken
+
+This one was sitting in my logs the whole time, several lines above the
+deprecation block, and I'd been reading past it for months:
+
+```text
+Failed to start Python task runner in internal mode. because Python 3 is
+missing from this system. Launching a Python runner in internal mode is
+intended only for debugging and is not recommended for production.
+```
+
+If any workflow of yours uses a **Python** Code node, it has not been running
+in internal mode at all — there's no Python in the stock image. JavaScript
+Code nodes are fine (a JS runner registers normally), which is why this can go
+unnoticed indefinitely: everything *looks* healthy.
+
+v2 makes the direction of travel explicit: switch to `external` mode and share
+an auth token with a separate launcher process.
+
+```env
+N8N_RUNNERS_MODE=external
+N8N_RUNNERS_AUTH_TOKEN=
+```
+
+**Lesson:** "internal mode is deprecated" is the headline, but the real
+finding is that a runner type can be *silently non-functional* for months.
+Check `docker logs` for the runner registration line, not just for up/down.
+
+## 4. Task timeout drops from 300s to 60s
+
+This is the one I'd flag hardest for anyone with slow workflows:
+
+```text
+N8N_RUNNERS_TASK_TIMEOUT -> The default for this variable will be reduced
+from 300 (5 minutes) to 60 (1 minute) in a future version.
+```
+
+Be honest about your own workloads. Do you have a Code node that loops over
+thousands of records, or an HTTP call to a slow upstream? Mine do — a nightly
+reconciliation walks every customer and calls an external API per record. On a
+future upgrade, that stops at sixty seconds with no config change on my side.
+
+Set it explicitly now, while you're already in the file:
+
+```env
+N8N_RUNNERS_TASK_TIMEOUT=300
+```
+
+The general principle for every deprecation of the form "the default will
+change": **if you rely on the current default, pin it explicitly.** Otherwise
+the upgrade is a silent behavior change, and you'll debug it as a bug rather
+than recognise it as a stale default.
+
+## 5 & 6. Two compression-node limits shrink (2 GiB → 256 MiB, 5000 → 1000 entries)
+
+These two travel together and matter only if you use compression/decompression
+nodes on large payloads — which is easy to end up doing when you're shuttling
+database dumps or archives through a workflow.
+
+```text
+N8N_COMPRESSION_NODE_MAX_DECOMPRESSED_SIZE_BYTES -> reduced from 2 GiB to
+256 MiB in a future version.
+N8N_COMPRESSION_NODE_MAX_ZIP_ENTRIES -> reduced from 5000 to 1000 in a
+future version.
+```
+
+An eighth of the memory ceiling and a fifth of the entry limit. Nothing errors;
+the node just refuses at a threshold you didn't set. Pin both if you're near
+either.
+
+## 7. The storage path renames in v3 — and you have a volume mounted at the old one
+
+Not a v2 breakage, but v2 warns about it, and it's the one with real data-planning
+implications:
+
+```text
+Deprecation warning: The storage directory "/home/node/.n8n/binaryData" will
+be renamed to "/home/node/.n8n/storage" in n8n v3. To migrate now, set
+N8N_MIGRATE_FS_STORAGE_PATH=true. If you have a volume mounted at the old
+path, update your mount configuration after migration.
+```
+
+Read that last sentence again: *if you have a volume mounted at the old path,
+update your mount configuration after migration.* If you set the migration flag
+and keep your old bind mount, you now have two directories and your binary data
+lives in whichever one the container is actually pointed at. Do the rename and
+the mount change in the same maintenance window — not one now, one "later".
+
+## The one that wasn't in the log: my AI sandbox disabled itself
+
+Here's the finding that had nothing to do with a deprecation notice, and that
+I'd never have caught without reading the full boot sequence:
+
+```text
+Sandbox: enabled=false provider=n8n-sandbox (DB override; env was enabled=true
+provider=n8n-sandbox)
+```
+
+My environment said enabled. The database said otherwise. **The database won.**
+
+The env var was `N8N_INSTANCE_AI_SANDBOX_ENABLED=true`, and it was still set
+correctly on the container. But a value persisted in n8n's own configuration
+store overrode it at startup — and the only place that conflict is reported is
+inside a parenthetical in a log line.
+
+This is a genuinely valuable debugging lesson beyond n8n: when a feature is off
+despite the env var being obviously right, suspect a **persisted settings layer
+that outranks your environment**. The container config is not always the last
+word. Grep for the feature name in the logs, and don't stop at the env value.
+
+## What I'd do differently
+
+1. **Read the boot log before declaring the upgrade done.** Every deprecation
+ that mattered to me was printed on startup, in one block, on the first run.
+ My v1 habit was to check "is it up, do the workflows run" — which is exactly
+ the check that misses all seven of these.
+2. **Treat "the default will change" as a to-do, not a warning.** Five of the
+ seven items are future-default changes. Pinning them now costs one edit and
+ converts a mystery outage later into a config diff.
+3. **Diff the container config against n8n's own stored config.** The sandbox
+ override taught me that env is one of two inputs, not the source of truth.
+ When behavior and configuration disagree, believe the behavior and go find
+ the higher-priority layer.
+4. **Pin the image tag and keep the previous one.** I jumped v1 → `2.40.1`
+ directly. Having the old image on disk is what makes a rollback a
+ `docker run` instead of a rebuild.
+
+## The result
+
+n8n is on 2.40.1 with the full deprecation block resolved: the schema-invalid
+value corrected, `N8N_WEBHOOK_URL` in place, the JS runner registering cleanly,
+and the task timeout and compression limits pinned so the next upgrade is a
+no-op rather than a surprise.
+
+Everything downstream kept working — the file-permission syncs, the nightly
+backups, the text-to-speech endpoint. That's the outcome worth aiming for with
+an upgrade like this: not "it came back up", but "it came back up *and* the
+next three upgrades are already paid for".
+
+The uncomfortable part is how much of it I could have known in advance. n8n
+handed me the entire list, unprompted, at startup. The upgrade was never the
+hard part — reading the output was.
+
+---
+
+## Running automation you'd rather not babysit?
+
+I build and maintain self-hosted automation — n8n workflows, Docker stacks,
+and the glue between apps that were never designed to talk to each other. If
+you're facing a major version upgrade, or you have automation that works until
+it doesn't, I plan the migration, do it in a maintenance window, and document
+every config decision so the next upgrade is boring.
+
+Reach me at [me@hoelee.com](mailto:me@hoelee.com?subject=n8n%20upgrade) or
+WhatsApp [+60 12-797 2969](https://wa.me/60127972969), or see what I do at
+[hoelee.com](https://hoelee.com).
diff --git a/src/content/posts/running-tts-as-a-service-with-token-sidecars.md b/src/content/posts/running-tts-as-a-service-with-token-sidecars.md
new file mode 100644
index 0000000..2de7a98
--- /dev/null
+++ b/src/content/posts/running-tts-as-a-service-with-token-sidecars.md
@@ -0,0 +1,307 @@
+---
+title: "I Ran a Text-to-Speech Service for a Year on Two Cron Containers"
+description: "A reading app needed TTS. Cloud free tiers expire tokens in 10 minutes and 1 hour. Here's the sidecar pattern that kept Azure and Google voices working for a year with no secrets in any workflow."
+pubDate: 2026-09-18
+category: ai
+tags: [n8n, tts, azure, google-cloud, docker, automation, sidecar]
+ogImage: /og/running-tts-as-a-service-with-token-sidecars.png
+banner: /banners/running-tts-as-a-service-with-token-sidecars.png
+---
+
+I run a self-hosted reading server for ebooks. It has a "read aloud" feature,
+and the built-in engines are serviceable but robotic. So I wired it up to
+proper neural voices — Azure Speech and Google Cloud TTS — through my n8n
+instance.
+
+That was about a year ago. It has been running ever since, and the design has
+barely changed. This is how it works, and more usefully, why it's shaped the
+way it is: the whole architecture exists to solve one specific problem that
+breaks naive implementations within the hour.
+
+## Why it matters
+
+Text-to-speech is a nice feature to add and a surprisingly annoying one to
+*keep* working. The voices are cheap or free at low volume, the quality is
+excellent, and the APIs are straightforward — right up until you discover that
+an OAuth access token is not a credential you can just store in a config file.
+
+Both providers issue short-lived access tokens, and the lifetimes are
+drastically different:
+
+- **Azure Speech** free tier: roughly **10 minutes**.
+- **Google Cloud**: roughly **1 hour**.
+
+If you put a token in a workflow variable, the integration works beautifully
+for ten minutes and then starts returning 401s forever. This is the classic
+shape of an integration that demos perfectly and fails in production — and the
+fix isn't "remember to refresh the token", because you will not.
+
+## The architecture: token refresh is not the workflow's job
+
+The design decision that makes this work is refusing to let the workflows
+manage credentials at all. Instead, two tiny containers own token lifecycle
+exclusively, and write the current token to a file on a shared volume. The
+workflows just read the file.
+
+```text
+┌──────────────────┐ GET /webhook/{mtts|gtts}?pass=…&text=…&speed=…
+│ Reading server │ ───────────────────────────────────────────────┐
+│ (httpTTS engine)│ │
+└──────────────────┘ ▼
+ ┌───────────────────────────┐
+ │ n8n │
+ │ ├ /mtts (Microsoft) │
+ │ └ /gtts (Google) │
+ └───────┬───────────────────┘
+ reads accesstoken.txt
+ ┌─────────────┴─────────────┐
+ ▼ ▼
+ Azure Speech (F0) Google Cloud TTS
+ southeastasia region cmn-CN Wavenet
+ │ │
+ └──────── WAV audio ────────┘
+ │
+ back to the player
+```
+
+Two cron sidecars keep the tokens fresh:
+
+| Container | Image | Interval | Writes |
+|---|---|---|---|
+| `cron-azure-refresh` | `curlimages/curl` | every ~570 s | `MicrosoftTTS/accesstoken.txt` |
+| `cron-gcloud-refresh` | `google/cloud-sdk:slim` | every ~3500 s | `GoogleTTS/accesstoken.txt` |
+
+Both bind-mount the **same host directory** that n8n reads through its
+Read/Write Files node. That shared volume is the entire interface between the
+credential layer and the workflow layer.
+
+Why the intervals are what they are: 570 seconds against a ~600-second Azure
+lifetime gives a 30-second safety margin, and refreshing slightly *early*
+forever is far more robust than refreshing exactly on expiry. Same reasoning
+for 3500 seconds against an hour.
+
+```yaml
+cron-azure-refresh:
+ image: curlimages/curl:8.10.1
+ restart: unless-stopped
+ volumes:
+ - /volume1/docker/n8n/file:/file
+ entrypoint: /bin/sh
+ command: >
+ -c 'while true; do
+ curl -s -X POST "https://southeastasia.api.cognitive.microsoft.com/sts/v1.0/issueToken"
+ -H "Ocp-Apim-Subscription-Key: $AZURE_SPEECH_KEY" > /file/MicrosoftTTS/accesstoken.txt;
+ sleep 570;
+ done'
+```
+
+## Why a file, and not the obvious alternatives
+
+**Why not store the token in n8n credentials and refresh in-workflow?** Because
+the refresh logic would then be copied into every workflow that needs a token,
+and each copy would need its own error handling. When a refresh fails at 3 AM,
+you want exactly one process to care.
+
+**Why not have the workflow call the token endpoint on every request?** It
+works, and it doubles the latency and the dependency surface of every read-aloud
+request. Worse, it means a token-endpoint hiccup becomes a TTS outage.
+
+**Why a file at all?** Because it's the simplest possible interface that both
+sides already support. n8n has a built-in Read/Write Files node; cron
+containers can write with `curl` and shell redirection. No queue, no database
+table, no shared library — just a file whose contents are always the current
+token.
+
+The tradeoff is honest: reading a file per request is a disk read in the hot
+path. At a reading app's request rate that is entirely free, and it buys a
+complete decoupling between credential lifecycle and request handling.
+
+## The n8n side: two workflows, one shape
+
+Both TTS workflows have the same skeleton, and it's worth walking through
+because the details are where it gets interesting.
+
+**1. Webhook with `responseMode: responseNode`.** The workflow must return raw
+audio bytes, not JSON, so the response is explicitly controlled by a Respond to
+Webhook node rather than n8n's default.
+
+**2. A password gate.** A query parameter is checked against an expected value,
+and mismatches return a real 403 rather than an empty 200:
+
+```text
+Respond to Webhook → text: "403 unauthorized", responseCode: 403
+```
+
+The `pass` value sits in the URL, which I'll address honestly in a moment.
+
+**3. Read the token file.** `Read/Write Files from Disk` reads
+`/home/user/file/MicrosoftTTS/accesstoken.txt`. Then two nodes clean it up:
+`Extract from File` (text mode) and a Set node that strips newlines, because a
+trailing `\n` in an `Authorization` header produces a maddening 401 that looks
+nothing like a whitespace problem:
+
+```js
+// Edit Fields node
+{{ $json.data.replace(/(\r\n|\n|\r)/g, '') }}
+```
+
+**4. Call the provider.** For Azure, the body is SSML with the voice and rate
+interpolated in:
+
+```xml
+
+
+
+ {{ $('Webhook').item.json.query.text }}
+
+
+
+```
+
+**5. Return the audio as a binary response** with the right content type:
+
+```text
+Respond to Webhook → binary, set
+ Content-Type: audio/wav
+ Content-Disposition: filename="output.wav"
+```
+
+## The mapping problem: the client speaks a different language
+
+Here's a detail that took more thought than the API calls. The reading app sends
+a `speed` value on its own scale — 5 to 50, because that's what its UI slider
+produces. Azure wants a prosody rate as a percentage, and Google wants a
+`speakingRate` multiplier around 1.0.
+
+Neither provider's scale matches the app's. So there's a deliberate translation
+step, and this one is worth copying because mapping a UI control onto an API
+parameter is a recurring chore:
+
+```js
+// Map the reader's 5–50 speed slider onto Azure's -20%…+150% rate range.
+const inMin = 5, inMax = 50;
+const outMin = -20, outMax = 150;
+
+// Clamp the input before mapping, so an out-of-range client value can't
+// produce an absurd prosody rate.
+if (speed < inMin) speed = inMin;
+if (speed > inMax) speed = inMax;
+
+const mapped = ((speed - inMin) / (inMax - inMin)) * (outMax - outMin) + outMin;
+// → rate: `${Math.round(mapped)}%`
+```
+
+The equivalent for Google is a straight division, since its scale is close to
+linear in the same range:
+
+```js
+speakingRate: speed / 25 // Google expects ~1.0, not a percentage
+```
+
+Two different providers, two different unit systems, one client-side concept.
+Keeping the mapping in the workflow (rather than asking the client to know
+about Azure percentages) is what lets the reading app stay provider-agnostic —
+and it's why I could add the second provider without touching the app at all.
+
+There's also a voice table, because the client sends an integer index rather
+than a voice name:
+
+```js
+const voices = [
+ "zh-CN-XiaochenMultilingualNeural", // 1
+ "zh-CN-XiaoxiaoMultilingualNeural", // 2
+ // ...
+ "zh-CN-XiaoshuangNeural", // 7 (female, child)
+ "zh-CN-XiaoyouNeural" // 8 (female, child)
+];
+```
+
+Eight voices — six adult, two child — selectable from the reading app's UI.
+The workflow clamps the index into range rather than trusting it, which is the
+same defensive habit as the speed clamp.
+
+## The security question, answered honestly
+
+The gate is a `pass` query parameter compared against a fixed string. I'm not
+going to dress that up: **it is a shared secret in a URL.** It stops casual
+abuse of an endpoint that costs me money per request. It does not stop anyone
+who can read the reading app's config, and it won't survive a serious attacker.
+
+I'm at peace with that because of what it's protecting. The worst outcome is
+that someone burns my free-tier TTS quota — an annoyance, not a data breach.
+There's no customer data behind this endpoint and no privileged access to
+anything. Paying for real authentication (OAuth, signed requests, per-user
+limits) would cost far more complexity than the exposure is worth.
+
+The transferable habit is being *explicit* about which tier a gate sits in:
+this is **abuse deterrence**, not authorization. Systems get into trouble when
+a deterrent is mistaken for a boundary. If this endpoint had touched customer
+records or file access, it would need real auth — and I'd have built it
+differently from the start.
+
+## The hidden cost: it was pinned to a dead branch
+
+The service ran for nearly a year with zero code changes. Which is the good
+news and the bad news:
+
+```text
+n8nio/n8n:1.123.72
+```
+
+That's the image the TTS documentation specified. It kept working, which meant
+nothing prompted me to revisit it — the classic failure mode of infrastructure
+that's *too* reliable. When I finally upgraded n8n to v2, this was the last
+1.x-era reference left in my stack, and the token-refresh containers are the
+part of the system most sensitive to platform behaviour changing underneath
+them.
+
+The lesson isn't "upgrade more often" — it's that **a service with no moving
+parts has no natural prompt to re-examine its assumptions.** Set a calendar
+reminder to review pinned dependency versions, because the system itself will
+never tell you.
+
+## What I'd do differently
+
+1. **Keep the mapping logic versioned with the app, not buried in a workflow.**
+ The speed mapping and the voice table encode the reading app's UI contract.
+ They live in JavaScript inside an n8n Code node, where they're invisible to
+ anyone working on the app — and if the slider range ever changes, nothing
+ will tell me.
+2. **Add a health endpoint that exercises a real synthesis.** All I can
+ currently check is that the containers are up. A token can be present in the
+ file and *expired anyway* (if a refresh silently failed), and there's no
+ cheap way to see that before a user hits it.
+3. **Track pinned versions somewhere I'll actually look.** `1.123.72` sat in a
+ markdown file for a year. A single "review pinned images" line in the
+ operations runbook would have surfaced it at the next maintenance window
+ instead of through a migration.
+
+## The result
+
+A year of uptime, two containers, one shared directory, and eight neural voices
+that a reading app can select from a slider. Requests complete in a couple of
+seconds including provider round-trip, and the whole thing costs nothing beyond
+the free tiers — because the token lifecycle problem was solved once, in the
+right place.
+
+The pattern generalises to any short-lived credential: **don't teach every
+caller to refresh a token — run one process whose only job is to keep a file
+current, and let everyone else read the file.** It's not clever, and that's the
+point. Clever credential handling is how you end up with four refresh
+implementations and three of them wrong.
+
+---
+
+## Need an AI feature wired into an app you already have?
+
+I build the unglamorous middle layer of AI integrations — the part that keeps
+working after the demo: token refresh, provider failover, voice and model
+mapping, and rate-limit handling. If you want text-to-speech, transcription, or
+an LLM feature added to an existing application and you'd like it to still work
+next year, that's the work I do.
+
+Reach me at [me@hoelee.com](mailto:me@hoelee.com?subject=TTS%20integration)
+or WhatsApp [+60 12-797 2969](https://wa.me/60127972969), or see what I do at
+[hoelee.com](https://hoelee.com).
diff --git a/src/content/posts/self-healing-digital-goods-entitlements.md b/src/content/posts/self-healing-digital-goods-entitlements.md
new file mode 100644
index 0000000..be1740f
--- /dev/null
+++ b/src/content/posts/self-healing-digital-goods-entitlements.md
@@ -0,0 +1,298 @@
+---
+title: "Self-Healing Access Control for Digital Goods: NocoDB, n8n, and AList"
+description: "How I built a five-workflow entitlement system that grants, expires, and continuously repairs customer file access — with a build-time code-sharing trick and a public endpoint designed to leak nothing."
+pubDate: 2026-09-18
+category: case-studies
+tags: [n8n, nocodb, alist, access-control, docker, automation, digital-goods]
+ogImage: /og/self-healing-digital-goods-entitlements.png
+banner: /banners/self-healing-digital-goods-entitlements.png
+---
+
+I sell digital products — files, artwork, licensed assets — and deliver them
+through a self-hosted file portal. The hard problem was never storage. It was
+**entitlement**: making sure that when someone buys, they get exactly the
+folders they paid for, for exactly as long as they paid for, and that when
+they stop paying, the access actually goes away.
+
+Doing that by hand works for the first ten customers. It does not work for a
+hundred, and it fails in a specific and nasty way: silently. Nothing errors
+when a subscription expires and the customer keeps downloading. You just keep
+serving files to someone who stopped paying months ago.
+
+So I built it as five n8n workflows that treat access as **derived state** —
+computed from my business database, applied to the file server, and
+continuously re-verified. Here's the architecture, the code-sharing trick that
+makes it maintainable, and the security reasoning behind the one public
+endpoint.
+
+## Why it matters
+
+For a digital-goods business, access control *is* the product. You're not
+protecting a warehouse; you're protecting the thing you sold. Two failure modes
+cost real money:
+
+- **Under-granting** — a paying customer can't reach what they bought, and you
+ find out from an angry message rather than a monitoring alert.
+- **Over-granting** — expired or revoked customers keep access, and the leak
+ is invisible until someone resells your catalog.
+
+Manual administration guarantees both, eventually. The fix is to stop treating
+a grant as a *thing you do* and start treating it as a *function of your data*:
+given the customer's current purchases, what should they be able to reach right
+now? Compute it, apply it, and then prove it's still true on a schedule.
+
+## The shape of the system
+
+Five workflows, each with one job:
+
+| Workflow | Trigger | Responsibility |
+|---|---|---|
+| **W1** — Customer Provision & Status Lifecycle | NocoDB webhook (customer row) | Create/update the file-server user; enable or disable on status change |
+| **W2** — CustomerProduct Sync | NocoDB webhook (purchase row) | Recompute and apply that customer's allowed paths |
+| **W3** — Expiry Sync | Cron, every 5 minutes | Sweep expiring entitlements; deselect paths that lapsed |
+| **W4** — Daily Full Reconciliation | Cron, 03:00 + manual webhook | Compare *desired* vs *actual* for everyone; repair drift; log |
+| **W5** — Footer Purchase Check | Public GET webhook | Let the portal show a customer their own purchases; read-only |
+
+The data lives in **NocoDB** (a self-hosted Airtable-style database) as
+Customers, CustomerProducts, Products, Resources, and AccessGrants. The file
+server is **AList**, which gives each customer a user and a role whose
+`permission_scopes` is a list of paths.
+
+The crucial design decision: **NocoDB is the business source of truth. AList is
+only enforcement state.** Sync runs one way. A manual tweak in the AList admin
+UI is not a configuration change — it's drift, and W4 repairs it back.
+
+## The algorithm: what *should* this customer have?
+
+Everything hinges on one function. Desired paths for a customer is the union of
+two sources:
+
+1. Every resource reachable from an **active, not-yet-expired** purchase.
+2. Every resource granted directly via an **active, not-yet-expired** manual grant.
+
+The interesting case is a resource reachable through *two* different products.
+If a customer bought Product A (expires in 30 days) and Product B (expires in
+200 days) and both include the same folder, the correct expiry is the **latest**
+one — buying more of something should never shorten your access to it.
+
+```js
+// For each product's resources, keep the MAX expiry per path.
+for (const c of cps) {
+ if (c.status && c.status !== 'active') continue;
+ if (c.expires_at && String(c.expires_at) <= today) continue; // expired
+ const pid = linkId(c.product);
+ if (!pid) continue;
+ const res = (await ncGet(ctx,
+ `/api/v2/tables/${T_PROD}/links/${LNK_PROD_RES}/records/${pid}`)).list || [];
+ const exp = c.expires_at;
+ for (const r of res) {
+ const p = resPath[r.Id];
+ if (!p) continue;
+ const cur = desired[p];
+ if (!cur || !cur.expires || (exp && exp > cur.expires)) {
+ desired[p] = { expires: exp || null, permission: cur?.permission ?? 0 };
+ }
+ }
+}
+```
+
+Note the boundary: `<= today`. **Expiring today counts as expired.** Off-by-one
+on an expiry check is the difference between a subscription period and a free
+extra day, and if you're not explicit about it you'll get it wrong in the
+customer's favour.
+
+## The problem: n8n Code nodes can't share code
+
+Here's the constraint that shaped the whole codebase. n8n Code nodes are
+self-contained: there's no `require`, no `import`, no filesystem access to a
+shared module. So the obvious structure — one grants algorithm, called by W1,
+W2, W3, and W4 — is exactly what the platform prevents.
+
+Copy-pasting that function into four nodes would have been a guaranteed
+maintenance disaster. Four copies of an expiry rule is four chances for them to
+disagree, and a system that computes access four different ways is worse than
+no system.
+
+The fix was to make the sharing happen at **build time** instead of run time:
+
+- `shared/effective-grants.js` is the single source of truth, and it's written
+ to be self-contained: no `require`, no `process.exit`, no top-level `return`
+ outside a function. It also takes its config through a `ctx` argument
+ (`{ $env, helpers }`) rather than reaching for globals — which has the happy
+ side effect that it's **unit-testable outside n8n entirely**.
+- `gen_w1.js`, `gen_w2.js`, `gen_w3.js`, `gen_w4.js` read that file, inline it
+ into the workflow's Code node body, and update the workflow.
+
+The result is one algorithm, four workflows, zero runtime dependencies — and a
+version of the function that I can test with plain Node before it ever touches
+a live system. The code is duplicated in *artifacts* but never in *sources*,
+which is the same tradeoff a bundler makes.
+
+One n8n-specific wrinkle worth knowing: **custom environment variables are only
+reliably readable via `$env` inside a Code node — `process.env` is not
+dependable there.** That's why every workflow has an explicit "Load Env" node
+that lifts the values it needs onto the item, rather than reading config
+wherever it's convenient.
+
+## Making expiry safe to automate
+
+W3 runs every five minutes and reconciles expiring entitlements. Two details
+make it safe to let a cron job modify live access:
+
+**It only disables when access genuinely ran out.** A naive "if no desired
+paths, disable the user" rule is dangerous — it will happily disable a
+brand-new customer who simply hasn't been granted anything yet. The guard is
+explicit:
+
+```js
+// Only disable when a customer that HAD scopes now has none.
+// currentPaths.length > 0 avoids killing a fresh, not-yet-granted customer.
+const shouldDisable = AUTO_DISABLE && nowEmpty
+ && cust.status === 'active' && currentPaths.length > 0;
+```
+
+**It has a dry-run mode.** `W3_DRY_RUN=true` on the container (or a query
+parameter on the manual webhook for W4) makes the sweep compute and report its
+plan while performing **zero writes**. Being able to ask "what would you do?"
+before letting a schedule do it is the single most useful safety feature I've
+added to any automation.
+
+```js
+const DRY_RUN = ($env.W3_DRY_RUN || '').toLowerCase() === 'true';
+```
+
+## The daily repair: assuming you'll drift
+
+W4 is the workflow I'd argue is the real product. It runs at 03:00, walks every
+customer, and compares desired state against actual state — then repairs the
+difference and **verifies the repair**.
+
+The drift matrix it handles:
+
+- **Should be active** → user must exist (looked up *by username first*, to
+ avoid creating duplicates if a stored ID was lost), be enabled, and carry
+ exactly the computed role scopes.
+- **Should be inactive** → disabled, with scopes emptied. Pending customers
+ stay without a user entirely — no auto-creation.
+- **Dangling references** → a stored user or role ID points at a record that no
+ longer exists. Look it up by name, adopt it if found, recreate if not.
+- **Username drift** → detected and *logged*, never destructive-migrated.
+
+Every repair is verified by re-reading the file server's state afterwards. A
+failed verification is logged as `verify_failed` rather than assumed successful,
+and the reconciliation log only receives **drift, repair, and error rows** —
+healthy customers produce nothing. That last choice is what makes the log
+usable: if it's empty, everything is fine, and you don't have to read past a
+thousand "no change" lines to find the one that matters.
+
+## The public endpoint, and why there's no HMAC
+
+W5 lets the file portal's footer show a signed-in customer their own purchases
+and expiry dates. It's a **public** webhook, and the security reasoning is the
+part I'm most deliberate about.
+
+The obvious instinct is to sign requests with an HMAC. I didn't, and the reason
+is worth stating plainly: **the key would have to ship to the browser, so the
+signature would be theatre.** A shared secret that every client holds protects
+nothing — it just adds ceremony that makes the endpoint *look* verified.
+
+So the endpoint relies on things that actually hold:
+
+- **CORS locked to one origin.** The response carries
+ `Access-Control-Allow-Origin: https://drive.example.com`, so only the portal's
+ own pages can read the response in a browser.
+- **Minimal data by design.** The response returns product name, expiry date,
+ display state, and public folder paths. No internal database IDs, no customer
+ PII, nothing about other customers.
+- **Rate limiting at the edge,** via a WAF rule on the request path, to blunt
+ username enumeration.
+
+And one subtle architectural choice: **the username comes from the caller's own
+session token, decoded client-side**, rather than from a parameter the client
+can set freely. The endpoint never authenticates to the file server, which
+avoids a whole class of connection state and device-registration side effects
+that a server-side login would introduce on every footer render.
+
+The honest framing: this endpoint is not a trust boundary, and I don't pretend
+it is. It shows a customer what they already know about themselves, over a
+response shape that's useless to anyone else.
+
+## The bug that taught me the most: public path, cascading failure
+
+W5 originally fetched data by calling NocoDB at its **public** hostname —
+through a Cloudflare tunnel. It worked in testing. Under real load it produced
+this chain:
+
+1. Public round-trip latency exceeded 60 seconds under load.
+2. nginx upstream timeout fired → **504**.
+3. The footer's client-side fetch had an 8-second timeout and retried **every
+ second**.
+4. Retries piled up connections.
+5. Those connections exhausted the pool → **503** for unrelated requests.
+
+A slow dependency became a cascading outage of a different service. The fix
+was to stop crossing the internet to reach something on the same Docker
+network:
+
+```js
+// n8n and NocoDB share the `bridge_hoelee` network; NocoDB listens on :10380.
+// The public route added CF-tunnel jitter and could exceed the proxy timeout.
+const NOCODB_URL = 'http://nocodb:10380';
+```
+
+**~30 ms internal versus 300 ms+ public, with no tunnel variance** — and the
+entire failure chain disappeared, because the trigger condition (multi-second
+latency) can no longer occur.
+
+The lesson generalises well beyond this stack: **when a service and its
+dependency are in the same container network, the public hostname is a bug
+waiting for load.** And when you see a 503 downstream of a 504, look for a
+client that retries aggressively — the retry loop is usually the amplifier, not
+the original problem.
+
+## What I'd do differently
+
+1. **Build the drift repair first, not last.** I wrote the grant path, then the
+ expiry sweep, then the reconciliation. In retrospect the reconciliation is
+ what makes the other two safe to operate, and it should have existed from
+ day one — because "assume you will drift" is a design stance, not a feature.
+2. **Write the desired-state function before any workflow.** Having it live in
+ `shared/` and be unit-testable *outside* n8n is why the whole system stayed
+ coherent across four workflows. If I'd started by pasting logic into nodes,
+ I'd have shipped four subtly different expiry rules.
+3. **Never cross the public internet between two containers on the same host.**
+ This one cost me a genuine outage, and it's a rule I now apply by default
+ rather than discovering per-integration.
+4. **Put the dry-run switch in from the beginning.** Adding `DRY_RUN` after the
+ fact was easy; operating a scheduled job that mutates access *without* one
+ was an unnecessary few weeks of nerves.
+
+## The result
+
+Five workflows run the full entitlement lifecycle: a purchase in the database
+becomes usable access within seconds, expiries sweep every five minutes, and a
+nightly full reconciliation repairs any drift and verifies each repair. The
+reconciliation log is empty when everything is healthy — which, most days, is
+what it says.
+
+The design principle worth taking away is the one that made it tractable:
+**stop administrating access, and start asserting it.** Define what a customer
+should have as a pure function of your business data, apply that function
+whenever the data changes, and re-assert it on a schedule to catch everything
+else. Then the system doesn't need you to be careful — it just needs you to be
+correct once, in one function.
+
+---
+
+## Want this for your business?
+
+If you sell digital products and you're still granting file access by hand —
+or you're not certain that expired customers lost access last month — I build
+exactly this: self-hosted entitlement systems where access is computed from
+your data, expires on its own, and repairs itself nightly. I work with
+NocoDB, n8n, AList, and Docker, and I hand it back documented so you can
+operate it without me.
+
+Reach me at [me@hoelee.com](mailto:me@hoelee.com?subject=Access%20control%20for%20digital%20goods)
+or WhatsApp [+60 12-797 2969](https://wa.me/60127972969), or see what I do at
+[hoelee.com](https://hoelee.com).
diff --git a/src/content/posts/zh/n8n-v1-to-v2-upgrade-gotchas.md b/src/content/posts/zh/n8n-v1-to-v2-upgrade-gotchas.md
new file mode 100644
index 0000000..fb7e610
--- /dev/null
+++ b/src/content/posts/zh/n8n-v1-to-v2-upgrade-gotchas.md
@@ -0,0 +1,190 @@
+---
+title: "n8n v1 升级到 v2:一个日志文件里的七项废弃警告"
+description: "我的自建 n8n 从 1.x 升级到 2.40.1,一次性暴露出七处静默失效——包括一个被 schema 校验拒绝的遥测事件、废弃的 webhook 变量,以及一个悄悄把 AI 沙盒关掉的数据库覆盖值。"
+pubDate: 2026-09-18
+category: devops
+tags: [n8n, docker, upgrade, self-hosting, debugging, automation]
+ogImage: /og/n8n-v1-to-v2-upgrade-gotchas.png
+banner: /banners/n8n-v1-to-v2-upgrade-gotchas.png
+---
+
+n8n 是我整个自建技术栈的自动化中枢——它负责文件交付权限、数据库备份,还有一套阅读应用依赖的语音合成 API。它一直停留在 `1.123.x` 版本线上快一年了,默默干活,没出过什么问题。
+
+然后我拉了 `n8nio/n8n:2.40.1` 镜像并重启容器。升级本身只花了大约九十秒。而搞清楚它*弄坏了什么*花了我剩下的一整晚——而其中几乎所有信息,n8n 在启动时就已经写在一个日志文件里了。只是我第一次读得不够仔细。
+
+这篇文章就是那个日志文件的解读,让你在规划自己的 v1 → v2 升级时能提前准备,而不是晚上十一点才发现问题。
+
+## 为什么这件事重要
+
+自动化平台的大版本升级,和其他服务升级的性质不一样。n8n 是*运行其他一切东西的那个东西*:它一旦起不来,你的备份、权限同步、内部 API 全部跟着停。更麻烦的是,v2 里坏掉的东西大多不会报错——它只打印一次废弃警告,然后就悄悄换了一种行为方式。
+
+下面这七项,都是在一个真实的、有点混乱的、生产形态的安装上实际命中的。其中三项改变了我的技术栈的行为。还有一项,悄悄把一个功能*关掉了*。
+
+## 从这里开始:n8n 在启动时会告诉你哪里不对
+
+在动任何工作流之前,先从头读容器日志。在 v2 全新的启动过程中,n8n 会明确打印出一整块废弃警告:
+
+```text
+There are deprecations related to your n8n setup. Please take the recommended
+actions to update your configuration:
+ - WEBHOOK_URL -> Use N8N_WEBHOOK_URL instead, which sets the base URL for
+ both test and production webhooks.
+ - N8N_UNVERIFIED_PACKAGES_ENABLED -> The default for this variable will
+ change to `false` in a future version.
+ - N8N_RUNNERS_MODE -> Internal task runner mode is deprecated and will be
+ removed in a future version.
+ - N8N_RUNNERS_TASK_TIMEOUT -> The default for this variable will be reduced
+ from 300 (5 minutes) to 60 (1 minute) in a future version.
+ - N8N_COMPRESSION_NODE_MAX_DECOMPRESSED_SIZE_BYTES -> The default will be
+ reduced from 2 GiB to 256 MiB in a future version.
+ - N8N_COMPRESSION_NODE_MAX_ZIP_ENTRIES -> The default will be reduced from
+ 5000 to 1000 in a future version.
+```
+
+这一块就是你的迁移清单。我那七项坑里的六项都在里面。
+
+## 1. 你的环境变量值现在可能过不了 schema 校验
+
+这是最让我困惑的一项,因为它看起来完全像一条无意义的报错:
+
+```text
+Telemetry event "Instance started" failed schema validation:
+execution_variables.executions_data_save_on_error: Invalid option:
+expected one of "all"|"none"
+```
+
+我当时设置的是 `EXECUTIONS_DATA_SAVE_ON_ERROR=error`——这个值在 v1 里完全合法,而且是我刻意选的,因为对一台繁忙的实例来说,「只保存失败的执行记录」是合理的默认策略。但在 v2 里这个值已经不在允许集合内了,现在的合法值是 `all` 或 `none`。
+
+失败方式才是关键。它没有崩溃。它甚至没有以一眼就能看出是错误的方式发出警告——它抛出的是一条 *telemetry schema validation* 消息,听起来像是 n8n 内部的问题,而不是我的配置问题。这个设置实际上被忽略了。
+
+修法是把你的意图放到一个说得通的地方:选一个合法的值,然后用 pruning 控制数据量。
+
+```env
+EXECUTIONS_DATA_SAVE_ON_ERROR=all
+EXECUTIONS_DATA_SAVE_ON_SUCCESS=none
+EXECUTIONS_DATA_PRUNE=true
+EXECUTIONS_DATA_MAX_AGE=336
+EXECUTIONS_DATA_PRUNE_MAX_COUNT=10000
+```
+
+**教训:** 在 v2 里,把你的环境变量当成一套有类型的、带 schema 的接口来看待。一个非法的值可能被静默丢弃,而不是大声拒绝。
+
+## 2. `WEBHOOK_URL` 已被 `N8N_WEBHOOK_URL` 取代
+
+如果你的 webhook 发布在反向代理后面——你几乎肯定是的,因为那是它们变得可访问的方式——那么这个 base URL 变量就是关键路径上的东西。它决定了 n8n 报告的是*公网* webhook 路径,还是 `http://localhost:5678/...`。
+
+旧名字目前还能用,所以这一项不会立刻咬你。但注意它的措辞:新变量为**测试和生产 webhook 同时**设置 base URL。在我的环境里这两者在行为上已经出现了分叉,而这次的合并正是为了消除这一类 bug。
+
+```env
+# 升级前(仍可用,但已废弃)
+WEBHOOK_URL=https://auto.example.com/
+
+# 升级后
+N8N_WEBHOOK_URL=https://auto.example.com/
+```
+
+## 3. Internal task runner 模式要取消了——而我的其实早就坏了
+
+这一项一直在我的日志里,就在废弃警告块上面几行,而我已经读过去好几个月了:
+
+```text
+Failed to start Python task runner in internal mode. because Python 3 is
+missing from this system. Launching a Python runner in internal mode is
+intended only for debugging and is not recommended for production.
+```
+
+如果你的任何工作流用了 **Python** 的 Code 节点,那它根本就没在 internal 模式下跑起来过——官方镜像里没有 Python。JavaScript 的 Code 节点是正常的(JS runner 会正常注册),所以这个问题可以无限期不被发现:一切*看起来*都是健康的。
+
+v2 把方向挑明了:切到 `external` 模式,和一个独立启动器进程共享一个 auth token。
+
+```env
+N8N_RUNNERS_MODE=external
+N8N_RUNNERS_AUTH_TOKEN=<一串足够长的随机字符串>
+```
+
+**教训:** 「internal 模式已废弃」是标题,但真正的发现是:某一类 runner 可能已经*静默失效*好几个月了。要检查 `docker logs` 里的 runner 注册那行,而不是只看容器起没起来。
+
+## 4. Task 超时默认值从 300 秒降到 60 秒
+
+这一项是我最想替所有有慢工作流的人标红的:
+
+```text
+N8N_RUNNERS_TASK_TIMEOUT -> The default for this variable will be reduced
+from 300 (5 minutes) to 60 (1 minute) in a future version.
+```
+
+对自己的负载诚实一点。你有没有 Code 节点要遍历几千条记录,或者要调一个很慢的上游接口?我有——一个每夜的对账流程会遍历每个客户,每条记录调一次外部 API。在未来的某次升级中,它会在六十秒处停下,而我这边没有任何配置变更。
+
+趁你人还在这个文件里,现在就显式设置:
+
+```env
+N8N_RUNNERS_TASK_TIMEOUT=300
+```
+
+对所有「默认值将会改变」形式的废弃警告,通用原则是:**如果你依赖当前的默认值,就把它显式钉死。** 否则这次升级就是一个静默的行为变更,而你会把它当成 bug 来排查,而不是认出它是一个过期的默认值。
+
+## 5 和 6. 两个压缩节点上限缩水(2 GiB → 256 MiB,5000 → 1000 条)
+
+这两项一起出现,只有当你在工作流里处理大负载的压缩/解压节点时才相关——而当你在工作流里搬运数据库导出或归档文件时,很容易就变成相关。
+
+```text
+N8N_COMPRESSION_NODE_MAX_DECOMPRESSED_SIZE_BYTES -> reduced from 2 GiB to
+256 MiB in a future version.
+N8N_COMPRESSION_NODE_MAX_ZIP_ENTRIES -> reduced from 5000 to 1000 in a
+future version.
+```
+
+内存上限降到八分之一,条目上限降到五分之一。不会报错;节点只是在一个你没设定的阈值处拒绝执行。如果你接近任何一个上限,两个都钉死。
+
+## 7. 存储路径将在 v3 改名——而你的卷正好挂在旧路径上
+
+这不是 v2 的破坏性变更,但 v2 会警告它,而且是真正涉及数据规划的那一项:
+
+```text
+Deprecation warning: The storage directory "/home/node/.n8n/binaryData" will
+be renamed to "/home/node/.n8n/storage" in n8n v3. To migrate now, set
+N8N_MIGRATE_FS_STORAGE_PATH=true. If you have a volume mounted at the old
+path, update your mount configuration after migration.
+```
+
+再读一遍最后一句:*如果你有一个卷挂载在旧路径上,请在迁移后更新你的挂载配置。* 如果你设了迁移标志却保留旧的 bind mount,你现在就有两个目录,而你的二进制数据住在容器实际指向的那一个里。把改名和挂载变更放在同一个维护窗口里做——不要一个现在、一个「以后」。
+
+## 那个不在日志里的:我的 AI 沙盒把自己关掉了
+
+这是跟废弃警告毫无关系的发现,也是如果我不读完整的启动序列就永远抓不到的一项:
+
+```text
+Sandbox: enabled=false provider=n8n-sandbox (DB override; env was enabled=true
+provider=n8n-sandbox)
+```
+
+我的环境变量说已启用。数据库说不是。**数据库赢了。**
+
+环境变量是 `N8N_INSTANCE_AI_SANDBOX_ENABLED=true`,在容器上依然设置正确。但有一个持久化在 n8n 自己的配置存储里的值,在启动时覆盖了它——而这个冲突唯一被报告的地方,就是一行日志里的一个括号。
+
+这是一个超越 n8n 的、非常有价值的排查教训:当一个功能明明环境变量设对了却是关闭状态,就怀疑存在一个**优先级高于环境的持久化设置层**。容器配置不总是最终答案。在日志里 grep 那个功能名,不要看到环境变量就停下。
+
+## 我会怎么做得不一样
+
+1. **在宣布升级完成之前先读启动日志。** 所有对我重要的废弃警告,都在第一次启动时、在一个块里、清清楚楚打印出来了。我的 v1 习惯是检查「它起来了吗、工作流跑得动吗」——而这恰恰是漏掉全部七项的检查方式。
+2. **把「默认值将会改变」当成待办事项,而不是警告。** 七项里有五项是未来的默认值变更。现在钉死它们只需要改一次配置,却能把以后的一次神秘故障转化为一次配置 diff。
+3. **把容器配置和 n8n 自己存储的配置做对比。** 沙盒覆盖这件事教会我:env 只是两个输入之一,不是真相来源。当行为和配置不一致时,相信行为,然后去找那个优先级更高的层。
+4. **钉死镜像 tag,并保留上一个。** 我是从 v1 直接跳到 `2.40.1` 的。本地留着旧镜像,是回滚能变成一条 `docker run` 而不是一次重新构建的原因。
+
+## 结果
+
+n8n 现在跑在 2.40.1 上,整块废弃警告都已处理:schema 非法的值已修正,`N8N_WEBHOOK_URL` 已就位,JS runner 正常注册,task 超时和压缩上限都已钉死,让下一次升级变成一次无操作而不是一次意外。
+
+下游一切都还在工作——文件权限同步、每夜备份、语音合成接口。这才是这类升级值得追求的结果:不是「它起来了」,而是「它起来了,*而且*接下来三次升级的成本已经预付了」。
+
+真正让人不舒服的地方在于,这些我本可以提前知道的信息占了多大比例。n8n 主动把整份清单递给了我,就在启动时。升级从来不是难的部分——读输出才是。
+
+---
+
+## 需要不用你天天盯着的自动化?
+
+我搭建并维护自建自动化——n8n 工作流、Docker 技术栈,以及那些从来不是设计来互相通信的应用之间的胶水层。如果你正面临一次大版本升级,或者你有那种「能用,直到不能用」的自动化,我会规划迁移、在维护窗口执行、并把每一个配置决策都记录下来,让下一次升级变得无聊。
+
+欢迎联系 [me@hoelee.com](mailto:me@hoelee.com?subject=n8n%20%E5%8D%87%E7%BA%A7)
+或 WhatsApp [+60 12-797 2969](https://wa.me/60127972969),也可以看看我在
+[hoelee.com](https://hoelee.com) 做什么。
diff --git a/src/content/posts/zh/running-tts-as-a-service-with-token-sidecars.md b/src/content/posts/zh/running-tts-as-a-service-with-token-sidecars.md
new file mode 100644
index 0000000..cd9b219
--- /dev/null
+++ b/src/content/posts/zh/running-tts-as-a-service-with-token-sidecars.md
@@ -0,0 +1,214 @@
+---
+title: "我用两个 Cron 容器跑了一年的语音合成服务"
+description: "一个阅读应用需要 TTS,而云服务免费版的 token 十分钟、一小时就过期。这是那套让 Azure 和 Google 语音稳定运行一年的边车模式——任何工作流里都没有存放密钥。"
+pubDate: 2026-09-18
+category: ai
+tags: [n8n, tts, azure, google-cloud, docker, automation, sidecar]
+ogImage: /og/running-tts-as-a-service-with-token-sidecars.png
+banner: /banners/running-tts-as-a-service-with-token-sidecars.png
+---
+
+我自建了一个电子书阅读服务器。它有个「朗读」功能,内置的引擎能用但很机械。所以我把它接到了真正的神经网络语音上——Azure Speech 和 Google Cloud TTS——通过我的 n8n 实例。
+
+那是大约一年前的事。它此后一直在运行,而且设计几乎没变过。这篇文章讲它怎么工作,更有用的是讲它为什么长成这样:整个架构的存在就是为了解决一个具体问题,而这个问题会在一小时之内击垮天真的实现方式。
+
+## 为什么这件事重要
+
+语音合成是个加进去很愉快、但维持起来出乎意料麻烦的功能。低用量下这些语音要么便宜要么免费,音质出色,API 也直截了当——直到你发现 OAuth access token 不是一种你可以直接存进配置文件的凭据。
+
+两家服务商都会签发短期 access token,而有效期差别巨大:
+
+- **Azure Speech** 免费层:大约 **10 分钟**。
+- **Google Cloud**:大约 **1 小时**。
+
+如果你把 token 放进一个工作流变量里,这个集成会漂亮地工作十分钟,然后开始永远返回 401。这就是那种「演示完美、生产失败」的集成的经典形状——而解法不是「记得刷新 token」,因为你不会记得。
+
+## 架构:刷新 token 不是工作流的事
+
+让这套东西能工作的设计决策,是拒绝让工作流管理凭据。取而代之,两个极小的容器独占 token 生命周期,把当前 token 写到一个共享卷上的文件里。工作流只负责读文件。
+
+```text
+┌──────────────────┐ GET /webhook/{mtts|gtts}?pass=…&text=…&speed=…
+│ 阅读服务器 │ ───────────────────────────────────────────────┐
+│ (httpTTS 引擎) │ │
+└──────────────────┘ ▼
+ ┌───────────────────────────┐
+ │ n8n │
+ │ ├ /mtts (Microsoft) │
+ │ └ /gtts (Google) │
+ └───────┬───────────────────┘
+ 读取 accesstoken.txt
+ ┌─────────────┴─────────────┐
+ ▼ ▼
+ Azure Speech (F0) Google Cloud TTS
+ southeastasia 区域 cmn-CN Wavenet
+ │ │
+ └──────── WAV 音频 ─────────┘
+ │
+ 回到播放器
+```
+
+两个 cron 边车容器负责保持 token 新鲜:
+
+| 容器 | 镜像 | 间隔 | 写入 |
+|---|---|---|---|
+| `cron-azure-refresh` | `curlimages/curl` | 每 ~570 秒 | `MicrosoftTTS/accesstoken.txt` |
+| `cron-gcloud-refresh` | `google/cloud-sdk:slim` | 每 ~3500 秒 | `GoogleTTS/accesstoken.txt` |
+
+两者都 bind-mount 了**同一个宿主目录**,n8n 通过它的 Read/Write Files 节点读取这个目录。这个共享卷就是凭据层与工作流层之间的完整接口。
+
+为什么间隔是这个数字:570 秒对约 600 秒的 Azure 有效期,留出 30 秒安全边际;而永远略微*提前*刷新,远比卡着到期点刷新稳健得多。Google 的 3500 秒对一小时是同样的道理。
+
+```yaml
+cron-azure-refresh:
+ image: curlimages/curl:8.10.1
+ restart: unless-stopped
+ volumes:
+ - /volume1/docker/n8n/file:/file
+ entrypoint: /bin/sh
+ command: >
+ -c 'while true; do
+ curl -s -X POST "https://southeastasia.api.cognitive.microsoft.com/sts/v1.0/issueToken"
+ -H "Ocp-Apim-Subscription-Key: $AZURE_SPEECH_KEY" > /file/MicrosoftTTS/accesstoken.txt;
+ sleep 570;
+ done'
+```
+
+## 为什么用文件,而不是那些显而易见的替代方案
+
+**为什么不把 token 存进 n8n 凭据、在工作流里刷新?** 因为刷新逻辑就会被复制进每一个需要 token 的工作流,而每一份拷贝都需要自己的错误处理。当凌晨三点刷新失败时,你希望只有一个进程需要关心这件事。
+
+**为什么不让工作流每次请求都去调 token 接口?** 可行,而且它让每次朗读请求的延迟和依赖面都翻倍。更糟的是,它意味着 token 接口的一次抖动就变成一次 TTS 故障。
+
+**那为什么用文件?** 因为它是双方都已经支持的最简单接口。n8n 有内置的 Read/Write Files 节点;cron 容器可以用 `curl` 和 shell 重定向写入。没有队列、没有数据库表、没有共享库——只有一个内容永远是当前 token 的文件。
+
+这笔取舍是诚实的:每次请求读一次文件,是热路径上的一次磁盘读。在一个阅读应用的请求频率下,这完全是免费的,而它换来的是凭据生命周期与请求处理之间的彻底解耦。
+
+## n8n 这一侧:两个工作流,一种形状
+
+两个 TTS 工作流骨架相同,值得走一遍,因为细节才是有意思的地方。
+
+**1. `responseMode: responseNode` 的 webhook。** 工作流必须返回原始音频字节而不是 JSON,所以响应由一个显式的 Respond to Webhook 节点控制,而不是 n8n 的默认行为。
+
+**2. 一道密码闸门。** 一个查询参数会与期望值比对,不匹配时返回真正的 403,而不是一个空的 200:
+
+```text
+Respond to Webhook → text: "403 unauthorized", responseCode: 403
+```
+
+`pass` 值就在 URL 里,这一点我稍后会诚实交代。
+
+**3. 读取 token 文件。** `Read/Write Files from Disk` 读取
+`/home/user/file/MicrosoftTTS/accesstoken.txt`。然后两个节点做清理:
+`Extract from File`(文本模式),以及一个删掉换行的 Set 节点——因为
+`Authorization` 头里一个尾随的 `\n` 会产生一个令人抓狂、且看起来完全不像空白字符问题的 401:
+
+```js
+// Edit Fields 节点
+{{ $json.data.replace(/(\r\n|\n|\r)/g, '') }}
+```
+
+**4. 调用服务商。** 对 Azure 来说,请求体是插入了语音和语速的 SSML:
+
+```xml
+
+
+
+ {{ $('Webhook').item.json.query.text }}
+
+
+
+```
+
+**5. 以二进制响应返回音频**,并带上正确的内容类型:
+
+```text
+Respond to Webhook → binary, set
+ Content-Type: audio/wav
+ Content-Disposition: filename="output.wav"
+```
+
+## 映射问题:客户端说的是另一种语言
+
+这个细节花的心思比 API 调用本身还多。阅读应用发送一个 `speed` 值,用的是它自己的刻度——5 到 50,因为那是它 UI 滑块产生的范围。Azure 想要的是百分比的 prosody rate,而 Google 想要的是一个约等于 1.0 的 `speakingRate` 乘数。
+
+两家服务商的刻度都和应用的刻度不一致。所以这里有一个刻意的转换步骤,而这一步值得照抄,因为把 UI 控件映射到 API 参数是一个反复出现的琐事:
+
+```js
+// 把阅读器的 5–50 速度滑块映射到 Azure 的 -20%…+150% 语速区间。
+const inMin = 5, inMax = 50;
+const outMin = -20, outMax = 150;
+
+// 映射之前先钳制输入,这样客户端一个越界的值不会产生荒谬的 prosody rate。
+if (speed < inMin) speed = inMin;
+if (speed > inMax) speed = inMax;
+
+const mapped = ((speed - inMin) / (inMax - inMin)) * (outMax - outMin) + outMin;
+// → rate: `${Math.round(mapped)}%`
+```
+
+Google 那边则是直接相除,因为它的刻度在同一区间里接近线性:
+
+```js
+speakingRate: speed / 25 // Google 期望约 1.0,而不是百分比
+```
+
+两家服务商、两套单位制、一个客户端概念。把映射留在工作流里(而不是要求客户端了解 Azure 的百分比),正是让阅读应用保持服务商无关的原因——也是我能在完全不动应用的情况下加上第二个服务商的原因。
+
+此外还有一张语音表,因为客户端发送的是整数索引而不是语音名:
+
+```js
+const voices = [
+ "zh-CN-XiaochenMultilingualNeural", // 1
+ "zh-CN-XiaoxiaoMultilingualNeural", // 2
+ // ...
+ "zh-CN-XiaoshuangNeural", // 7(女声,儿童)
+ "zh-CN-XiaoyouNeural" // 8(女声,儿童)
+];
+```
+
+八种语音——六种成人、两种儿童——可从阅读应用 UI 选择。工作流会把索引钳制进范围,而不是信任它,这和语速钳制是同一个防御习惯。
+
+## 安全方面,诚实作答
+
+闸门是一个 `pass` 查询参数,比对一个固定字符串。我不打算美化它:**这就是一个放在 URL 里的共享密钥。** 它阻止了针对一个每次请求都要花我钱的接口的随意滥用。它阻止不了任何能读到阅读应用配置的人,也扛不住认真的攻击者。
+
+我能接受这一点,是因为它所保护的东西。最坏的结果是有人烧掉我的免费层 TTS 配额——一件烦人事,不是数据泄露。这个接口后面没有客户数据,也没有对任何东西的特权访问。为真正的认证(OAuth、签名请求、按用户限流)付出的代价,远超这点暴露所值。
+
+可迁移的习惯是:**明确说出**一道闸门属于哪一层级——这是**滥用威慑**,不是授权。系统出问题,往往是因为把威慑误当成了边界。如果这个接口碰到客户记录或文件访问,它就需要真正的认证——而我会从一开始就设计得不一样。
+
+## 隐形成本:它被钉在一个早已过时的分支上
+
+这项服务跑了将近一年,零代码改动。这既是好消息也是坏消息:
+
+```text
+n8nio/n8n:1.123.72
+```
+
+那是 TTS 文档里写明的镜像。它一直能用,所以也就没有任何东西促使我重新审视它——这正是「*过于*可靠」的基础设施的经典失败模式。当我终于把 n8n 升到 v2 时,这是整个技术栈里最后一处 1.x 时代的引用,而 token 刷新容器恰恰是最容易受平台行为悄悄变化影响的那部分。
+
+教训不是「升级更勤一点」——而是:**一个没有活动部件的服务,没有任何自然契机去重审它的假设。** 给自己设个日历提醒去复查钉死的依赖版本,因为系统本身永远不会告诉你。
+
+## 我会怎么做得不一样
+
+1. **把映射逻辑和应用一起版本化,而不是埋在某个工作流里。** 语速映射和语音表编码的是阅读应用的 UI 契约。它们活在 n8n Code 节点里的 JavaScript 中,任何做应用的人在那边都看不见——而如果滑块范围哪天变了,没有任何东西会告诉我。
+2. **加一个真正做一次合成请求的健康检查端点。** 我目前能查的只有容器起没起来。token 可以存在于文件里但*依然*是过期的(如果某次刷新静默失败了),而在用户撞上之前,没有便宜的办法发现这一点。
+3. **把钉死的版本记录在一个我真的会看的地方。** `1.123.72` 在一个 markdown 文件里躺了一年。运维手册里加一行「复查钉死的镜像」,就能让它在下一个维护窗口浮现,而不是通过一次迁移。
+
+## 结果
+
+一年可用性、两个容器、一个共享目录,以及八种阅读应用可以用滑块选择的神经语音。请求含服务商往返在内几秒内完成,而整个东西除免费层之外零成本——因为 token 生命周期问题被一次性解决在了正确的地方。
+
+这个模式可以泛化到任何短期凭据:**别教会每一个调用方去刷新 token——跑一个唯一职责就是让某个文件保持最新的进程,让其他所有人读这个文件。** 它不聪明,而这就是重点。聪明的凭据处理,正是你最后会得到四份刷新实现、其中三份是错的的原因。
+
+---
+
+## 需要在现有应用里接上 AI 功能?
+
+我搭建 AI 集成里那些不性感的中层——也就是演示之后还能继续工作的那部分:token 刷新、服务商故障转移、语音与模型映射、限流处理。如果你想把语音合成、转写或 LLM 功能加进一个现有应用,并且希望它明年还能用,那就是我在做的事。
+
+欢迎联系 [me@hoelee.com](mailto:me@hoelee.com?subject=TTS%20%E9%9B%86%E6%88%90)
+或 WhatsApp [+60 12-797 2969](https://wa.me/60127972969),也可以看看我在
+[hoelee.com](https://hoelee.com) 做什么。
diff --git a/src/content/posts/zh/self-healing-digital-goods-entitlements.md b/src/content/posts/zh/self-healing-digital-goods-entitlements.md
new file mode 100644
index 0000000..37dfe18
--- /dev/null
+++ b/src/content/posts/zh/self-healing-digital-goods-entitlements.md
@@ -0,0 +1,181 @@
+---
+title: "数字商品的自愈式访问控制:NocoDB、n8n 与 AList"
+description: "我如何用五个工作流搭出一套权限系统,自动授予、到期失效并持续修复客户的文件访问——包括一个构建期代码共享技巧,以及一个设计上刻意不泄露任何信息的公开接口。"
+pubDate: 2026-09-18
+category: case-studies
+tags: [n8n, nocodb, alist, access-control, docker, automation, digital-goods]
+ogImage: /og/self-healing-digital-goods-entitlements.png
+banner: /banners/self-healing-digital-goods-entitlements.png
+---
+
+我卖数字产品——文件、美术素材、授权资源——并通过一个自建的文件门户交付。真正的难题从来不是存储,而是**授权**:确保客户在购买时拿到恰好他们买下的目录,拿到恰好他们付费的时长,并且在停止付费时访问权限真的被收回。
+
+手工做这件事,前十个客户没问题。一百个就不行了,而且失败的方�式很具体、也很讨厌:静默失败。订阅到期而客户继续下载时,不会有任何报错。你只是一直在给一个几个月前就停止付费的人供文件。
+
+所以我把它做成五个 n8n 工作流,把访问权限当作**派生状态**来处理——从我的业务数据库计算得出,应用到文件服务器上,并持续重新校验。下面是它的架构、让它可维护的代码共享技巧,以及那个公开接口背后的安全推理。
+
+## 为什么这件事重要
+
+对一个数字商品生意来说,访问控制*就是*产品本身。你保护的不是一座仓库,你保护的是你卖掉的那个东西。两种失败模式会真金白银地亏钱:
+
+- **授予不足**——付费客户拿不到他买的东西,而你是从愤怒的消息里而不是监控告警里知道的。
+- **授予过度**——已到期或已撤销的客户继续有访问权,而泄露会一直不可见,直到有人转卖你的目录。
+
+手工管理最终会必然地导致这两种。解法是不要再把一次授权当成*你做的一件事*,而把它当成*你数据的一个函数*:给定客户当前的购买记录,他现在应该能访问什么?算出来、应用它,然后按计划证明它仍然成立。
+
+## 系统的形状
+
+五个工作流,每个只做一件事:
+
+| 工作流 | 触发器 | 职责 |
+|---|---|---|
+| **W1** — Customer Provision & Status Lifecycle | NocoDB webhook(客户行) | 创建/更新文件服务器用户;状态变更时启用或禁用 |
+| **W2** — CustomerProduct Sync | NocoDB webhook(购买行) | 重新计算并应用该客户被允许的路径 |
+| **W3** — Expiry Sync | Cron,每 5 分钟 | 扫描即将到期的授权;移除已失效的路径 |
+| **W4** — Daily Full Reconciliation | Cron 03:00 + 手动 webhook | 对比全体客户的*期望*与*实际*;修复漂移;记录日志 |
+| **W5** — Footer Purchase Check | 公开 GET webhook | 让门户展示客户自己的购买记录;只读 |
+
+数据存在 **NocoDB**(一个自建的 Airtable 类数据库)里,包括 Customers、CustomerProducts、Products、Resources 和 AccessGrants 几张表。文件服务器是 **AList**,它为每个客户提供一个用户和一个角色,角色的 `permission_scopes` 就是一串路径。
+
+至关重要的设计决策:**NocoDB 是业务真相来源,AList 只是执行状态。** 同步是单向的。在 AList 管理后台手动改一笔不算配置变更——那叫漂移,W4 会把它修复回去。
+
+## 算法:这个客户*应该*有什么?
+
+一切都系于一个函数。客户的期望路径是两个来源的并集:
+
+1. 所有可从**有效的、未过期**购买记录到达的资源。
+2. 所有通过**有效的、未过期**手工授权直接授予的资源。
+
+有意思的情况是:同一个资源可以通过*两个不同*产品到达。如果客户买了产品 A(30 天后到期)和产品 B(200 天后到期),而两者都包含同一个目录,那么正确的到期时间是**较晚**的那一个——多买一样东西永远不该缩短你对它的访问权。
+
+```js
+// 对每个产品的资源,按路径保留最大的 expires_at。
+for (const c of cps) {
+ if (c.status && c.status !== 'active') continue;
+ if (c.expires_at && String(c.expires_at) <= today) continue; // 已过期
+ const pid = linkId(c.product);
+ if (!pid) continue;
+ const res = (await ncGet(ctx,
+ `/api/v2/tables/${T_PROD}/links/${LNK_PROD_RES}/records/${pid}`)).list || [];
+ const exp = c.expires_at;
+ for (const r of res) {
+ const p = resPath[r.Id];
+ if (!p) continue;
+ const cur = desired[p];
+ if (!cur || !cur.expires || (exp && exp > cur.expires)) {
+ desired[p] = { expires: exp || null, permission: cur?.permission ?? 0 };
+ }
+ }
+}
+```
+
+注意边界:`<= today`。**当天到期就算已过期。** 到期检查上的一个差一错误,就是订阅期与白送一天之间的区别;如果你不把它写明确,你一定会写错,而且会错在对客户有利的那一边。
+
+## 问题所在:n8n 的 Code 节点无法共享代码
+
+这个约束塑造了整个代码库。n8n 的 Code 节点是自包含的:没有 `require`、没有 `import`、也没有访问共享模块的文件系统权限。所以最自然的结构——一个授权算法,被 W1、W2、W3、W4 调用——恰恰是这个平台不让你做的事。
+
+把那个函数复制粘贴进四个节点,注定是一场维护灾难。四份到期规则的拷贝,就是四次它们互相不一致的机会,而一个用四种方式计算访问权限的系统,比没有系统更糟。
+
+解法是让共享发生在**构建期**而不是运行期:
+
+- `shared/effective-grants.js` 是唯一真相来源,而且它被写成自包含的:没有 `require`、没有 `process.exit`、函数外没有顶层 `return`。它通过一个 `ctx` 参数(`{ $env, helpers }`)接收配置,而不是去抓全局变量——这带来一个令人愉快的副作用:它**可以在 n8n 之外做单元测试**。
+- `gen_w1.js`、`gen_w2.js`、`gen_w3.js`、`gen_w4.js` 读取这个文件,把它内联进工作流的 Code 节点主体,并更新工作流。
+
+结果是:一个算法、四个工作流、零运行期依赖——而且这个函数可以在碰到任何线上系统之前,用纯 Node 测试。代码在*产物*里重复,但在*源码*里从不重复,这和打包器做的是同一笔交易。
+
+一个 n8n 特有的细节值得知道:**自定义环境变量只有在 Code 节点里通过 `$env` 才可靠可读——`process.env` 在那里不可靠。** 这就是为什么每个工作流都有一个显式的「Load Env」节点,把它需要的值提升到 item 上,而不是哪里方便就在哪里读配置。
+
+## 让到期处理可以安全自动化
+
+W3 每五分钟跑一次,对即将到期的授权做对账。有两个细节让一个 cron 任务可以安全地改动线上权限:
+
+**只有当访问权限真的用完时才禁用。** 一个天真的「如果没有期望路径就禁用用户」规则是危险的——它会乐于禁用一位只是还没被授予任何东西的新客户。守卫条件是显式的:
+
+```js
+// 只在「曾经有权限、现在一个都没有」的客户身上禁用。
+// currentPaths.length > 0 避免干掉一个全新的、尚未授权的客户。
+const shouldDisable = AUTO_DISABLE && nowEmpty
+ && cust.status === 'active' && currentPaths.length > 0;
+```
+
+**它有 dry-run 模式。** 容器上的 `W3_DRY_RUN=true`(或 W4 手动 webhook 上的查询参数)会让扫描计算并报告它的计划,同时**执行零写入**。能在让一个计划任务动手之前先问「你会做什么?」,是我在任何自动化里加过的最有用的安全功能。
+
+```js
+const DRY_RUN = ($env.W3_DRY_RUN || '').toLowerCase() === 'true';
+```
+
+## 每日修复:假定你一定会漂移
+
+W4 是那个我会说才是真正产品的工作流。它在 03:00 运行,遍历每个客户,对比期望状态与实际状态——然后修复差异并**验证修复**。
+
+它处理的漂移矩阵:
+
+- **应当是有效的** → 用户必须存在(**先用用户名查找**,以避免在存储的 ID 丢失时创建重复用户)、处于启用状态,并且恰好带有计算出的角色权限范围。
+- **应当是无效的** → 禁用,并清空权限范围。待处理客户完全不建用户——不自动创建。
+- **悬空引用** → 存储的用户或角色 ID 指向一个已不存在的记录。按名字查找,找到就收编,找不到就重建。
+- **用户名漂移** → 只检测并*记录日志*,绝不做破坏性迁移。
+
+每一次修复之后都会重新读取文件服务器状态来验证。失败验证会被记为 `verify_failed` 而不是假定成功,而对账日志只接收**漂移、修复和错误三类记录**——健康客户不产生任何行。最后这个选择正是让日志可用的原因:如果它是空的,一切正常,你不必读着一千行「无变化」去找那一条重要的。
+
+## 那个公开接口,以及为什么没有 HMAC
+
+W5 让文件门户的页脚能展示已登录客户自己的购买记录和到期日期。它是一个**公开** webhook,而它的安全推理是我最刻意对待的部分。
+
+直觉是用 HMAC 给请求签名。我没这么做,理由值得直说:**密钥必须发到浏览器,所以签名只是表演。** 一个每个客户端都持有的共享密钥什么都保护不了——它只增加了一层仪式,让这个接口*看起来*经过验证。
+
+所以这个接口依赖的是真正成立的东西:
+
+- **CORS 锁定单一来源。** 响应带有
+ `Access-Control-Allow-Origin: https://drive.example.com`,所以只有门户自己的页面能在浏览器里读取响应。
+- **数据最小化是设计出来的。** 响应只返回产品名、到期日期、展示状态和公开目录路径。没有内部数据库 ID、没有客户个人信息、没有任何其他客户的信息。
+- **边缘限流**,通过请求路径上的一条 WAF 规则,削弱用户名枚举。
+
+还有一个细微的架构选择:**用户名来自调用者自己的会话 token,在客户端解码**,而不是来自一个客户端可以自由设置的参数。这个接口从不向文件服务器认证,从而避开了一整类连接状态与设备注册副作用——否则每次页脚渲染都会引入这些副作用。
+
+诚实的说法是:这个接口不是信任边界,我也不假装它是。它向客户展示的只是他们本来就知道的关于自己的信息,走的是一个对任何其他人无用的响应形状。
+
+## 教会我最多的那个 bug:公网路径与级联故障
+
+W5 最初是通过 NocoDB 的**公网**主机名——穿过一个 Cloudflare 隧道——去取数据的。测试中它工作良好。在真实负载下,它产生了这样一条链:
+
+1. 公网往返延迟在负载下超过 60 秒。
+2. nginx 上游超时触发 → **504**。
+3. 页脚客户端 fetch 的 8 秒超时**每秒重试一次**。
+4. 重试堆积了连接。
+5. 这些连接耗尽连接池 → 无关请求收到 **503**。
+
+一个慢依赖变成了一次不同服务的级联故障。修法是停止跨过整个互联网去访问同一个 Docker 网络上的东西:
+
+```js
+// n8n 与 NocoDB 同处 bridge_hoelee 网络;NocoDB 监听 :10380。
+// 走公网路由会引入 CF 隧道抖动,且可能超过代理超时。
+const NOCODB_URL = 'http://nocodb:10380';
+```
+
+**内部约 30 ms,公网 300 ms 以上,且没有隧道抖动**——整条故障链消失了,因为触发条件(数秒级延迟)已不可能出现。
+
+这个教训可以很好地泛化到这个技术栈之外:**当一个服务和它的依赖在同一个容器网络里时,用公网主机名就是一个等着负载来触发的 bug。** 而当你看到一个 503 出现在一个 504 的下游时,去找那个激进重试的客户端——重试循环通常才是放大器,而不是原始问题。
+
+## 我会怎么做得不一样
+
+1. **先做漂移修复,而不是最后做。** 我先写授权路径,然后到期扫描,最后才对账。回头看,对账才是让另外两个可以安全运行的东西,它应该从第一天就存在——因为「假定你一定会漂移」是一种设计立场,不是一个功能。
+2. **在任何工作流之前先写期望状态函数。** 它能待在 `shared/` 里、能在 n8n *之外*做单元测试,正是整个系统在四个工作流之间保持连贯的原因。如果我一开始就把逻辑粘贴进节点,我就会发布四份微妙的、彼此不同的到期规则。
+3. **永远不要在同一台主机的两个容器之间走公网。** 这条让我真的经历了一次故障,而它现在是我默认应用的规则,而不是每个集成都要重新发现一次。
+4. **从第一天就把 dry-run 开关放进去。** 事后加 `DRY_RUN` 很容易;而在没有它的情况下运行一个会改动权限的计划任务,是几周完全不必要的提心吊胆。
+
+## 结果
+
+五个工作流跑完整个授权生命周期:数据库里的一笔购买在几秒内变成可用访问,到期每五分钟扫一次,每夜的完整对账修复任何漂移并验证每次修复。一切健康时对账日志是空的——而大多数日子里,它说的就是这件事。
+
+值得带走的设计原则,是让这件事变得可控的那一条:**别再「管理」访问权限,开始「断言」它。** 把客户应该拥有什么定义成业务数据的纯函数,在数据变化时应用这个函数,并按计划重新断言它以捕获其他所有情况。这样系统就不需要你小心翼翼——它只需要你在一个函数里,一次性地正确。
+
+---
+
+## 想让你的生意也用上这套?
+
+如果你卖数字产品,还在手工授予文件访问权限——或者你不确定上个月到期的客户是否真的已经失去访问权——我搭的正是这套东西:自建授权系统,访问权限从你的数据计算得出、自动到期、每夜自我修复。我熟悉 NocoDB、n8n、AList 和 Docker,并且会把文档一并交付,让你不用依赖我也能运维。
+
+欢迎联系 [me@hoelee.com](mailto:me@hoelee.com?subject=%E6%95%B0%E5%AD%97%E5%95%86%E5%93%81%E8%AE%BF%E9%97%AE%E6%8E%A7%E5%88%B6)
+或 WhatsApp [+60 12-797 2969](https://wa.me/60127972969),也可以看看我在
+[hoelee.com](https://hoelee.com) 做什么。