diff --git a/public/banners/stop-writing-agent-prompts-for-deterministic-work.png b/public/banners/stop-writing-agent-prompts-for-deterministic-work.png new file mode 100644 index 0000000..48e1044 Binary files /dev/null and b/public/banners/stop-writing-agent-prompts-for-deterministic-work.png differ diff --git a/public/og/stop-writing-agent-prompts-for-deterministic-work.png b/public/og/stop-writing-agent-prompts-for-deterministic-work.png new file mode 100644 index 0000000..fdcc179 Binary files /dev/null and b/public/og/stop-writing-agent-prompts-for-deterministic-work.png differ diff --git a/scripts/banner-gen/generate.mjs b/scripts/banner-gen/generate.mjs index 8082d49..1d6f3c6 100644 --- a/scripts/banner-gen/generate.mjs +++ b/scripts/banner-gen/generate.mjs @@ -1070,6 +1070,27 @@ BANNERS['loop-engineering-without-a-coding-agent'] = { ], }; +BANNERS['stop-writing-agent-prompts-for-deterministic-work'] = { + titlebar: 'root@hermes — deterministic check, no model', + lines: [ + { t: 'prompt', text: '$' }, { t: 'cmd', text: 'registry.npmjs.org//latest vs local version' }, + { t: 'prompt', text: 'WARN' }, { t: 'err', text: 'agent mode: [drift_skip] × 29 — died before it could speak' }, + { t: 'prompt', text: '' }, { t: 'dim', text: 'silent-because-healthy == silent-because-crashed' }, + { t: 'prompt', text: '$' }, { t: 'cmd', text: 'rewrite: exit 0 = measured · exit 1 = CANNOT MEASURE' }, + { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'script stdout delivered as the notice · no LLM in the path' }, + { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'the test: write down what success AND failure print' }, + { t: 'prompt', text: 'INFO' }, { t: 'cmd', text: 'number or string → script · "it depends" → model + verifier' }, + { t: 'prompt', text: '' }, { t: 'ok', text: '→ 8 of 11 loops now carry no model ✓' }, + ], + flow: [ + { n: '1', label: 'compare' }, + { n: '2', label: 'agent ✗', err: true }, + { n: '3', label: 'script' }, + { n: '4', label: 'exit 1 ≠ silent' }, + { n: '5', label: 'green ✓' }, + ], +}; + // ---------- read frontmatter ---------- const postPath = join(ROOT, 'src', 'content', 'posts', `${slug}.md`); let category = 'devops'; diff --git a/scripts/og-gen/generate.mjs b/scripts/og-gen/generate.mjs index 99605f9..d8dfd83 100644 --- a/scripts/og-gen/generate.mjs +++ b/scripts/og-gen/generate.mjs @@ -326,6 +326,11 @@ TERMINALS['loop-engineering-without-a-coding-agent'] = `
 agent-mode job: [drift_skip] × 29 runs — silent broken and silent healthy
$rewrite as a script · exit 1 = CANNOT MEASURE→ 1,618 ticks ✓
`; +TERMINALS['stop-writing-agent-prompts-for-deterministic-work'] = ` +
$hermes cron: CLI version check · agent mode
+
 [drift_skip] × 29 runs — silent, and silence reads as "nothing to report"
+
$rewrite as a script · stdout IS the notification→ green since ✓
`; + // ---------- read frontmatter ---------- const postPath = join(ROOT, 'src', 'content', 'posts', `${slug}.md`); if (!existsSync(postPath)) { diff --git a/src/content/posts/stop-writing-agent-prompts-for-deterministic-work.md b/src/content/posts/stop-writing-agent-prompts-for-deterministic-work.md new file mode 100644 index 0000000..7b6e46f --- /dev/null +++ b/src/content/posts/stop-writing-agent-prompts-for-deterministic-work.md @@ -0,0 +1,91 @@ +--- +title: "Stop Writing Agent Prompts for Deterministic Work" +description: "One of my scheduled jobs failed 29 times in a row because I put a model in a loop that only needed a string comparison. Here is the rule I use now." +pubDate: 2026-10-01 +category: notes +tags: ["loop-engineering", "cron", "ai-agents", "automation"] +ogImage: "/og/stop-writing-agent-prompts-for-deterministic-work.png" +banner: "/banners/stop-writing-agent-prompts-for-deterministic-work.png" +draft: false +--- + +One of my scheduled jobs failed 29 times in a row. Nobody told me, because +it was designed to speak only when it had something to report — and a run +that dies before it can speak looks exactly like a run that found nothing +wrong. + +The job was a version check: ask npm for the latest published version of a +CLI I use, compare it with the one installed here, and tell me if there is a +newer one. That is the entire task. It is two strings and an inequality. + +I had built it as an agent job, because that was the pattern I had in my +head at the time. Every tick, the model would read the injected script +output and decide whether to write a notification. It worked for a while. +Then my global default model changed, and the job — which records the model +it was created with — stopped running entirely. It printed `[drift_skip]` +and exited. Twenty-nine times. + +The fix was not a better prompt. It was deleting the model from the loop. + +## The rule + +**If the stop condition is a comparison — a string, a number, a status code — a model in that loop can only add failure modes.** + +There is no judgment to make in "is `3.8.50` newer than `3.8.50`". The model +was not deciding anything the code could not; it was adding a dependency (a +model route, a provider, a config snapshot, a token bill) and a new way to +fail. So I rewrote the checker as 125 lines of Python with an explicit exit +path for every branch, and flipped the job to script mode: the script's +stdout is delivered as the notification, with no LLM anywhere in the path. + +It has run clean ever since. Anthropic's own guidance says the same thing in +one line — *use scripts for deterministic work; running a script is cheaper +than reasoning through the steps* — but the cost was never the interesting +part for me. The reliability was. + +## Where the model does earn its place + +Of the eleven loops I run, eight now have no model in the path. The three +that do all handle something I genuinely cannot express as code: + +- **Is this email important?** A mailbox produces promos, DMARC reports, container alerts and a customer asking for a quote. Fetching mail is deterministic. Ranking it is not. +- **Is this shop voucher about to overspend?** The arithmetic is trivial; deciding what a finding means for money I have already committed is not. +- **Is this project stalled?** Reading two markdown files is a script's job. Noticing that the plan has not moved in three weeks is not. + +The test I apply to a new loop: **write down what the check would print on +success and on failure.** If both answers are a number or a string, it is a +script. If the answer is "it depends what it says", it needs a model — and +then it needs a verifier I trust more than the agent, because I am no longer +able to predict the output. + +## The part that actually bit me + +The failure mode was not the drift. It was the **silence**. + +A loop that speaks only when something is wrong is the only kind of loop I +can tolerate — I do not want a daily status report from eleven jobs. But +"silent because healthy" and "silent because it crashed before it could +speak" are the same observation from my side of the phone. Twenty-nine runs +of evidence were sitting in the job's execution log the whole time; I was +trusting the absence of a notification instead of reading the history. + +So every script in the fleet now has a second exit path: exit 0 means +*measured, here is what I found*, and exit 1 means *I could not measure* — +and exit 1 is never silent. Missing config, unparseable state, an unexpected +exception: all of them print a line and exit 1. That one change would have +surfaced this bug on the first run instead of the twenty-ninth. + +If you have a loop that has been "quiet for a while", go and read its last +ten runs before you conclude it is working. That advice cost me three weeks. + +## The short version + +- Deterministic check → script. Judgment → model. Do not mix them for convenience. +- Give the silent-on-success convention one exception: "I could not measure" must always be loud. +- Read the runner's own execution history, not just your notifications. + +Want a version of this for your own site or shop — uptime, certificates, +backups, mail that only pings you when it matters? +**[WhatsApp +60 12-797 2969](https://wa.me/60127972969)** · +**[me@hoelee.com](mailto:me@hoelee.com?subject=Self-hosted%20monitoring)** · +**[hoelee.com](https://hoelee.com)** diff --git a/src/content/posts/zh/stop-writing-agent-prompts-for-deterministic-work.md b/src/content/posts/zh/stop-writing-agent-prompts-for-deterministic-work.md new file mode 100644 index 0000000..f20e5b1 --- /dev/null +++ b/src/content/posts/zh/stop-writing-agent-prompts-for-deterministic-work.md @@ -0,0 +1,54 @@ +--- +title: "确定性的活,别再用 Agent Prompt 去跑" +description: "我有一个定时任务连续失败 29 次,原因是我在一个只需要字符串比较的循环里塞了个模型。这是我现在用的规则。" +pubDate: 2026-10-01 +category: notes +tags: ["loop-engineering", "cron", "ai-agents", "automation"] +ogImage: "/og/stop-writing-agent-prompts-for-deterministic-work.png" +banner: "/banners/stop-writing-agent-prompts-for-deterministic-work.png" +draft: false +--- + +我有一个定时任务连续失败了 29 次。没人告诉我,因为它被设计成只在有东西可报的时候才开口——而一个还没来得及开口就崩掉的运行,看起来和一个「什么都没发现」的运行一模一样。 + +那个任务是个版本检查:问 npm 某个我常用的 CLI 最新发布版本是多少,和本机装的版本比一下,有更新就告诉我。整个任务就这些。它就是两个字符串加一次比较。 + +我当初把它做成了 agent 任务,因为那阵子我脑子里的默认模式就是这个。每次 tick,模型读一遍注入的脚本输出,然后判断要不要写一条通知。它正常跑了一阵子。后来我改了全局默认模型,而这个任务会记录自己创建时的模型,于是它干脆整个不跑了,打印一行 `[drift_skip]` 就退出。连续 29 次。 + +修法不是写一个更好的 prompt,而是把模型从循环里删掉。 + +## 规则 + +**如果停止条件是一次比较——一个字符串、一个数字、一个状态码——那么循环里的模型只能带来新的失败模式。** + +「`3.8.50` 是不是比 `3.8.50` 新」这件事里没有任何判断可做。模型并没有在决定代码决定不了的事;它带来的是一条依赖(一条模型路由、一个提供商、一份配置快照、一张 token 账单)和一种全新的出错方式。所以我把它重写成 125 行 Python,每条分支都有明确出口,再把任务切成脚本模式:脚本的 stdout 直接作为通知内容,整条路径上没有任何 LLM。 + +从此一路干净。Anthropic 自己的建议也是一句话——*确定性的活用脚本,跑脚本比让模型去推理步骤便宜*——但对我来说成本从来不是重点,可靠性才是。 + +## 模型该用在哪 + +我跑的 11 个循环里,现在有 8 个整条路径上没有模型。用模型的那 3 个,处理的都是我确实写不成代码的东西: + +- **这封邮件重不重要?** 一个邮箱里同时有促销、DMARC 报告、容器告警,和一位来问报价的客户。抓邮件是确定性的,给它排优先级不是。 +- **这张店铺优惠券是不是快超支了?** 算术很简单;判断一个发现对我已经承诺出去的钱意味着什么,不简单。 +- **这个项目是不是卡住了?** 读两个 markdown 文件是脚本的活。发现「计划已经三周没动过」不是。 + +我给新循环做的测试是:**写下这个检查在成功和失败时分别会打印什么。** 如果两个答案都是一个数字或一个字符串,那它就是脚本。如果答案是「要看它说了什么」,那它需要模型——而这时候它就需要一个我比 agent 更信任的验证器,因为我已经无法预测输出了。 + +## 真正咬到我的是哪一部分 + +失败模式不是 drift,是**沉默**。 + +一个只在出事时才开口的循环,是我唯一能忍受的那种——我不想让 11 个任务每天给我发状态报告。但「健康所以安静」和「崩了所以来不及开口」,从我手机这端看是同一个现象。29 次运行的证据一直躺在那个任务的执行日志里;我却一直在相信「没有通知」这件事本身,而没去读历史。 + +所以现在这套脚本都多了一条出口:退出码 0 表示*量过了,这是结果*,退出码 1 表示*我量不了*——而退出码 1 永远不静默。配置缺失、状态文件解析失败、没预料到的异常:全都打印一行并退出 1。就这一处改动,本该让这个 bug 在第 1 次就暴露,而不是第 29 次。 + +如果你有一个循环「已经安静一阵子」了,先去翻它最近十次运行,再下结论说它正常。这条建议花了我三个星期才换到。 + +## 简短版 + +- 确定性的检查 → 脚本。判断 → 模型。别为了方便把两者混在一起。 +- 给「成功则静默」这条约定开一个例外:「我量不了」必须永远大声。 +- 去读 runner 自己的执行历史,而不是只看你收到的通知。 + +想给你的网站或网店也做一套这样的东西——可用性、证书、备份、以及只在真的有事时才 ping 你的邮件?**[WhatsApp +60 12-797 2969](https://wa.me/60127972969)** · **[me@hoelee.com](mailto:me@hoelee.com?subject=Self-hosted%20monitoring)** · **[hoelee.com](https://hoelee.com)**