Harden against Carousell soft-blocks; stop one flaky watch failing the tick

Two problems found while verifying the pagination fix on the live stack:

1. fetch_listings() did state['SearchListing']['listingCards'] with no guard.
   Carousell serves HTTP 200 with listingCards=null under soft rate limiting,
   so the tick died with TypeError: 'NoneType' object is not iterable ->
   ok:false -> container unhealthy, repeatedly.
2. run_tick() set ok = (no failures at all), so a single soft-blocked watch
   out of 7 marked the whole monitor failed. That flaps on transient blocks
   and (now that failures alert) would spam Telegram.

- fetch_listings: explicit null check -> clear retryable RuntimeError
- add FAILURE_RATIO_THRESHOLD (default 1.0 = all watches must fail); partial
  failures are reported as '[partial n/N] ...' without failing the tick
- health gains failed_watches
- wire the knob into compose/.env.example/DOCUMENTATION/COMPOSE-SETUP
- 15 new checks: null vs empty cards, real card still parses, threshold edges
This commit is contained in:
2026-09-22 03:52:14 +08:00
parent b88f6d0963
commit f99a0d878e
6 changed files with 122 additions and 4 deletions
+2
View File
@@ -15,3 +15,5 @@ FETCH_GAP_SECONDS=1
HEALTH_STALE_SECONDS=600
# 连续失败几次才发 Telegram 故障告警(去抖):
ERROR_ALERT_AFTER=3
# 失败 watch 占比达到该值才判定整轮故障(1.0 = 全部失败):
FAILURE_RATIO_THRESHOLD=1.0
+2
View File
@@ -98,6 +98,7 @@ build context is the repo directory).
FETCH_GAP_SECONDS: ${FETCH_GAP_SECONDS:-1}
HEALTH_STALE_SECONDS: ${HEALTH_STALE_SECONDS:-600}
ERROR_ALERT_AFTER: ${ERROR_ALERT_AFTER:-3}
FAILURE_RATIO_THRESHOLD: ${FAILURE_RATIO_THRESHOLD:-1.0}
TZ: Asia/Kuala_Lumpur
```
@@ -112,6 +113,7 @@ build context is the repo directory).
| `FETCH_GAP_SECONDS` | `1` | Minimum pause (s) between watch URL fetches within one tick — prevents request bursts (default 1; set `0` to disable). |
| `HEALTH_STALE_SECONDS` | `600` | Docker healthcheck tolerance: if last tick older than this → unhealthy. |
| `ERROR_ALERT_AFTER` | `3` | Consecutive failed ticks before a Telegram failure alert fires (debounce). Sent once on entering the failure state and once on recovery. |
| `FAILURE_RATIO_THRESHOLD` | `1.0` | Fraction of watches that must fail for the tick to be reported as failed (`1.0` = all of them; `0` disables the check). |
| `TZ` | `Asia/Kuala_Lumpur` | Container clock (mostly cosmetic; timestamps are written in UTC deliberately for NocoDB). |
`${VAR:-default}` syntax: compose substitutes the value from the environment /
+1
View File
@@ -60,6 +60,7 @@ Credentials are documented in `SECRETS.md` there.
| `FETCH_GAP_SECONDS` | `1` | min pause (s) between watch URL fetches within one tick — anti-burst |
| `HEALTH_STALE_SECONDS` | `600` | healthcheck staleness window |
| `ERROR_ALERT_AFTER` | `3` | consecutive failed ticks before a Telegram failure alert is sent (debounce); a recovery notice is sent when it clears |
| `FAILURE_RATIO_THRESHOLD` | `1.0` | fraction of watches that must fail for the tick to count as failed (`1.0` = all). Shields the healthcheck and alerts from one flaky watch being soft-blocked. |
**Secrets = env vars (`.env`). Operational knobs = NocoDB Settings table.**
Speed, enable/disable, and notify on/off are all changed from the NocoDB UI — no
+1
View File
@@ -21,6 +21,7 @@ services:
FETCH_GAP_SECONDS: ${FETCH_GAP_SECONDS:-1}
HEALTH_STALE_SECONDS: ${HEALTH_STALE_SECONDS:-600}
ERROR_ALERT_AFTER: ${ERROR_ALERT_AFTER:-3}
FAILURE_RATIO_THRESHOLD: ${FAILURE_RATIO_THRESHOLD:-1.0}
TZ: Asia/Kuala_Lumpur
volumes:
- carousell-data:/data
+23 -4
View File
@@ -40,6 +40,9 @@ HEALTH_STALE_SECONDS = int(os.environ.get("HEALTH_STALE_SECONDS", "600"))
# 错误告警去抖:连续失败达到该次数才发 Telegram,恢复时发一条恢复通知。
# 目的:单个 watch 偶发 403/超时不会刷屏,但持续故障一定通知到人。
ERROR_ALERT_AFTER = int(os.environ.get("ERROR_ALERT_AFTER", "3"))
# 单个 watch 偶发失败(Carousell 软限流)不该把整轮判为故障:只有失败占比
# 达到该比例(默认全挂)才 ok=false。设为 1.0 = 全部失败才算故障;0.0 关闭。
FAILURE_RATIO_THRESHOLD = float(os.environ.get("FAILURE_RATIO_THRESHOLD", "1.0"))
DEFAULT_INTERVAL_MIN = int(os.environ.get("DEFAULT_INTERVAL_MIN", "5"))
# 同一 tick 内逐条抓取 watch URL 之间的最小间隔秒数(防瞬时并发打爆 Carousell)。
FETCH_GAP_SECONDS = float(os.environ.get("FETCH_GAP_SECONDS", "1"))
@@ -352,7 +355,15 @@ def fetch_listings(search_url):
if not blobs:
raise RuntimeError("no application/json state found (blocked/ratelimited?)")
state = json.loads(max(blobs, key=len))
cards = state["SearchListing"]["listingCards"]
# listingCards 会在 Carousell 软限流/挑战页时是 null(HTTP 仍是 200)。
# 不加判断会抛 TypeError: 'NoneType' object is not iterable,把整个 tick
# 拖垮 → health ok:false → 容器 unhealthy。这里转成清晰的可重试错误。
sl = state.get("SearchListing") or {}
cards = sl.get("listingCards")
if cards is None:
err = sl.get("error")
raise RuntimeError(
f"listingCards null (soft-block/ratelimit? error={err!r})")
out = []
for c in cards:
try:
@@ -814,9 +825,17 @@ def run_tick(listings_tid, settings_tid, ignored_sellers_tid, ignored_keywords_t
send_pending_notifications(listings_tid, settings_tid, ignored_sellers_tid,
ignored_keywords_tid, kw_fk_col)
ok = len(failures) == 0
return ok, ("; ".join(failures) if failures else ""), {
"watch_count": len(watches), "new_this_tick": new_total}
# 只有失败占比达到阈值才算整轮故障;单个 watch 偶发软限流不报故障。
# 阈值 1.0 = 全部 watch 失败才 ok=false(默认);0.0 = 任何失败都算。
n = len(watches)
ratio = (len(failures) / n) if n else 0.0
ok = ratio < FAILURE_RATIO_THRESHOLD if FAILURE_RATIO_THRESHOLD > 0 else True
err = "; ".join(failures) if failures else ""
if failures and ok:
err = f"[partial {len(failures)}/{n}] " + err
return ok, err, {
"watch_count": n, "new_this_tick": new_total,
"failed_watches": len(failures)}
def main():
+93
View File
@@ -430,6 +430,99 @@ except Exception as e:
check("backlog on page 2 detected end-to-end", False, f"{type(e).__name__}: {e}")
# --------------------------------------------------------------------------- #
print("\n[11] fetch_listings tolerates null listingCards (soft-block, HTTP 200)")
mon = load_monitor()
def html_with(cards_literal, error="null"):
state = ('{"SearchListing":{"listingCards":%s,"error":%s},'
'"RateLimit":{"timestamps":{}}}' % (cards_literal, error))
return 200, '<html><script type="application/json">' + state + '</script></html>'
mon._http = lambda method, url, **kw: html_with("null")
try:
mon.fetch_listings("https://x/search")
check("null listingCards raises a clear error", False, "no exception")
except RuntimeError as e:
check("null listingCards raises RuntimeError (not TypeError)", True, str(e)[:60])
except TypeError as e:
check("null listingCards raises RuntimeError (not TypeError)", False,
f"got TypeError: {e}")
except Exception as e:
check("null listingCards raises a clear error", False, f"{type(e).__name__}: {e}")
# empty list is valid -> zero listings, not an error
mon2 = load_monitor()
mon2._http = lambda method, url, **kw: html_with("[]")
try:
got = mon2.fetch_listings("https://x/search")
check("empty listingCards -> [] (not an error)", got == [])
except Exception as e:
check("empty listingCards -> [] (not an error)", False, f"{type(e).__name__}: {e}")
# a real card still parses
mon3 = load_monitor()
card = ('[{"listingID":123,"aboveFold":[],"belowFold":'
'[{"component":"header_1","stringContent":"Widget"},'
'{"component":"header_2","stringContent":"RM 10"}],'
'"seller":{"username":"someone"},"thumbnailURL":"http://i/x.jpg"}]')
mon3._http = lambda method, url, **kw: html_with(card)
try:
got = mon3.fetch_listings("https://x/search")
check("a real card still parses", len(got) == 1 and got[0]["listing_id"] == 123,
f"{got}")
except Exception as e:
check("a real card still parses", False, f"{type(e).__name__}: {e}")
# --------------------------------------------------------------------------- #
print("\n[12] FAILURE_RATIO_THRESHOLD: one flaky watch must not fail the tick")
def tick_with(fail_count, total, threshold):
m = load_monitor({"FAILURE_RATIO_THRESHOLD": str(threshold)})
watches = [{"Id": i, "title": f"w{i}", "url": f"https://x/{i}",
"enabled": 1, "check_interval_minutes": 1} for i in range(total)]
m.load_watches = lambda tid: watches
m.update_checked = lambda *a, **k: None
def fake_fetch(url):
idx = int(url.rsplit("/", 1)[1])
if idx < fail_count:
raise RuntimeError("soft-block")
return []
m.fetch_listings = fake_fetch
m.send_pending_notifications = lambda *a, **k: None
m.load_seen = lambda tid: set()
return m.run_tick("L", "S", "IS", "IK", None, set(), {})
ok, err, extra = tick_with(fail_count=1, total=7, threshold=1.0)
check("1 of 7 failing -> ok=True (partial, threshold 1.0)", ok is True, f"err={err[:40]!r}")
check("partial error is still reported in message", err.startswith("[partial 1/7]"), err[:30])
ok, err, extra = tick_with(fail_count=7, total=7, threshold=1.0)
check("7 of 7 failing -> ok=False", ok is False)
check("failed_watches surfaced in health extra", extra.get("failed_watches") == 7, f"{extra}")
ok, err, extra = tick_with(fail_count=3, total=7, threshold=0.5)
check("3 of 7 failing w/ threshold 0.5 -> ok=True", ok is True)
ok, err, extra = tick_with(fail_count=4, total=7, threshold=0.5)
check("4 of 7 failing w/ threshold 0.5 -> ok=False", ok is False)
ok, err, extra = tick_with(fail_count=0, total=7, threshold=1.0)
check("0 failing -> ok=True, no error", ok is True and err == "")
ok, err, extra = tick_with(fail_count=2, total=7, threshold=0.0)
check("threshold 0 disables the check -> ok=True", ok is True)
print("\n" + "=" * 62)
if FAILURES:
print(f"FAILED ({len(FAILURES)}): " + "; ".join(FAILURES))