Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/sdlc-lint.yml
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@ jobs:
name: lint # stable name — referenced by branch-protection required checks
runs-on: ubuntu-latest
env:
GOTOOLCHAIN: go1.26.5 # pin to the same toolchain as the QA job
GOTOOLCHAIN: go1.26.6 # pin to the same toolchain as the QA job

steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
Expand Down
2 changes: 1 addition & 1 deletion .github/workflows/sdlc-vuln.yml
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,7 @@ jobs:
name: vuln # stable check name — referenced by branch-protection required checks
runs-on: ubuntu-latest
env:
GOTOOLCHAIN: go1.26.5 # pin to the toolchain that fixes GO-2026-5856 (crypto/tls ECH leak) and all stdlib CVEs through go1.26.5
GOTOOLCHAIN: go1.26.6 # pin to the toolchain that fixes GO-2026-6088/6089/6090/6091 and all stdlib CVEs through go1.26.6
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2

Expand Down
2 changes: 1 addition & 1 deletion e2e/quickstart/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,4 +2,4 @@ module github.com/dshakes/lantern/e2e/quickstart

go 1.23

toolchain go1.26.5
toolchain go1.26.6
2 changes: 1 addition & 1 deletion e2e/runtime/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,4 +2,4 @@ module github.com/dshakes/lantern/e2e/runtime

go 1.23

toolchain go1.26.5
toolchain go1.26.6
2 changes: 1 addition & 1 deletion gen/go/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module github.com/dshakes/lantern/gen/go

go 1.25.0

toolchain go1.26.5
toolchain go1.26.6

require (
google.golang.org/grpc v1.82.1
Expand Down
5 changes: 5 additions & 0 deletions infra/docker/docker-compose.yml
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,9 @@ x-common-env: &common-env
services:
postgres:
image: pgvector/pgvector:pg16
# ponytail: Docker's own restart policy is the durable fix — containers
# come back whenever the daemon does, with no launchd job in the loop.
restart: unless-stopped
ports:
- "5432:5432"
environment:
Expand All @@ -26,6 +29,7 @@ services:

redis:
image: redis:7-alpine
restart: unless-stopped
ports:
- "6379:6379"
healthcheck:
Expand All @@ -36,6 +40,7 @@ services:

minio:
image: minio/minio:latest
restart: unless-stopped
ports:
- "9000:9000"
- "9001:9001"
Expand Down
14 changes: 14 additions & 0 deletions packages/bridge-core/src/life-events.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -64,6 +64,20 @@ describe("life-events — undated travel/appointment must NOT nudge (placeholder
const e = ev({ kind: "otp", urgency: "now", fields: { code: "123456" }, rawText: "Your verification code is 123456" });
assert.notEqual(proactiveDecision(e, {}, NOW).route, "suppress", "otp must still surface");
});

// REGRESSION: a codeless OTP interpolated `undefined` into the owner DM and
// shipped "🔑 your code is undefined" 11 times over 36 days in production.
test("otp with no code never renders the literal string undefined", () => {
const e = ev({ kind: "otp", urgency: "now", fields: {}, rawText: "Your verification code is expiring" });
const msg = proactiveDecision(e, {}, NOW).ownerMessage;
assert.ok(!/undefined/.test(msg), `ownerMessage leaked "undefined": ${msg}`);
assert.ok(msg.length > 0, "codeless otp must still say something");
});

test("otp WITH a code still reports it", () => {
const e = ev({ kind: "otp", urgency: "now", fields: { code: "611586" }, rawText: "code 611586" });
assert.match(proactiveDecision(e, {}, NOW).ownerMessage, /611586/);
});
});

describe("life-events — recency gate: only surface RECENT + actionable (stale-Gmail-nudge bug)", () => {
Expand Down
5 changes: 5 additions & 0 deletions packages/bridge-core/src/life-events.ts
Original file line number Diff line number Diff line change
Expand Up @@ -716,6 +716,11 @@ function buildOwnerMessage(event: LifeEvent, actions: ProactiveAction[]): string
return `⚠️ ${who} flagged a declined/suspicious charge — might be fraud.${phone ? ` want the number (${phone})?` : " want the number?"}`;
}
case "otp": {
// A missing code interpolated as the literal string "undefined" and
// shipped "🔑 your code is undefined" 11 times over 36 days. An OTP
// ping with no code carries no information, so say what we do know
// rather than inventing a value we never extracted.
if (!f.code) return `🔑 a login code arrived${f.merchant ? ` from ${f.merchant}` : ""} — check the original message.`;
return `🔑 your code is ${f.code} — (i won't share this with anyone).`;
}
case "appointment": {
Expand Down
2 changes: 1 addition & 1 deletion packages/cli/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module github.com/dshakes/lantern/packages/cli

go 1.25.0

toolchain go1.26.5
toolchain go1.26.6

require (
github.com/dshakes/lantern/gen/go v0.0.0
Expand Down
2 changes: 1 addition & 1 deletion packages/sdk-go/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,4 +2,4 @@ module github.com/dshakes/lantern/packages/sdk-go

go 1.23

toolchain go1.26.5
toolchain go1.26.6
52 changes: 52 additions & 0 deletions scripts/launchd/dev.lantern.dashboard-reload.plist
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
<key>Label</key>
<string>dev.lantern.dashboard-reload</string>

<!-- Restart the dashboard whenever a new build lands.

Why a separate job: launchd's WatchPaths only LAUNCHES a job that is
not running. The dashboard sets KeepAlive=true, so it is always
running and WatchPaths on it is a no-op (verified: BUILD_ID changed,
PID did not). This job is NOT KeepAlive, so a BUILD_ID change starts
it, it kickstarts the dashboard, and it exits.

The bug it prevents: Next.js content-hashes chunk filenames per build
and the dashboard wrapper deliberately skips rebuilding when a bundle
exists. A server left running across a rebuild therefore answers HTML
referencing chunks it never built — every /_next/static/chunks/*.js
404s or 400s and the page dies with "Application error: a client-side
exception has occurred". It also serves the OLD NEXT_PUBLIC_API_URL,
which is inlined at build time, so a phone hitting it gets a login
pointed at the wrong host and reports "invalid credentials". Both were
hit in production before this existed. -->
<key>ProgramArguments</key>
<array>
<string>/bin/launchctl</string>
<string>kickstart</string>
<string>-k</string>
<string>gui/__UID__/dev.lantern.dashboard</string>
</array>

<key>WatchPaths</key>
<array>
<string>__REPO_ROOT__/apps/web/.next/BUILD_ID</string>
</array>

<!-- Do NOT RunAtLoad: that would kick the dashboard on every login for no
reason. Only a BUILD_ID change should trigger it. -->
<key>RunAtLoad</key>
<false/>

<!-- A build writes BUILD_ID once, but coalesce any burst into one restart. -->
<key>ThrottleInterval</key>
<integer>30</integer>

<key>StandardOutPath</key>
<string>__HOME__/Library/Logs/Lantern/dashboard-reload.out.log</string>
<key>StandardErrorPath</key>
<string>__HOME__/Library/Logs/Lantern/dashboard-reload.err.log</string>
</dict>
</plist>
5 changes: 5 additions & 0 deletions scripts/launchd/dev.lantern.dashboard.plist
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,11 @@ Logs: ~/Library/Logs/Lantern/dashboard.{out,err}.log
<key>ThrottleInterval</key>
<integer>15</integer>

<!-- Restart-on-rebuild lives in dev.lantern.dashboard-reload.plist, NOT
here: launchd's WatchPaths only LAUNCHES a stopped job, so on this
KeepAlive=true job it is a no-op (verified — BUILD_ID changed, PID did
not). See that plist for what goes wrong without it. -->

<key>StandardOutPath</key>
<string>__HOME__/Library/Logs/Lantern/dashboard.out.log</string>
<key>StandardErrorPath</key>
Expand Down
15 changes: 13 additions & 2 deletions scripts/launchd/dev.lantern.infra.plist
Original file line number Diff line number Diff line change
Expand Up @@ -25,10 +25,21 @@ Logs: ~/Library/Logs/Lantern/infra.{out,err}.log
<key>WorkingDirectory</key>
<string>__REPO_ROOT__</string>

<!-- RunAtLoad fires once per login. KeepAlive is OFF — docker-compose
is a one-shot 'bring up' command, not a long-running daemon. -->
<!-- RunAtLoad fires once per login. The wrapper exits 1 when Docker never
becomes ready, so KeepAlive-on-failure is the retry: without it a single
cold-boot race left infra down for 13 days while every service sat in a
connection-refused loop. Compose itself is still one-shot — a successful
run exits 0 and is NOT restarted. Containers carry restart:unless-stopped
so Docker, not launchd, keeps them alive after that. -->
<key>RunAtLoad</key>
<true/>
<key>KeepAlive</key>
<dict>
<key>SuccessfulExit</key>
<false/>
</dict>
<key>ThrottleInterval</key>
<integer>300</integer>

<key>StandardOutPath</key>
<string>__HOME__/Library/Logs/Lantern/infra.out.log</string>
Expand Down
4 changes: 3 additions & 1 deletion scripts/launchd/install.sh
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,8 @@ LOG_DIR="$HOME/Library/Logs/Lantern"
# Order matters — infra brings up docker, api waits for postgres,
# dashboard waits for api. LaunchAgents run in parallel but each
# wrapper waits for its upstream dependency before launching.
ALL_SERVICES=( "infra" "api" "dashboard" "whatsapp-bridge" "imessage-bridge" \
ALL_SERVICES=( "infra" "api" "dashboard" "dashboard-reload" \
"whatsapp-bridge" "imessage-bridge" \
"model-router" "runtime-manager" "runtime-scheduler" \
"workflow-engine" "gateway" "surface-gateway" )

Expand Down Expand Up @@ -108,6 +109,7 @@ for s in "${ALL_SERVICES[@]}"; do
-e "s|__NODE__|$NODE_BIN|g" \
-e "s|__REPO_ROOT__|$REPO_ROOT|g" \
-e "s|__HOME__|$HOME|g" \
-e "s|__UID__|$(id -u)|g" \
"$SRC" > "$TMP"
mv "$TMP" "$DST"

Expand Down
2 changes: 1 addition & 1 deletion services/billing/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module github.com/dshakes/lantern/services/billing

go 1.25.0

toolchain go1.26.5
toolchain go1.26.6

require (
github.com/dshakes/lantern/gen/go v0.0.0
Expand Down
2 changes: 1 addition & 1 deletion services/control-plane/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module github.com/dshakes/lantern/services/control-plane

go 1.25.0

toolchain go1.26.5
toolchain go1.26.6

require (
github.com/golang-jwt/jwt/v5 v5.3.0
Expand Down
2 changes: 1 addition & 1 deletion services/data-plane-agent/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module github.com/dshakes/lantern/services/data-plane-agent

go 1.25.0

toolchain go1.26.5
toolchain go1.26.6

require (
github.com/dshakes/lantern/gen/go v0.0.0
Expand Down
66 changes: 64 additions & 2 deletions services/imessage-bridge/src/session.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1383,6 +1383,21 @@ export class IMessageSession {
// This must NOT be called for NORMAL non-responses — "not the owner",
// "not addressed in a group", "low-confidence draft held", "trivial
// chatter ack". Those are by-design silence, not drops.
// Release a life-event idempotency claim taken before an emission that then
// failed to deliver. Without this, a claim-then-fail permanently suppresses
// the event: the retry would see hasActed() and stay silent forever. Only
// ever called on paths where nothing reached the owner.
private async releaseLifeEventClaim(key: string | undefined, why: string): Promise<void> {
if (!key) return;
try {
const { unmarkActed } = await import("@lantern/bridge-core/life-events-store");
unmarkActed(key);
this.logger.warn({ idempotencyKey: key, why }, "life-event claim released — not delivered, will retry");
} catch (err) {
this.logger.warn({ err, idempotencyKey: key }, "failed to release life-event claim");
}
}

private notifyOwnerOfDrop(message: string, dedupeKey: string): void {
try {
const now = Date.now();
Expand Down Expand Up @@ -4443,8 +4458,37 @@ export class IMessageSession {

if (decision.route === "suppress") return true; // owned + dropped — do NOT emit

// Dedup EVERY owner-facing emission, not just auto-act. autoActLifeEvent
// has always had a hasActed guard, but the ping/digest routes below did
// not — so a re-delivered or re-classified inbound fired the same DM again
// with nothing to stop it. Observed in production: one fraud alert sent
// 22×, another 27×, and one reply 15× inside 32 seconds (several in the
// same second). That is the "spammy/repetitive" report. The key is already
// computed above for auto-act; reuse it so all three routes share one
// guard rather than each growing its own.
// Claim the key BEFORE emitting so two concurrent classifications of the
// same inbound can't both pass the check and both send — that race is the
// storm. But a claim that is never delivered must be RELEASED: marking
// up-front and then failing the send would suppress the alert forever, and
// a fraud notice silently dropped is worse than one sent twice. Every
// non-delivering path below calls releaseLifeEventClaim().
if (auto.idempotencyKey) {
const { hasActed, markActed } = await import("@lantern/bridge-core/life-events-store");
if (hasActed(auto.idempotencyKey)) {
this.logger.info(
{ kind: event.kind, idempotencyKey: auto.idempotencyKey, route: decision.route },
"life-event emit skipped — already surfaced (idempotent)",
);
return true;
}
markActed(auto.idempotencyKey);
}

if (!owner) {
// No self-chat target — surface to the dashboard feed so it isn't invisible.
// The owner DM never happened, so release the claim: once a self-chat
// target resolves, this event should still be able to reach them.
await this.releaseLifeEventClaim(auto.idempotencyKey, "no owner self-chat target");
this.broadcast({ type: "activity", data: { kind: "system", summary: decision.ownerMessage, timestamp: Date.now() } });
return true;
}
Expand All @@ -4468,7 +4512,10 @@ export class IMessageSession {
}

// ping — DM the owner now + arm the one-tap offer for the top action.
await this.send(owner, decision.ownerMessage).catch(() => {});
// Release the claim if the DM did not land, so a transient send error
// cannot suppress this event forever.
const delivered = await this.send(owner, decision.ownerMessage).then(() => true).catch(() => false);
if (!delivered) await this.releaseLifeEventClaim(auto.idempotencyKey, "owner DM send failed");
this.lastLifeEventKind = event.kind;
const top = decision.actions.find((a) => a.kind !== "snooze" && a.kind !== "none");
if (top && top.offerAction) {
Expand Down Expand Up @@ -6454,6 +6501,12 @@ export class IMessageSession {
// chat_identifier == own handle) and remember.
private selfChatRowIds: Set<number> = new Set();

// Per-chat timestamp of the last "I'm still working on it" fallback. The
// fallback fires whenever the agent returns nothing, and with no throttle it
// went out 11× in production — 8 of them inside 55 seconds. Repeating
// "give me another minute" eight times says nothing the first one didn't.
private lastAgentStallNoticeAt: Map<string, number> = new Map();

// Per-chat timestamp of the last doc query. Lets us recognize
// short follow-ups ("send it", "yes", "the first one") as
// continuations within a 5-minute window.
Expand Down Expand Up @@ -10263,7 +10316,16 @@ export class IMessageSession {
if (!draft) {
// Graceful fallback — never "couldn't reach the agent — try
// again". The bot OWNS the retry; the user should not have to.
void this.send(jid, "hmm, that one took longer than I'd like. give me another minute and ask again");
// Say it once per 5 minutes per chat. Beyond that the owner already
// knows the agent is struggling, and repeating it is the noise they
// reported — better to stay quiet than to apologise on a loop.
const lastStall = this.lastAgentStallNoticeAt.get(jid) ?? 0;
if (Date.now() - lastStall > 5 * 60_000) {
this.lastAgentStallNoticeAt.set(jid, Date.now());
void this.send(jid, "hmm, that one took longer than I'd like. give me another minute and ask again");
} else {
this.logger.info({ jid }, "agent stall notice suppressed — already sent within 5m");
}
return;
}

Expand Down
2 changes: 1 addition & 1 deletion services/memory/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module github.com/dshakes/lantern/services/memory

go 1.25.0

toolchain go1.26.5
toolchain go1.26.6

require (
github.com/jackc/pgx/v5 v5.7.4
Expand Down
2 changes: 1 addition & 1 deletion services/notifier/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module github.com/dshakes/lantern/services/notifier

go 1.25.0

toolchain go1.26.5
toolchain go1.26.6

require (
github.com/jackc/pgx/v5 v5.7.4
Expand Down
2 changes: 1 addition & 1 deletion services/runtime-scheduler/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module github.com/dshakes/lantern/services/runtime-scheduler

go 1.25.0

toolchain go1.26.5
toolchain go1.26.6

require (
github.com/dshakes/lantern/gen/go v0.0.0
Expand Down
2 changes: 1 addition & 1 deletion services/scheduler/go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module github.com/dshakes/lantern/services/scheduler

go 1.25.0

toolchain go1.26.5
toolchain go1.26.6

require (
github.com/dshakes/lantern/gen/go v0.0.0
Expand Down
Loading
Loading