Skip to content
← Public packages

@kentcdodds/sentry-triage

Sentry triage wakes Cole (Grok Bot) per issue: one active wake per repo (lease + queue), loop-safe Discord. Cole may spawn Cursor for isolated repo work.

src/process-sentry-webhook.ts

758 lines · 23.7 KB · TypeScript
import { packageStorage, workflows } from 'kody:runtime'
import {
	postSentryReport,
	editSentryReport,
} from './format-discord-report.ts'
import getIssue from 'kody:@kentcdodds/sentry/get-issue'
import { buildTriageAgentPrompt } from './agent-prompt.ts'
import { gatherSpawnContext } from './spawn-snapshot.ts'
import handleSeerWebhook from './handle-seer-webhook.ts'
import {
	acquireRepoLease,
	wakeCole,
	enqueueRepoIssue,
	findReposNeedingQueueFlush,
	findStaleAwaitingIssues,
	flushRepoQueue,
	getRepoQueueDepth,
	isRepoLeaseHeld,
	releaseRepoLease,
	seerRunsToPr,
	triggerSeerRca,
	updateRepoLeaseHolder,
} from './triage-core.ts'
import {
	alertSentKey,
	hourBucket,
	issueAnomalyThresholdPerHour,
	issueStorageKey,
	issuesSeenCounterKey,
	maxAgentsPerHour,
	seerFirstEnabled,
	seerGraceMs,
	seerPrGraceMs,
	spawnedCounterKey,
	truncate,
} from './shared.ts'
import {
	promptOwnerFields,
	readTriageSettings,
	resolveTriageProjectForProjectId,
	resolveTriageProjectForSlug,
} from './settings.ts'
import type {
	ProcessSentryWebhookInput,
	ProcessSentryWebhookResult,
	SentryWebhookPayload,
} from './types.ts'

/**
 * Resolve which triage project an inbound delivery belongs to. Issue-resource
 * webhooks carry `issue.project.slug`; alert-rule ("event_alert") webhooks
 * carry the numeric project id on the event; as a last resort, look the issue
 * up in the Sentry API. Returns null when the project is not in the registry.
 */
async function resolveTriageProject(issue, payload) {
	const slug = issue.project?.slug ?? null
	if (slug) return { project: await resolveTriageProjectForSlug(slug), slug }
	const eventProjectId = payload?.data?.event?.project ?? null
	if (eventProjectId != null) {
		const project = await resolveTriageProjectForProjectId(eventProjectId)
		if (project) return { project, slug: project.slug }
	}
	try {
		const data = await getIssue({ issueId: issue.id })
		const fetchedSlug = data?.project?.slug ?? null
		if (fetchedSlug) {
			return { project: await resolveTriageProjectForSlug(fetchedSlug), slug: fetchedSlug }
		}
	} catch {
		// Fall through to unresolved.
	}
	return { project: null, slug: eventProjectId != null ? String(eventProjectId) : null }
}

async function bumpCounter(storage, key) {
	const current = Number((await storage.get(key)) ?? 0)
	const next = current + 1
	await storage.set(key, next)
	return next
}

async function alertOnce(storage, kind, content) {
	const key = alertSentKey(kind)
	if (await storage.get(key)) return false
	await storage.set(key, new Date().toISOString())
	await postSentryReport({
		status: 'queued',
		title: kind === 'anomaly' ? 'Sentry triage paused' : 'Sentry triage alert',
		summary: content,
	})
	return true
}

/**
 * Durable Sentry webhook processor (issue + Seer). Invoked by
 * `./handle-sentry-webhook` via `workflows.create` so the ingress export can
 * acknowledge quickly while prefetch / Discord / Cole wake run with a
 * workflow budget. Failures throw so the workflow run surfaces as errored.
 *
 * Workflow dispatches pass `{ payloadKey, issueId, resource, action }` only;
 * the full Sentry body is loaded from packageStorage. Missing staged payloads
 * throw (visible workflow failure) — never silently succeed.
 */
export default async function processSentryWebhook(
	input: ProcessSentryWebhookInput = {},
): Promise<ProcessSentryWebhookResult> {
	const storage = packageStorage()
	const payloadKey =
		typeof input?.payloadKey === 'string' && input.payloadKey
			? input.payloadKey
			: null

	let payload = input?.request?.json
	let resource =
		input?.request?.headers?.['sentry-hook-resource'] ??
		input?.resource ??
		null

	if (payloadKey) {
		const staged = await storage.get(payloadKey).catch(() => null)
		if (staged == null || staged.json == null) {
			throw new Error(`webhook-payload-missing:${payloadKey}`)
		}
		payload = staged.json
		resource =
			staged.headers?.['sentry-hook-resource'] ?? resource ?? null
	}

	const result = await processResolvedWebhook({
		storage,
		payload,
		resource,
		input,
		payloadKey,
	})
	// Drop the staged body after a successful run so the bucket does not grow
	// unbounded; leave it in place on throw so workflow retries can still load.
	if (payloadKey) {
		await storage.delete(payloadKey).catch(() => {})
	}
	return result
}

async function processResolvedWebhook({
	storage,
	payload,
	resource,
	input,
	payloadKey,
}: {
	storage: ReturnType<typeof packageStorage>
	payload: SentryWebhookPayload | undefined
	resource: string | null
	input: ProcessSentryWebhookInput
	payloadKey: string | null
}): Promise<ProcessSentryWebhookResult> {
	// A Sentry internal integration has a single webhook URL, so Seer autofix
	// lifecycle events (sentry-hook-resource: seer) arrive here alongside
	// issue events. Route them to the Seer handler, which seeds Cole with
	// the completed root-cause analysis instead of investigating from scratch.
	if (resource === 'seer' || String(payload?.action ?? '').startsWith('root_cause') || String(payload?.action ?? '').startsWith('solution') || String(payload?.action ?? '').startsWith('coding') || payload?.action === 'pr_created') {
		const seerInput = payloadKey
			? {
					request: {
						headers: {
							'sentry-hook-resource': resource ?? 'seer',
						},
						json: payload,
					},
				}
			: input
		const seerResult = await handleSeerWebhook(seerInput)
		if (seerResult?.ok === false) {
			throw new Error(
				`seer-webhook-failed:${seerResult.error ?? 'unknown'}:${seerResult.issueId ?? ''}`,
			)
		}
		return seerResult
	}

	// Two Sentry delivery shapes are supported:
	// - "issue" resource webhooks: { action: 'created', data: { issue } }
	// - "event_alert" (alert-rule) webhooks: { action: 'triggered', data: { event } }
	let issue = payload?.data?.issue ?? null
	let action = payload?.action ?? null
	if (!issue?.id && payload?.data?.event?.issue_id) {
		const event = payload.data.event
		issue = {
			id: String(event.issue_id),
			shortId: null,
			title: event.title ?? event.message ?? 'Unknown issue',
			culprit: event.culprit ?? null,
			level: event.level ?? 'error',
			permalink: event.web_url ?? null,
			project: null,
		}
		action = 'created'
	}
	if (!issue?.id) {
		return { ok: true, skipped: 'not-an-issue-event', resource, action }
	}
	if (action !== 'created') {
		return { ok: true, skipped: `action-${action}`, issueId: issue.id }
	}
	// Backfill support for direct invocations only (Sentry never sends this
	// key; HMAC verification means real webhook payloads come from Sentry):
	// embeds a curated sibling group + coordination note into the prompt so
	// Cole triages pre-automation issues together in one run.
	const rawBackfill = payload?.kodyBackfill ?? null
	const backfill = rawBackfill
		? {
				siblings: (Array.isArray(rawBackfill.siblings)
					? rawBackfill.siblings
					: []
				)
					.slice(0, 20)
					.map((sibling) => ({
						id: String(sibling.id ?? ''),
						shortId: sibling.shortId ?? null,
						title: truncate(sibling.title ?? '', 140),
						culprit: truncate(sibling.culprit ?? '', 110),
						permalink: sibling.permalink ?? null,
					})),
				note:
					typeof rawBackfill.note === 'string'
						? truncate(rawBackfill.note, 600)
						: null,
			}
		: null
	const resolved = await resolveTriageProject(issue, payload)
	const project = resolved.project
	if (!project) {
		return { ok: true, skipped: 'unconfigured-project', projectSlug: resolved.slug }
	}

	const issueKey = issueStorageKey(issue.id)
	const existing = await storage.get(issueKey)

	// Side-effect-free routing probe (direct-invocation test only; Sentry
	// never sends kodyDryRun because payloads are HMAC-verified).
	if (payload?.kodyDryRun === true) {
		const leaseHeld = await isRepoLeaseHeld(storage, project.slug).catch(
			() => false,
		)
		const queueDepth = await getRepoQueueDepth(storage, project.slug).catch(
			() => 0,
		)
		const seerAutomationOwnsRun = await seerRunsToPr(project.slug).catch(
			() => false,
		)
		return {
			ok: true,
			dryRun: true,
			issueId: issue.id,
			projectSlug: project.slug,
			seerFirstEnabled,
			existingStatus: existing?.status ?? null,
			leaseHeld,
			queueDepth,
			seerAutomationOwnsRun,
			wouldDeferToSeer:
				seerFirstEnabled &&
				!backfill &&
				existing?.status !== 'agent-spawned' &&
				!['fixed', 'filtered', 'ignored'].includes(existing?.status),
		}
	}

	// Lazy sweeper: if a prior Cole wake never reported, its lease goes
	// stale (>3h) while issues may still sit in the per-repo queue — or a
	// crash between release and flush left a queue with no lease at all.
	// One cheap JOIN; no-op in the common case. After dry-run so the probe
	// stays side-effect-free; before claim/wake so a stranded queue does
	// not wait on this issue's outcome.
	try {
		const needsFlush = await findReposNeedingQueueFlush(storage)
		for (const slug of needsFlush) {
			await flushRepoQueue(storage, slug).catch(() => {})
		}
	} catch {
		// Schema/storage hiccups must not block the inbound issue path.
	}

	// Awaiting-rescue sweeper: with automation-owned Seer runs, an issue
	// Sentry's automation declines gets NO later webhook of its own — the
	// grace-window fallback needs a driver. The 15-minute
	// `./sweep-stale-awaiting` job is the quiet-period driver; this inbound
	// path is opportunistic and redelivers up to two stale awaiting records
	// (isRetriage → investigation agent seeded with any stored RCA). Daily
	// reconcile remains the completeness backstop. kodySweeperRescue bounds
	// recursion to depth one; Sentry never sends it (payloads are HMAC-verified).
	if (payload?.kodySweeperRescue !== true) {
		try {
			const staleAwaiting = await findStaleAwaitingIssues(storage, 2)
			for (const record of staleAwaiting) {
				const rescueAnchor =
					record.seerPrAwaitedAt ??
					record.seerRequestedAt ??
					record.seenAt ??
					'unknown'
				await workflows
					.create({
						exportName: './process-sentry-webhook',
						idempotencyKey: `sentry-triage:issue:${record.issueId}:rescue:${rescueAnchor}`,
						params: {
							request: {
								headers: { 'sentry-hook-resource': 'issue' },
								json: {
									action: 'created',
									kodySweeperRescue: true,
									data: {
										issue: {
											id: String(record.issueId),
											shortId: record.shortId ?? null,
											title: record.title ?? null,
											culprit: null,
											level: 'error',
											permalink: record.link ?? null,
											project: { slug: record.projectSlug },
										},
									},
								},
							},
						},
					})
					.catch(() => {})
			}
		} catch {
			// Rescue is best-effort; never block the inbound issue path.
		}
	}

	// A prior terminal outcome that has gone stale means this delivery is a
	// regression (the alert rule fires on regressed issues) or a retry after
	// a failure — re-triage instead of skipping forever. Fresh records still
	// dedupe; in-flight records go stale after 3h (wakes that were interrupted or
	// cancelled must not block the issue and its Discord message forever).
	// awaiting-seer records go stale after the Seer grace window: a lost
	// seer.root_cause_completed webhook must not strand the issue (observed
	// in production — Seer completed but the completion delivery was lost,
	// so the issue sat awaiting-seer forever).
	const retriageAfterMs = 6 * 60 * 60 * 1000
	const inFlightStaleMs = 3 * 60 * 60 * 1000
	let isRetriage = false
	if (existing) {
		const finishedAt =
			existing.completedAt ?? existing.spawnedAt ?? existing.seenAt ?? null
		const ageMs =
			finishedAt != null ? Date.now() - Date.parse(finishedAt) : 0
		const awaitingAgeMs =
			Date.now() -
			(Date.parse(existing.seerRequestedAt ?? finishedAt ?? '') || Date.now())
		const inFlight = existing.status === 'agent-spawned'
		// Spawn failures retry quickly (the reconciler redelivers after a
		// 15-minute cooldown); a 6h hold would strand a transient Cursor API
		// error for most of a workday.
		const spawnFailedRetryMs = 15 * 60 * 1000
		const awaitingPrAgeMs =
			Date.now() -
			(Date.parse(existing.seerPrAwaitedAt ?? finishedAt ?? '') || Date.now())
		const stale =
			existing.status === 'awaiting-seer'
				? awaitingAgeMs > seerGraceMs
				: existing.status === 'awaiting-seer-pr'
					? awaitingPrAgeMs > seerPrGraceMs
					: existing.status === 'spawn-failed'
						? ageMs > spawnFailedRetryMs
						: inFlight
							? ageMs > inFlightStaleMs
							: ageMs > retriageAfterMs
		if (!stale) {
			return { ok: true, skipped: 'already-handled', issueId: issue.id }
		}
		isRetriage = true
	}

	// Atomic claim: Sentry can deliver the same issue twice concurrently
	// (issue webhook + retries). The package storage DO serializes SQL, so a
	// plain INSERT on a primary key makes exactly one delivery win the race.
	// Retriage claims re-arm via a staleness-guarded UPDATE with one winner.
	await storage.sql(
		`CREATE TABLE IF NOT EXISTS triage_claims (
			issue_id TEXT PRIMARY KEY,
			claimed_at TEXT NOT NULL
		)`,
	)
	if (isRetriage) {
		// The re-arm guard only needs to defeat CONCURRENT duplicate
		// deliveries (seconds apart) — retriage cadence is already enforced
		// by the record-status staleness gates above. A 1h guard here
		// silently blocked awaiting-seer rescues, whose grace window (15m)
		// is shorter than an hour.
		const guardWindowAgo = new Date(Date.now() - 10 * 60 * 1000).toISOString()
		const claim = await storage.sql(
			`UPDATE triage_claims SET claimed_at = ?2
			WHERE issue_id = ?1 AND claimed_at < ?3`,
			[String(issue.id), new Date().toISOString(), guardWindowAgo],
		)
		if ((claim?.rowsWritten ?? 0) === 0) {
			// No claim row yet (record predates the claims table): insert one.
			try {
				await storage.sql(
					`INSERT INTO triage_claims (issue_id, claimed_at) VALUES (?1, ?2)`,
					[String(issue.id), new Date().toISOString()],
				)
			} catch {
				return { ok: true, skipped: 'concurrent-duplicate', issueId: issue.id }
			}
		}
	} else {
		try {
			await storage.sql(
				`INSERT INTO triage_claims (issue_id, claimed_at) VALUES (?1, ?2)`,
				[String(issue.id), new Date().toISOString()],
			)
		} catch {
			return { ok: true, skipped: 'concurrent-duplicate', issueId: issue.id }
		}
	}

	const issuesSeen = await bumpCounter(storage, issuesSeenCounterKey())
	if (issuesSeen > issueAnomalyThresholdPerHour) {
		await alertOnce(
			storage,
			'anomaly',
			`🛑 **sentry-triage paused**: ${issuesSeen} new Sentry issues this hour (threshold ${issueAnomalyThresholdPerHour}). This smells like an incident or a loop — not waking Cole. Check Sentry directly.`,
		)
		await storage.set(issueKey, {
			issueId: issue.id,
			projectSlug: project.slug,
			status: 'skipped-anomaly',
			seenAt: new Date().toISOString(),
		})
		return { ok: true, skipped: 'anomaly-breaker', issuesSeen }
	}

	const spawned = Number((await storage.get(spawnedCounterKey())) ?? 0)
	if (spawned >= maxAgentsPerHour) {
		await alertOnce(
			storage,
			'cap',
			`⏸️ **sentry-triage hourly cap reached** (${maxAgentsPerHour} Cole wakes this hour). Further new issues will be recorded but not triaged until the next hour.`,
		)
		await storage.set(issueKey, {
			issueId: issue.id,
			projectSlug: project.slug,
			status: 'skipped-cap',
			seenAt: new Date().toISOString(),
		})
		return { ok: true, skipped: 'hourly-cap', spawned }
	}

	const title = truncate(issue.title ?? 'Unknown issue', 200)
	const link =
		issue.permalink ?? `https://sentry.io/organizations/issues/${issue.id}/`

	// Seer-first: request a Seer RCA and defer waking Cole to the
	// `seer.root_cause_completed` webhook (which seeds an RCA-backed Cole wake).
	// Skips for re-triage (a prior fix regressed — investigate directly, the
	// RCA already exists) and for issues stuck awaiting Seer past the grace
	// window (Seer stalled — fall back to from-scratch so nothing is lost).
	const awaitingSeer =
		existing?.status === 'awaiting-seer' ||
		existing?.status === 'awaiting-seer-pr'
	const awaitingSince = awaitingSeer
		? Date.parse(
				existing.seerPrAwaitedAt ??
					existing.seerRequestedAt ??
					existing.seenAt ??
					'',
			) || 0
		: 0
	const awaitingGrace =
		existing?.status === 'awaiting-seer-pr' ? seerPrGraceMs : seerGraceMs
	const awaitingStale =
		awaitingSeer && Date.now() - awaitingSince > awaitingGrace
	if (seerFirstEnabled && !isRetriage && !awaitingStale && !backfill) {
		// Competing-automations guard (confirmed by the Sentry team): when a
		// project's Seer stopping point is "open_pr", Sentry's own automation
		// runs triage → RCA → draft PR, but it SKIPS issues with recent Seer
		// activity — so manually triggering an RCA here would freeze the run
		// at root cause and no PR would ever arrive. For those projects we
		// trigger nothing and let their automation own the run; the existing
		// awaiting-seer grace window still covers issues their automation
		// declines (not fixable enough), falling back to our investigation
		// agent — which routes agent spend by Sentry's own fixability triage.
		if (await seerRunsToPr(project.slug)) {
			const posted = await postSentryReport({
				status: 'queued',
				title,
				shortId: issue.shortId,
				issueId: issue.id,
				projectSlug: project.slug,
				link,
				summary: "Waiting for Sentry's Seer automation (triage → RCA → draft PR)…",
			})
			await storage.set(issueKey, {
				issueId: issue.id,
				projectSlug: project.slug,
				shortId: issue.shortId ?? null,
				title,
				link,
				status: 'awaiting-seer',
				seerAutomation: true,
				discordMessageId: posted.id,
				seerRequestedAt: new Date().toISOString(),
				seenAt: new Date().toISOString(),
			})
			return {
				ok: true,
				issueId: issue.id,
				projectSlug: project.slug,
				deferredToSeerAutomation: true,
			}
		}
		const posted = await postSentryReport({
			status: 'queued',
			title,
			shortId: issue.shortId,
			issueId: issue.id,
			projectSlug: project.slug,
			link,
			summary: 'Requesting Seer root-cause analysis…',
		})
		const trigger = await triggerSeerRca(issue.id)
		if (trigger.ok) {
			await storage.set(issueKey, {
				issueId: issue.id,
				projectSlug: project.slug,
				shortId: issue.shortId ?? null,
				title,
				link,
				status: 'awaiting-seer',
				seerRunId: trigger.runId ?? null,
				discordMessageId: posted.id,
				seerRequestedAt: new Date().toISOString(),
				seenAt: new Date().toISOString(),
			})
			return {
				ok: true,
				issueId: issue.id,
				projectSlug: project.slug,
				deferredToSeer: true,
				seerRunId: trigger.runId ?? null,
			}
		}
		// Seer could not run for this issue — fall through to from-scratch
		// triage, reusing the message we already posted.
		await editSentryReport(posted.id, {
			status: 'queued',
			title,
			shortId: issue.shortId,
			issueId: issue.id,
			projectSlug: project.slug,
			link,
			summary: `Seer RCA unavailable (${trigger.status ?? trigger.error ?? 'no run'}); triaging directly…`,
		}).catch(() => {})
		return await spawnFromScratch({
			storage,
			issue,
			project,
			link,
			title,
			isRetriage,
			existing,
			backfill,
			issueKey,
			discordMessageId: posted.id,
		})
	}

	return await spawnFromScratch({
		storage,
		issue,
		project,
		link,
		title,
		isRetriage,
		existing,
		backfill,
		issueKey,
		// Reuse the awaiting-seer message on the stale fallback so a stuck
		// issue still ends up with a single Discord message.
		discordMessageId: awaitingStale ? (existing?.discordMessageId ?? null) : null,
	})
}

/**
 * From-scratch triage: post (or reuse) the Discord status message, prefetch
 * Sentry context, and wake Cole to investigate without a Seer
 * RCA. Used for re-triage, backfill, and the Seer-unavailable fallback.
 */
async function spawnFromScratch({
	storage,
	issue,
	project,
	link,
	title,
	isRetriage,
	existing,
	backfill,
	issueKey,
	discordMessageId,
}) {
	const posted = discordMessageId
		? { id: discordMessageId }
		: await postSentryReport({
				status: 'queued',
				title,
				shortId: issue.shortId,
				issueId: issue.id,
				projectSlug: project.slug,
				link,
				summary: 'Waking Cole…',
			})

	// Per-repo slots: up to N concurrent Cole wakes per repository. When all
	// slots are busy, enqueue this issue; record-outcome (or the lazy sweeper)
	// fills free slots with one Cole wake per queued issue.
	const acquired = await acquireRepoLease(storage, project.slug, String(issue.id))
	if (!acquired) {
		await enqueueRepoIssue(storage, project.slug, issue.id, {
			issue: {
				id: String(issue.id),
				shortId: issue.shortId ?? null,
				title: issue.title ?? null,
				culprit: issue.culprit ?? null,
				level: issue.level ?? 'error',
				permalink: issue.permalink ?? link,
			},
			discordMessageId: posted.id,
			isRetriage: Boolean(isRetriage),
			// A stored RCA (from an awaiting-seer-pr rescue) still seeds the
			// eventual Cole wake.
			seerRootCause: existing?.seerRootCause ?? null,
		})
		await storage.set(issueKey, {
			issueId: issue.id,
			projectSlug: project.slug,
			backfill: Boolean(backfill),
			shortId: issue.shortId ?? null,
			title,
			link,
			status: 'queued',
			discordMessageId: posted.id,
			seenAt: new Date().toISOString(),
		})
		await editSentryReport(posted.id, {
			status: 'queued',
			title,
			shortId: issue.shortId,
			issueId: issue.id,
			projectSlug: project.slug,
			link,
			summary: 'Queued — Cole slots for this repo are full…',
		}).catch(() => {})
		return {
			ok: true,
			queued: true,
			issueId: issue.id,
			projectSlug: project.slug,
		}
	}

	// Persist the Discord message id before wake so retries edit in place.
	await storage.set(issueKey, {
		issueId: issue.id,
		projectSlug: project.slug,
		backfill: Boolean(backfill),
		shortId: issue.shortId ?? null,
		title,
		link,
		status: 'spawn-failed',
		wakePending: true,
		discordMessageId: posted.id,
		seenAt: existing?.seenAt ?? new Date().toISOString(),
	})

	const spawnContext = await gatherSpawnContext(issue.id, project)

	let agent = null
	try {
		agent = await wakeCole({
			promptText: buildTriageAgentPrompt({
				issue,
				project,
				discordMessageId: posted.id,
				context: spawnContext.sentryContext,
				issueState: spawnContext.issueState,
				priorRecord: isRetriage ? existing : null,
				...promptOwnerFields(await readTriageSettings()),
				// Rescued awaiting-seer-pr issues carry the completed RCA even
				// though Seer never delivered its PR — keep the head start.
				seerRootCause: existing?.seerRootCause ?? null,
				backfill,
			}),
			discordMessageId: posted.id,
			discordChannelId: (await readTriageSettings()).discordChannelId,
		})
	} catch (error) {
		// Release before recording spawn-failed or the repo deadlocks for 3h.
		await releaseRepoLease(storage, project.slug, String(issue.id))
		const message = truncate(
			error instanceof Error ? error.message : String(error),
			300,
		)
		await editSentryReport(posted.id, {
			status: 'queued',
			title,
			shortId: issue.shortId,
			issueId: issue.id,
			projectSlug: project.slug,
			link,
			summary: `Cole wake failed — will retry: ${message}`,
		})
		await storage.set(issueKey, {
			issueId: issue.id,
			projectSlug: project.slug,
			status: 'spawn-failed',
			discordMessageId: posted.id,
			seenAt: new Date().toISOString(),
		})
		// Throw so the durable workflow run is visibly errored (reconciler
		// still sees spawn-failed and can re-deliver after cooldown).
		throw new Error(`cole-wake-failed:${issue.id}:${message}`)
	}

	await updateRepoLeaseHolder(storage, project.slug, String(issue.id))
	await bumpCounter(storage, spawnedCounterKey())
	await storage.set(issueKey, {
		issueId: issue.id,
		projectSlug: project.slug,
		backfill: Boolean(backfill),
		shortId: issue.shortId ?? null,
		title,
		link,
		status: 'agent-spawned',
		wakePending: false,
		agentId: 'cole',
		agentUrl: 'grok-bot:cole',
		discordMessageId: posted.id,
		spawnedAt: new Date().toISOString(),
		hourBucket: hourBucket(),
	})

	await editSentryReport(posted.id, {
		status: 'investigating',
		title,
		shortId: issue.shortId,
		issueId: issue.id,
		projectSlug: project.slug,
		link,
		summary: 'Cole is triaging',
		agentUrl: 'grok-bot:cole',
	})

	return { ok: true, issueId: issue.id, projectSlug: project.slug, agentId: 'cole' }
}