← Public packages
@kentcdodds/sentry-triage
Sentry triage wakes Cole (Grok Bot) per issue: one active wake per repo (lease + queue), loop-safe Discord. Cole may spawn Cursor for isolated repo work.
src/process-sentry-webhook.ts
758 lines · 23.7 KB · TypeScriptimport { packageStorage, workflows } from 'kody:runtime'
import {
postSentryReport,
editSentryReport,
} from './format-discord-report.ts'
import getIssue from 'kody:@kentcdodds/sentry/get-issue'
import { buildTriageAgentPrompt } from './agent-prompt.ts'
import { gatherSpawnContext } from './spawn-snapshot.ts'
import handleSeerWebhook from './handle-seer-webhook.ts'
import {
acquireRepoLease,
wakeCole,
enqueueRepoIssue,
findReposNeedingQueueFlush,
findStaleAwaitingIssues,
flushRepoQueue,
getRepoQueueDepth,
isRepoLeaseHeld,
releaseRepoLease,
seerRunsToPr,
triggerSeerRca,
updateRepoLeaseHolder,
} from './triage-core.ts'
import {
alertSentKey,
hourBucket,
issueAnomalyThresholdPerHour,
issueStorageKey,
issuesSeenCounterKey,
maxAgentsPerHour,
seerFirstEnabled,
seerGraceMs,
seerPrGraceMs,
spawnedCounterKey,
truncate,
} from './shared.ts'
import {
promptOwnerFields,
readTriageSettings,
resolveTriageProjectForProjectId,
resolveTriageProjectForSlug,
} from './settings.ts'
import type {
ProcessSentryWebhookInput,
ProcessSentryWebhookResult,
SentryWebhookPayload,
} from './types.ts'
/**
* Resolve which triage project an inbound delivery belongs to. Issue-resource
* webhooks carry `issue.project.slug`; alert-rule ("event_alert") webhooks
* carry the numeric project id on the event; as a last resort, look the issue
* up in the Sentry API. Returns null when the project is not in the registry.
*/
async function resolveTriageProject(issue, payload) {
const slug = issue.project?.slug ?? null
if (slug) return { project: await resolveTriageProjectForSlug(slug), slug }
const eventProjectId = payload?.data?.event?.project ?? null
if (eventProjectId != null) {
const project = await resolveTriageProjectForProjectId(eventProjectId)
if (project) return { project, slug: project.slug }
}
try {
const data = await getIssue({ issueId: issue.id })
const fetchedSlug = data?.project?.slug ?? null
if (fetchedSlug) {
return { project: await resolveTriageProjectForSlug(fetchedSlug), slug: fetchedSlug }
}
} catch {
// Fall through to unresolved.
}
return { project: null, slug: eventProjectId != null ? String(eventProjectId) : null }
}
async function bumpCounter(storage, key) {
const current = Number((await storage.get(key)) ?? 0)
const next = current + 1
await storage.set(key, next)
return next
}
async function alertOnce(storage, kind, content) {
const key = alertSentKey(kind)
if (await storage.get(key)) return false
await storage.set(key, new Date().toISOString())
await postSentryReport({
status: 'queued',
title: kind === 'anomaly' ? 'Sentry triage paused' : 'Sentry triage alert',
summary: content,
})
return true
}
/**
* Durable Sentry webhook processor (issue + Seer). Invoked by
* `./handle-sentry-webhook` via `workflows.create` so the ingress export can
* acknowledge quickly while prefetch / Discord / Cole wake run with a
* workflow budget. Failures throw so the workflow run surfaces as errored.
*
* Workflow dispatches pass `{ payloadKey, issueId, resource, action }` only;
* the full Sentry body is loaded from packageStorage. Missing staged payloads
* throw (visible workflow failure) — never silently succeed.
*/
export default async function processSentryWebhook(
input: ProcessSentryWebhookInput = {},
): Promise<ProcessSentryWebhookResult> {
const storage = packageStorage()
const payloadKey =
typeof input?.payloadKey === 'string' && input.payloadKey
? input.payloadKey
: null
let payload = input?.request?.json
let resource =
input?.request?.headers?.['sentry-hook-resource'] ??
input?.resource ??
null
if (payloadKey) {
const staged = await storage.get(payloadKey).catch(() => null)
if (staged == null || staged.json == null) {
throw new Error(`webhook-payload-missing:${payloadKey}`)
}
payload = staged.json
resource =
staged.headers?.['sentry-hook-resource'] ?? resource ?? null
}
const result = await processResolvedWebhook({
storage,
payload,
resource,
input,
payloadKey,
})
// Drop the staged body after a successful run so the bucket does not grow
// unbounded; leave it in place on throw so workflow retries can still load.
if (payloadKey) {
await storage.delete(payloadKey).catch(() => {})
}
return result
}
async function processResolvedWebhook({
storage,
payload,
resource,
input,
payloadKey,
}: {
storage: ReturnType<typeof packageStorage>
payload: SentryWebhookPayload | undefined
resource: string | null
input: ProcessSentryWebhookInput
payloadKey: string | null
}): Promise<ProcessSentryWebhookResult> {
// A Sentry internal integration has a single webhook URL, so Seer autofix
// lifecycle events (sentry-hook-resource: seer) arrive here alongside
// issue events. Route them to the Seer handler, which seeds Cole with
// the completed root-cause analysis instead of investigating from scratch.
if (resource === 'seer' || String(payload?.action ?? '').startsWith('root_cause') || String(payload?.action ?? '').startsWith('solution') || String(payload?.action ?? '').startsWith('coding') || payload?.action === 'pr_created') {
const seerInput = payloadKey
? {
request: {
headers: {
'sentry-hook-resource': resource ?? 'seer',
},
json: payload,
},
}
: input
const seerResult = await handleSeerWebhook(seerInput)
if (seerResult?.ok === false) {
throw new Error(
`seer-webhook-failed:${seerResult.error ?? 'unknown'}:${seerResult.issueId ?? ''}`,
)
}
return seerResult
}
// Two Sentry delivery shapes are supported:
// - "issue" resource webhooks: { action: 'created', data: { issue } }
// - "event_alert" (alert-rule) webhooks: { action: 'triggered', data: { event } }
let issue = payload?.data?.issue ?? null
let action = payload?.action ?? null
if (!issue?.id && payload?.data?.event?.issue_id) {
const event = payload.data.event
issue = {
id: String(event.issue_id),
shortId: null,
title: event.title ?? event.message ?? 'Unknown issue',
culprit: event.culprit ?? null,
level: event.level ?? 'error',
permalink: event.web_url ?? null,
project: null,
}
action = 'created'
}
if (!issue?.id) {
return { ok: true, skipped: 'not-an-issue-event', resource, action }
}
if (action !== 'created') {
return { ok: true, skipped: `action-${action}`, issueId: issue.id }
}
// Backfill support for direct invocations only (Sentry never sends this
// key; HMAC verification means real webhook payloads come from Sentry):
// embeds a curated sibling group + coordination note into the prompt so
// Cole triages pre-automation issues together in one run.
const rawBackfill = payload?.kodyBackfill ?? null
const backfill = rawBackfill
? {
siblings: (Array.isArray(rawBackfill.siblings)
? rawBackfill.siblings
: []
)
.slice(0, 20)
.map((sibling) => ({
id: String(sibling.id ?? ''),
shortId: sibling.shortId ?? null,
title: truncate(sibling.title ?? '', 140),
culprit: truncate(sibling.culprit ?? '', 110),
permalink: sibling.permalink ?? null,
})),
note:
typeof rawBackfill.note === 'string'
? truncate(rawBackfill.note, 600)
: null,
}
: null
const resolved = await resolveTriageProject(issue, payload)
const project = resolved.project
if (!project) {
return { ok: true, skipped: 'unconfigured-project', projectSlug: resolved.slug }
}
const issueKey = issueStorageKey(issue.id)
const existing = await storage.get(issueKey)
// Side-effect-free routing probe (direct-invocation test only; Sentry
// never sends kodyDryRun because payloads are HMAC-verified).
if (payload?.kodyDryRun === true) {
const leaseHeld = await isRepoLeaseHeld(storage, project.slug).catch(
() => false,
)
const queueDepth = await getRepoQueueDepth(storage, project.slug).catch(
() => 0,
)
const seerAutomationOwnsRun = await seerRunsToPr(project.slug).catch(
() => false,
)
return {
ok: true,
dryRun: true,
issueId: issue.id,
projectSlug: project.slug,
seerFirstEnabled,
existingStatus: existing?.status ?? null,
leaseHeld,
queueDepth,
seerAutomationOwnsRun,
wouldDeferToSeer:
seerFirstEnabled &&
!backfill &&
existing?.status !== 'agent-spawned' &&
!['fixed', 'filtered', 'ignored'].includes(existing?.status),
}
}
// Lazy sweeper: if a prior Cole wake never reported, its lease goes
// stale (>3h) while issues may still sit in the per-repo queue — or a
// crash between release and flush left a queue with no lease at all.
// One cheap JOIN; no-op in the common case. After dry-run so the probe
// stays side-effect-free; before claim/wake so a stranded queue does
// not wait on this issue's outcome.
try {
const needsFlush = await findReposNeedingQueueFlush(storage)
for (const slug of needsFlush) {
await flushRepoQueue(storage, slug).catch(() => {})
}
} catch {
// Schema/storage hiccups must not block the inbound issue path.
}
// Awaiting-rescue sweeper: with automation-owned Seer runs, an issue
// Sentry's automation declines gets NO later webhook of its own — the
// grace-window fallback needs a driver. The 15-minute
// `./sweep-stale-awaiting` job is the quiet-period driver; this inbound
// path is opportunistic and redelivers up to two stale awaiting records
// (isRetriage → investigation agent seeded with any stored RCA). Daily
// reconcile remains the completeness backstop. kodySweeperRescue bounds
// recursion to depth one; Sentry never sends it (payloads are HMAC-verified).
if (payload?.kodySweeperRescue !== true) {
try {
const staleAwaiting = await findStaleAwaitingIssues(storage, 2)
for (const record of staleAwaiting) {
const rescueAnchor =
record.seerPrAwaitedAt ??
record.seerRequestedAt ??
record.seenAt ??
'unknown'
await workflows
.create({
exportName: './process-sentry-webhook',
idempotencyKey: `sentry-triage:issue:${record.issueId}:rescue:${rescueAnchor}`,
params: {
request: {
headers: { 'sentry-hook-resource': 'issue' },
json: {
action: 'created',
kodySweeperRescue: true,
data: {
issue: {
id: String(record.issueId),
shortId: record.shortId ?? null,
title: record.title ?? null,
culprit: null,
level: 'error',
permalink: record.link ?? null,
project: { slug: record.projectSlug },
},
},
},
},
},
})
.catch(() => {})
}
} catch {
// Rescue is best-effort; never block the inbound issue path.
}
}
// A prior terminal outcome that has gone stale means this delivery is a
// regression (the alert rule fires on regressed issues) or a retry after
// a failure — re-triage instead of skipping forever. Fresh records still
// dedupe; in-flight records go stale after 3h (wakes that were interrupted or
// cancelled must not block the issue and its Discord message forever).
// awaiting-seer records go stale after the Seer grace window: a lost
// seer.root_cause_completed webhook must not strand the issue (observed
// in production — Seer completed but the completion delivery was lost,
// so the issue sat awaiting-seer forever).
const retriageAfterMs = 6 * 60 * 60 * 1000
const inFlightStaleMs = 3 * 60 * 60 * 1000
let isRetriage = false
if (existing) {
const finishedAt =
existing.completedAt ?? existing.spawnedAt ?? existing.seenAt ?? null
const ageMs =
finishedAt != null ? Date.now() - Date.parse(finishedAt) : 0
const awaitingAgeMs =
Date.now() -
(Date.parse(existing.seerRequestedAt ?? finishedAt ?? '') || Date.now())
const inFlight = existing.status === 'agent-spawned'
// Spawn failures retry quickly (the reconciler redelivers after a
// 15-minute cooldown); a 6h hold would strand a transient Cursor API
// error for most of a workday.
const spawnFailedRetryMs = 15 * 60 * 1000
const awaitingPrAgeMs =
Date.now() -
(Date.parse(existing.seerPrAwaitedAt ?? finishedAt ?? '') || Date.now())
const stale =
existing.status === 'awaiting-seer'
? awaitingAgeMs > seerGraceMs
: existing.status === 'awaiting-seer-pr'
? awaitingPrAgeMs > seerPrGraceMs
: existing.status === 'spawn-failed'
? ageMs > spawnFailedRetryMs
: inFlight
? ageMs > inFlightStaleMs
: ageMs > retriageAfterMs
if (!stale) {
return { ok: true, skipped: 'already-handled', issueId: issue.id }
}
isRetriage = true
}
// Atomic claim: Sentry can deliver the same issue twice concurrently
// (issue webhook + retries). The package storage DO serializes SQL, so a
// plain INSERT on a primary key makes exactly one delivery win the race.
// Retriage claims re-arm via a staleness-guarded UPDATE with one winner.
await storage.sql(
`CREATE TABLE IF NOT EXISTS triage_claims (
issue_id TEXT PRIMARY KEY,
claimed_at TEXT NOT NULL
)`,
)
if (isRetriage) {
// The re-arm guard only needs to defeat CONCURRENT duplicate
// deliveries (seconds apart) — retriage cadence is already enforced
// by the record-status staleness gates above. A 1h guard here
// silently blocked awaiting-seer rescues, whose grace window (15m)
// is shorter than an hour.
const guardWindowAgo = new Date(Date.now() - 10 * 60 * 1000).toISOString()
const claim = await storage.sql(
`UPDATE triage_claims SET claimed_at = ?2
WHERE issue_id = ?1 AND claimed_at < ?3`,
[String(issue.id), new Date().toISOString(), guardWindowAgo],
)
if ((claim?.rowsWritten ?? 0) === 0) {
// No claim row yet (record predates the claims table): insert one.
try {
await storage.sql(
`INSERT INTO triage_claims (issue_id, claimed_at) VALUES (?1, ?2)`,
[String(issue.id), new Date().toISOString()],
)
} catch {
return { ok: true, skipped: 'concurrent-duplicate', issueId: issue.id }
}
}
} else {
try {
await storage.sql(
`INSERT INTO triage_claims (issue_id, claimed_at) VALUES (?1, ?2)`,
[String(issue.id), new Date().toISOString()],
)
} catch {
return { ok: true, skipped: 'concurrent-duplicate', issueId: issue.id }
}
}
const issuesSeen = await bumpCounter(storage, issuesSeenCounterKey())
if (issuesSeen > issueAnomalyThresholdPerHour) {
await alertOnce(
storage,
'anomaly',
`🛑 **sentry-triage paused**: ${issuesSeen} new Sentry issues this hour (threshold ${issueAnomalyThresholdPerHour}). This smells like an incident or a loop — not waking Cole. Check Sentry directly.`,
)
await storage.set(issueKey, {
issueId: issue.id,
projectSlug: project.slug,
status: 'skipped-anomaly',
seenAt: new Date().toISOString(),
})
return { ok: true, skipped: 'anomaly-breaker', issuesSeen }
}
const spawned = Number((await storage.get(spawnedCounterKey())) ?? 0)
if (spawned >= maxAgentsPerHour) {
await alertOnce(
storage,
'cap',
`⏸️ **sentry-triage hourly cap reached** (${maxAgentsPerHour} Cole wakes this hour). Further new issues will be recorded but not triaged until the next hour.`,
)
await storage.set(issueKey, {
issueId: issue.id,
projectSlug: project.slug,
status: 'skipped-cap',
seenAt: new Date().toISOString(),
})
return { ok: true, skipped: 'hourly-cap', spawned }
}
const title = truncate(issue.title ?? 'Unknown issue', 200)
const link =
issue.permalink ?? `https://sentry.io/organizations/issues/${issue.id}/`
// Seer-first: request a Seer RCA and defer waking Cole to the
// `seer.root_cause_completed` webhook (which seeds an RCA-backed Cole wake).
// Skips for re-triage (a prior fix regressed — investigate directly, the
// RCA already exists) and for issues stuck awaiting Seer past the grace
// window (Seer stalled — fall back to from-scratch so nothing is lost).
const awaitingSeer =
existing?.status === 'awaiting-seer' ||
existing?.status === 'awaiting-seer-pr'
const awaitingSince = awaitingSeer
? Date.parse(
existing.seerPrAwaitedAt ??
existing.seerRequestedAt ??
existing.seenAt ??
'',
) || 0
: 0
const awaitingGrace =
existing?.status === 'awaiting-seer-pr' ? seerPrGraceMs : seerGraceMs
const awaitingStale =
awaitingSeer && Date.now() - awaitingSince > awaitingGrace
if (seerFirstEnabled && !isRetriage && !awaitingStale && !backfill) {
// Competing-automations guard (confirmed by the Sentry team): when a
// project's Seer stopping point is "open_pr", Sentry's own automation
// runs triage → RCA → draft PR, but it SKIPS issues with recent Seer
// activity — so manually triggering an RCA here would freeze the run
// at root cause and no PR would ever arrive. For those projects we
// trigger nothing and let their automation own the run; the existing
// awaiting-seer grace window still covers issues their automation
// declines (not fixable enough), falling back to our investigation
// agent — which routes agent spend by Sentry's own fixability triage.
if (await seerRunsToPr(project.slug)) {
const posted = await postSentryReport({
status: 'queued',
title,
shortId: issue.shortId,
issueId: issue.id,
projectSlug: project.slug,
link,
summary: "Waiting for Sentry's Seer automation (triage → RCA → draft PR)…",
})
await storage.set(issueKey, {
issueId: issue.id,
projectSlug: project.slug,
shortId: issue.shortId ?? null,
title,
link,
status: 'awaiting-seer',
seerAutomation: true,
discordMessageId: posted.id,
seerRequestedAt: new Date().toISOString(),
seenAt: new Date().toISOString(),
})
return {
ok: true,
issueId: issue.id,
projectSlug: project.slug,
deferredToSeerAutomation: true,
}
}
const posted = await postSentryReport({
status: 'queued',
title,
shortId: issue.shortId,
issueId: issue.id,
projectSlug: project.slug,
link,
summary: 'Requesting Seer root-cause analysis…',
})
const trigger = await triggerSeerRca(issue.id)
if (trigger.ok) {
await storage.set(issueKey, {
issueId: issue.id,
projectSlug: project.slug,
shortId: issue.shortId ?? null,
title,
link,
status: 'awaiting-seer',
seerRunId: trigger.runId ?? null,
discordMessageId: posted.id,
seerRequestedAt: new Date().toISOString(),
seenAt: new Date().toISOString(),
})
return {
ok: true,
issueId: issue.id,
projectSlug: project.slug,
deferredToSeer: true,
seerRunId: trigger.runId ?? null,
}
}
// Seer could not run for this issue — fall through to from-scratch
// triage, reusing the message we already posted.
await editSentryReport(posted.id, {
status: 'queued',
title,
shortId: issue.shortId,
issueId: issue.id,
projectSlug: project.slug,
link,
summary: `Seer RCA unavailable (${trigger.status ?? trigger.error ?? 'no run'}); triaging directly…`,
}).catch(() => {})
return await spawnFromScratch({
storage,
issue,
project,
link,
title,
isRetriage,
existing,
backfill,
issueKey,
discordMessageId: posted.id,
})
}
return await spawnFromScratch({
storage,
issue,
project,
link,
title,
isRetriage,
existing,
backfill,
issueKey,
// Reuse the awaiting-seer message on the stale fallback so a stuck
// issue still ends up with a single Discord message.
discordMessageId: awaitingStale ? (existing?.discordMessageId ?? null) : null,
})
}
/**
* From-scratch triage: post (or reuse) the Discord status message, prefetch
* Sentry context, and wake Cole to investigate without a Seer
* RCA. Used for re-triage, backfill, and the Seer-unavailable fallback.
*/
async function spawnFromScratch({
storage,
issue,
project,
link,
title,
isRetriage,
existing,
backfill,
issueKey,
discordMessageId,
}) {
const posted = discordMessageId
? { id: discordMessageId }
: await postSentryReport({
status: 'queued',
title,
shortId: issue.shortId,
issueId: issue.id,
projectSlug: project.slug,
link,
summary: 'Waking Cole…',
})
// Per-repo slots: up to N concurrent Cole wakes per repository. When all
// slots are busy, enqueue this issue; record-outcome (or the lazy sweeper)
// fills free slots with one Cole wake per queued issue.
const acquired = await acquireRepoLease(storage, project.slug, String(issue.id))
if (!acquired) {
await enqueueRepoIssue(storage, project.slug, issue.id, {
issue: {
id: String(issue.id),
shortId: issue.shortId ?? null,
title: issue.title ?? null,
culprit: issue.culprit ?? null,
level: issue.level ?? 'error',
permalink: issue.permalink ?? link,
},
discordMessageId: posted.id,
isRetriage: Boolean(isRetriage),
// A stored RCA (from an awaiting-seer-pr rescue) still seeds the
// eventual Cole wake.
seerRootCause: existing?.seerRootCause ?? null,
})
await storage.set(issueKey, {
issueId: issue.id,
projectSlug: project.slug,
backfill: Boolean(backfill),
shortId: issue.shortId ?? null,
title,
link,
status: 'queued',
discordMessageId: posted.id,
seenAt: new Date().toISOString(),
})
await editSentryReport(posted.id, {
status: 'queued',
title,
shortId: issue.shortId,
issueId: issue.id,
projectSlug: project.slug,
link,
summary: 'Queued — Cole slots for this repo are full…',
}).catch(() => {})
return {
ok: true,
queued: true,
issueId: issue.id,
projectSlug: project.slug,
}
}
// Persist the Discord message id before wake so retries edit in place.
await storage.set(issueKey, {
issueId: issue.id,
projectSlug: project.slug,
backfill: Boolean(backfill),
shortId: issue.shortId ?? null,
title,
link,
status: 'spawn-failed',
wakePending: true,
discordMessageId: posted.id,
seenAt: existing?.seenAt ?? new Date().toISOString(),
})
const spawnContext = await gatherSpawnContext(issue.id, project)
let agent = null
try {
agent = await wakeCole({
promptText: buildTriageAgentPrompt({
issue,
project,
discordMessageId: posted.id,
context: spawnContext.sentryContext,
issueState: spawnContext.issueState,
priorRecord: isRetriage ? existing : null,
...promptOwnerFields(await readTriageSettings()),
// Rescued awaiting-seer-pr issues carry the completed RCA even
// though Seer never delivered its PR — keep the head start.
seerRootCause: existing?.seerRootCause ?? null,
backfill,
}),
discordMessageId: posted.id,
discordChannelId: (await readTriageSettings()).discordChannelId,
})
} catch (error) {
// Release before recording spawn-failed or the repo deadlocks for 3h.
await releaseRepoLease(storage, project.slug, String(issue.id))
const message = truncate(
error instanceof Error ? error.message : String(error),
300,
)
await editSentryReport(posted.id, {
status: 'queued',
title,
shortId: issue.shortId,
issueId: issue.id,
projectSlug: project.slug,
link,
summary: `Cole wake failed — will retry: ${message}`,
})
await storage.set(issueKey, {
issueId: issue.id,
projectSlug: project.slug,
status: 'spawn-failed',
discordMessageId: posted.id,
seenAt: new Date().toISOString(),
})
// Throw so the durable workflow run is visibly errored (reconciler
// still sees spawn-failed and can re-deliver after cooldown).
throw new Error(`cole-wake-failed:${issue.id}:${message}`)
}
await updateRepoLeaseHolder(storage, project.slug, String(issue.id))
await bumpCounter(storage, spawnedCounterKey())
await storage.set(issueKey, {
issueId: issue.id,
projectSlug: project.slug,
backfill: Boolean(backfill),
shortId: issue.shortId ?? null,
title,
link,
status: 'agent-spawned',
wakePending: false,
agentId: 'cole',
agentUrl: 'grok-bot:cole',
discordMessageId: posted.id,
spawnedAt: new Date().toISOString(),
hourBucket: hourBucket(),
})
await editSentryReport(posted.id, {
status: 'investigating',
title,
shortId: issue.shortId,
issueId: issue.id,
projectSlug: project.slug,
link,
summary: 'Cole is triaging',
agentUrl: 'grok-bot:cole',
})
return { ok: true, issueId: issue.id, projectSlug: project.slug, agentId: 'cole' }
}