← Public packages
@kentcdodds/sentry-triage
Sentry triage wakes Cole (Grok Bot) per issue: one active wake per repo (lease + queue), loop-safe Discord. Cole may spawn Cursor for isolated repo work.
src/reconcile.ts
118 lines · 4.1 KB · TypeScriptimport { packageStorage } from 'kody:runtime'
import searchIssues from 'kody:@kentcdodds/sentry/search-issues'
import handleSentryWebhook from './handle-sentry-webhook.ts'
import {
issueStorageKey,
seerGraceMs,
seerPrGraceMs,
truncate,
} from './shared.ts'
import { readTriageSettings } from './settings.ts'
import type { ReconcileInput, ReconcileResult } from './types.ts'
/**
* Webhook-loss backstop: webhooks give triage its latency, this reconciler
* gives it completeness. Deliveries can be lost with no trace on either side
* (observed in production: issue webhooks and seer.root_cause_completed
* webhooks silently dropped overnight — most plausibly during worker
* deploys, since Sentry internal-integration webhooks are not retried
* aggressively). For every registered project, list recently-first-seen
* unresolved issues and re-deliver any that triage never recorded — plus any
* stuck awaiting a Seer RCA past the grace window — through the normal
* webhook handler, so every gate (dedupe, claims, caps, anomaly breaker,
* repo lease) still applies. Quiet-period grace-window rescue is owned by
* `./sweep-stale-awaiting` (15m). Runs from the package manifest job
* `reconcile-daily` (cron, package runtime — a standalone kody job cannot
* invoke package exports); safe to invoke manually for an immediate backfill.
*/
export default async function reconcile(
input: ReconcileInput = {},
): Promise<ReconcileResult> {
const storage = packageStorage()
const { sentryOrgSlug, projects: triageProjects } = await readTriageSettings()
const lookbackHours = Math.min(
72,
Math.max(1, Math.floor(Number(input?.lookbackHours) || 30)),
)
const actions = []
for (const project of triageProjects) {
let issues = []
try {
issues = await searchIssues({
orgSlug: sentryOrgSlug,
project: project.projectId,
query: `is:unresolved firstSeen:-${lookbackHours}h`,
limit: 25,
})
} catch (error) {
actions.push({
project: project.slug,
error: truncate(
error instanceof Error ? error.message : String(error),
200,
),
})
continue
}
for (const issue of Array.isArray(issues) ? issues : []) {
const record = await storage.get(issueStorageKey(issue.id))
const awaitingStale =
(record?.status === 'awaiting-seer' &&
Date.now() -
(Date.parse(record.seerRequestedAt ?? record.seenAt ?? '') || 0) >
seerGraceMs) ||
(record?.status === 'awaiting-seer-pr' &&
Date.now() -
(Date.parse(record.seerPrAwaitedAt ?? record.seenAt ?? '') || 0) >
seerPrGraceMs)
// Wake failures (Cole wake hiccups) are exactly what a reconciler
// should retry; the cooldown avoids hot-looping on a persistent
// failure while still recovering within the daily cadence (or the next webhook).
const spawnFailedCooldownMs = 15 * 60 * 1000
const spawnFailedStale =
record?.status === 'spawn-failed' &&
Date.now() - (Date.parse(record.seenAt ?? '') || 0) >
spawnFailedCooldownMs
if (record && !awaitingStale && !spawnFailedStale) continue
// Re-deliver through the real handler (not a wake shortcut) so the
// claim, cap, anomaly, Seer, and repo-lease logic all run normally.
const delivery = await handleSentryWebhook({
request: {
headers: { 'sentry-hook-resource': 'issue' },
json: {
action: 'created',
data: {
issue: {
id: String(issue.id),
shortId: issue.shortId ?? null,
title: issue.title ?? null,
culprit: issue.culprit ?? null,
level: issue.level ?? 'error',
permalink: issue.permalink ?? null,
project: { slug: project.slug },
},
},
},
},
}).catch((error) => ({
ok: false,
error: truncate(
error instanceof Error ? error.message : String(error),
200,
),
}))
actions.push({
project: project.slug,
issueId: String(issue.id),
shortId: issue.shortId ?? null,
reason: !record
? 'missing-record'
: awaitingStale
? 'awaiting-seer-stale'
: 'spawn-failed-stale',
delivery,
})
}
}
return { ok: true, lookbackHours, reconciled: actions.length, actions }
}