Skip to content
← Public packages

@kentcdodds/sentry-triage

Sentry triage wakes Cole (Grok Bot) per issue: one active wake per repo (lease + queue), loop-safe Discord. Cole may spawn Cursor for isolated repo work.

src/reconcile.ts

118 lines · 4.1 KB · TypeScript
import { packageStorage } from 'kody:runtime'
import searchIssues from 'kody:@kentcdodds/sentry/search-issues'
import handleSentryWebhook from './handle-sentry-webhook.ts'
import {
	issueStorageKey,
	seerGraceMs,
	seerPrGraceMs,
	truncate,
} from './shared.ts'
import { readTriageSettings } from './settings.ts'
import type { ReconcileInput, ReconcileResult } from './types.ts'

/**
 * Webhook-loss backstop: webhooks give triage its latency, this reconciler
 * gives it completeness. Deliveries can be lost with no trace on either side
 * (observed in production: issue webhooks and seer.root_cause_completed
 * webhooks silently dropped overnight — most plausibly during worker
 * deploys, since Sentry internal-integration webhooks are not retried
 * aggressively). For every registered project, list recently-first-seen
 * unresolved issues and re-deliver any that triage never recorded — plus any
 * stuck awaiting a Seer RCA past the grace window — through the normal
 * webhook handler, so every gate (dedupe, claims, caps, anomaly breaker,
 * repo lease) still applies. Quiet-period grace-window rescue is owned by
 * `./sweep-stale-awaiting` (15m). Runs from the package manifest job
 * `reconcile-daily` (cron, package runtime — a standalone kody job cannot
 * invoke package exports); safe to invoke manually for an immediate backfill.
 */
export default async function reconcile(
	input: ReconcileInput = {},
): Promise<ReconcileResult> {
	const storage = packageStorage()
	const { sentryOrgSlug, projects: triageProjects } = await readTriageSettings()
	const lookbackHours = Math.min(
		72,
		Math.max(1, Math.floor(Number(input?.lookbackHours) || 30)),
	)
	const actions = []
	for (const project of triageProjects) {
		let issues = []
		try {
			issues = await searchIssues({
				orgSlug: sentryOrgSlug,
				project: project.projectId,
				query: `is:unresolved firstSeen:-${lookbackHours}h`,
				limit: 25,
			})
		} catch (error) {
			actions.push({
				project: project.slug,
				error: truncate(
					error instanceof Error ? error.message : String(error),
					200,
				),
			})
			continue
		}
		for (const issue of Array.isArray(issues) ? issues : []) {
			const record = await storage.get(issueStorageKey(issue.id))
			const awaitingStale =
				(record?.status === 'awaiting-seer' &&
					Date.now() -
						(Date.parse(record.seerRequestedAt ?? record.seenAt ?? '') || 0) >
						seerGraceMs) ||
				(record?.status === 'awaiting-seer-pr' &&
					Date.now() -
						(Date.parse(record.seerPrAwaitedAt ?? record.seenAt ?? '') || 0) >
						seerPrGraceMs)
			// Wake failures (Cole wake hiccups) are exactly what a reconciler
			// should retry; the cooldown avoids hot-looping on a persistent
			// failure while still recovering within the daily cadence (or the next webhook).
			const spawnFailedCooldownMs = 15 * 60 * 1000
			const spawnFailedStale =
				record?.status === 'spawn-failed' &&
				Date.now() - (Date.parse(record.seenAt ?? '') || 0) >
					spawnFailedCooldownMs
			if (record && !awaitingStale && !spawnFailedStale) continue
			// Re-deliver through the real handler (not a wake shortcut) so the
			// claim, cap, anomaly, Seer, and repo-lease logic all run normally.
			const delivery = await handleSentryWebhook({
				request: {
					headers: { 'sentry-hook-resource': 'issue' },
					json: {
						action: 'created',
						data: {
							issue: {
								id: String(issue.id),
								shortId: issue.shortId ?? null,
								title: issue.title ?? null,
								culprit: issue.culprit ?? null,
								level: issue.level ?? 'error',
								permalink: issue.permalink ?? null,
								project: { slug: project.slug },
							},
						},
					},
				},
			}).catch((error) => ({
				ok: false,
				error: truncate(
					error instanceof Error ? error.message : String(error),
					200,
				),
			}))
			actions.push({
				project: project.slug,
				issueId: String(issue.id),
				shortId: issue.shortId ?? null,
				reason: !record
					? 'missing-record'
					: awaitingStale
						? 'awaiting-seer-stale'
						: 'spawn-failed-stale',
				delivery,
			})
		}
	}
	return { ok: true, lookbackHours, reconciled: actions.length, actions }
}