Surface unresolved critical payment states as admin notifications

Adds the scan-critical-payment-logs job (daily 2:45am) that surfaces stale pending payments/till-sales and refunds at the retry cap as admin_notifications with reason='critical_payment_log' — the app has no log/alert pipeline (Gap Backlog T14), so money events that would otherwise sit in un-watched CRITICAL log lines now reach the owner's in-app notification centre. Dedup is NULL-safe (IS NOT DISTINCT FROM) and re-surfaces acknowledged-but-still-unresolved rows. Stopgap until a real alerting pipeline lands.
This commit is contained in:
2026-08-22 00:34:49 +01:00
parent 8a3a7ec062
commit 5605402e13
3 changed files with 319 additions and 2 deletions
+80
View File
@@ -4,6 +4,7 @@ import (
"context"
"errors"
"fmt"
"log"
"log/slog"
"time"
@@ -231,6 +232,21 @@ func RegisterAll(s *Scheduler) {
Concurrency: 1,
Handler: SweepSquareWebhookEvents,
})
// Surfaces unresolved money events (stale pending payments/till sales,
// refunds at the retry cap) as admin notifications so the owner sees the
// "money may have moved at Square but the DB couldn't record it" situations
// that would otherwise live only in un-watched CRITICAL log lines (Gap
// Backlog T14). Retire this job when a proper log/alert pipeline (T14
// Sentry) lands — the notifications page is the stopgap while the app has
// none. Staggered at 2:45am between the 2am and 3am daily batches.
s.Register(Job{
Name: "scan-critical-payment-logs",
Schedule: "45 2 * * *", // Daily at 2:45am
Timeout: 30 * time.Second,
Concurrency: 1,
Handler: ScanCriticalPaymentLogs,
})
}
// SweepSquareWebhookEvents deletes square_webhook_events rows older than 90
@@ -259,3 +275,67 @@ func SweepSquareWebhookEvents(ctx context.Context) (int, error) {
return int(tag.RowsAffected()), tx.Commit(ctx)
}
// criticalPaymentStaleAge is how long a pending payment / till sale may sit
// unresolved before the scan surfaces it to the admin notification centre. A
// pending row with no update in this window is exactly the "money may have
// moved at Square but the DB couldn't record it" situation the CRITICAL payment
// logs describe (handlers.go, giftcards.go, refunds.go, till.go). Deliberately
// shorter than the 24h stale-pending sweep (handlers/payments/sweep.go) so the
// owner hears about it while the row could still be rescued.
const criticalPaymentStaleAge = "2 hours"
// ScanCriticalPaymentLogs surfaces unresolved critical payment states as admin
// notifications (reason='critical_payment_log'), giving the owner an in-app
// view of money events that would otherwise be visible only in un-watched
// CRITICAL log lines (the app has no log/alert pipeline — Gap Backlog T14).
//
// Candidates:
// - payments / till_sales rows still 'pending' with no update for >2h: a
// lost-response charge the DB never recorded.
// - refunds rows still 'pending' at the 3-attempt retry cap: a refund the
// sweep could not resolve (money may have moved at Square).
//
// Each candidate inserts one admin_notifications row keyed on (reason,
// booking_id) — but ONLY if no unacknowledged notification for that key
// already exists (NULL-safe via IS NOT DISTINCT FROM, matching the
// insertRefundFailedNotifications dedup in handlers/payments/refunds.go), so
// the bell never floods. Acknowledging the notification re-arms the scan:
// while the row stays unresolved it is surfaced again on the next run.
//
// This is the "grep CRITICAL" the audit asked for, but DB-backed since the app
// has no log pipeline. When a proper log/alert pipeline (T14 Sentry) lands,
// this job can be retired.
func ScanCriticalPaymentLogs(ctx context.Context) (int, error) {
tag, err := db.Conn.Exec(ctx, `
INSERT INTO admin_notifications (reason, booking_id, created_at)
SELECT DISTINCT 'critical_payment_log'::admin_notification_reason, src.booking_id, NOW()
FROM (
-- Lost-response payments: still pending with no update past the threshold.
SELECT id, booking_id FROM payments
WHERE status = 'pending' AND updated_at < NOW() - INTERVAL '`+criticalPaymentStaleAge+`'
UNION ALL
-- Lost-response till sales (no booking linkage — booking_id NULL).
SELECT id, NULL::CHAR(12) FROM till_sales
WHERE status = 'pending' AND updated_at < NOW() - INTERVAL '`+criticalPaymentStaleAge+`'
UNION ALL
-- Refunds stuck at the 3-attempt retry cap.
SELECT id, booking_id FROM refunds
WHERE status = 'pending' AND refund_attempts >= 3
) src
WHERE NOT EXISTS (
SELECT 1 FROM admin_notifications an
WHERE an.reason = 'critical_payment_log'
AND an.booking_id IS NOT DISTINCT FROM src.booking_id
AND an.acknowledged_at IS NULL
)
`)
if err != nil {
return 0, fmt.Errorf("failed to scan critical payment logs: %w", err)
}
n := int(tag.RowsAffected())
if n > 0 {
log.Printf("[SCAN] Inserted %d admin_notification(s) for unresolved critical payment states (pending payments/till sales > %s, refunds at retry cap)", n, criticalPaymentStaleAge)
}
return n, nil
}