Surface unresolved critical payment states as admin notifications
Adds the scan-critical-payment-logs job (daily 2:45am) that surfaces stale pending payments/till-sales and refunds at the retry cap as admin_notifications with reason='critical_payment_log' — the app has no log/alert pipeline (Gap Backlog T14), so money events that would otherwise sit in un-watched CRITICAL log lines now reach the owner's in-app notification centre. Dedup is NULL-safe (IS NOT DISTINCT FROM) and re-surfaces acknowledged-but-still-unresolved rows. Stopgap until a real alerting pipeline lands.
This commit is contained in:
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"log/slog"
|
||||
"time"
|
||||
|
||||
@@ -231,6 +232,21 @@ func RegisterAll(s *Scheduler) {
|
||||
Concurrency: 1,
|
||||
Handler: SweepSquareWebhookEvents,
|
||||
})
|
||||
|
||||
// Surfaces unresolved money events (stale pending payments/till sales,
|
||||
// refunds at the retry cap) as admin notifications so the owner sees the
|
||||
// "money may have moved at Square but the DB couldn't record it" situations
|
||||
// that would otherwise live only in un-watched CRITICAL log lines (Gap
|
||||
// Backlog T14). Retire this job when a proper log/alert pipeline (T14
|
||||
// Sentry) lands — the notifications page is the stopgap while the app has
|
||||
// none. Staggered at 2:45am between the 2am and 3am daily batches.
|
||||
s.Register(Job{
|
||||
Name: "scan-critical-payment-logs",
|
||||
Schedule: "45 2 * * *", // Daily at 2:45am
|
||||
Timeout: 30 * time.Second,
|
||||
Concurrency: 1,
|
||||
Handler: ScanCriticalPaymentLogs,
|
||||
})
|
||||
}
|
||||
|
||||
// SweepSquareWebhookEvents deletes square_webhook_events rows older than 90
|
||||
@@ -259,3 +275,67 @@ func SweepSquareWebhookEvents(ctx context.Context) (int, error) {
|
||||
|
||||
return int(tag.RowsAffected()), tx.Commit(ctx)
|
||||
}
|
||||
|
||||
// criticalPaymentStaleAge is how long a pending payment / till sale may sit
|
||||
// unresolved before the scan surfaces it to the admin notification centre. A
|
||||
// pending row with no update in this window is exactly the "money may have
|
||||
// moved at Square but the DB couldn't record it" situation the CRITICAL payment
|
||||
// logs describe (handlers.go, giftcards.go, refunds.go, till.go). Deliberately
|
||||
// shorter than the 24h stale-pending sweep (handlers/payments/sweep.go) so the
|
||||
// owner hears about it while the row could still be rescued.
|
||||
const criticalPaymentStaleAge = "2 hours"
|
||||
|
||||
// ScanCriticalPaymentLogs surfaces unresolved critical payment states as admin
|
||||
// notifications (reason='critical_payment_log'), giving the owner an in-app
|
||||
// view of money events that would otherwise be visible only in un-watched
|
||||
// CRITICAL log lines (the app has no log/alert pipeline — Gap Backlog T14).
|
||||
//
|
||||
// Candidates:
|
||||
// - payments / till_sales rows still 'pending' with no update for >2h: a
|
||||
// lost-response charge the DB never recorded.
|
||||
// - refunds rows still 'pending' at the 3-attempt retry cap: a refund the
|
||||
// sweep could not resolve (money may have moved at Square).
|
||||
//
|
||||
// Each candidate inserts one admin_notifications row keyed on (reason,
|
||||
// booking_id) — but ONLY if no unacknowledged notification for that key
|
||||
// already exists (NULL-safe via IS NOT DISTINCT FROM, matching the
|
||||
// insertRefundFailedNotifications dedup in handlers/payments/refunds.go), so
|
||||
// the bell never floods. Acknowledging the notification re-arms the scan:
|
||||
// while the row stays unresolved it is surfaced again on the next run.
|
||||
//
|
||||
// This is the "grep CRITICAL" the audit asked for, but DB-backed since the app
|
||||
// has no log pipeline. When a proper log/alert pipeline (T14 Sentry) lands,
|
||||
// this job can be retired.
|
||||
func ScanCriticalPaymentLogs(ctx context.Context) (int, error) {
|
||||
tag, err := db.Conn.Exec(ctx, `
|
||||
INSERT INTO admin_notifications (reason, booking_id, created_at)
|
||||
SELECT DISTINCT 'critical_payment_log'::admin_notification_reason, src.booking_id, NOW()
|
||||
FROM (
|
||||
-- Lost-response payments: still pending with no update past the threshold.
|
||||
SELECT id, booking_id FROM payments
|
||||
WHERE status = 'pending' AND updated_at < NOW() - INTERVAL '`+criticalPaymentStaleAge+`'
|
||||
UNION ALL
|
||||
-- Lost-response till sales (no booking linkage — booking_id NULL).
|
||||
SELECT id, NULL::CHAR(12) FROM till_sales
|
||||
WHERE status = 'pending' AND updated_at < NOW() - INTERVAL '`+criticalPaymentStaleAge+`'
|
||||
UNION ALL
|
||||
-- Refunds stuck at the 3-attempt retry cap.
|
||||
SELECT id, booking_id FROM refunds
|
||||
WHERE status = 'pending' AND refund_attempts >= 3
|
||||
) src
|
||||
WHERE NOT EXISTS (
|
||||
SELECT 1 FROM admin_notifications an
|
||||
WHERE an.reason = 'critical_payment_log'
|
||||
AND an.booking_id IS NOT DISTINCT FROM src.booking_id
|
||||
AND an.acknowledged_at IS NULL
|
||||
)
|
||||
`)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("failed to scan critical payment logs: %w", err)
|
||||
}
|
||||
n := int(tag.RowsAffected())
|
||||
if n > 0 {
|
||||
log.Printf("[SCAN] Inserted %d admin_notification(s) for unresolved critical payment states (pending payments/till sales > %s, refunds at retry cap)", n, criticalPaymentStaleAge)
|
||||
}
|
||||
return n, nil
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user