diff --git a/cmd/main.go b/cmd/main.go
index 8876a29..ebd0b34 100644
--- a/cmd/main.go
+++ b/cmd/main.go
@@ -33,6 +33,13 @@ func main() {
if err != nil {
logger.Error("fail to init checker", zap.Error(err))
}
+
+ // Wire the checker's publisher to the app's single delivery worker so
+ // checker-driven transitions wake it immediately (same shared queue).
+ if ch != nil {
+ ch.Publisher().SetNotify(s.NotifyFunc())
+ }
+
stopCh := make(chan struct{})
ctx, done := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM)
diff --git a/db/migrations/000008_notification.down.sql b/db/migrations/000008_notification.down.sql
new file mode 100644
index 0000000..6d664d1
--- /dev/null
+++ b/db/migrations/000008_notification.down.sql
@@ -0,0 +1 @@
+DROP TABLE IF EXISTS notification_outbox;
diff --git a/db/migrations/000008_notification.up.sql b/db/migrations/000008_notification.up.sql
new file mode 100644
index 0000000..f884e0b
--- /dev/null
+++ b/db/migrations/000008_notification.up.sql
@@ -0,0 +1,33 @@
+CREATE TABLE IF NOT EXISTS notification_outbox (
+ id SERIAL PRIMARY KEY,
+ kind VARCHAR(64) NOT NULL,
+ incident_id INTEGER NOT NULL REFERENCES incident(id),
+ recipient VARCHAR(255) NOT NULL,
+ payload JSONB NOT NULL,
+ change_id UUID NOT NULL,
+ dedup_key VARCHAR(255) NOT NULL,
+ status VARCHAR(20) NOT NULL DEFAULT 'pending',
+ attempts INTEGER NOT NULL DEFAULT 0,
+ next_attempt_at TIMESTAMPTZ NULL,
+ locked_by VARCHAR(255) NULL,
+ locked_at TIMESTAMPTZ NULL,
+ last_error TEXT NULL,
+ created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
+ updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
+);
+
+CREATE INDEX IF NOT EXISTS idx_outbox_dispatch
+ ON notification_outbox (next_attempt_at)
+ WHERE status = 'pending';
+
+CREATE INDEX IF NOT EXISTS idx_outbox_stale_processing
+ ON notification_outbox (locked_at)
+ WHERE status = 'processing';
+
+CREATE UNIQUE INDEX IF NOT EXISTS idx_outbox_dedup
+ ON notification_outbox (dedup_key);
+
+-- Supports retention pruning and the sent-count ops stat.
+CREATE INDEX IF NOT EXISTS idx_outbox_retention
+ ON notification_outbox (updated_at)
+ WHERE status = 'sent';
diff --git a/docs/notifications/architecture.md b/docs/notifications/architecture.md
new file mode 100644
index 0000000..4180f3a
--- /dev/null
+++ b/docs/notifications/architecture.md
@@ -0,0 +1,509 @@
+# Architecture: Maintenance Email Notifications
+
+This document describes how maintenance email notifications work — the design and the reasoning
+behind it. For deployment settings see [configuration.md](configuration.md); for planned work see
+[improvements.md](improvements.md).
+
+The goal is simple: **when a maintenance event is created or its status changes, send an email to
+the right people.** Everything below explains how that happens without slowing down the API.
+
+---
+
+## 1. Who gets notified
+
+There are two audiences:
+
+1. **Review audience** — notified while the maintenance still needs a human decision. It consists of:
+ - the RBAC roles that can review and approve maintenance: **Operator** and **Admin**
+ (see [../auth/rbac.md](../auth/rbac.md) and [../auth/permissions.md](../auth/permissions.md)),
+ via per-role email lists in configuration; plus
+ - a fixed **SMOD team** address (`SD_NOTIFICATIONS_SMOD_EMAIL`, for example `support@com.com`).
+ None of these come from the request — they are all predefined in configuration.
+2. **Creator** — the maintenance contact address, stored in `incident.contact_email`. It is
+ resolved once at creation time from two sources, in this order:
+ 1. the `contact_email` field of the create request, when supplied;
+ 2. otherwise the verified `email` claim of the creator's token.
+
+ The explicit field wins because "notify the team mailbox, not me personally" is a legitimate
+ request. The token is the fallback so that omitting the field still reaches a real person, and
+ because a typo in a hand-typed address is unfixable later: `contact_email` cannot be changed
+ after creation. Tokens without an `email` claim (HMAC, service-to-service) simply leave the
+ address empty, which narrows the audience but never blocks the request.
+
+ The address is resolved at creation because notifications continue for the whole lifecycle,
+ long after the creator's token is gone: an operator approves with *their* token, and the checker
+ transitions with none at all.
+
+The recipients are decided by the **resulting maintenance status**:
+
+| Maintenance goes to… | Review audience (Operator + Admin + SMOD team) | Creator |
+|----------------------|:----------------------------------------------:|:-------:|
+| `pending_review` (just created for review) | ✅ | ✅ |
+| `reviewed` (approved) | ✅ | ✅ |
+| `planned` | ❌ | ✅ |
+| `in_progress` | ❌ | ✅ |
+| `completed` | ❌ | ✅ |
+| `cancelled` | ❌ | ✅ |
+
+In words:
+
+- While the maintenance still needs a human decision (`pending_review`, `reviewed`),
+ the review audience (operators, admins, and the fixed SMOD team address) and the creator are
+ informed.
+- Once it is an ordinary lifecycle change (`planned` → `in_progress` → `completed`, or `cancelled`),
+ only the creator is informed.
+
+This rule is the same whether the status changed through the API or automatically through the
+checker.
+
+---
+
+## 2. How it works, end to end
+
+```mermaid
+flowchart TB
+ subgraph Producers["Where changes happen"]
+ H["API handlers
(create / patch maintenance)"]
+ C["Checker
(automatic status changes)"]
+ end
+
+ P["Publisher"]
+ RULES["Recipient rules"]
+ OUT[("notification_outbox
(email to-do list)")]
+ W["Delivery worker"]
+ REN["Renderer"]
+ S["SMTP sender"]
+ MAIL["Mail server"]
+
+ H -->|"same DB transaction"| P
+ C -->|"same DB transaction"| P
+ P --> RULES
+ RULES -->|"one row per recipient"| OUT
+ P -. "signal after commit" .-> W
+ W -->|"claims due rows (on signal or rare sweep)"| OUT
+ W --> REN
+ W --> S
+ S --> MAIL
+```
+
+The flow has two independent halves:
+
+1. **Recording the intent** (fast, inside the request): a maintenance change writes one email task
+ per recipient into the `notification_outbox` table, in the **same transaction** as the change.
+2. **Sending the email** (background): the worker is triggered **right after the change commits**
+ and sends immediately. A **low-frequency safety sweep** catches anything the immediate path
+ missed — retries and rows orphaned by a pod crash.
+
+The API never waits for the mail server. If email fails, the maintenance change is unaffected.
+Because dispatch is event-driven, there is **no constant polling**: on an idle system (our volume is
+about 41 maintenances in 2 months) the worker does almost nothing.
+
+---
+
+## 3. The components, explained
+
+Each component below has one clear job.
+
+### Publisher
+**What it does:** writes email tasks into the outbox.
+**Why it exists:** it guarantees the email task is saved *together* with the maintenance change.
+If the maintenance change is rolled back, the email task is rolled back too — so we never send an
+email about something that did not actually happen, and we never lose an email for something that
+did.
+
+### Recipient rules
+**What it does:** looks at the resulting maintenance status and produces the recipient list
+(review audience — operators, admins, and the fixed SMOD team address — the creator, or both) using
+the table in §1.
+**Why it exists:** it keeps the "who gets what" logic in one small, testable place instead of
+scattered across handlers. The operator and admin parts map to the RBAC roles that can approve
+maintenance; their addresses and the SMOD team address come from configuration, not from the
+requester's token.
+
+### notification_outbox (table)
+**What it does:** a durable to-do list of emails. One row = one email to one recipient.
+**Why it exists:** it decouples "we decided to send an email" from "the email was actually sent".
+It survives restarts, so nothing is lost if a pod dies.
+
+### Delivery worker
+**What it does:** runs in every pod. On the happy path it is triggered right after a maintenance
+change commits and sends the emails immediately. It also runs a **low-frequency safety sweep** to
+pick up retries and rows orphaned by a pod crash.
+**Why it exists:** it moves the slow network work (SMTP) out of the API request path and retries
+failures on its own. Because dispatch is event-driven rather than a tight polling loop, an idle
+system consumes almost no resources — which matters at our low event volume. Crucially, retry state
+lives in the outbox row (`next_attempt_at`), **not in memory**, so pending retries survive a pod
+restart and are shared across pods.
+
+### Renderer
+**What it does:** turns a stored task into a real email — subject, body, and a link to the
+maintenance.
+**Why it exists:** it keeps email formatting (templates) separate from delivery logic.
+
+### SMTP sender
+**What it does:** connects to the mail server and sends one email.
+**Why it exists:** it isolates the only part that talks to the outside world, with its own timeout,
+authentication, and TLS settings.
+**Implementation note:** the Gin backend speaks SMTP **directly** to the OTC Secure Mail Gateway
+(`otc-de-out.mms.t-systems-service.com`, port 25, authenticated over STARTTLS). No HTTP mail API is
+involved. The sender uses a maintained Go mail library (`github.com/wneessen/go-mail`) rather than
+bare `net/smtp`, for robust MIME, auth, and TLS handling, and lets the library negotiate the auth
+mechanism instead of pinning one — relays differ in what they advertise.
+
+### Where the code lives
+
+`internal/notification/` holds the feature itself; storage and HTTP entry points stay in their
+existing packages.
+
+| File | Responsibility |
+|------|----------------|
+| `notification.go` | `Config` parsed from `conf.Config`, status-to-kind mapping, retry backoff |
+| `resolver.go` | `Change` input type, recipient rules, one outbox row per recipient, dedup keys |
+| `renderer.go` | Outbox row to subject and body; decides whether the wording is a creation or a change |
+| `templates/subject.tmpl`, `templates/body.tmpl` | Embedded message text |
+| `smtp.go` | `Sender` interface and its SMTP implementation, including permanent-failure detection |
+| `publisher.go` | Enqueues rows inside the caller's transaction; `contact_email` domain check |
+| `worker.go` | Claim, send, record; stale recovery and retention |
+| `metrics.go` | Delivery counters and the DB-backed queue-depth collector |
+
+| Outside the package | Responsibility |
+|---------------------|----------------|
+| [internal/db/notification.go](../../internal/db/notification.go) | Outbox CRUD: enqueue, claim, mark sent/failed, recover stale |
+| [internal/db/notification_ops.go](../../internal/db/notification_ops.go) | Queue statistics, failed listing, re-drive, retention pruning |
+| [internal/api/v2/notifications.go](../../internal/api/v2/notifications.go) | Admin ops endpoints |
+| [internal/conf/conf.go](../../internal/conf/conf.go) | Raw settings and startup validation |
+| [internal/app/app.go](../../internal/app/app.go) | Wiring: worker, metrics listener, publisher signal |
+
+Delivery timing lives in `worker.go` and the storage layer, never in the producers: handlers and the
+checker only record intent.
+
+---
+
+## 4. Data model — one table
+
+Migration `000008_notification.up.sql` / `.down.sql`, following the existing `golang-migrate`
+layout under `db/migrations/`.
+
+### `notification_outbox`
+
+One row represents one email to one recipient.
+
+| Column | Type | Meaning |
+|--------|------|---------|
+| `id` | `SERIAL PK` | row id |
+| `kind` | `VARCHAR(64)` | why the email is sent (`pending_review`, `reviewed`, `status_changed`) |
+| `incident_id` | `INTEGER` | the maintenance event (FK → `incident(id)`) |
+| `recipient` | `VARCHAR(255)` | the single email address |
+| `payload` | `JSONB` | data needed to render the email (title, old/new status, actor, link) |
+| `change_id` | `UUID` | one id per maintenance change, used to avoid duplicates |
+| `dedup_key` | `VARCHAR(255)` | unique key `change_id + kind + recipient` |
+| `status` | `VARCHAR(20)` | `pending` → `processing` → `sent` / `failed` |
+| `attempts` | `INTEGER` | how many times we tried to send |
+| `next_attempt_at` | `TIMESTAMPTZ` | when the row becomes eligible again (backoff) |
+| `locked_by` | `VARCHAR(255)` | which pod is currently sending it |
+| `locked_at` | `TIMESTAMPTZ` | when that pod claimed it (used to recover crashes) |
+| `last_error` | `TEXT` | last failure reason |
+| `created_at` / `updated_at` | `TIMESTAMPTZ` | timestamps |
+
+```sql
+CREATE INDEX idx_outbox_dispatch
+ ON notification_outbox (next_attempt_at)
+ WHERE status = 'pending';
+
+CREATE INDEX idx_outbox_stale_processing
+ ON notification_outbox (locked_at)
+ WHERE status = 'processing';
+
+CREATE UNIQUE INDEX idx_outbox_dedup
+ ON notification_outbox (dedup_key);
+
+-- Supports retention pruning and the sent-count observability stat.
+CREATE INDEX idx_outbox_retention
+ ON notification_outbox (updated_at)
+ WHERE status = 'sent';
+```
+
+### Column groups, explained
+
+The columns fall into five groups:
+
+1. **What the email is about:** `kind`, `incident_id`, `recipient`, `payload`.
+ `payload` is a snapshot of the data needed to render the email, so the worker never has to read
+ the maintenance again and the email is unaffected by later changes.
+2. **Duplicate protection:** `change_id` + `dedup_key`. `change_id` is generated once per successful
+ maintenance change and shared by all rows of that change. `dedup_key` (`change_id : kind :
+ recipient`) is unique, so the same email to the same address for the same change cannot be
+ inserted twice — even under retries or a race between pods.
+3. **Delivery state:** `status`, `attempts`, `next_attempt_at`, `last_error`. These also serve as the
+ audit trail, which is why no separate log table is needed.
+4. **Multi-pod coordination:** `locked_by`, `locked_at`. They record which pod is sending a row and
+ when it claimed it, so a crashed pod's stuck row can be recovered after the lease expires.
+5. **Bookkeeping:** `created_at`, `updated_at`.
+
+### Example: one maintenance change becomes several rows
+
+Maintenance #42 is created in `pending_review`. Recipients are the SMOD team address, the operator
+list, the admin list, and the creator. The producer generates one `change_id` and inserts one row
+per recipient in the same transaction as the maintenance creation:
+
+| id | kind | incident_id | recipient | change_id | status |
+|----|------|:-----------:|-----------|-----------|--------|
+| 1 | `pending_review` | 42 | support@com.com | `a1b2…` | `pending` |
+| 2 | `pending_review` | 42 | ops@com.com | `a1b2…` | `pending` |
+| 3 | `pending_review` | 42 | admin@com.com | `a1b2…` | `pending` |
+| 4 | `pending_review` | 42 | creator@com.com | `a1b2…` | `pending` |
+
+The worker then processes them one at a time:
+
+1. Claims a row → `status = processing`, `locked_by = pod-1`, `locked_at = now`, `attempts++`.
+2. Commits that claim, then sends the email.
+3. On success → `status = sent`. On a temporary failure → back to `pending` with a later
+ `next_attempt_at`, or `failed` once attempts run out. On a permanent rejection (`5xx`) → `failed`
+ straight away.
+
+Four separate rows give **independent retries**: if the email to the admin list fails, only that
+row is retried — the SMOD team, operator, and creator emails are not sent again.
+
+### Do we need a separate `notification_log` table?
+
+**No.** The outbox row already records everything an audit needs:
+
+- final `status` (`sent` / `failed`),
+- `attempts`,
+- `last_error`,
+- `updated_at` (when it reached that state).
+
+A separate log table would just duplicate this for our small scope. Sent rows are kept for a
+retention period (for audit and re-drive) and then cleaned up. If richer per-attempt history is
+ever required, a log table can be added later without changing the delivery design.
+
+---
+
+## 5. Sending the email (the worker)
+
+Every pod runs one worker. Because there are multiple pods, they coordinate through PostgreSQL so
+the same email is not sent twice at the same time.
+
+**What wakes the worker:**
+
+1. **A signal after commit (happy path).** When a maintenance change commits, the publisher signals
+ the in-process worker (via a channel) to send right away. No waiting for a poll tick.
+2. **A low-frequency safety-sweep ticker.** Every few minutes the worker also scans for due rows —
+ `pending` rows whose `next_attempt_at` has passed (retries) and rows stuck in `processing` after
+ a crash. On our volume this sweep almost always finds zero rows, so its cost is negligible; it
+ exists purely to guarantee nothing is stranded if a signal was missed (e.g. the sending pod
+ restarted).
+
+This keeps the design cheap when idle **and** durable: retries are driven by `next_attempt_at` in
+the database, not by in-memory timers, so a pod restart never loses a pending retry.
+
+```mermaid
+stateDiagram-v2
+ [*] --> Pending: enqueue
+ Pending --> Processing: a pod claims it
+ Processing --> Sent: mail server accepted
+ Processing --> Pending: failed, retries left
+ Processing --> Failed: failed, no retries left
+ Sent --> [*]
+ Failed --> [*]
+```
+
+The loop, in plain steps:
+
+1. **Recover stuck rows.** If a pod claimed a row and then crashed, the row stays in `processing`.
+ After a lease timeout it is returned to `pending` (or set to `failed` if it already used all
+ attempts).
+2. **Claim one row.** Select the next due `pending` row with `FOR UPDATE SKIP LOCKED`, mark it
+ `processing`, set `locked_by`/`locked_at`, and increment `attempts`. `SKIP LOCKED` guarantees two
+ pods never grab the same row.
+3. **Commit, then send.** The database transaction ends *before* the email is sent — a DB lock is
+ never held while waiting on the network.
+4. **Record the result.** On success mark `sent`. On failure either keep it `pending` with a longer
+ `next_attempt_at`, or mark it `failed` — see §5.1.
+5. **Stay safe.** Each send runs inside a `recover()` guard so one bad email cannot crash the
+ worker.
+6. **Repeat** until no due rows remain.
+
+**Why one row at a time.** The lease starts when a row is claimed, but sends are sequential. Had the
+worker claimed a batch of N rows, the last one would begin sending up to `N × smtp_timeout` after
+its lease started — long past expiry. The stale-recovery path would then hand that row to another
+pod (or to the next pass of the same one) while the first send was still in flight, producing
+duplicate emails. Claiming one row keeps `claim → send → record` inside a single lease, so
+correctness no longer depends on tuning `lease_timeout` against the batch size. The extra queries
+cost nothing next to the network round-trip they accompany.
+
+**Timing rule:** the lease timeout must be longer than the SMTP timeout so a slow-but-alive send is
+never reclaimed by another pod. This is enforced at startup.
+
+### 5.1 Permanent versus temporary failures
+
+Not every rejection is worth retrying. The sender inspects the SMTP reply:
+
+| Reply | Meaning | Action |
+|-------|---------|--------|
+| `5xx` | The server refuses this message — unknown recipient, blocked sender | `failed` immediately |
+| `4xx` | Temporary — greylisting, mailbox full, rate limit | retry with backoff |
+| No reply (transport error) | Unknown — DNS, connection refused, timeout | retry with backoff |
+
+A permanent rejection is terminal on the first attempt: repeating it cannot change the outcome, and
+failing fast surfaces a typo in a recipient address within seconds instead of hiding it behind hours
+of backoff. Anything without a definite `5xx` is treated as temporary, so an ambiguous error never
+causes a lost notification.
+
+### 5.2 Retry backoff
+
+A temporary failure schedules `next_attempt_at` at `base × 2^(attempts-1)`, capped at 2 hours, with
+a random spread of ±20%.
+
+The spread matters because failures correlate: when a relay goes down, every queued row fails within
+the same second. Without jitter all of them would retry at the same instant, hitting the recovering
+server with a synchronised burst — and repeating that burst on every subsequent attempt. Jitter
+spreads the load and prevents this thundering herd.
+
+### Design note: why retries live in the database, not in memory
+
+At our volume (about 41 maintenances in 2 months) a tempting simplification is to skip the outbox
+row and, right after saving the maintenance, send the email in a goroutine — keeping failed emails
+in memory and retrying every 15 minutes. We deliberately do **not** do this. The retry *interval* is
+kept (a failed row becomes eligible again after a backoff delay), but the retry *state* lives in the
+outbox row (`attempts`, `next_attempt_at`, `last_error`), not in process memory, for two reasons:
+
+1. **Pod restarts lose in-memory state.** In Kubernetes pods restart routinely (deploys, OOM,
+ rescheduling). An in-memory retry timer would silently drop every email waiting to be retried —
+ exactly the "problem with retries after a few failed attempts" we need to avoid. A row in the
+ database survives the restart and is picked up by the safety sweep.
+2. **Multiple pods cannot share memory.** With ≥2 pods, an in-memory queue in one pod is invisible to
+ the others, so retries cannot be coordinated and the same email could be retried twice or not at
+ all. The shared outbox table plus `FOR UPDATE SKIP LOCKED` gives one owner per row across all
+ pods.
+
+This costs almost nothing extra: the durable row is the same record the "save then send" idea would
+keep anyway — we simply reuse it as the retry source instead of adding a separate in-memory
+mechanism.
+
+---
+
+## 6. Configuration
+
+Settings extend `conf.Config` in [internal/conf/conf.go](../../internal/conf/conf.go), using the
+existing `envconfig` + `.env` mechanism. **The full reference lives in
+[configuration.md](configuration.md);** this section only records the design decisions behind it.
+
+**Everything is off by default.** `SD_NOTIFICATIONS_ENABLED` defaults to `false`, and when the
+feature is off no SMTP setting is required, no worker starts, and no metrics listener opens. The
+feature can be absent from an installation entirely.
+
+**Invalid configuration fails at startup, not at send time.** When the feature is enabled the
+validator parses the sender address, every review address, the SMTP port range and all durations. A
+typo in an operator address would otherwise stay invisible until the first maintenance, then produce
+failures on every message. Review addresses come from the operator, so they are validated as
+strictly as the user-supplied `contact_email`.
+
+**The lease must outlast the SMTP timeout,** or a slow-but-alive send would be reclaimed by another
+pod. This relationship is checked at startup because it cannot be detected safely at runtime.
+
+**The creator recipient is not configured.** It is the maintenance `contact_email` stored in the
+database. The SMOD address plus the operator and admin lists form the review audience, and none of
+them come from the requester's token.
+
+**Two lists constrain recipients.** `SD_NOTIFICATIONS_ALLOWED_DOMAINS` restricts the domains a user
+may type into `contact_email`; without it any authenticated creator could make the service mail an
+arbitrary external address from a trusted corporate domain. It applies only to the explicit field,
+since the token address is already verified. `SD_NOTIFICATIONS_EXCLUDED_EMAILS` drops specific
+addresses from *every* recipient list, so an exclusion cannot be bypassed by passing the address as
+`contact_email`.
+
+**Transport:** a direct SMTP connection to the OTC (Open Telekom Cloud) endpoint. No external mail
+gateway or HTTP mail API is involved. SMTP secrets are masked in logs.
+
+---
+
+## 7. Guarantees and trade-offs
+
+1. **Nothing is lost.** The email task and the maintenance change are saved in one transaction, so a
+ committed change always has its email task.
+2. **The API never blocks.** Sending happens in the background.
+3. **Failures are isolated.** A mail server problem is recorded on the outbox row and never turns
+ into an API error. A permanent rejection stops after one attempt; anything ambiguous is retried.
+4. **At-least-once delivery.** In a rare case (the mail server accepts the email but the pod dies
+ before writing `sent`), the email may be sent twice after recovery. This is accepted: a rare
+ duplicate is better than a lost notification, and plain SMTP offers no safe way to avoid it.
+ Note this is the *only* remaining duplicate window — claiming one row per lease removed the
+ batch-expiry case.
+5. **Cheap when idle.** Dispatch is event-driven; the only recurring background activity is a rare
+ safety sweep that returns nothing on an idle system. Retry state lives in the outbox row
+ (`next_attempt_at`), not in memory, so retries survive pod restarts and are coordinated across
+ pods — unlike an in-memory retry timer, which would lose pending emails on restart.
+
+---
+
+## 8. Startup and lifecycle
+
+```mermaid
+flowchart TB
+ M[cmd/main.go] --> APP["app: API + publisher"]
+ M --> CHK["checker + publisher"]
+ M --> WRK["notification worker"]
+ M --> MET["metrics listener
(own port)"]
+ M --> SD["graceful shutdown of all"]
+```
+
+The worker reuses the application's existing database connection pool — it must **not** open its
+own. With multiple pods, extra pools would multiply PostgreSQL connections. Budget connections as
+`connections_per_pod * number_of_pods`.
+
+The metrics listener is a second `http.Server` bound to `SD_METRICS_PORT`. It starts only when
+notifications are enabled and is shut down together with the API server.
+
+On shutdown the worker stops claiming new rows and finishes in-flight sends; anything unfinished is
+recovered by the lease mechanism on the next run.
+
+---
+
+## 9. Observability and retention
+
+The outbox row is the single source of truth for delivery state, so observability reads directly
+from it — there is no separate metrics store. Two interfaces expose the same data:
+
+### Prometheus `/metrics`
+
+Served on a **separate listener** (`SD_METRICS_PORT`, default `9090`), not on the public API port,
+and only when notifications are enabled. Queue depth and failure counts are operational detail that
+should not be readable by anyone who can reach the dashboard, so the port is meant to stay internal
+to the cluster — scraped by Prometheus, never published through an Ingress. The registry is
+dedicated, so the endpoint exposes notification series only.
+
+Two kinds of series:
+
+- **Worker counters/histogram** (updated as rows are delivered): `notification_sent_total{kind}`,
+ `notification_failed_total{kind}`, `notification_attempts_total`,
+ `notification_stale_recovered_total`, `notification_delivery_duration_seconds`.
+- **Queue-depth gauges** (pulled on each scrape by a DB-backed collector, so they always reflect
+ current state): `notification_outbox_pending`, `_processing`, `_failed`, `_stale_processing`,
+ `_retry_backlog`, `_oldest_pending_age_seconds`.
+
+The scrape query is bounded by a timeout: Prometheus scrapes on a schedule regardless of how the
+previous attempt went, so an unbounded query against a stalled database would accumulate goroutines
+and connections. A failed collection increments `notification_collector_errors_total` rather than
+silently omitting the gauges, which would look identical to a healthy empty queue.
+
+### Admin ops API (`/v2/notifications/…`)
+
+For manual inspection and recovery, admin-only:
+
+- `GET /stats` — the same queue-depth snapshot as JSON.
+- `GET /failed` — the most recent terminal-`failed` rows.
+- `POST /redrive` — reset `failed` rows back to `pending` (optionally by id) and wake the worker.
+
+### Retention
+
+Delivery outcome lives on the outbox row, which also serves as the audit trail. To keep the table
+(and the count queries above) small, the worker prunes on its safety sweep:
+
+- `sent` rows older than **30 days** are deleted in batches (`idx_outbox_retention` supports it).
+- `failed` rows are **kept indefinitely** — they are unfinished work: queryable via the ops API and
+ re-drivable. (At our volume this is negligible; a longer sent-retention window is a one-constant
+ change if a longer audit history is ever wanted.)
+
+Structured logs carry `outbox_id`, `incident_id`, `recipient`, `kind`, and `attempts` for
+per-delivery tracing.
diff --git a/docs/notifications/configuration.md b/docs/notifications/configuration.md
new file mode 100644
index 0000000..bdceaba
--- /dev/null
+++ b/docs/notifications/configuration.md
@@ -0,0 +1,321 @@
+# Configuration Guide: Maintenance Email Notifications
+
+How to configure, deploy and troubleshoot maintenance email notifications.
+
+For the design and its reasoning see [architecture.md](architecture.md).
+
+---
+
+## Quick start
+
+Notifications are **disabled by default**. A minimal working configuration:
+
+```bash
+SD_NOTIFICATIONS_ENABLED=true
+SD_SMTP_HOST=otc-de-out.mms.t-systems-service.com
+SD_SMTP_PORT=25
+SD_SMTP_FROM=status-dashboard@example.com
+SD_SMTP_TLS=true
+SD_SMTP_USER=
+SD_SMTP_PASSWORD=
+SD_NOTIFICATIONS_SMOD_EMAIL=smod@example.com
+```
+
+Everything else has a default. The application refuses to start if the configuration is incomplete
+or malformed, so a successful startup means the settings are valid.
+
+---
+
+## Reference
+
+### Feature switch
+
+| Variable | Default | Description |
+|----------|---------|-------------|
+| `SD_NOTIFICATIONS_ENABLED` | `false` | Master on/off switch. When `false`, no SMTP setting is required, no worker runs and no metrics listener opens. |
+
+### SMTP transport
+
+| Variable | Required | Default | Description |
+|----------|:--------:|---------|-------------|
+| `SD_SMTP_HOST` | **yes** | — | Mail server hostname. |
+| `SD_SMTP_PORT` | **yes** | — | Mail server port, `1`–`65535`. `25` for the OTC Secure Mail Gateway. |
+| `SD_SMTP_FROM` | **yes** | — | Sender address. Must be a valid address **and** permitted for the account, or the relay rejects every message. |
+| `SD_SMTP_USER` | no | — | SMTP login. **Omit entirely** when the relay authorises by IP. |
+| `SD_SMTP_PASSWORD` | no | — | SMTP password. Store in a secret, never in a ConfigMap. |
+| `SD_SMTP_TLS` | no | `false` | `true` requires STARTTLS; `false` uses it opportunistically. |
+| `SD_SMTP_TIMEOUT` | no | `30s` | Connect + send timeout (Go duration). |
+
+Required fields apply only when the feature is enabled.
+
+### OTC Secure Mail Gateway
+
+The production relay is the OTC Secure Mail Gateway
+([docs](https://docs.otc.t-systems.com/secure-mail-gateway/umn/)):
+
+| Setting | Value |
+|---------|-------|
+| Smarthost | `otc-de-out.mms.t-systems-service.com` |
+| Port | `25` — the only port open for mail acceptance |
+| Authentication | Required; credentials are issued by Cloud Handling Support (`service@open-telekom-cloud.com`) |
+| TLS | `SD_SMTP_TLS=true`, so credentials are never sent before STARTTLS |
+
+The gateway is a paid service (~23 €/month, invoiced via OTC), so request access before
+planning a rollout.
+
+The authentication mechanism is negotiated: the sender advertises auto-discovery and picks
+whatever the gateway offers in its EHLO reply. Hard-coding `PLAIN` would break against a
+LOGIN-only server, and the gateway documentation does not state which mechanisms it
+supports.
+
+### Recipients
+
+| Variable | Default | Description |
+|----------|---------|-------------|
+| `SD_NOTIFICATIONS_SMOD_EMAIL` | — | Fixed SMOD team review address. |
+| `SD_NOTIFICATIONS_EMAILS_OPERATORS` | — | Review addresses for the Operator role, comma-separated. |
+| `SD_NOTIFICATIONS_EMAILS_ADMINS` | — | Review addresses for the Admin role, comma-separated. |
+| `SD_NOTIFICATIONS_ALLOWED_DOMAINS` | — | Domains accepted in `contact_email`, comma-separated. Empty allows any. |
+| `SD_NOTIFICATIONS_EXCLUDED_EMAILS` | — | Addresses that never receive mail, comma-separated. |
+
+At least one of the three review variables must be set. Addresses are trimmed, lowercased and
+deduplicated.
+
+The creator recipient is **not** configured here. It is resolved at creation time from the
+`contact_email` field of the request, or from the `email` claim of the creator's token when the
+field is omitted.
+
+`SD_NOTIFICATIONS_ALLOWED_DOMAINS` applies only to the address a user types in; the token address is
+already verified. Matching is exact, so `example.com` does not cover `mail.example.com`.
+Rejected requests get `400` naming the permitted domains.
+
+`SD_NOTIFICATIONS_EXCLUDED_EMAILS` is applied to every recipient, including review addresses and the
+creator, so a blocked address cannot slip through via `contact_email`.
+
+### Delivery tuning
+
+| Variable | Default | Description |
+|----------|---------|-------------|
+| `SD_NOTIFICATIONS_LEASE_TIMEOUT` | `60s` | How long a claimed row stays owned by a pod. **Must exceed `SD_SMTP_TIMEOUT`.** |
+| `SD_NOTIFICATIONS_MAX_ATTEMPTS` | `5` | Attempts before a row becomes terminally `failed`. |
+| `SD_NOTIFICATIONS_BACKOFF_INTERVAL` | `5m` | Base retry delay. Doubles per attempt, capped at 2h, spread by ±20%. |
+
+### Observability
+
+| Variable | Default | Description |
+|----------|---------|-------------|
+| `SD_METRICS_PORT` | `9090` | Port for the private `/metrics` listener. Must differ from `SD_PORT`. |
+
+---
+
+## Deployment notes
+
+For the `sd3-test` Kubernetes environment, where settings are rendered from Vault rather than set
+in the manifest, see [stackmon-config-deployment.md](stackmon-config-deployment.md).
+
+### Which SMTP account to use
+
+For the OTC Secure Mail Gateway the credentials come from Cloud Handling Support; there is
+no IP-based alternative. For any other relay:
+
+| Relay setup | What to configure |
+|-------------|-------------------|
+| Requires authentication | Service account, never a personal one; password from a secret |
+| Functional mailbox | The mailbox login |
+| Authorises by IP or subnet | Omit `SD_SMTP_USER` and `SD_SMTP_PASSWORD` entirely |
+
+Do not set `SD_SMTP_USER=""` explicitly. An empty value behaves like an unset one, but the redundant
+key invites confusion later.
+
+Agree `SD_SMTP_FROM` together with the account: relays verify that the sender is entitled to the
+address and answer `550 sender address rejected` otherwise.
+
+### Distribution lists over individual addresses
+
+Prefer one distribution list per role:
+
+```yaml
+SD_NOTIFICATIONS_SMOD_EMAIL: smod@example.com
+SD_NOTIFICATIONS_EMAILS_OPERATORS: sd-operators@example.com
+SD_NOTIFICATIONS_EMAILS_ADMINS: sd-admins@example.com
+```
+
+Membership then lives in the mail system and changes without a redeploy, instead of requiring a
+config change and a pod restart for every staffing update.
+
+### Metrics port
+
+`/metrics` is served on its own listener so that queue depth and failure counts are not reachable
+from the public API port. Keep it internal to the cluster:
+
+```yaml
+ports:
+ - name: http
+ containerPort: 8000
+ - name: metrics
+ containerPort: 9090 # scraped by Prometheus, not exposed through Ingress
+```
+
+### Secrets
+
+`SD_SMTP_PASSWORD` belongs in a Kubernetes Secret. It is masked in application logs, but a ConfigMap
+or a committed `.env` would expose it.
+
+---
+
+## Validation at startup
+
+Migrations are applied out of band, so the application also refuses to start when
+notifications are enabled and the `notification_outbox` table is missing:
+
+```
+notification_outbox table is missing: apply the pending database migrations
+```
+
+When the feature is enabled the application refuses to start unless:
+
+- `SD_SMTP_HOST`, `SD_SMTP_PORT` and `SD_SMTP_FROM` are set;
+- `SD_SMTP_PORT` is a number in `1`–`65535`;
+- `SD_SMTP_FROM` parses as an email address;
+- at least one review address is configured, and **every** configured review address parses;
+- every address in `SD_NOTIFICATIONS_EXCLUDED_EMAILS` parses;
+- every entry in `SD_NOTIFICATIONS_ALLOWED_DOMAINS` is a bare domain, not a full address;
+- `SD_SMTP_TIMEOUT`, `SD_NOTIFICATIONS_LEASE_TIMEOUT` and `SD_NOTIFICATIONS_BACKOFF_INTERVAL` parse
+ as Go durations;
+- `SD_NOTIFICATIONS_LEASE_TIMEOUT` is greater than `SD_SMTP_TIMEOUT`;
+- `SD_NOTIFICATIONS_MAX_ATTEMPTS` is a positive integer;
+- `SD_METRICS_PORT` is in `1024`–`65535` and differs from `SD_PORT`.
+
+Failing here is deliberate: a typo in a review address would otherwise stay invisible until the
+first maintenance, then break every review notification.
+
+---
+
+## Verifying the setup
+
+### Admin API
+
+All three endpoints require the `admin` role.
+
+```bash
+curl -H "Authorization: Bearer $TOKEN" https:///v2/notifications/stats
+```
+
+```json
+{
+ "pending": 0, "processing": 0, "sent": 42, "failed": 0,
+ "stale_processing": 0, "retry_backlog": 0,
+ "oldest_pending_age_seconds": 0
+}
+```
+
+| Endpoint | Purpose |
+|----------|---------|
+| `GET /v2/notifications/stats` | Queue snapshot |
+| `GET /v2/notifications/failed` | Recent rows; `?status=` and `?limit=` select the rest of the queue |
+| `POST /v2/notifications/redrive` | Reset failed rows to `pending` and wake the worker |
+
+`?status=` accepts `pending`, `processing`, `sent` or `failed` and defaults to `failed`.
+`?limit=` accepts `1`–`1000` and defaults to `100`. Listing `pending` is the quickest way to
+see rows a broken relay is holding up:
+
+```bash
+curl -H "Authorization: Bearer $TOKEN" \
+ "https:///v2/notifications/failed?status=pending&limit=20"
+```
+
+Re-drive everything, or selected rows:
+
+```bash
+curl -X POST -H "Authorization: Bearer $TOKEN" -H 'Content-Type: application/json' \
+ -d '{}' https:///v2/notifications/redrive
+curl -X POST -H "Authorization: Bearer $TOKEN" -H 'Content-Type: application/json' \
+ -d '{"ids":[12,13]}' https:///v2/notifications/redrive
+```
+
+### Metrics worth alerting on
+
+| Series | Signal |
+|--------|--------|
+| `notification_outbox_oldest_pending_age_seconds` | Rising steadily → delivery is stuck |
+| `notification_outbox_failed` | Growing → recipients or relay misconfigured |
+| `notification_outbox_stale_processing` | Non-zero → pods crashing mid-send |
+| `notification_collector_errors_total` | Increasing → the gauges below it are unreliable |
+
+---
+
+## Local development
+
+Use [Mailpit](https://mailpit.axllent.org/) as a catcher — it accepts everything and offers no
+authentication:
+
+```bash
+podman run -d --rm --name status-dashboard-mailpit \
+ -p 127.0.0.1:1025:1025 -p 127.0.0.1:8025:8025 \
+ docker.io/axllent/mailpit:latest --verbose
+```
+
+```bash
+SD_NOTIFICATIONS_ENABLED=true
+SD_SMTP_HOST=127.0.0.1
+SD_SMTP_PORT=1025
+SD_SMTP_FROM=status-dashboard@local.test
+SD_SMTP_TLS=false
+# No SD_SMTP_USER — Mailpit offers no AUTH
+SD_NOTIFICATIONS_SMOD_EMAIL=smod@local.test
+SD_NOTIFICATIONS_EMAILS_OPERATORS=operators@local.test
+SD_NOTIFICATIONS_EMAILS_ADMINS=admins@local.test
+```
+
+Inbox at `http://127.0.0.1:8025`; `--verbose` logs every SMTP session, which distinguishes "the app
+never connected" from "the message was rejected".
+
+---
+
+## Troubleshooting
+
+### Nothing arrives, and the outbox is empty
+
+The change never reached the publisher. Confirm the event is a **maintenance** (other incident types
+never notify) and that `SD_NOTIFICATIONS_ENABLED=true`.
+
+### Rows stay `pending` with a rising `attempts`
+
+Delivery is failing and being retried. Read `last_error`:
+
+```sql
+SELECT id, recipient, status, attempts, left(last_error, 80)
+FROM notification_outbox ORDER BY id DESC LIMIT 20;
+```
+
+| `last_error` contains | Cause | Fix |
+|-----------------------|-------|-----|
+| `dial failed` / `connection refused` | Host, port or firewall | Check `SD_SMTP_HOST`/`SD_SMTP_PORT` and egress rules |
+| `server does not support SMTP AUTH` | Credentials sent to a relay without AUTH | Unset `SD_SMTP_USER` |
+| `550 sender address rejected` | `SD_SMTP_FROM` not allowed for the account | Align the sender with the account |
+| `context deadline exceeded` | Relay too slow | Raise `SD_SMTP_TIMEOUT`, then `SD_NOTIFICATIONS_LEASE_TIMEOUT` above it |
+
+### Rows go straight to `failed` on the first attempt
+
+A permanent `5xx` rejection — usually an unknown recipient. Check the address, then `redrive` after
+fixing it.
+
+### Retries seem slow
+
+By design: `5m → 10m → 20m → 40m → 80m`, capped at 2h and jittered by ±20%. `redrive` bypasses the
+wait, but only for rows already in `failed`.
+
+### Application will not start
+
+The message names the offending variable, for example:
+
+```
+SD_NOTIFICATIONS_EMAILS_OPERATORS contains an invalid address "ops at example.com"
+SD_NOTIFICATIONS_LEASE_TIMEOUT (30s) must be greater than SD_SMTP_TIMEOUT (30s)
+SD_METRICS_PORT must differ from SD_PORT
+```
+
+### Metrics are unreachable
+
+They are on `SD_METRICS_PORT` (default `9090`), not on the API port, and only when notifications are
+enabled.
diff --git a/docs/notifications/improvements.md b/docs/notifications/improvements.md
new file mode 100644
index 0000000..d955660
--- /dev/null
+++ b/docs/notifications/improvements.md
@@ -0,0 +1,171 @@
+# Notifications — Development Roadmap
+
+Open improvements for the maintenance email notification feature. Nothing listed here is
+implemented; sections are ordered by priority, and each records the reasoning so the
+decision does not have to be rediscovered later.
+
+Related: [architecture.md](architecture.md), [configuration.md](configuration.md).
+
+---
+
+## 1. Frontend: prefill `contact_email`
+
+**Priority: high (the backend fallback is unreachable without it)**
+
+Lives in the separate [StatusDashboard-V3](https://github.com/stackmon/StatusDashboard-V3)
+repository.
+
+### Problem
+
+The backend accepts a maintenance without `contact_email` and falls back to the `email`
+claim of the creator's token. The create form still marks the field as required and blocks
+submission while it is empty:
+
+```ts
+if (type === EventType.Maintenance && !value) {
+ setValContactEmail("Contact Email is required for maintenance.");
+}
+```
+
+So the fallback can never trigger through the UI. Users keep typing an address by hand,
+with the same risk of a typo that cannot be corrected afterwards.
+
+### Recommendation
+
+Prefill the field from the signed-in user instead of relaxing the requirement.
+`NewForm.tsx` already calls `useAuth()`, and `profile.email` is present in the ID token,
+so no extra request is needed:
+
+```ts
+const userEmail = useAuth().user?.profile.email;
+useEffect(() => {
+ if (userEmail && !contactEmail) {
+ _setContactEmail(userEmail);
+ }
+}, [userEmail]);
+```
+
+Keep the field required and editable. Making it optional would match the backend fallback,
+but the user would no longer see where the notifications are going, and "send to the team
+mailbox instead of me" is a routine request that must stay one edit away.
+
+### Note
+
+The `email` claim reaches the **access token** only through a mapper on the client
+(`Add to access token`); the frontend reads it from the ID token, where it is present by
+default. The two paths are independent, so the frontend prefill works even where the
+backend fallback does not — and vice versa.
+
+---
+
+## 2. Review audience via distribution lists
+
+**Priority: medium (operational, no code change)**
+
+### Problem
+
+`SD_NOTIFICATIONS_EMAILS_OPERATORS` and `SD_NOTIFICATIONS_EMAILS_ADMINS` hold individual
+addresses, so every staffing change requires a config change and a pod restart. The same
+membership information already exists in Keycloak groups, duplicated by hand.
+
+### Proposal
+
+Point each variable at one distribution list instead of a list of people:
+
+```yaml
+SD_NOTIFICATIONS_SMOD_EMAIL: smod@company.com
+SD_NOTIFICATIONS_EMAILS_OPERATORS: sd-operators@company.com
+SD_NOTIFICATIONS_EMAILS_ADMINS: sd-admins@company.com
+```
+
+Membership then lives in the mail system, owned by the people who already own the groups.
+The application keeps three stable addresses that change once every few years.
+
+### Alternative considered: Keycloak Admin API
+
+Resolving group members at send time looks natural — the groups are already there — but it
+requires a service account with user-read permissions, pagination handling, a cache with
+invalidation, and a defined behaviour when Keycloak is unreachable mid-delivery. That
+inserts a distributed dependency into the mail path to buy what a distribution list
+provides for free. Not recommended.
+
+---
+
+## 3. Editable `contact_email`
+
+**Priority: medium**
+
+### Problem
+
+`contact_email` is set once at creation and cannot be changed afterwards — the field does
+not exist in `PatchIncidentData`. Only the syntax is validated, so `user@gmial.com` is
+accepted and every notification for that maintenance is delivered to a stranger, or
+nowhere, for the entire lifecycle of the event. The only repair is a manual `UPDATE` in
+the database.
+
+### Proposal
+
+Add the field to `PatchIncidentData` and apply the same checks as on creation: address
+syntax plus the `SD_NOTIFICATIONS_ALLOWED_DOMAINS` allow-list.
+
+Restrict the change to the roles that may already patch the maintenance. Note that the
+creator's own permission is derived from `created_by`, so a creator editing their own
+event keeps working without extra rules.
+
+### Trade-off
+
+Changing the address mid-flight means rows already queued keep the old recipient, since
+the payload is a snapshot. That is acceptable: the alternative — rewriting pending rows —
+would blur the audit trail for no practical gain.
+
+---
+
+## 4. Ops API: `503` when the feature is disabled
+
+**Priority: low**
+
+With `SD_NOTIFICATIONS_ENABLED=false` the three admin endpoints still respond `200` with
+zeroed statistics, so an operator cannot tell "nothing to send" from "feature switched
+off". Both look like a perfectly healthy empty queue.
+
+**Proposal:** return `503` with an explicit body when the feature is disabled.
+
+---
+
+## 5. SMTP transport: implicit TLS (port 465)
+
+**Priority: very low — not needed for the current relay**
+
+`SD_SMTP_TLS=true` maps to `mail.TLSMandatory`, which is *mandatory STARTTLS* on a plain
+port. Relays that expect TLS negotiated at connection time (SMTPS, port 465) are not
+supported — the handshake never happens and the connection fails.
+
+The OTC Secure Mail Gateway accepts mail on **port 25 only**, so this does not affect the
+planned deployment. Keep the item for a future relay that requires SMTPS.
+
+**Proposal:** add `SD_SMTP_TLS_MODE` with values `starttls` (default), `implicit`
+(`mail.WithSSL()`), and `none`, deprecating the boolean. Keep the boolean working for one
+release to avoid breaking deployments.
+
+---
+
+## 6. Delivery throughput
+
+**Priority: low**
+
+Two related inefficiencies, neither affecting correctness.
+
+**A new connection per message.** `DialAndSendWithContext` opens and closes an SMTP
+session for every recipient, so a queue of 50 messages performs 50 TCP and TLS handshakes.
+Corporate relays often rate-limit connections per source address and may temporarily block
+a sender that reconnects too eagerly. `go-mail` supports `DialWithContext` followed by
+several `Send` calls on one session.
+
+**Single-threaded sending.** The worker sends one message at a time, so throughput is
+capped at one email per round-trip. A small bounded pool (3–5 senders) would remove the
+ceiling. This became straightforward only after claiming moved to one row per lease — with
+batch claiming, concurrency would have widened the duplicate window described in
+[architecture.md](architecture.md) §5.
+
+Both are worth doing only if the queue is observed to lag: at the current volume
+(~41 maintenances in 2 months) neither is measurable.
diff --git a/docs/notifications/stackmon-config-deployment.md b/docs/notifications/stackmon-config-deployment.md
new file mode 100644
index 0000000..b881d9d
--- /dev/null
+++ b/docs/notifications/stackmon-config-deployment.md
@@ -0,0 +1,228 @@
+# Agent Instructions: Enable Notifications in `stackmon-config` (test)
+
+Task description for an agent working in the **`stackmon-config`** repository. Prepared from the
+Status Dashboard side; the agent editing `stackmon-config` has no context about this service, so
+everything it needs is stated explicitly below.
+
+---
+
+## Goal
+
+Deploy the Status Dashboard build that includes maintenance email notifications to the **sd3-test**
+environment, and wire it to the corporate SMTP relay.
+
+## Repository and branch
+
+- Repository: `opentelekomcloud-infra/stackmon-config`
+- Branch: `sd-test-branch`
+- Scope: `kustomize/sd3/api/` only. Do **not** touch `overlays/prod`, `kustomize/sd3-ch` or
+ `kustomize/sdb`.
+
+## How this deployment works
+
+Environment variables are **not** declared in the Deployment manifest. A Vault agent init container
+renders a shell file and the application sources it at startup:
+
+```yaml
+args: ['source /secrets/sd3-api-env && "/usr/src/app/app"']
+```
+
+The file is produced by the template in `overlays/test/vault-agent.hcl` from the Vault secret
+`secret/data/statusdashboard/sd3-test`.
+
+Consequence: adding a setting requires **two** steps — a key in Vault, and an `export` line in the
+template. A key without a template line has no effect; a template line without a key renders empty.
+
+## Relevant files
+
+| File | Purpose |
+|---|---|
+| `kustomize/sd3/api/overlays/test/vault-agent.hcl` | Renders env vars from Vault |
+| `kustomize/sd3/api/overlays/test/kustomization.yaml` | Image tag, namespace `sd3-test`, ingress |
+| `kustomize/sd3/api/base/deployment.yaml` | Container ports, resources, vault init container |
+
+---
+
+## Prerequisites (must be supplied before starting)
+
+The relay is the OTC Secure Mail Gateway
+([docs](https://docs.otc.t-systems.com/secure-mail-gateway/umn/)). Host and port are fixed;
+the credentials and the sender address are not in this repository and must not be invented.
+
+| Value | Vault key | Source |
+|---|---|---|
+| `otc-de-out.mms.t-systems-service.com` | `smtphost` | Gateway documentation |
+| `25` (the only port open for mail acceptance) | `smtpport` | Gateway documentation |
+| Allowed sender address | `smtpfrom` | Cloud Handling Support |
+| SMTP login | `smtpuser` | Cloud Handling Support |
+| SMTP password | `smtppassword` | Cloud Handling Support |
+| SMOD test recipient | `notificationssmodemail` | Team decision |
+| Operator test recipients | `notificationsemailsoperators` | Team decision |
+| Admin test recipients | `notificationsemailsadmins` | Team decision |
+
+Authentication is **mandatory** on this gateway — there is no IP-based alternative.
+Credentials are issued by Cloud Handling Support (`service@open-telekom-cloud.com`), and
+the service is billed at roughly 23 €/month, so request access early.
+
+Also required: the image tag of a Status Dashboard build that contains the notification feature
+(`quay.io/stackmon/status-dashboard-v3:sha-`). The tag currently referenced,
+`sha-17d25aa`, predates the feature.
+
+---
+
+## Change 1 — `overlays/test/vault-agent.hcl`
+
+Add to the `template` block, inside the existing `{{ with secret ... }}` section:
+
+```hcl
+export SD_NOTIFICATIONS_ENABLED=true
+export SD_SMTP_HOST={{ .Data.data.smtphost }}
+export SD_SMTP_PORT={{ .Data.data.smtpport }}
+export SD_SMTP_FROM={{ .Data.data.smtpfrom }}
+export SD_SMTP_TLS=true
+export SD_SMTP_TIMEOUT=30s
+export SD_SMTP_USER="{{ .Data.data.smtpuser }}"
+export SD_SMTP_PASSWORD="{{ .Data.data.smtppassword }}"
+export SD_NOTIFICATIONS_LEASE_TIMEOUT=60s
+export SD_NOTIFICATIONS_MAX_ATTEMPTS=5
+export SD_NOTIFICATIONS_BACKOFF_INTERVAL=5m
+export SD_NOTIFICATIONS_SMOD_EMAIL="{{ .Data.data.notificationssmodemail }}"
+export SD_NOTIFICATIONS_EMAILS_OPERATORS="{{ .Data.data.notificationsemailsoperators }}"
+export SD_NOTIFICATIONS_EMAILS_ADMINS="{{ .Data.data.notificationsemailsadmins }}"
+export SD_METRICS_PORT=9090
+```
+
+`SD_SMTP_TLS=true` makes STARTTLS mandatory, so the credentials are never sent over an
+unencrypted connection. The gateway requires authentication, so both credential lines are
+always present here; against a relay that authorises by IP they would be omitted entirely
+rather than set to empty strings, since a non-empty user makes the sender negotiate AUTH.
+
+### Quoting rule — do not skip
+
+The rendered file is executed with `source`, so any value containing a space, comma or shell
+metacharacter must be quoted. Unquoted:
+
+```bash
+export SD_NOTIFICATIONS_EMAILS_ADMINS=a@x.com, b@x.com
+# shell parses: export "a@x.com," then tries to run "b@x.com"
+```
+
+All comma-separated lists and the password must use `"{{ ... }}"`.
+
+**Pre-existing issue worth fixing in the same PR:** `SD_RBAC_GROUPS_ADMINS`,
+`SD_RBAC_GROUPS_OPERATORS` and `SD_RBAC_GROUPS_CREATORS` are currently unquoted. If the Vault value
+holds a list such as `sd-admins, status-dashboard`, it is silently truncated. Quote them.
+
+### Dead variable
+
+`SD_AUTH_GROUP` is exported by the template but no longer read by the application. Safe to remove;
+mention it in the PR description rather than removing it silently.
+
+---
+
+## Change 2 — `base/deployment.yaml`
+
+Declare the metrics port next to the existing one:
+
+```yaml
+ ports:
+ - containerPort: 8000
+ - containerPort: 9090
+ name: metrics
+```
+
+Do **not** add it to the Ingress. The endpoint exposes queue depth and failure counts and is meant
+to stay reachable only inside the cluster.
+
+---
+
+## Change 3 — `overlays/test/kustomization.yaml`
+
+Update the image tag:
+
+```yaml
+images:
+ - name: sd3-api
+ newName: quay.io/stackmon/status-dashboard-v3
+ newTag: sha-
+```
+
+Leave namespace, ingress host and TLS secret unchanged.
+
+---
+
+## Out of scope
+
+- Vault key creation — performed by whoever holds Vault access, not through git.
+- `overlays/prod` — production stays on the current build until the test run succeeds.
+- Ingress changes.
+
+---
+
+## Database migration — required before rollout
+
+Migrations are **not** applied by the application, and the image contains neither the
+`db/migrations` directory nor the `migrate` CLI. Migration `000008` (which creates
+`notification_outbox`) must be applied to the `sd3-test` database out of band, by whoever
+normally runs migrations for this environment.
+
+The application refuses to start when notifications are enabled and the table is absent:
+
+```
+notification_outbox table is missing: apply the pending database migrations
+```
+
+This is deliberate. Without the check the pod would come up healthy and only fail on the
+first maintenance change, turning a deployment mistake into a user-visible error.
+
+Order therefore matters: **apply the migration first, then roll out the new image.**
+
+---
+
+## Verification after rollout
+
+```bash
+kubectl -n sd3-test get pods
+kubectl -n sd3-test logs deploy/sd3-api | head -30
+```
+
+The application refuses to start on invalid notification settings, so a running pod already proves
+the configuration parsed. Typical startup failures name the offending variable:
+
+```
+notifications enabled: SD_SMTP_HOST, SD_SMTP_PORT and SD_SMTP_FROM are required
+SD_NOTIFICATIONS_EMAILS_OPERATORS contains an invalid address "ops at example.com"
+SD_NOTIFICATIONS_LEASE_TIMEOUT (30s) must be greater than SD_SMTP_TIMEOUT (30s)
+```
+
+Check relay connectivity and the queue:
+
+```bash
+kubectl -n sd3-test exec deploy/sd3-api -- nc -zv otc-de-out.mms.t-systems-service.com 25
+kubectl -n sd3-test port-forward deploy/sd3-api 9090:9090
+curl -s http://localhost:9090/metrics | grep notification_outbox
+```
+
+Queue snapshot over the API (requires an admin token):
+
+```bash
+curl -H "Authorization: Bearer $TOKEN" \
+ https://api.test.status.otc-service.com/v2/notifications/stats
+```
+
+A healthy idle queue reports `pending: 0` and `failed: 0`. Rows stuck in `pending` with
+rising `attempts` mean delivery is failing; list them and read the SMTP error:
+
+```bash
+curl -H "Authorization: Bearer $TOKEN" \
+ "https://api.test.status.otc-service.com/v2/notifications/failed?status=pending"
+```
+
+---
+
+## Rollback
+
+Revert the image tag in `overlays/test/kustomization.yaml`, or set
+`export SD_NOTIFICATIONS_ENABLED=false` in the template. With the feature off the application starts
+without any SMTP setting, no worker runs and no mail is sent; queued rows stay in the database
+untouched.
diff --git a/go.mod b/go.mod
index c17139f..d69a9bd 100644
--- a/go.mod
+++ b/go.mod
@@ -6,17 +6,20 @@ require (
github.com/DATA-DOG/go-sqlmock v1.5.2
github.com/coreos/go-oidc/v3 v3.16.0
github.com/gin-gonic/gin v1.11.0
- github.com/golang-jwt/jwt/v5 v5.3.0
+ github.com/golang-jwt/jwt/v5 v5.3.1
github.com/golang-migrate/migrate/v4 v4.19.0
+ github.com/google/uuid v1.6.0
github.com/gorilla/feeds v1.2.0
github.com/joho/godotenv v1.5.1
github.com/kelseyhightower/envconfig v1.4.0
+ github.com/prometheus/client_golang v1.24.1
github.com/stretchr/testify v1.11.1
github.com/swaggo/files v1.0.1
github.com/testcontainers/testcontainers-go v0.39.0
github.com/testcontainers/testcontainers-go/modules/postgres v0.39.0
+ github.com/wneessen/go-mail v0.8.1
go.uber.org/zap v1.27.0
- golang.org/x/oauth2 v0.32.0
+ golang.org/x/oauth2 v0.36.0
gopkg.in/yaml.v3 v3.0.1
gorm.io/driver/postgres v1.6.0
gorm.io/gorm v1.31.0
@@ -27,10 +30,12 @@ require (
dario.cat/mergo v1.0.2 // indirect
github.com/Azure/go-ansiterm v0.0.0-20250102033503-faa5f7b0171c // indirect
github.com/Microsoft/go-winio v0.6.2 // indirect
+ github.com/beorn7/perks v1.0.1 // indirect
github.com/bytedance/gopkg v0.1.3 // indirect
github.com/bytedance/sonic v1.14.2 // indirect
github.com/bytedance/sonic/loader v0.4.0 // indirect
github.com/cenkalti/backoff/v4 v4.3.0 // indirect
+ github.com/cespare/xxhash/v2 v2.3.0 // indirect
github.com/cloudwego/base64x v0.1.6 // indirect
github.com/containerd/errdefs v1.0.0 // indirect
github.com/containerd/errdefs/pkg v0.3.0 // indirect
@@ -55,7 +60,6 @@ require (
github.com/go-playground/validator/v10 v10.28.0 // indirect
github.com/goccy/go-json v0.10.5 // indirect
github.com/goccy/go-yaml v1.18.0 // indirect
- github.com/google/uuid v1.6.0 // indirect
github.com/hashicorp/errwrap v1.1.0 // indirect
github.com/hashicorp/go-multierror v1.1.1 // indirect
github.com/jackc/pgpassfile v1.0.0 // indirect
@@ -65,7 +69,7 @@ require (
github.com/jinzhu/inflection v1.0.0 // indirect
github.com/jinzhu/now v1.1.5 // indirect
github.com/json-iterator/go v1.1.12 // indirect
- github.com/klauspost/compress v1.18.1 // indirect
+ github.com/klauspost/compress v1.19.1 // indirect
github.com/klauspost/cpuid/v2 v2.3.0 // indirect
github.com/leodido/go-urn v1.4.0 // indirect
github.com/lib/pq v1.10.9 // indirect
@@ -82,12 +86,16 @@ require (
github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect
github.com/modern-go/reflect2 v1.0.2 // indirect
github.com/morikuni/aec v1.0.0 // indirect
+ github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect
github.com/opencontainers/go-digest v1.0.0 // indirect
github.com/opencontainers/image-spec v1.1.1 // indirect
github.com/pelletier/go-toml/v2 v2.2.4 // indirect
github.com/pkg/errors v0.9.1 // indirect
github.com/pmezard/go-difflib v1.0.0 // indirect
github.com/power-devops/perfstat v0.0.0-20240221224432-82ca36839d55 // indirect
+ github.com/prometheus/client_model v0.6.2 // indirect
+ github.com/prometheus/common v0.70.1 // indirect
+ github.com/prometheus/procfs v0.21.1 // indirect
github.com/quic-go/qpack v0.5.1 // indirect
github.com/quic-go/quic-go v0.55.0 // indirect
github.com/shirou/gopsutil/v4 v4.25.9 // indirect
@@ -106,13 +114,13 @@ require (
go.uber.org/mock v0.6.0 // indirect
go.uber.org/multierr v1.11.0 // indirect
golang.org/x/arch v0.22.0 // indirect
- golang.org/x/crypto v0.43.0 // indirect
- golang.org/x/mod v0.29.0 // indirect
- golang.org/x/net v0.46.0 // indirect
- golang.org/x/sync v0.17.0 // indirect
- golang.org/x/sys v0.37.0 // indirect
- golang.org/x/text v0.30.0 // indirect
+ golang.org/x/crypto v0.54.0 // indirect
+ golang.org/x/mod v0.37.0 // indirect
+ golang.org/x/net v0.57.0 // indirect
+ golang.org/x/sync v0.22.0 // indirect
+ golang.org/x/sys v0.47.0 // indirect
+ golang.org/x/text v0.40.0 // indirect
golang.org/x/time v0.8.0 // indirect
- golang.org/x/tools v0.38.0 // indirect
- google.golang.org/protobuf v1.36.10 // indirect
+ golang.org/x/tools v0.47.0 // indirect
+ google.golang.org/protobuf v1.36.11 // indirect
)
diff --git a/go.sum b/go.sum
index e927d03..f1d572c 100644
--- a/go.sum
+++ b/go.sum
@@ -9,6 +9,8 @@ github.com/DATA-DOG/go-sqlmock v1.5.2/go.mod h1:88MAG/4G7SMwSE3CeA0ZKzrT5CiOU3OJ
github.com/Microsoft/go-winio v0.6.2 h1:F2VQgta7ecxGYO8k3ZZz3RS8fVIXVxONVUPlNERoyfY=
github.com/Microsoft/go-winio v0.6.2/go.mod h1:yd8OoFMLzJbo9gZq8j5qaps8bJ9aShtEA8Ipt1oGCvU=
github.com/benbjohnson/clock v1.1.0/go.mod h1:J11/hYXuz8f4ySSvYwY0FKfm+ezbsZBKZxNJlLklBHA=
+github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM=
+github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw=
github.com/bytedance/gopkg v0.1.3 h1:TPBSwH8RsouGCBcMBktLt1AymVo2TVsBVCY4b6TnZ/M=
github.com/bytedance/gopkg v0.1.3/go.mod h1:576VvJ+eJgyCzdjS+c4+77QF3p7ubbtiKARP3TxducM=
github.com/bytedance/sonic v1.14.2 h1:k1twIoe97C1DtYUo+fZQy865IuHia4PR5RPiuGPPIIE=
@@ -17,6 +19,8 @@ github.com/bytedance/sonic/loader v0.4.0 h1:olZ7lEqcxtZygCK9EKYKADnpQoYkRQxaeY2N
github.com/bytedance/sonic/loader v0.4.0/go.mod h1:AR4NYCk5DdzZizZ5djGqQ92eEhCCcdf5x77udYiSJRo=
github.com/cenkalti/backoff/v4 v4.3.0 h1:MyRJ/UdXutAwSAT+s3wNd7MfTIcy71VQueUuFK343L8=
github.com/cenkalti/backoff/v4 v4.3.0/go.mod h1:Y3VNntkOUPxTVeUxJ/G5vcM//AlwfmyYozVcomhLiZE=
+github.com/cespare/xxhash/v2 v2.3.0 h1:UL815xU9SqsFlibzuggzjXhog7bL6oX9BbNZnL2UFvs=
+github.com/cespare/xxhash/v2 v2.3.0/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs=
github.com/cloudwego/base64x v0.1.6 h1:t11wG9AECkCDk5fMSoxmufanudBtJ+/HemLstXDLI2M=
github.com/cloudwego/base64x v0.1.6/go.mod h1:OFcloc187FXDaYHvrNIjxSe8ncn0OOM8gEHfghB2IPU=
github.com/containerd/errdefs v1.0.0 h1:tg5yIfIlQIrxYtu9ajqY42W3lpS19XqdxRQeEwYG8PI=
@@ -78,8 +82,8 @@ github.com/goccy/go-json v0.10.5 h1:Fq85nIqj+gXn/S5ahsiTlK3TmC85qgirsdTP/+DeaC4=
github.com/goccy/go-json v0.10.5/go.mod h1:oq7eo15ShAhp70Anwd5lgX2pLfOS3QCiwU/PULtXL6M=
github.com/goccy/go-yaml v1.18.0 h1:8W7wMFS12Pcas7KU+VVkaiCng+kG8QiFeFwzFb+rwuw=
github.com/goccy/go-yaml v1.18.0/go.mod h1:XBurs7gK8ATbW4ZPGKgcbrY1Br56PdM69F7LkFRi1kA=
-github.com/golang-jwt/jwt/v5 v5.3.0 h1:pv4AsKCKKZuqlgs5sUmn4x8UlGa0kEVt/puTpKx9vvo=
-github.com/golang-jwt/jwt/v5 v5.3.0/go.mod h1:fxCRLWMO43lRc8nhHWY6LGqRcf+1gQWArsqaEUEa5bE=
+github.com/golang-jwt/jwt/v5 v5.3.1 h1:kYf81DTWFe7t+1VvL7eS+jKFVWaUnK9cB1qbwn63YCY=
+github.com/golang-jwt/jwt/v5 v5.3.1/go.mod h1:fxCRLWMO43lRc8nhHWY6LGqRcf+1gQWArsqaEUEa5bE=
github.com/golang-migrate/migrate/v4 v4.19.0 h1:RcjOnCGz3Or6HQYEJ/EEVLfWnmw9KnoigPSjzhCuaSE=
github.com/golang-migrate/migrate/v4 v4.19.0/go.mod h1:9dyEcu+hO+G9hPSw8AIg50yg622pXJsoHItQnDGZkI0=
github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8=
@@ -116,8 +120,8 @@ github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHm
github.com/kelseyhightower/envconfig v1.4.0 h1:Im6hONhd3pLkfDFsbRgu68RDNkGF1r3dvMUtDTo2cv8=
github.com/kelseyhightower/envconfig v1.4.0/go.mod h1:cccZRl6mQpaq41TPp5QxidR+Sa3axMbJDNb//FQX6Gg=
github.com/kisielk/sqlstruct v0.0.0-20201105191214-5f3e10d3ab46/go.mod h1:yyMNCyc/Ib3bDTKd379tNMpB/7/H5TjM2Y9QJ5THLbE=
-github.com/klauspost/compress v1.18.1 h1:bcSGx7UbpBqMChDtsF28Lw6v/G94LPrrbMbdC3JH2co=
-github.com/klauspost/compress v1.18.1/go.mod h1:ZQFFVG+MdnR0P+l6wpXgIL4NTtwiKIdBnrBd8Nrxr+0=
+github.com/klauspost/compress v1.19.1 h1:VsB4HPswih7mmZ8WleSFQ75c/Ui1M4trX5oAsJnhSlk=
+github.com/klauspost/compress v1.19.1/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
github.com/klauspost/cpuid/v2 v2.3.0 h1:S4CRMLnYUhGeDFDqkGriYKdfoFlDnMtqTiI/sFzhA9Y=
github.com/klauspost/cpuid/v2 v2.3.0/go.mod h1:hqwkgyIinND0mEev00jJYCxPNVRVXFQeu1XKlok6oO0=
github.com/kr/pretty v0.1.0/go.mod h1:dAy3ld7l9f0ibDNOQOHHMYYIIbhfbHSm3C4ZsoJORNo=
@@ -127,6 +131,8 @@ github.com/kr/pty v1.1.1/go.mod h1:pFQYn66WHrOpPYNljwOMqo10TkYh1fy3cYio2l3bCsQ=
github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI=
github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE=
+github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc=
+github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw=
github.com/leodido/go-urn v1.4.0 h1:WT9HwE9SGECu3lg4d/dIA+jxlljEa1/ffXKmRjqdmIQ=
github.com/leodido/go-urn v1.4.0/go.mod h1:bvxc+MVxLKB4z00jd1z+Dvzr47oO32F/QSNjSBOlFxI=
github.com/lib/pq v1.10.9 h1:YXG7RB+JIjhP29X+OtkiDnYaXQwpS4JEWq7dtCCRUEw=
@@ -162,6 +168,8 @@ github.com/modern-go/reflect2 v1.0.2 h1:xBagoLtFs94CBntxluKeaWgTMpvLxC4ur3nMaC9G
github.com/modern-go/reflect2 v1.0.2/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk=
github.com/morikuni/aec v1.0.0 h1:nP9CBfwrvYnBRgY6qfDQkygYDmYwOilePFkwzv4dU8A=
github.com/morikuni/aec v1.0.0/go.mod h1:BbKIizmSmc5MMPqRYbxO4ZU0S0+P200+tUnFx7PXmsc=
+github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA=
+github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ=
github.com/opencontainers/go-digest v1.0.0 h1:apOUWs51W5PlhuyGyz9FCeeBIOUDA/6nW8Oi/yOhh5U=
github.com/opencontainers/go-digest v1.0.0/go.mod h1:0JzlMkj0TRzQZfJkVvzbP0HBR3IKzErnv2BNG4W4MAM=
github.com/opencontainers/image-spec v1.1.1 h1:y0fUlFfIZhPF1W537XOLg0/fcx6zcHCJwooC2xJA040=
@@ -175,6 +183,14 @@ github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZb
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
github.com/power-devops/perfstat v0.0.0-20240221224432-82ca36839d55 h1:o4JXh1EVt9k/+g42oCprj/FisM4qX9L3sZB3upGN2ZU=
github.com/power-devops/perfstat v0.0.0-20240221224432-82ca36839d55/go.mod h1:OmDBASR4679mdNQnz2pUhc2G8CO2JrUAVFDRBDP/hJE=
+github.com/prometheus/client_golang v1.24.1 h1:JnJkREXzWxUdCuPFpIWZiPispT9xVV59uiuyR2bPlnU=
+github.com/prometheus/client_golang v1.24.1/go.mod h1:F+oSRECHg4sse5ucfYpYDeIv/hu68Zo0uoHKetWnzcE=
+github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk=
+github.com/prometheus/client_model v0.6.2/go.mod h1:y3m2F6Gdpfy6Ut/GBsUqTWZqCUvMVzSfMLjcu6wAwpE=
+github.com/prometheus/common v0.70.1 h1:1HvjP4D5oL3t8RsPlwxA9onvvStjtIHYE5XuuwOi/PY=
+github.com/prometheus/common v0.70.1/go.mod h1:VdFUQDMZK3VLkurFUVhia6uys/0suUp86TJz5qbJRhc=
+github.com/prometheus/procfs v0.21.1 h1:GljZCt+zSTS+NZq88cyQ1LjZ+RCHp3uVuabBWA5+OJI=
+github.com/prometheus/procfs v0.21.1/go.mod h1:aB55Cww9pdSJVHk0hUf0inxWyyjPogFIjmHKYgMKmtY=
github.com/quic-go/qpack v0.5.1 h1:giqksBPnT/HDtZ6VhtFKgoLOWmlyo9Ei6u9PqzIMbhI=
github.com/quic-go/qpack v0.5.1/go.mod h1:+PC4XFrEskIVkcLzpEkbLqq1uCoxPhQuvK5rH1ZgaEg=
github.com/quic-go/quic-go v0.55.0 h1:zccPQIqYCXDt5NmcEabyYvOnomjs8Tlwl7tISjJh9Mk=
@@ -212,6 +228,8 @@ github.com/twitchyliquid64/golang-asm v0.15.1 h1:SU5vSMR7hnwNxj24w34ZyCi/FmDZTkS
github.com/twitchyliquid64/golang-asm v0.15.1/go.mod h1:a1lVb/DtPvCB8fslRZhAngC2+aY1QWCk3Cedj/Gdt08=
github.com/ugorji/go/codec v1.3.1 h1:waO7eEiFDwidsBN6agj1vJQ4AG7lh2yqXyOXqhgQuyY=
github.com/ugorji/go/codec v1.3.1/go.mod h1:pRBVtBSKl77K30Bv8R2P+cLSGaTtex6fsA2Wjqmfxj4=
+github.com/wneessen/go-mail v0.8.1 h1:tVcncj02/QySVFw3zr/kXOzZcuFQqBNT6K+Rbgm/pcM=
+github.com/wneessen/go-mail v0.8.1/go.mod h1:dWZ61zadzCIyvB4y1/YzC5O7MrbbzBfPkARmbosdf8w=
github.com/yuin/goldmark v1.3.5/go.mod h1:mwnBkeHKe2W/ZEtQ+71ViKU8L12m81fl3OWwC1Zlc8k=
github.com/yuin/goldmark v1.4.13/go.mod h1:6yULJ656Px+3vBD8DxQVa3kxgyrAnzto9xy5taEt/CY=
github.com/yusufpapurcu/wmi v1.2.4 h1:zFUKzehAFReQwLys1b/iSMl+JQGSCSjtVqQn9bBrPo0=
@@ -249,18 +267,20 @@ go.uber.org/multierr v1.11.0/go.mod h1:20+QtiLqy0Nd6FdQB9TLXag12DsQkrbs3htMFfDN8
go.uber.org/zap v1.21.0/go.mod h1:wjWOCqI0f2ZZrJF/UufIOkiC8ii6tm1iqIsLo76RfJw=
go.uber.org/zap v1.27.0 h1:aJMhYGrd5QSmlpLMr2MftRKl7t8J8PTZPA732ud/XR8=
go.uber.org/zap v1.27.0/go.mod h1:GB2qFLM7cTU87MWRP2mPIjqfIDnGu+VIO4V/SdhGo2E=
+go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ=
+go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ=
golang.org/x/arch v0.22.0 h1:c/Zle32i5ttqRXjdLyyHZESLD/bB90DCU1g9l/0YBDI=
golang.org/x/arch v0.22.0/go.mod h1:dNHoOeKiyja7GTvF9NJS1l3Z2yntpQNzgrjh1cU103A=
golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w=
golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI=
golang.org/x/crypto v0.0.0-20210921155107-089bfa567519/go.mod h1:GvvjBRRGRdwPK5ydBHafDWAxML/pGHZbMvKqRZ5+Abc=
-golang.org/x/crypto v0.43.0 h1:dduJYIi3A3KOfdGOHX8AVZ/jGiyPa3IbBozJ5kNuE04=
-golang.org/x/crypto v0.43.0/go.mod h1:BFbav4mRNlXJL4wNeejLpWxB7wMbc79PdRGhWKncxR0=
+golang.org/x/crypto v0.54.0 h1:YLIA59K4fiNzHzjnZt2tUJQjQtUWfWbeHBqKtk3eScw=
+golang.org/x/crypto v0.54.0/go.mod h1:KWL8ny2AZdGR2cWmzeHrp2azQPGogOv+HeQaVEXC2dk=
golang.org/x/lint v0.0.0-20190930215403-16217165b5de/go.mod h1:6SW0HCj/g11FgYtHlgUYUwCkIfeOF89ocIRzGO/8vkc=
golang.org/x/mod v0.4.2/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA=
golang.org/x/mod v0.6.0-dev.0.20220419223038-86c51ed26bb4/go.mod h1:jJ57K6gSWd91VN4djpZkiMVwK6gcyfeH4XE8wZrZaV4=
-golang.org/x/mod v0.29.0 h1:HV8lRxZC4l2cr3Zq1LvtOsi/ThTgWnUk/y64QSs8GwA=
-golang.org/x/mod v0.29.0/go.mod h1:NyhrlYXJ2H4eJiRy/WDBO6HMqZQ6q9nk4JzS3NuCK+w=
+golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ=
+golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0=
golang.org/x/net v0.0.0-20190311183353-d8887717615a/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg=
golang.org/x/net v0.0.0-20190404232315-eb5bcb51f2a3/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg=
golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s=
@@ -268,15 +288,15 @@ golang.org/x/net v0.0.0-20210226172049-e18ecbb05110/go.mod h1:m0MpNAwzfU5UDzcl9v
golang.org/x/net v0.0.0-20210405180319-a5a99cb37ef4/go.mod h1:p54w0d4576C0XHj96bSt6lcn1PtDYWL6XObtHCRCNQM=
golang.org/x/net v0.0.0-20220722155237-a158d28d115b/go.mod h1:XRhObCWvk6IyKnWLug+ECip1KBveYUHfp+8e9klMJ9c=
golang.org/x/net v0.7.0/go.mod h1:2Tu9+aMcznHK/AK1HMvgo6xiTLG5rD5rZLDS+rp2Bjs=
-golang.org/x/net v0.46.0 h1:giFlY12I07fugqwPuWJi68oOnpfqFnJIJzaIIm2JVV4=
-golang.org/x/net v0.46.0/go.mod h1:Q9BGdFy1y4nkUwiLvT5qtyhAnEHgnQ/zd8PfU6nc210=
-golang.org/x/oauth2 v0.32.0 h1:jsCblLleRMDrxMN29H3z/k1KliIvpLgCkE6R8FXXNgY=
-golang.org/x/oauth2 v0.32.0/go.mod h1:lzm5WQJQwKZ3nwavOZ3IS5Aulzxi68dUSgRHujetwEA=
+golang.org/x/net v0.57.0 h1:K5+3DljvIuDG9/Jv9rvyMywYNFCQ9RSUY6OOTTkT+tE=
+golang.org/x/net v0.57.0/go.mod h1:KpXc8iv+r3XplLAG/f7Jsf9RPszJzdR0f58q9vGOuEU=
+golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs=
+golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q=
golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
golang.org/x/sync v0.0.0-20210220032951-036812b2e83c/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
golang.org/x/sync v0.0.0-20220722155255-886fb9371eb4/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
-golang.org/x/sync v0.17.0 h1:l60nONMj9l5drqw6jlhIELNv9I0A4OFgRsG9k2oT9Ug=
-golang.org/x/sync v0.17.0/go.mod h1:9KTHXmSnoGruLpwFjVSX0lNNA75CykiMECbovNTZqGI=
+golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
+golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
golang.org/x/sys v0.0.0-20190412213103-97732733099d/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20190916202348-b4ddaad3f8a3/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
@@ -292,19 +312,19 @@ golang.org/x/sys v0.0.0-20220722155257-8c9f86f7a55f/go.mod h1:oPkhp1MJrh7nUepCBc
golang.org/x/sys v0.1.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.5.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
-golang.org/x/sys v0.37.0 h1:fdNQudmxPjkdUTPnLn5mdQv7Zwvbvpaxqs831goi9kQ=
-golang.org/x/sys v0.37.0/go.mod h1:OgkHotnGiDImocRcuBABYBEXf8A9a87e/uXjp9XT3ks=
+golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
+golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
golang.org/x/term v0.5.0/go.mod h1:jMB1sMXY+tzblOD4FWmEbocvup2/aLOaQEp7JmGp78k=
-golang.org/x/term v0.36.0 h1:zMPR+aF8gfksFprF/Nc/rd1wRS1EI6nDBGyWAvDzx2Q=
-golang.org/x/term v0.36.0/go.mod h1:Qu394IJq6V6dCBRgwqshf3mPF85AqzYEzofzRdZkWss=
+golang.org/x/term v0.45.0 h1:NwWyBmoJCbfTHpxrWoZ9C6/VxOf7ic219I8xZZFdrf0=
+golang.org/x/term v0.45.0/go.mod h1:9aqxs0blBcrm/n0L9QW0aRVD+ktan8ssZromtqJC43w=
golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
golang.org/x/text v0.3.7/go.mod h1:u+2+/6zg+i71rQMx5EYifcz6MCKuco9NR6JIITiCfzQ=
golang.org/x/text v0.7.0/go.mod h1:mrYo+phRRbMaCq/xk9113O4dZlRixOauAjOtrjsXDZ8=
-golang.org/x/text v0.30.0 h1:yznKA/E9zq54KzlzBEAWn1NXSQ8DIp/NYMy88xJjl4k=
-golang.org/x/text v0.30.0/go.mod h1:yDdHFIX9t+tORqspjENWgzaCVXgk0yYnYuSZ8UzzBVM=
+golang.org/x/text v0.40.0 h1:Ub2Z6/xjgF1WrYQz2nuITOEegKFtiIy+rieRJ5lHZKs=
+golang.org/x/text v0.40.0/go.mod h1:hpnzDAfGV753zIKo+wk3u1bVKCGPbrnF7+7LBF/UHVY=
golang.org/x/time v0.8.0 h1:9i3RxcPv3PZnitoVGMPDKZSq1xW1gK1Xy3ArNOGZfEg=
golang.org/x/time v0.8.0/go.mod h1:3BpzKBy/shNhVucY/MWOyx10tF3SFh9QdLuxbVysPQM=
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
@@ -312,8 +332,8 @@ golang.org/x/tools v0.0.0-20190311212946-11955173bddd/go.mod h1:LCzVGOaR6xXOjkQ3
golang.org/x/tools v0.0.0-20191119224855-298f0cb1881e/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo=
golang.org/x/tools v0.1.5/go.mod h1:o0xws9oXOQQZyjljx8fwUC0k7L1pTE6eaCbjGeHmOkk=
golang.org/x/tools v0.1.12/go.mod h1:hNGJHUnrk76NpqgfD5Aqm5Crs+Hm0VOH/i9J2+nxYbc=
-golang.org/x/tools v0.38.0 h1:Hx2Xv8hISq8Lm16jvBZ2VQf+RLmbd7wVUsALibYI/IQ=
-golang.org/x/tools v0.38.0/go.mod h1:yEsQ/d/YK8cjh0L6rZlY8tgtlKiBNTL14pGDJPJpYQs=
+golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q=
+golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA=
golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
golang.org/x/xerrors v0.0.0-20191011141410-1b5146add898/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
@@ -324,8 +344,8 @@ google.golang.org/genproto/googleapis/rpc v0.0.0-20241104194629-dd2ea8efbc28 h1:
google.golang.org/genproto/googleapis/rpc v0.0.0-20241104194629-dd2ea8efbc28/go.mod h1:GX3210XPVPUjJbTUbvwI8f2IpZDMZuPJWDzDuebbviI=
google.golang.org/grpc v1.67.1 h1:zWnc1Vrcno+lHZCOofnIMvycFcc0QRGIzm9dhnDX68E=
google.golang.org/grpc v1.67.1/go.mod h1:1gLDyUQU7CTLJI90u3nXZ9ekeghjeM7pTDZlqFNg2AA=
-google.golang.org/protobuf v1.36.10 h1:AYd7cD/uASjIL6Q9LiTjz8JLcrh/88q5UObnmY3aOOE=
-google.golang.org/protobuf v1.36.10/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
+google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE=
+google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/check.v1 v1.0.0-20180628173108-788fd7840127/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk=
diff --git a/internal/api/api.go b/internal/api/api.go
index 2ddaeb8..b03663d 100644
--- a/internal/api/api.go
+++ b/internal/api/api.go
@@ -11,6 +11,7 @@ import (
"github.com/stackmon/otc-status-dashboard/internal/api/rbac"
"github.com/stackmon/otc-status-dashboard/internal/conf"
"github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/notification"
)
type API struct {
@@ -20,6 +21,7 @@ type API struct {
oa2Prov *auth.Provider
secretKeyV1 string
rbac *rbac.Service
+ notifier *notification.Publisher
}
func New(cfg *conf.Config, log *zap.Logger, database *db.DB) (*API, error) {
@@ -41,11 +43,17 @@ func New(cfg *conf.Config, log *zap.Logger, database *db.DB) (*API, error) {
r := gin.New()
r.Use(Logger(log), gin.Recovery())
r.Use(ErrorHandle())
+ r.Use(SecurityHeaders())
r.Use(CORSMiddleware())
r.NoRoute(errors.Return404)
rbacService := rbac.New(cfg.RBAC.Creators, cfg.RBAC.Operators, cfg.RBAC.Admins)
+ ncfg, err := notification.ConfigFromConf(cfg)
+ if err != nil {
+ return nil, fmt.Errorf("could not parse notification config: %w", err)
+ }
+
a := &API{
r: r,
db: database,
@@ -53,8 +61,9 @@ func New(cfg *conf.Config, log *zap.Logger, database *db.DB) (*API, error) {
oa2Prov: oa2Prov,
secretKeyV1: cfg.SecretKeyV1,
rbac: rbacService,
+ notifier: notification.NewPublisher(ncfg, database),
}
- if err := a.InitRoutes(cfg.OpenAPISpecPath); err != nil {
+ if err = a.InitRoutes(cfg.OpenAPISpecPath); err != nil {
return nil, fmt.Errorf("init routes: %w", err)
}
return a, nil
@@ -63,3 +72,9 @@ func New(cfg *conf.Config, log *zap.Logger, database *db.DB) (*API, error) {
func (a *API) Router() *gin.Engine {
return a.r
}
+
+// Publisher returns the notification publisher so the delivery worker's Notify can
+// be wired in during app startup.
+func (a *API) Publisher() *notification.Publisher {
+ return a.notifier
+}
diff --git a/internal/api/errors/incident.go b/internal/api/errors/incident.go
index 1b16d82..934f832 100644
--- a/internal/api/errors/incident.go
+++ b/internal/api/errors/incident.go
@@ -1,6 +1,10 @@
package errors
-import "errors"
+import (
+ "errors"
+ "fmt"
+ "strings"
+)
var ErrIncidentDSNotExist = errors.New("event does not exist")
var ErrIncidentEndDateShouldBeEmpty = errors.New("event end_date should be empty")
@@ -48,6 +52,22 @@ var ErrMaintenanceContactEmailInvalid = errors.New("contact_email has invalid fo
var ErrMaintenanceEndDateBeforeStart = errors.New("end_date must be after start_date")
var ErrMaintenanceDescriptionRequired = errors.New("description is required for maintenance")
+// NewErrMaintenanceContactEmailDomain names the permitted domains so the caller can
+// fix the request without reading the deployment configuration.
+func NewErrMaintenanceContactEmailDomain(allowed []string) error {
+ return fmt.Errorf("contact_email domain is not allowed, permitted domains: %s", strings.Join(allowed, ", "))
+}
+
+// NewErrNotificationStatusInvalid names the statuses accepted by ?status=.
+func NewErrNotificationStatusInvalid(allowed []string) error {
+ return fmt.Errorf("invalid status, expected one of: %s", strings.Join(allowed, ", "))
+}
+
+// NewErrNotificationLimitInvalid states the accepted range for ?limit=.
+func NewErrNotificationLimitInvalid(maxLimit int) error {
+ return fmt.Errorf("invalid limit, expected a number in range 1:%d", maxLimit)
+}
+
// Errors for extract restrictions
var ErrExtractForbiddenRole = errors.New("extract is only available for operators and admins")
diff --git a/internal/api/middleware.go b/internal/api/middleware.go
index 839d59d..2df9b58 100644
--- a/internal/api/middleware.go
+++ b/internal/api/middleware.go
@@ -25,6 +25,7 @@ const (
const (
usernameClaim = "preferred_username"
+ emailClaim = "email"
groupsClaim = "groups"
)
@@ -157,6 +158,8 @@ func validateAndSetClaims(
return apiErrors.ErrAuthTokenInvalid
}
+ setUserEmailFromClaims(claims, c)
+
username, _ := c.Get(v2.UsernameContextKey)
usernameStr, _ := username.(string)
@@ -230,6 +233,19 @@ func setUserIDFromClaims(claims jwt.MapClaims, c *gin.Context, logger *zap.Logge
return nil
}
+// setUserEmailFromClaims stores the "email" claim when present. It is optional on
+// purpose: HMAC and service tokens carry no email, and that must not block the request.
+func setUserEmailFromClaims(claims jwt.MapClaims, c *gin.Context) {
+ email, exists := claims[emailClaim]
+ if !exists {
+ return
+ }
+
+ if emailStr, ok := email.(string); ok && emailStr != "" {
+ c.Set(v2.UserEmailContextKey, emailStr)
+ }
+}
+
// setGroupsFromClaims extracts the "groups" claim from JWT as a string slice.
func setGroupsFromClaims(claims jwt.MapClaims, c *gin.Context, logger *zap.Logger) error {
groupsCl, exists := claims[groupsClaim]
@@ -380,6 +396,14 @@ func Logger(log *zap.Logger) gin.HandlerFunc {
}
}
+func SecurityHeaders() gin.HandlerFunc {
+ return func(c *gin.Context) {
+ c.Writer.Header().Set("X-Frame-Options", "DENY")
+ c.Writer.Header().Set("Content-Security-Policy", "frame-ancestors 'none'")
+ c.Next()
+ }
+}
+
func CORSMiddleware() gin.HandlerFunc {
return func(c *gin.Context) {
c.Writer.Header().Set("Access-Control-Allow-Origin", "*")
diff --git a/internal/api/middleware_test.go b/internal/api/middleware_test.go
index 26086b0..9ca492c 100644
--- a/internal/api/middleware_test.go
+++ b/internal/api/middleware_test.go
@@ -571,3 +571,20 @@ func TestAuthAudit_DoesNotPanic(t *testing.T) {
authAudit(logger, "authorization", "denied", "", "user2", "no_matching_rbac_group")
})
}
+
+func TestSecurityHeaders(t *testing.T) {
+ gin.SetMode(gin.TestMode)
+ r := gin.New()
+ r.Use(SecurityHeaders())
+ r.GET("/test", func(c *gin.Context) {
+ c.String(http.StatusOK, "ok")
+ })
+
+ w := httptest.NewRecorder()
+ req := httptest.NewRequest(http.MethodGet, "/test", nil)
+ r.ServeHTTP(w, req)
+
+ assert.Equal(t, http.StatusOK, w.Code)
+ assert.Equal(t, "DENY", w.Header().Get("X-Frame-Options"))
+ assert.Equal(t, "frame-ancestors 'none'", w.Header().Get("Content-Security-Policy"))
+}
diff --git a/internal/api/routes.go b/internal/api/routes.go
index 6fa3247..8c11739 100644
--- a/internal/api/routes.go
+++ b/internal/api/routes.go
@@ -64,7 +64,7 @@ func (a *API) InitRoutes(openAPISpecPath string) error {
AuthenticationMW(a.oa2Prov, a.log, a.secretKeyV1),
RBACAuthorizationMW(a.rbac, a.log),
ValidateComponentsMW(a.db, a.log),
- v2.PostIncidentHandler(a.db, a.log),
+ v2.PostIncidentHandler(a.db, a.log, a.notifier),
)
v2API.GET("incidents/:eventID",
SetJWTClaims(a.oa2Prov, a.log, a.secretKeyV1),
@@ -73,7 +73,7 @@ func (a *API) InitRoutes(openAPISpecPath string) error {
AuthenticationMW(a.oa2Prov, a.log, a.secretKeyV1),
RBACAuthorizationMW(a.rbac, a.log),
CheckEventExistenceMW(a.db, a.log),
- v2.PatchIncidentHandler(a.db, a.log))
+ v2.PatchIncidentHandler(a.db, a.log, a.notifier))
v2API.POST("incidents/:eventID/extract",
AuthenticationMW(a.oa2Prov, a.log, a.secretKeyV1),
RBACAuthorizationMW(a.rbac, a.log),
@@ -95,7 +95,7 @@ func (a *API) InitRoutes(openAPISpecPath string) error {
AuthenticationMW(a.oa2Prov, a.log, a.secretKeyV1),
RBACAuthorizationMW(a.rbac, a.log),
ValidateComponentsMW(a.db, a.log),
- v2.PostIncidentHandler(a.db, a.log))
+ v2.PostIncidentHandler(a.db, a.log, a.notifier))
v2API.GET("events/:eventID",
SetJWTClaims(a.oa2Prov, a.log, a.secretKeyV1),
v2.GetIncidentHandler(a.db, a.log, a.rbac))
@@ -103,7 +103,7 @@ func (a *API) InitRoutes(openAPISpecPath string) error {
AuthenticationMW(a.oa2Prov, a.log, a.secretKeyV1),
RBACAuthorizationMW(a.rbac, a.log),
CheckEventExistenceMW(a.db, a.log),
- v2.PatchIncidentHandler(a.db, a.log))
+ v2.PatchIncidentHandler(a.db, a.log, a.notifier))
v2API.POST("events/:eventID/extract",
AuthenticationMW(a.oa2Prov, a.log, a.secretKeyV1),
RBACAuthorizationMW(a.rbac, a.log),
@@ -118,6 +118,20 @@ func (a *API) InitRoutes(openAPISpecPath string) error {
// Availability section.
v2API.GET("availability", v2.GetComponentsAvailabilityHandler(a.db, a.log))
+ // Notifications operations (admin only): queue stats, failed rows, re-drive.
+ v2API.GET("notifications/stats",
+ AuthenticationMW(a.oa2Prov, a.log, a.secretKeyV1),
+ RBACAuthorizationMW(a.rbac, a.log),
+ v2.GetNotificationStatsHandler(a.db, a.log))
+ v2API.GET("notifications/failed",
+ AuthenticationMW(a.oa2Prov, a.log, a.secretKeyV1),
+ RBACAuthorizationMW(a.rbac, a.log),
+ v2.GetFailedNotificationsHandler(a.db, a.log))
+ v2API.POST("notifications/redrive",
+ AuthenticationMW(a.oa2Prov, a.log, a.secretKeyV1),
+ RBACAuthorizationMW(a.rbac, a.log),
+ v2.RedriveNotificationsHandler(a.db, a.log, a.notifier))
+
// For testing purposes only.
v2API.GET("rss/", newRSS.HandleRSS(a.db, a.log))
}
diff --git a/internal/api/v2/notifications.go b/internal/api/v2/notifications.go
new file mode 100644
index 0000000..fcc5ed7
--- /dev/null
+++ b/internal/api/v2/notifications.go
@@ -0,0 +1,149 @@
+package v2
+
+import (
+ "net/http"
+ "slices"
+ "strconv"
+ "time"
+
+ "github.com/gin-gonic/gin"
+ "go.uber.org/zap"
+
+ apiErrors "github.com/stackmon/otc-status-dashboard/internal/api/errors"
+ "github.com/stackmon/otc-status-dashboard/internal/api/rbac"
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/notification"
+)
+
+// statsStaleThreshold marks a processing row as stuck when its lease is older than
+// this — comfortably beyond the default lease timeout so live sends are not flagged.
+const statsStaleThreshold = 2 * time.Minute
+
+// defaultFailedListLimit bounds the failed-rows listing.
+const defaultFailedListLimit = 100
+
+// maxFailedListLimit caps ?limit= so one request cannot dump the whole outbox.
+const maxFailedListLimit = 1000
+
+// requireAdmin ensures the caller resolved to the Admin role. It writes the error
+// response and returns false when not.
+func requireAdmin(c *gin.Context, logger *zap.Logger) bool {
+ role, ok := getRoleFromContext(c, logger)
+ if !ok {
+ return false
+ }
+ if role != rbac.Admin {
+ apiErrors.RaiseForbiddenErr(c, apiErrors.ErrAuthForbidden)
+ return false
+ }
+ return true
+}
+
+// GetNotificationStatsHandler returns the outbox queue statistics (admin only).
+func GetNotificationStatsHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
+ return func(c *gin.Context) {
+ if !requireAdmin(c, logger) {
+ return
+ }
+
+ stats, err := dbInst.GetNotificationStats(c.Request.Context(), statsStaleThreshold)
+ if err != nil {
+ logger.Error("failed to get notification stats", zap.Error(err))
+ apiErrors.RaiseInternalErr(c, err)
+ return
+ }
+ c.JSON(http.StatusOK, stats)
+ }
+}
+
+// GetFailedNotificationsHandler lists recent outbox rows (admin only). It defaults to
+// the failed ones, and accepts ?status= and ?limit= to inspect the rest of the queue.
+func GetFailedNotificationsHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
+ return func(c *gin.Context) {
+ if !requireAdmin(c, logger) {
+ return
+ }
+
+ status := c.DefaultQuery("status", db.NotificationStatusFailed)
+ if !isListableStatus(status) {
+ apiErrors.RaiseBadRequestErr(c, apiErrors.NewErrNotificationStatusInvalid(listableStatuses()))
+ return
+ }
+
+ limit, err := parseListLimit(c.Query("limit"))
+ if err != nil {
+ apiErrors.RaiseBadRequestErr(c, err)
+ return
+ }
+
+ rows, err := dbInst.ListNotificationsByStatus(c.Request.Context(), status, limit)
+ if err != nil {
+ logger.Error("failed to list notifications", zap.String("status", status), zap.Error(err))
+ apiErrors.RaiseInternalErr(c, err)
+ return
+ }
+ c.JSON(http.StatusOK, gin.H{"data": rows, "status": status, "limit": limit})
+ }
+}
+
+// listableStatuses is the set accepted by ?status=, in queue order.
+func listableStatuses() []string {
+ return []string{
+ db.NotificationStatusPending,
+ db.NotificationStatusProcessing,
+ db.NotificationStatusSent,
+ db.NotificationStatusFailed,
+ }
+}
+
+func isListableStatus(status string) bool {
+ return slices.Contains(listableStatuses(), status)
+}
+
+// parseListLimit bounds the page so a large queue cannot be dumped in one response.
+func parseListLimit(raw string) (int, error) {
+ if raw == "" {
+ return defaultFailedListLimit, nil
+ }
+
+ limit, err := strconv.Atoi(raw)
+ if err != nil || limit < 1 || limit > maxFailedListLimit {
+ return 0, apiErrors.NewErrNotificationLimitInvalid(maxFailedListLimit)
+ }
+
+ return limit, nil
+}
+
+// RedriveNotificationsData is the optional re-drive request body.
+type RedriveNotificationsData struct {
+ // IDs limits the re-drive to specific outbox rows; empty means all failed rows.
+ IDs []uint `json:"ids"`
+}
+
+// RedriveNotificationsHandler resets failed rows back to pending (admin only) and
+// wakes the worker to retry them immediately.
+func RedriveNotificationsHandler(dbInst *db.DB, logger *zap.Logger, pub ...*notification.Publisher) gin.HandlerFunc {
+ publisher := optionalPublisher(pub)
+ return func(c *gin.Context) {
+ if !requireAdmin(c, logger) {
+ return
+ }
+
+ var body RedriveNotificationsData
+ // A missing/empty body is valid: re-drive everything.
+ _ = c.ShouldBindBodyWithJSON(&body)
+
+ count, err := dbInst.RedriveFailed(c.Request.Context(), body.IDs...)
+ if err != nil {
+ logger.Error("failed to re-drive notifications", zap.Error(err))
+ apiErrors.RaiseInternalErr(c, err)
+ return
+ }
+
+ if count > 0 {
+ publisher.Notify() // wake the worker to pick up the re-driven rows
+ }
+ logger.Info("re-drove failed notifications", zap.Int64("count", count))
+ c.JSON(http.StatusOK, gin.H{"redriven": count})
+ }
+}
diff --git a/internal/api/v2/v2.go b/internal/api/v2/v2.go
index ae90ce7..4bcac7b 100644
--- a/internal/api/v2/v2.go
+++ b/internal/api/v2/v2.go
@@ -1,6 +1,7 @@
package v2
import (
+ "context"
"errors"
"fmt"
"net/http"
@@ -10,11 +11,13 @@ import (
"github.com/gin-gonic/gin"
"go.uber.org/zap"
+ "gorm.io/gorm"
apiErrors "github.com/stackmon/otc-status-dashboard/internal/api/errors"
"github.com/stackmon/otc-status-dashboard/internal/api/rbac"
"github.com/stackmon/otc-status-dashboard/internal/db"
"github.com/stackmon/otc-status-dashboard/internal/event"
+ "github.com/stackmon/otc-status-dashboard/internal/notification"
)
const (
@@ -25,11 +28,11 @@ const (
const (
authorizedView = true
- publicView = false
)
const (
UsernameContextKey = "userID"
+ UserEmailContextKey = "userEmail"
UserIDGroupsContextKey = "userIDGroups"
RoleContextKey = "role"
)
@@ -361,7 +364,8 @@ func toAPIEvent(inc *db.Incident, isAuth bool) *Incident {
return &Incident{IncidentID{ID: int(inc.ID)}, incData}
}
-func PostIncidentHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
+func PostIncidentHandler(dbInst *db.DB, logger *zap.Logger, pub ...*notification.Publisher) gin.HandlerFunc {
+ publisher := optionalPublisher(pub)
return func(c *gin.Context) {
var incData IncidentData
if err := c.ShouldBindBodyWithJSON(&incData); err != nil {
@@ -370,7 +374,7 @@ func PostIncidentHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
return
}
- if !prepareIncidentCreate(c, logger, &incData) {
+ if !prepareIncidentCreate(c, logger, &incData, publisher) {
logger.Warn("incident creation failed: validation or authorization error")
return
}
@@ -383,7 +387,7 @@ func PostIncidentHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
incData.System = &system
}
- result, err := routeIncidentCreation(c, dbInst, log, incData)
+ result, err := routeIncidentCreation(c, dbInst, log, incData, publisher)
if err != nil {
if errors.Is(err, apiErrors.ErrIncidentSystemCreationWrongType) {
logger.Warn("incident creation failed: invalid system incident type", zap.Error(err))
@@ -407,19 +411,19 @@ func PostIncidentHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
}
func routeIncidentCreation(
- c *gin.Context, dbInst *db.DB, log *zap.Logger, incData IncidentData,
+ c *gin.Context, dbInst *db.DB, log *zap.Logger, incData IncidentData, pub *notification.Publisher,
) ([]*ProcessComponentResp, error) {
if *incData.System {
log.Info("system incident detected, using system incident creation logic")
- return handleSystemIncidentCreation(dbInst, log, incData)
+ return handleSystemIncidentCreation(dbInst, log, incData, pub)
}
log.Info("regular incident detected, using regular incident creation logic")
userID := getUserIDFromContext(c)
- return handleRegularIncidentCreation(dbInst, log, incData, userID)
+ return handleRegularIncidentCreation(dbInst, log, incData, userID, pub)
}
func handleSystemIncidentCreation(
- dbInst *db.DB, log *zap.Logger, incData IncidentData,
+ dbInst *db.DB, log *zap.Logger, incData IncidentData, _ *notification.Publisher,
) ([]*ProcessComponentResp, error) {
if incData.Type != event.TypeIncident {
log.Info("system incident must be of type 'incident'")
@@ -648,7 +652,7 @@ func addComponentToSystemIncident(
Components: []db.Component{*comp},
}
- if err := createEvent(dbInst, log, &incIn, nil); err != nil {
+ if err := createEvent(dbInst, log, &incIn, nil, nil); err != nil {
return nil, err
}
@@ -737,7 +741,7 @@ func moveComponentFromToSystemIncidents(
}
func handleRegularIncidentCreation(
- dbInst *db.DB, log *zap.Logger, incData IncidentData, userID *string,
+ dbInst *db.DB, log *zap.Logger, incData IncidentData, userID *string, pub *notification.Publisher,
) ([]*ProcessComponentResp, error) {
components := make([]db.Component, len(incData.Components))
for i, comp := range incData.Components {
@@ -774,7 +778,7 @@ func handleRegularIncidentCreation(
log.Info("opened incidents and maintenances retrieved", zap.Any("openedIncidents", openedIncidents))
- if err = createEvent(dbInst, log, &incIn, userID); err != nil {
+ if err = createEvent(dbInst, log, &incIn, userID, pub); err != nil {
return nil, err
}
@@ -957,61 +961,135 @@ func validateEventCreationTimes(incData IncidentData) error {
return nil
}
-func createEvent(dbInst *db.DB, log *zap.Logger, inc *db.Incident, userID *string) error {
+func createEvent(dbInst *db.DB, log *zap.Logger, inc *db.Incident, userID *string, pub *notification.Publisher) error {
log.Info("start to save an event to the database")
- id, err := dbInst.SaveIncident(inc)
- if err != nil {
- return err
+
+ err := dbInst.WithTx(context.Background(), func(tx *gorm.DB) error {
+ id, err := dbInst.SaveIncidentTx(tx, inc)
+ if err != nil {
+ return err
+ }
+
+ inc.ID = id
+
+ log.Info("add initial status to the event", zap.Uint("eventID", inc.ID))
+ var statusText string
+ var status event.Status
+ timestamp := time.Now().UTC()
+ // Sometimes we have a gap between the start date and the current time.
+ // Example: the incident was created now, but we add an update with a detected status since 1-2 seconds.
+ // And on the FE it looks like the incident was created in the past.
+ // it doesn't affect planned events, like maintenance or info, because they have a start date in the future.
+ // However, if someone creates an incident with a start date in the past,
+ // we should set up the right timestamp for the status update.
+ if inc.StartDate.Before(timestamp) {
+ timestamp = *inc.StartDate
+ }
+
+ switch inc.Type {
+ case event.TypeInformation:
+ statusText = event.InfoPlannedStatusText()
+ status = event.InfoPlanned
+ case event.TypeMaintenance:
+ if inc.Status == event.MaintenancePendingReview {
+ statusText = event.MaintenancePendingReviewStatusText()
+ status = event.MaintenancePendingReview
+ } else {
+ statusText = event.MaintenancePlannedStatusText()
+ status = event.MaintenancePlanned
+ }
+ case event.TypeIncident:
+ statusText = event.IncidentDetectedStatusText()
+ status = event.IncidentDetected
+ }
+
+ inc.Statuses = append(inc.Statuses, db.IncidentStatus{
+ IncidentID: inc.ID,
+ Status: status,
+ Text: statusText,
+ Timestamp: timestamp,
+ CreatedBy: userID,
+ })
+ inc.Status = status
+
+ if err = dbInst.ModifyIncidentTx(tx, inc); err != nil {
+ return err
+ }
+
+ // A newly created maintenance has no previous status.
+ return publishMaintenanceChange(context.Background(), tx, pub, inc, "", userID)
+ })
+ if err == nil && inc.Type == event.TypeMaintenance {
+ pub.Notify() // wake the worker after the commit
+ }
+ return err
+}
+
+// optionalPublisher extracts the single optional publisher from a variadic arg.
+func optionalPublisher(pub []*notification.Publisher) *notification.Publisher {
+ if len(pub) > 0 {
+ return pub[0]
}
+ return nil
+}
- inc.ID = id
-
- log.Info("add initial status to the event", zap.Uint("eventID", inc.ID))
- var statusText string
- var status event.Status
- timestamp := time.Now().UTC()
- // Sometimes we have a gap between the start date and the current time.
- // Example: the incident was created now, but we add an update with a detected status since 1-2 seconds.
- // And on the FE it looks like the incident was created in the past.
- // it doesn't affect planned events, like maintenance or info, because they have a start date in the future.
- // However, if someone creates an incident with a start date in the past,
- // we should set up the right timestamp for the status update.
- if inc.StartDate.Before(timestamp) {
- timestamp = *inc.StartDate
- }
-
- switch inc.Type {
- case event.TypeInformation:
- statusText = event.InfoPlannedStatusText()
- status = event.InfoPlanned
- case event.TypeMaintenance:
- if inc.Status == event.MaintenancePendingReview {
- statusText = event.MaintenancePendingReviewStatusText()
- status = event.MaintenancePendingReview
- } else {
- statusText = event.MaintenancePlannedStatusText()
- status = event.MaintenancePlanned
- }
- case event.TypeIncident:
- statusText = event.IncidentDetectedStatusText()
- status = event.IncidentDetected
- }
-
- inc.Statuses = append(inc.Statuses, db.IncidentStatus{
- IncidentID: inc.ID,
- Status: status,
- Text: statusText,
- Timestamp: timestamp,
- CreatedBy: userID,
+// publishMaintenanceChange enqueues notification rows for a committed maintenance
+// change inside tx. It is a no-op for non-maintenance events or a disabled publisher.
+func publishMaintenanceChange(
+ ctx context.Context, tx *gorm.DB, pub *notification.Publisher,
+ inc *db.Incident, oldStatus event.Status, userID *string,
+) error {
+ if !pub.Enabled() || inc.Type != event.TypeMaintenance {
+ return nil
+ }
+ return pub.PublishTx(ctx, tx, notification.Change{
+ IncidentID: inc.ID,
+ Title: strDeref(inc.Text),
+ OldStatus: oldStatus,
+ NewStatus: inc.Status,
+ ContactEmail: strDeref(inc.ContactEmail),
+ Actor: strDeref(userID),
})
- inc.Status = status
+}
+
+// strDeref returns the pointed-to string, or "" for a nil pointer.
+func strDeref(s *string) string {
+ if s == nil {
+ return ""
+ }
+ return *s
+}
- err = dbInst.ModifyIncident(inc)
+// persistIncidentPatch writes the modification and its notification in one
+// transaction, mapping failures to HTTP responses. It returns false when the
+// caller should stop (an error response was already written).
+func persistIncidentPatch(
+ c *gin.Context, dbInst *db.DB, logger *zap.Logger, publisher *notification.Publisher,
+ storedIncident *db.Incident, oldStatus event.Status, userID *string,
+) bool {
+ err := dbInst.WithTx(c.Request.Context(), func(tx *gorm.DB) error {
+ if e := dbInst.ModifyIncidentTx(tx, storedIncident); e != nil {
+ return e
+ }
+ return publishMaintenanceChange(c.Request.Context(), tx, publisher, storedIncident, oldStatus, userID)
+ })
if err != nil {
- return err
+ if errors.Is(err, db.ErrVersionConflict) {
+ logger.Warn("incident patch failed: version conflict",
+ zap.Uint("event_id", storedIncident.ID))
+ apiErrors.RaiseConflictErr(c, apiErrors.ErrVersionConflict)
+ return false
+ }
+ logger.Error("incident patch failed: database error",
+ zap.Uint("event_id", storedIncident.ID), zap.Error(err))
+ apiErrors.RaiseInternalErr(c, err)
+ return false
}
- return nil
+ if storedIncident.Type == event.TypeMaintenance {
+ publisher.Notify() // wake the worker after the commit
+ }
+ return true
}
type PatchIncidentData struct {
@@ -1027,7 +1105,8 @@ type PatchIncidentData struct {
Version *int `json:"version"`
}
-func PatchIncidentHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
+func PatchIncidentHandler(dbInst *db.DB, logger *zap.Logger, pub ...*notification.Publisher) gin.HandlerFunc {
+ publisher := optionalPublisher(pub)
return func(c *gin.Context) {
logger.Debug("update incident")
@@ -1038,6 +1117,9 @@ func PatchIncidentHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
return
}
+ // Capture the pre-change status before any mutation for the notification summary.
+ oldStatus := storedIncident.Status
+
var incData PatchIncidentData
if err := c.ShouldBindBodyWithJSON(&incData); err != nil {
logger.Warn("incident patch failed: invalid request body", zap.Error(err))
@@ -1075,17 +1157,7 @@ func PatchIncidentHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
storedIncident.Version = incData.Version
}
- err := dbInst.ModifyIncident(storedIncident)
- if err != nil {
- if errors.Is(err, db.ErrVersionConflict) {
- logger.Warn("incident patch failed: version conflict",
- zap.Uint("event_id", storedIncident.ID))
- apiErrors.RaiseConflictErr(c, apiErrors.ErrVersionConflict)
- return
- }
- logger.Error("incident patch failed: database error",
- zap.Uint("event_id", storedIncident.ID), zap.Error(err))
- apiErrors.RaiseInternalErr(c, err)
+ if !persistIncidentPatch(c, dbInst, logger, publisher, storedIncident, oldStatus, userID) {
return
}
@@ -1096,7 +1168,7 @@ func PatchIncidentHandler(dbInst *db.DB, logger *zap.Logger) gin.HandlerFunc {
zap.Time("timestamp", incData.UpdateDate),
)
- if err = reopenIncident(c, dbInst, logger, storedIncident, incData.Status); err != nil {
+ if err := reopenIncident(c, dbInst, logger, storedIncident, incData.Status); err != nil {
return
}
@@ -1846,6 +1918,17 @@ func getUserIDFromContext(c *gin.Context) *string {
return nil
}
+// getUserEmailFromContext returns the verified address from the token, or nil when the
+// token carries no email claim (HMAC and service tokens).
+func getUserEmailFromContext(c *gin.Context) *string {
+ if email, exists := c.Get(UserEmailContextKey); exists {
+ if e, ok := email.(string); ok && e != "" {
+ return &e
+ }
+ }
+ return nil
+}
+
func resolveMaintenanceCreateStatus(role rbac.Role) (event.Status, error) {
switch {
case role >= rbac.Operator:
@@ -1913,7 +1996,9 @@ func allowMaintenancePatchAsCreator(
return false
}
-func prepareIncidentCreate(c *gin.Context, logger *zap.Logger, incData *IncidentData) bool {
+func prepareIncidentCreate(
+ c *gin.Context, logger *zap.Logger, incData *IncidentData, pub *notification.Publisher,
+) bool {
incData.StartDate = incData.StartDate.UTC()
if incData.EndDate != nil {
*incData.EndDate = incData.EndDate.UTC()
@@ -1930,6 +2015,10 @@ func prepareIncidentCreate(c *gin.Context, logger *zap.Logger, incData *Incident
return false
}
+ if !resolveContactEmail(c, incData, pub) {
+ return false
+ }
+
role, ok := getRoleFromContext(c, logger)
if !ok {
return false
@@ -1945,6 +2034,26 @@ func prepareIncidentCreate(c *gin.Context, logger *zap.Logger, incData *Incident
return true
}
+// resolveContactEmail settles the creator address: an explicitly supplied one wins and
+// is domain-checked, otherwise the verified address from the token is used. Tokens
+// without an email claim leave it empty, which only limits who gets notified.
+func resolveContactEmail(c *gin.Context, incData *IncidentData, pub *notification.Publisher) bool {
+ if incData.ContactEmail == "" {
+ if email := getUserEmailFromContext(c); email != nil {
+ incData.ContactEmail = *email
+ }
+
+ return true
+ }
+
+ if !pub.AllowsDomain(incData.ContactEmail) {
+ apiErrors.RaiseBadRequestErr(c, apiErrors.NewErrMaintenanceContactEmailDomain(pub.AllowedDomains()))
+ return false
+ }
+
+ return true
+}
+
func prepareIncidentPatch(
c *gin.Context, logger *zap.Logger, storedIncident *db.Incident, incData *PatchIncidentData,
) bool {
diff --git a/internal/api/v2/v2_rbac_test.go b/internal/api/v2/v2_rbac_test.go
index c15c8cf..513fb5c 100644
--- a/internal/api/v2/v2_rbac_test.go
+++ b/internal/api/v2/v2_rbac_test.go
@@ -560,7 +560,7 @@ func TestPrepareIncidentCreateNonMaintenance(t *testing.T) {
Type: event.TypeIncident,
}
- result := prepareIncidentCreate(c, logger, incData)
+ result := prepareIncidentCreate(c, logger, incData, nil)
assert.True(t, result, "non-maintenance should pass without RBAC check")
assert.Empty(t, incData.Status, "status should not be set for non-maintenance")
diff --git a/internal/api/v2/v2_validation_test.go b/internal/api/v2/v2_validation_test.go
index aa379c6..c8da3b5 100644
--- a/internal/api/v2/v2_validation_test.go
+++ b/internal/api/v2/v2_validation_test.go
@@ -34,15 +34,14 @@ func TestValidateMaintenanceCreation(t *testing.T) {
expectError: false,
},
{
- name: "Missing contact_email",
+ name: "Missing contact_email is allowed, resolved from the token later",
incData: IncidentData{
ContactEmail: "",
StartDate: futureTime,
EndDate: &laterTime,
Description: "Scheduled maintenance",
},
- expectError: true,
- errorMsg: "contact_email is required",
+ expectError: false,
},
{
name: "Invalid contact_email format",
diff --git a/internal/api/v2/validation.go b/internal/api/v2/validation.go
index 7e3c8e2..68d7d02 100644
--- a/internal/api/v2/validation.go
+++ b/internal/api/v2/validation.go
@@ -108,11 +108,11 @@ func parseAndSetTypes(queryTypes *string, params *db.IncidentsParams) error {
// validateMaintenanceCreation validates maintenance-specific fields at creation time.
// Note: EndDate nil check is handled by validateEventCreation before this function is called.
func validateMaintenanceCreation(incData IncidentData) error {
- if incData.ContactEmail == "" {
- return apiErrors.ErrMaintenanceContactEmailRequired
- }
- if _, err := mail.ParseAddress(incData.ContactEmail); err != nil {
- return apiErrors.ErrMaintenanceContactEmailInvalid
+ // Optional: prepareIncidentCreate falls back to the address from the token.
+ if incData.ContactEmail != "" {
+ if _, err := mail.ParseAddress(incData.ContactEmail); err != nil {
+ return apiErrors.ErrMaintenanceContactEmailInvalid
+ }
}
if incData.EndDate != nil && !incData.EndDate.After(incData.StartDate) {
diff --git a/internal/app/app.go b/internal/app/app.go
index e416eb8..d7f4c2a 100644
--- a/internal/app/app.go
+++ b/internal/app/app.go
@@ -2,19 +2,25 @@ package app
import (
"context"
+ "errors"
"fmt"
"net/http"
"time"
"go.uber.org/zap"
+ "github.com/prometheus/client_golang/prometheus"
+ "github.com/prometheus/client_golang/prometheus/promhttp"
"github.com/stackmon/otc-status-dashboard/internal/api"
"github.com/stackmon/otc-status-dashboard/internal/conf"
"github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/notification"
)
const (
readHeaderTimeout = 3 * time.Second
+ // notificationStaleThreshold flags a processing row as stuck in the /metrics gauge.
+ notificationStaleThreshold = 2 * time.Minute
)
type App struct {
@@ -28,6 +34,11 @@ type App struct {
DB *db.DB
// http server
srv *http.Server
+ // metrics server, listening on its own port (nil when notifications are disabled)
+ metricsSrv *http.Server
+ // notification delivery worker (nil when notifications are disabled)
+ worker *notification.Worker
+ workerCancel context.CancelFunc
}
func New(c *conf.Config, log *zap.Logger) (*App, error) {
@@ -41,22 +52,108 @@ func New(c *conf.Config, log *zap.Logger) (*App, error) {
return nil, err
}
+ // Build the delivery worker on the app's shared pool — it must NOT open its own.
+ // Budget PostgreSQL connections as max_open_conns_per_pod * number_of_pods.
+ worker, metricsHandler, err := buildWorker(c, log, dbNew, apiNew)
+ if err != nil {
+ return nil, err
+ }
+
s := &http.Server{
Addr: fmt.Sprintf(":%s", c.Port),
Handler: apiNew.Router(),
ReadHeaderTimeout: readHeaderTimeout,
}
- a := &App{api: apiNew, Log: log, conf: c, DB: dbNew, srv: s}
+ a := &App{api: apiNew, Log: log, conf: c, DB: dbNew, srv: s, worker: worker}
+
+ if metricsHandler != nil {
+ a.metricsSrv = &http.Server{
+ Addr: fmt.Sprintf(":%s", c.MetricsPort),
+ Handler: metricsHandler,
+ ReadHeaderTimeout: readHeaderTimeout,
+ }
+ }
return a, nil
}
+// buildWorker constructs the delivery worker and the handler for the private metrics
+// listener, and wires the API publisher's hot-path signal to the worker. Both results
+// are nil when notifications are disabled.
+func buildWorker(
+ c *conf.Config, log *zap.Logger, dbNew *db.DB, apiNew *api.API,
+) (*notification.Worker, http.Handler, error) {
+ ncfg, err := notification.ConfigFromConf(c)
+ if err != nil {
+ return nil, nil, err
+ }
+ if !ncfg.Enabled {
+ return nil, nil, nil
+ }
+
+ if err = dbNew.EnsureNotificationSchema(); err != nil {
+ return nil, nil, err
+ }
+
+ sender, err := notification.NewSMTPSender(ncfg)
+ if err != nil {
+ return nil, nil, fmt.Errorf("build smtp sender: %w", err)
+ }
+
+ // Dedicated registry so /metrics exposes just the notification signals.
+ reg := prometheus.NewRegistry()
+ metrics := notification.NewMetrics()
+ metrics.MustRegister(reg)
+ reg.MustRegister(notification.NewStatsCollector(dbNew, notificationStaleThreshold))
+
+ mux := http.NewServeMux()
+ mux.Handle("/metrics", promhttp.HandlerFor(reg, promhttp.HandlerOpts{}))
+
+ worker, err := notification.NewWorker(ncfg, dbNew, sender, log, metrics)
+ if err != nil {
+ return nil, nil, fmt.Errorf("build notification worker: %w", err)
+ }
+
+ apiNew.Publisher().SetNotify(worker.Notify)
+ return worker, mux, nil
+}
+
+// NotifyFunc returns the worker's wake-up callback, or nil when notifications are
+// disabled. Used to wire the checker's publisher to the same worker.
+func (a *App) NotifyFunc() func() {
+ if a.worker == nil {
+ return nil
+ }
+ return a.worker.Notify
+}
+
func (a *App) Run() error {
+ if a.worker != nil {
+ var ctx context.Context
+ ctx, a.workerCancel = context.WithCancel(context.Background())
+ go a.worker.Run(ctx)
+ }
+ if a.metricsSrv != nil {
+ go func() {
+ a.Log.Info("metrics server started", zap.String("addr", a.metricsSrv.Addr))
+ if err := a.metricsSrv.ListenAndServe(); err != nil && !errors.Is(err, http.ErrServerClosed) {
+ a.Log.Error("metrics server failed", zap.Error(err))
+ }
+ }()
+ }
return a.srv.ListenAndServe()
}
func (a *App) Shutdown(ctx context.Context) error {
+ if a.workerCancel != nil {
+ a.workerCancel()
+ }
+ if a.metricsSrv != nil {
+ if err := a.metricsSrv.Shutdown(ctx); err != nil {
+ a.Log.Error("metrics server shutdown", zap.Error(err))
+ }
+ }
// TODO: add a proper shutdown for a database
return a.srv.Shutdown(ctx)
}
diff --git a/internal/checker/checker.go b/internal/checker/checker.go
index bb8095e..6579f7a 100644
--- a/internal/checker/checker.go
+++ b/internal/checker/checker.go
@@ -8,13 +8,15 @@ import (
"github.com/stackmon/otc-status-dashboard/internal/conf"
"github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/notification"
)
const defaultPeriod = time.Minute * 2
type Checker struct {
- db *db.DB
- log *zap.Logger
+ db *db.DB
+ log *zap.Logger
+ notifier *notification.Publisher
// lastIDs are the earliest planned or in_progress maintenance/info events ID.
lastMntID uint
lastInfoID uint
@@ -25,7 +27,11 @@ func New(c *conf.Config, log *zap.Logger) (*Checker, error) {
if err != nil {
return nil, err
}
- return &Checker{db: dbNew, log: log}, nil
+ ncfg, err := notification.ConfigFromConf(c)
+ if err != nil {
+ return nil, err
+ }
+ return &Checker{db: dbNew, log: log, notifier: notification.NewPublisher(ncfg, dbNew)}, nil
}
func (ch *Checker) Check() {
@@ -73,3 +79,14 @@ func (ch *Checker) Shutdown(done chan struct{}) error {
close(done)
return ch.db.Close()
}
+
+// Close releases the checker's database pool without going through the Run loop.
+func (ch *Checker) Close() error {
+ return ch.db.Close()
+}
+
+// Publisher returns the checker's notification publisher so the delivery worker's
+// Notify can be wired in during app startup.
+func (ch *Checker) Publisher() *notification.Publisher {
+ return ch.notifier
+}
diff --git a/internal/checker/maintenance.go b/internal/checker/maintenance.go
index 0acb4a7..77d43c8 100644
--- a/internal/checker/maintenance.go
+++ b/internal/checker/maintenance.go
@@ -1,14 +1,17 @@
package checker
import (
+ "context"
"fmt"
"slices"
"time"
"go.uber.org/zap"
+ "gorm.io/gorm"
"github.com/stackmon/otc-status-dashboard/internal/db"
"github.com/stackmon/otc-status-dashboard/internal/event"
+ "github.com/stackmon/otc-status-dashboard/internal/notification"
)
type MntStatusHistory struct {
@@ -116,16 +119,41 @@ func (ch *Checker) processMaintenance(mn *db.Incident, activeMaintenances *[]uin
actualStatus := ch.evaluateAndFixMntStatus(mn)
if mn.Status != actualStatus {
+ oldStatus := mn.Status
mn.Status = actualStatus
- if modErr := ch.db.ModifyIncident(mn); modErr != nil {
- return fmt.Errorf("update maintenance %d: %w", mn.ID, modErr)
+ // The modify + enqueue share one transaction: on a version conflict the
+ // whole thing rolls back and no notification is published.
+ txErr := ch.db.WithTx(context.Background(), func(tx *gorm.DB) error {
+ if modErr := ch.db.ModifyIncidentTx(tx, mn); modErr != nil {
+ return modErr
+ }
+ return ch.notifier.PublishTx(context.Background(), tx, notification.Change{
+ IncidentID: mn.ID,
+ Title: strDeref(mn.Text),
+ OldStatus: oldStatus,
+ NewStatus: mn.Status,
+ ContactEmail: strDeref(mn.ContactEmail),
+ Actor: notification.ActorChecker,
+ })
+ })
+ if txErr != nil {
+ return fmt.Errorf("update maintenance %d: %w", mn.ID, txErr)
}
+ ch.notifier.Notify() // wake the worker after the commit
}
trackActiveMaintenance(actualStatus, mn.ID, activeMaintenances)
return nil
}
+// strDeref returns the pointed-to string, or "" for a nil pointer.
+func strDeref(s *string) string {
+ if s == nil {
+ return ""
+ }
+ return *s
+}
+
func (ch *Checker) evaluateAndFixMntStatus(mn *db.Incident) event.Status {
sHistory := calculateMntStatusHistory(mn)
actualStatus := calculateCurrentMntStatus(sHistory, mn)
diff --git a/internal/conf/conf.go b/internal/conf/conf.go
index d00065a..5c6c372 100644
--- a/internal/conf/conf.go
+++ b/internal/conf/conf.go
@@ -3,10 +3,12 @@ package conf
import (
"errors"
"fmt"
+ "net/mail"
"net/url"
"reflect"
"strconv"
"strings"
+ "time"
"github.com/joho/godotenv"
"github.com/kelseyhightower/envconfig"
@@ -20,29 +22,48 @@ const (
DefaultWebURL = "http://localhost:9000"
DefaultHostname = "localhost"
DefaultPort = "8000"
+ DefaultMetricsPort = "9090"
DefaultOpenAPISpecPath = "openapi.yaml"
// MinSecretKeyLength is the minimum required length for the HMAC secret key.
// HMAC-SHA256 requires at least 32 bytes for cryptographic strength.
MinSecretKeyLength = 32
+
+ // MaxPortNumber is the highest valid TCP port.
+ MaxPortNumber = 65535
+)
+
+// Notification delivery defaults, applied when notifications are enabled.
+const (
+ DefaultSMTPTimeout = "30s"
+ DefaultLeaseTimeout = "60s"
+ DefaultMaxAttempts = "5"
+ DefaultBackoffInterval = "5m"
)
type Config struct {
+ // Single-word fields below carry no envconfig tag on purpose: envconfig falls back
+ // to the bare tag name when the prefixed variable is unset, so a tag of "HOSTNAME"
+ // would inherit the container's $HOSTNAME. Field names yield the same SD_* keys.
+
// DB connection uri
// format is `postgresql://user:pass@host:port/db_name`
- DB string `envconfig:"DB"`
+ DB string
// Cache connection uri
// It can be redis format or internal
- Cache string `envconfig:"CACHE"`
+ Cache string
// Keycloak settings
Keycloak *Keycloak `envconfig:"KEYCLOAK"`
// Log level for verbosity
LogLevel string `envconfig:"LOG_LEVEL"`
// App port
- Port string `envconfig:"PORT"`
+ Port string
+ // MetricsPort serves /metrics on its own listener so the queue telemetry is not
+ // reachable from the public API port.
+ MetricsPort string `envconfig:"METRICS_PORT"`
// Hostname for the app, used to generate a callback URL for keycloak
// Example: https://api.example.com
- Hostname string `envconfig:"HOSTNAME"`
+ Hostname string
// Web URL for the app
// Example: https://web.example.com
WebURL string `envconfig:"WEB_URL"`
@@ -55,6 +76,49 @@ type Config struct {
OpenAPISpecPath string `envconfig:"OPENAPI_SPEC_PATH"`
// RBAC configuration
RBAC RBACConfig `envconfig:"RBAC"`
+ // SMTP transport settings for outgoing mail
+ SMTP SMTPConfig `envconfig:"SMTP"`
+ // Notifications feature settings
+ Notifications NotificationsConfig `envconfig:"NOTIFICATIONS"`
+}
+
+// SMTPConfig holds the direct OTC SMTP transport settings.
+//
+// No envconfig tags: a tag like "USER" makes envconfig fall back to the shell's $USER
+// when SD_SMTP_USER is unset. Field names yield the same keys without that fallback.
+type SMTPConfig struct {
+ Host string
+ Port string
+ From string
+ User string
+ Password string
+ TLS bool
+ // Timeout is a Go duration string (e.g. "30s") for the SMTP connect/send.
+ Timeout string
+}
+
+// NotificationsConfig holds the maintenance email notification settings.
+type NotificationsConfig struct {
+ // Enabled is the master on/off switch. Untagged for the same reason as SMTPConfig.
+ Enabled bool
+ // LeaseTimeout is a Go duration string; must exceed the SMTP timeout.
+ LeaseTimeout string `envconfig:"LEASE_TIMEOUT"`
+ // MaxAttempts is the finite retry limit before a row is marked failed.
+ MaxAttempts string `envconfig:"MAX_ATTEMPTS"`
+ // BackoffInterval is the base delay (Go duration string) for retry backoff.
+ BackoffInterval string `envconfig:"BACKOFF_INTERVAL"`
+ // SmodEmail is the fixed SMOD team review recipient.
+ SmodEmail string `envconfig:"SMOD_EMAIL"`
+ // EmailsOperators is the review recipient list for the Operator role.
+ EmailsOperators string `envconfig:"EMAILS_OPERATORS"`
+ // EmailsAdmins is the review recipient list for the Admin role.
+ EmailsAdmins string `envconfig:"EMAILS_ADMINS"`
+ // AllowedDomains restricts the user-supplied contact_email to these domains,
+ // comma-separated. Empty means any domain is accepted.
+ AllowedDomains string `envconfig:"ALLOWED_DOMAINS"`
+ // ExcludedEmails never receive notifications, comma-separated. Applied to every
+ // recipient so an exclusion cannot be bypassed via contact_email.
+ ExcludedEmails string `envconfig:"EXCLUDED_EMAILS"`
}
type RBACConfig struct {
@@ -82,6 +146,10 @@ func (c *Config) Validate() error {
return fmt.Errorf("wrong port for http server")
}
+ if err = c.validateMetricsPort(p); err != nil {
+ return err
+ }
+
if provErr := c.validateProviders(); provErr != nil {
return provErr
}
@@ -90,6 +158,149 @@ func (c *Config) Validate() error {
return rbacErr
}
+ if notifErr := c.validateNotifications(); notifErr != nil {
+ return notifErr
+ }
+
+ return nil
+}
+
+// validateMetricsPort keeps the metrics listener on its own port; sharing apiPort
+// would put the queue telemetry back on the public API. An empty value is left to
+// FillDefaults, which LoadConf runs before validating.
+func (c *Config) validateMetricsPort(apiPort int) error {
+ if c.MetricsPort == "" {
+ return nil
+ }
+
+ p, err := strconv.Atoi(c.MetricsPort)
+ if err != nil || p < 1024 || p > MaxPortNumber {
+ return fmt.Errorf("wrong SD_METRICS_PORT format, should be a number in range 1024:%d", MaxPortNumber)
+ }
+ if p == apiPort {
+ return fmt.Errorf("SD_METRICS_PORT must differ from SD_PORT")
+ }
+
+ return nil
+}
+
+// validateNotifications enforces SMTP and review-audience requirements when
+// notifications are enabled. When disabled, the feature stays inert and no
+// notification settings are required.
+func (c *Config) validateNotifications() error {
+ if !c.Notifications.Enabled {
+ return nil
+ }
+
+ if err := c.validateSMTP(); err != nil {
+ return err
+ }
+
+ if err := c.validateReviewAudience(); err != nil {
+ return err
+ }
+
+ smtpTimeout, err := time.ParseDuration(c.SMTP.Timeout)
+ if err != nil {
+ return fmt.Errorf("invalid SD_SMTP_TIMEOUT: %w", err)
+ }
+
+ leaseTimeout, err := time.ParseDuration(c.Notifications.LeaseTimeout)
+ if err != nil {
+ return fmt.Errorf("invalid SD_NOTIFICATIONS_LEASE_TIMEOUT: %w", err)
+ }
+
+ if leaseTimeout <= smtpTimeout {
+ return fmt.Errorf("SD_NOTIFICATIONS_LEASE_TIMEOUT (%s) must be greater than SD_SMTP_TIMEOUT (%s)",
+ leaseTimeout, smtpTimeout)
+ }
+
+ attempts, err := strconv.Atoi(c.Notifications.MaxAttempts)
+ if err != nil || attempts < 1 {
+ return fmt.Errorf("SD_NOTIFICATIONS_MAX_ATTEMPTS must be a positive integer")
+ }
+
+ if _, parseErr := time.ParseDuration(c.Notifications.BackoffInterval); parseErr != nil {
+ return fmt.Errorf("invalid SD_NOTIFICATIONS_BACKOFF_INTERVAL: %w", parseErr)
+ }
+
+ return nil
+}
+
+// validateSMTP checks the transport settings. The sender address is parsed here
+// because a malformed From is rejected by the relay on every single message.
+func (c *Config) validateSMTP() error {
+ if c.SMTP.Host == "" || c.SMTP.Port == "" || c.SMTP.From == "" {
+ return fmt.Errorf("notifications enabled: SD_SMTP_HOST, SD_SMTP_PORT and SD_SMTP_FROM are required")
+ }
+
+ port, err := strconv.Atoi(c.SMTP.Port)
+ if err != nil || port < 1 || port > MaxPortNumber {
+ return fmt.Errorf("SD_SMTP_PORT must be a number in range 1:%d", MaxPortNumber)
+ }
+
+ if _, err = mail.ParseAddress(c.SMTP.From); err != nil {
+ return fmt.Errorf("invalid SD_SMTP_FROM %q: %w", c.SMTP.From, err)
+ }
+
+ return nil
+}
+
+// validateReviewAudience requires at least one review address and rejects malformed
+// ones: unlike contact_email these come from the operator, so a typo would silently
+// break every review notification.
+func (c *Config) validateReviewAudience() error {
+ if c.Notifications.SmodEmail == "" &&
+ c.Notifications.EmailsOperators == "" &&
+ c.Notifications.EmailsAdmins == "" {
+ return fmt.Errorf("notifications enabled: at least one review address must be set " +
+ "(SD_NOTIFICATIONS_SMOD_EMAIL, SD_NOTIFICATIONS_EMAILS_OPERATORS or SD_NOTIFICATIONS_EMAILS_ADMINS)")
+ }
+
+ lists := map[string]string{
+ "SD_NOTIFICATIONS_SMOD_EMAIL": c.Notifications.SmodEmail,
+ "SD_NOTIFICATIONS_EMAILS_OPERATORS": c.Notifications.EmailsOperators,
+ "SD_NOTIFICATIONS_EMAILS_ADMINS": c.Notifications.EmailsAdmins,
+ "SD_NOTIFICATIONS_EXCLUDED_EMAILS": c.Notifications.ExcludedEmails,
+ }
+ for envName, raw := range lists {
+ if err := validateEmailList(envName, raw); err != nil {
+ return err
+ }
+ }
+
+ return validateDomainList("SD_NOTIFICATIONS_ALLOWED_DOMAINS", c.Notifications.AllowedDomains)
+}
+
+// validateDomainList checks a comma-separated domain list. Entries are bare domains,
+// so a stray "@" usually means a full address was pasted in by mistake.
+func validateDomainList(envName, raw string) error {
+ for _, part := range strings.Split(raw, ",") {
+ domain := strings.TrimSpace(part)
+ if domain == "" {
+ continue
+ }
+ if strings.ContainsAny(domain, "@ ") || !strings.Contains(domain, ".") {
+ return fmt.Errorf("%s contains an invalid domain %q, expected e.g. \"example.com\"", envName, domain)
+ }
+ }
+
+ return nil
+}
+
+// validateEmailList parses a comma-separated recipient list, mirroring how
+// notification.splitEmails will later read it.
+func validateEmailList(envName, raw string) error {
+ for _, part := range strings.Split(raw, ",") {
+ addr := strings.TrimSpace(part)
+ if addr == "" {
+ continue
+ }
+ if _, err := mail.ParseAddress(addr); err != nil {
+ return fmt.Errorf("%s contains an invalid address %q: %w", envName, addr, err)
+ }
+ }
+
return nil
}
@@ -127,6 +338,10 @@ func (c *Config) FillDefaults() {
c.Port = DefaultPort
}
+ if c.MetricsPort == "" {
+ c.MetricsPort = DefaultMetricsPort
+ }
+
if c.Hostname == "" {
c.Hostname = DefaultHostname
}
@@ -138,6 +353,21 @@ func (c *Config) FillDefaults() {
if c.OpenAPISpecPath == "" {
c.OpenAPISpecPath = DefaultOpenAPISpecPath
}
+
+ if c.Notifications.Enabled {
+ if c.SMTP.Timeout == "" {
+ c.SMTP.Timeout = DefaultSMTPTimeout
+ }
+ if c.Notifications.LeaseTimeout == "" {
+ c.Notifications.LeaseTimeout = DefaultLeaseTimeout
+ }
+ if c.Notifications.MaxAttempts == "" {
+ c.Notifications.MaxAttempts = DefaultMaxAttempts
+ }
+ if c.Notifications.BackoffInterval == "" {
+ c.Notifications.BackoffInterval = DefaultBackoffInterval
+ }
+ }
}
// LoadConf loads configuration from .env file and environment.
@@ -169,6 +399,16 @@ var ErrInvalidDataMerge = errors.New("could not merge config, the obj must be a
const envConfigTag = "envconfig"
+// envKeyPart mirrors envconfig's key derivation: the tag when present, the field
+// name otherwise. Untagged fields are how we avoid envconfig's bare-name fallback.
+func envKeyPart(field reflect.StructField) string {
+ if tag := field.Tag.Get(envConfigTag); tag != "" {
+ return tag
+ }
+
+ return field.Name
+}
+
// mergeConfigs allow to merge config params from env variables and .env file.
// It checks the Config struct and if the value is missing, it set up the value from .env file.
func mergeConfigs(env map[string]string, obj any, prefix string) error { //nolint:gocognit
@@ -196,8 +436,7 @@ func mergeConfigs(env map[string]string, obj any, prefix string) error { //nolin
// Handle pointer to struct (e.g., *Keycloak)
if value.Kind() == reflect.Ptr && value.Elem().Kind() == reflect.Struct {
- envValueTag := field.Tag.Get(envConfigTag)
- confPrefix := fmt.Sprintf("%s_%s", prefix, envValueTag)
+ confPrefix := fmt.Sprintf("%s_%s", prefix, envKeyPart(field))
err := mergeConfigs(env, value.Interface(), confPrefix)
if err != nil {
return err
@@ -209,8 +448,7 @@ func mergeConfigs(env map[string]string, obj any, prefix string) error { //nolin
// Handle embedded struct (e.g., RBACConfig)
// For struct values (not pointers), we need to pass a pointer
if value.Kind() == reflect.Struct {
- envValueTag := field.Tag.Get(envConfigTag)
- confPrefix := fmt.Sprintf("%s_%s", prefix, envValueTag)
+ confPrefix := fmt.Sprintf("%s_%s", prefix, envKeyPart(field))
err := mergeConfigs(env, value.Addr().Interface(), confPrefix)
if err != nil {
return err
@@ -220,8 +458,7 @@ func mergeConfigs(env map[string]string, obj any, prefix string) error { //nolin
}
if value.IsZero() && value.IsValid() && value.CanSet() {
- envValueTag := field.Tag.Get(envConfigTag)
- mapKey := strings.ToUpper(fmt.Sprintf("%s_%s", prefix, envValueTag))
+ mapKey := strings.ToUpper(fmt.Sprintf("%s_%s", prefix, envKeyPart(field)))
switch value.Kind() {
case reflect.String:
@@ -294,4 +531,14 @@ func (c *Config) Log(logger *zap.Logger) {
zap.String("client_secret", maskSecret(c.Keycloak.ClientSecret)),
)
}
+
+ logger.Info("Notifications configuration",
+ zap.Bool("enabled", c.Notifications.Enabled),
+ zap.String("smtp_host", c.SMTP.Host),
+ zap.String("smtp_port", c.SMTP.Port),
+ zap.String("smtp_from", c.SMTP.From),
+ zap.String("smtp_user", c.SMTP.User),
+ zap.String("smtp_password", maskSecret(c.SMTP.Password)),
+ zap.Bool("smtp_tls", c.SMTP.TLS),
+ )
}
diff --git a/internal/conf/conf_test.go b/internal/conf/conf_test.go
index 539c3cb..6124dd6 100644
--- a/internal/conf/conf_test.go
+++ b/internal/conf/conf_test.go
@@ -337,6 +337,246 @@ func TestMergeConfigs(t *testing.T) {
assert.Equal(t, "http://kc.local", c.Keycloak.URL)
assert.Equal(t, "test", c.Keycloak.Realm)
})
+
+ t.Run("merges untagged SMTP fields by field name", func(t *testing.T) {
+ c := &Config{Keycloak: &Keycloak{}}
+ env := map[string]string{
+ "SD_SMTP_HOST": "smtp.local",
+ "SD_SMTP_USER": "mailer",
+ "SD_SMTP_TIMEOUT": "15s",
+ }
+ err := mergeConfigs(env, c, "SD")
+ require.NoError(t, err)
+ assert.Equal(t, "smtp.local", c.SMTP.Host)
+ assert.Equal(t, "mailer", c.SMTP.User)
+ assert.Equal(t, "15s", c.SMTP.Timeout)
+ })
+}
+
+// TestLoadConf_IgnoresBareEnvNames guards against envconfig's fallback to the bare tag
+// name: a tag of "USER" would otherwise inherit the shell's $USER and enable SMTP AUTH
+// against a server that offers none, and "HOSTNAME" would adopt the container's name.
+func TestLoadConf_IgnoresBareEnvNames(t *testing.T) {
+ t.Setenv("USER", "shell-user")
+ t.Setenv("PASSWORD", "shell-password")
+ t.Setenv("HOSTNAME", "pod-7f9c8d4b6-xk2wl")
+ t.Setenv("PORT", "8080")
+ t.Setenv("DB", "postgresql://wrong/db")
+ t.Setenv("CACHE", "redis://wrong")
+ t.Setenv("SD_SECRET_KEY", "my-secret-key-that-is-32-chars!!")
+ t.Setenv("SD_RBAC_GROUPS_ADMINS", "sd_admins")
+ t.Setenv("SD_SMTP_HOST", "127.0.0.1")
+
+ c, err := LoadConf()
+ require.NoError(t, err)
+
+ assert.Empty(t, c.SMTP.User)
+ assert.Empty(t, c.SMTP.Password)
+ assert.Equal(t, "127.0.0.1", c.SMTP.Host)
+
+ assert.Equal(t, DefaultHostname, c.Hostname, "must not adopt the container hostname")
+ assert.Equal(t, DefaultPort, c.Port, "must not adopt a platform-injected PORT")
+ assert.Empty(t, c.DB)
+ assert.Empty(t, c.Cache)
+}
+
+func TestLoadConf_PrefixedNamesStillApply(t *testing.T) {
+ t.Setenv("HOSTNAME", "pod-7f9c8d4b6-xk2wl")
+ t.Setenv("SD_HOSTNAME", "https://api.example.com")
+ t.Setenv("SD_PORT", "9000")
+ t.Setenv("SD_DB", "postgresql://localhost/sd")
+ t.Setenv("SD_CACHE", "internal")
+ t.Setenv("SD_SECRET_KEY", "my-secret-key-that-is-32-chars!!")
+ t.Setenv("SD_RBAC_GROUPS_ADMINS", "sd_admins")
+
+ c, err := LoadConf()
+ require.NoError(t, err)
+
+ assert.Equal(t, "https://api.example.com", c.Hostname)
+ assert.Equal(t, "9000", c.Port)
+ assert.Equal(t, "postgresql://localhost/sd", c.DB)
+ assert.Equal(t, "internal", c.Cache)
+}
+
+func baseNotifConfig() Config {
+ return Config{
+ Port: "8000",
+ MetricsPort: DefaultMetricsPort,
+ SecretKeyV1: "my-secret-key-that-is-32-chars!!", // 32 chars
+ RBAC: RBACConfig{Admins: "sd_admins"},
+ }
+}
+
+func TestValidateNotifications(t *testing.T) {
+ tests := []struct {
+ name string
+ mutate func(c *Config)
+ expectErr bool
+ errSubstr string
+ }{
+ {
+ name: "disabled skips all notification checks",
+ mutate: func(c *Config) { c.Notifications.Enabled = false },
+ expectErr: false,
+ },
+ {
+ name: "enabled with full valid config passes",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{
+ Enabled: true,
+ LeaseTimeout: "60s",
+ MaxAttempts: "5",
+ BackoffInterval: "5m",
+ SmodEmail: "support@com.com",
+ }
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "587", From: "sd@com.com", Timeout: "30s"}
+ },
+ expectErr: false,
+ },
+ {
+ name: "enabled without SMTP host fails",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{Enabled: true, SmodEmail: "support@com.com"}
+ c.SMTP = SMTPConfig{Port: "587", From: "sd@com.com", Timeout: "30s"}
+ },
+ expectErr: true,
+ errSubstr: "SD_SMTP_HOST",
+ },
+ {
+ name: "enabled without any review address fails",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "5m"}
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "587", From: "sd@com.com", Timeout: "30s"}
+ },
+ expectErr: true,
+ errSubstr: "at least one review address",
+ },
+ {
+ name: "malformed smtp from fails",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "5m", SmodEmail: "support@com.com",
+ }
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "587", From: "not-an-address", Timeout: "30s"}
+ },
+ expectErr: true,
+ errSubstr: "SD_SMTP_FROM",
+ },
+ {
+ name: "smtp port out of range fails",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "5m", SmodEmail: "support@com.com",
+ }
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "70000", From: "sd@com.com", Timeout: "30s"}
+ },
+ expectErr: true,
+ errSubstr: "SD_SMTP_PORT",
+ },
+ {
+ name: "malformed review address fails",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "5m",
+ EmailsOperators: "ops@com.com, broken at com.com",
+ }
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "587", From: "sd@com.com", Timeout: "30s"}
+ },
+ expectErr: true,
+ errSubstr: "SD_NOTIFICATIONS_EMAILS_OPERATORS",
+ },
+ {
+ name: "multi-address review lists pass",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "5m",
+ SmodEmail: "support@com.com",
+ EmailsOperators: "ops1@com.com, ops2@com.com",
+ EmailsAdmins: "admin@com.com",
+ }
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "587", From: "sd@com.com", Timeout: "30s"}
+ },
+ expectErr: false,
+ },
+ {
+ name: "lease timeout not greater than smtp timeout fails",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{
+ Enabled: true, LeaseTimeout: "30s", MaxAttempts: "5", BackoffInterval: "5m", SmodEmail: "support@com.com",
+ }
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "587", From: "sd@com.com", Timeout: "30s"}
+ },
+ expectErr: true,
+ errSubstr: "must be greater than",
+ },
+ {
+ name: "invalid smtp timeout fails",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "5m", SmodEmail: "support@com.com",
+ }
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "587", From: "sd@com.com", Timeout: "notaduration"}
+ },
+ expectErr: true,
+ errSubstr: "SD_SMTP_TIMEOUT",
+ },
+ {
+ name: "non-positive max attempts fails",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "0", BackoffInterval: "5m", SmodEmail: "support@com.com",
+ }
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "587", From: "sd@com.com", Timeout: "30s"}
+ },
+ expectErr: true,
+ errSubstr: "SD_NOTIFICATIONS_MAX_ATTEMPTS",
+ },
+ {
+ name: "invalid backoff interval fails",
+ mutate: func(c *Config) {
+ c.Notifications = NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "bad", SmodEmail: "support@com.com",
+ }
+ c.SMTP = SMTPConfig{Host: "smtp.otc", Port: "587", From: "sd@com.com", Timeout: "30s"}
+ },
+ expectErr: true,
+ errSubstr: "SD_NOTIFICATIONS_BACKOFF_INTERVAL",
+ },
+ }
+
+ for _, tc := range tests {
+ t.Run(tc.name, func(t *testing.T) {
+ cfg := baseNotifConfig()
+ tc.mutate(&cfg)
+ err := cfg.Validate()
+ if tc.expectErr {
+ require.Error(t, err)
+ assert.Contains(t, err.Error(), tc.errSubstr)
+ } else {
+ require.NoError(t, err)
+ }
+ })
+ }
+}
+
+func TestFillDefaults_Notifications(t *testing.T) {
+ t.Run("fills notification timing defaults when enabled", func(t *testing.T) {
+ c := &Config{Notifications: NotificationsConfig{Enabled: true}}
+ c.FillDefaults()
+
+ assert.Equal(t, DefaultSMTPTimeout, c.SMTP.Timeout)
+ assert.Equal(t, DefaultLeaseTimeout, c.Notifications.LeaseTimeout)
+ assert.Equal(t, DefaultMaxAttempts, c.Notifications.MaxAttempts)
+ assert.Equal(t, DefaultBackoffInterval, c.Notifications.BackoffInterval)
+ })
+
+ t.Run("leaves notification timing empty when disabled", func(t *testing.T) {
+ c := &Config{Notifications: NotificationsConfig{Enabled: false}}
+ c.FillDefaults()
+
+ assert.Empty(t, c.SMTP.Timeout)
+ assert.Empty(t, c.Notifications.LeaseTimeout)
+ })
}
func TestConfig_Log(t *testing.T) {
diff --git a/internal/db/db.go b/internal/db/db.go
index 6356f12..5e0b93e 100644
--- a/internal/db/db.go
+++ b/internal/db/db.go
@@ -1,6 +1,7 @@
package db
import (
+ "context"
"errors"
"fmt"
"time"
@@ -266,17 +267,29 @@ func (db *DB) GetIncident(id int) (*Incident, error) {
return &inc, nil
}
-func (db *DB) SaveIncident(inc *Incident) (uint, error) {
- r := db.g.Create(inc)
+// WithTx runs fn inside a single transaction on the shared connection pool.
+// Callers use it to write a business change and enqueue its notification atomically.
+func (db *DB) WithTx(ctx context.Context, fn func(tx *gorm.DB) error) error {
+ return db.g.WithContext(ctx).Transaction(func(tx *gorm.DB) error {
+ return fn(tx)
+ })
+}
- if r.Error != nil {
- return 0, r.Error
+// SaveIncidentTx creates an incident using the provided transaction.
+func (db *DB) SaveIncidentTx(tx *gorm.DB, inc *Incident) (uint, error) {
+ if err := tx.Create(inc).Error; err != nil {
+ return 0, err
}
-
return inc.ID, nil
}
-func (db *DB) ModifyIncident(inc *Incident) error {
+func (db *DB) SaveIncident(inc *Incident) (uint, error) {
+ return db.SaveIncidentTx(db.g, inc)
+}
+
+// ModifyIncidentTx applies a modification (with maintenance optimistic locking and
+// new status inserts) using the provided transaction.
+func (db *DB) ModifyIncidentTx(tx *gorm.DB, inc *Incident) error {
if inc.Version == nil {
return errors.New("version is required for event modification")
}
@@ -285,36 +298,40 @@ func (db *DB) ModifyIncident(inc *Incident) error {
newVersion := expectedVersion + 1
inc.Version = &newVersion
- return db.g.Transaction(func(tx *gorm.DB) error {
- query := tx.Model(&Incident{}).Where("id = ?", inc.ID)
+ query := tx.Model(&Incident{}).Where("id = ?", inc.ID)
- if inc.Type == event.TypeMaintenance {
- query = query.Where("version = ?", expectedVersion)
- }
+ if inc.Type == event.TypeMaintenance {
+ query = query.Where("version = ?", expectedVersion)
+ }
- r := query.Omit("Statuses", "Components").Updates(inc)
+ r := query.Omit("Statuses", "Components").Updates(inc)
- if r.Error != nil {
- return r.Error
- }
+ if r.Error != nil {
+ return r.Error
+ }
- if inc.Type == event.TypeMaintenance && r.RowsAffected == 0 {
- return ErrVersionConflict
- }
+ if inc.Type == event.TypeMaintenance && r.RowsAffected == 0 {
+ return ErrVersionConflict
+ }
- for i := range inc.Statuses {
- if inc.Statuses[i].ID != 0 {
- continue
- }
- if inc.Statuses[i].IncidentID == 0 {
- inc.Statuses[i].IncidentID = inc.ID
- }
- if err := tx.Create(&inc.Statuses[i]).Error; err != nil {
- return err
- }
+ for i := range inc.Statuses {
+ if inc.Statuses[i].ID != 0 {
+ continue
}
+ if inc.Statuses[i].IncidentID == 0 {
+ inc.Statuses[i].IncidentID = inc.ID
+ }
+ if err := tx.Create(&inc.Statuses[i]).Error; err != nil {
+ return err
+ }
+ }
- return nil
+ return nil
+}
+
+func (db *DB) ModifyIncident(inc *Incident) error {
+ return db.g.Transaction(func(tx *gorm.DB) error {
+ return db.ModifyIncidentTx(tx, inc)
})
}
@@ -795,10 +812,12 @@ func (db *DB) GetEventUpdates(incidentID uint) ([]IncidentStatus, error) {
return updates, nil
}
-func (db *DB) ModifyEventUpdate(update IncidentStatus) (IncidentStatus, error) {
+// ModifyEventUpdateTx patches an event status update's text using the provided
+// transaction and returns the updated row.
+func (db *DB) ModifyEventUpdateTx(tx *gorm.DB, update IncidentStatus) (IncidentStatus, error) {
now := time.Now().UTC()
var updated IncidentStatus
- r := db.g.Model(&IncidentStatus{}).
+ r := tx.Model(&IncidentStatus{}).
Clauses(clause.Returning{}).
Where("id = ? AND incident_id = ?", update.ID, update.IncidentID).
Updates(map[string]interface{}{
@@ -816,3 +835,7 @@ func (db *DB) ModifyEventUpdate(update IncidentStatus) (IncidentStatus, error) {
return updated, nil
}
+
+func (db *DB) ModifyEventUpdate(update IncidentStatus) (IncidentStatus, error) {
+ return db.ModifyEventUpdateTx(db.g, update)
+}
diff --git a/internal/db/errors.go b/internal/db/errors.go
index abaa14f..ea3370d 100644
--- a/internal/db/errors.go
+++ b/internal/db/errors.go
@@ -8,3 +8,5 @@ var ErrDBIncidentDSNotExist = errors.New("incident does not exist")
var ErrDBEventUpdateDSNotExist = errors.New("update does not exist")
var ErrDBIncidentFilterActiveFalse = errors.New("filter for inactive incidents is restricted")
var ErrVersionConflict = errors.New("version conflict")
+var ErrNotificationSchemaMissing = errors.New(
+ "notification_outbox table is missing: apply the pending database migrations")
diff --git a/internal/db/models.go b/internal/db/models.go
index 8c1e84d..e557a5b 100644
--- a/internal/db/models.go
+++ b/internal/db/models.go
@@ -176,3 +176,26 @@ func (is *IncidentStatus) BeforeUpdate(_ *gorm.DB) error {
is.ModifiedAt = &now
return nil
}
+
+// NotificationOutbox stores one email task per recipient.
+type NotificationOutbox struct {
+ ID uint `json:"id" gorm:"primaryKey;autoIncrement:true"`
+ Kind string `json:"kind" gorm:"type:varchar(64);not null"`
+ IncidentID uint `json:"incident_id" gorm:"not null"`
+ Recipient string `json:"recipient" gorm:"type:varchar(255);not null"`
+ Payload map[string]any `json:"payload" gorm:"type:jsonb;not null;serializer:json"`
+ ChangeID string `json:"change_id" gorm:"type:uuid;not null"`
+ DedupKey string `json:"dedup_key" gorm:"column:dedup_key;type:varchar(255);not null;uniqueIndex"`
+ Status string `json:"status" gorm:"type:varchar(20);not null;default:pending"`
+ Attempts int `json:"attempts" gorm:"not null;default:0"`
+ NextAttemptAt *time.Time `json:"next_attempt_at" gorm:"type:timestamptz"`
+ LockedBy *string `json:"locked_by" gorm:"type:varchar(255)"`
+ LockedAt *time.Time `json:"locked_at" gorm:"type:timestamptz"`
+ LastError *string `json:"last_error" gorm:"type:text"`
+ CreatedAt time.Time `json:"created_at" gorm:"autoCreateTime"`
+ UpdatedAt time.Time `json:"updated_at" gorm:"autoUpdateTime"`
+}
+
+func (no *NotificationOutbox) TableName() string {
+ return "notification_outbox"
+}
diff --git a/internal/db/notification.go b/internal/db/notification.go
new file mode 100644
index 0000000..3ae3a32
--- /dev/null
+++ b/internal/db/notification.go
@@ -0,0 +1,237 @@
+package db
+
+import (
+ "context"
+ "errors"
+ "time"
+
+ "gorm.io/gorm"
+ "gorm.io/gorm/clause"
+)
+
+const (
+ NotificationStatusPending = "pending"
+ NotificationStatusProcessing = "processing"
+ NotificationStatusSent = "sent"
+ NotificationStatusFailed = "failed"
+
+ NotificationKindPendingReview = "pending_review"
+ NotificationKindReviewed = "reviewed"
+ NotificationKindStatusChanged = "status_changed"
+)
+
+var (
+ ErrNotificationDuplicate = errors.New("notification: duplicate outbox row")
+ ErrNotificationNotFound = errors.New("notification: not found")
+)
+
+// rowExists checks whether a row with the same dedup key already exists.
+func (db *DB) rowExists(tx *gorm.DB, dedupKey string) (bool, error) {
+ q := db.g
+ if tx != nil {
+ q = tx
+ }
+
+ var row NotificationOutbox
+ if err := q.Select("id").Where("dedup_key = ?", dedupKey).First(&row).Error; err != nil {
+ if errors.Is(err, gorm.ErrRecordNotFound) {
+ return false, nil
+ }
+ return false, err
+ }
+ return true, nil
+}
+
+// Enqueue inserts one outbox row for a single recipient.
+// The row must be written in the same DB transaction as the business change.
+func (db *DB) Enqueue(ctx context.Context, tx *gorm.DB, row NotificationOutbox) error {
+ if row.DedupKey == "" {
+ return errors.New("notification: dedup_key is required")
+ }
+
+ return db.execWithTx(ctx, tx, func(gtx *gorm.DB) error {
+ exists, err := db.rowExists(gtx, row.DedupKey)
+ if err != nil {
+ return err
+ }
+ if exists {
+ return ErrNotificationDuplicate
+ }
+ return gtx.Create(&row).Error
+ })
+}
+
+// ClaimPending claims a batch of due rows for processing.
+// It must use `FOR UPDATE SKIP LOCKED` semantics and mark rows as processing,
+// increment attempts, and store lease metadata.
+func (db *DB) ClaimPending(
+ ctx context.Context, tx *gorm.DB, limit int, leaseOwner string, _ time.Duration,
+) ([]NotificationOutbox, error) {
+ var rows []NotificationOutbox
+
+ return rows, db.execWithTx(ctx, tx, func(gtx *gorm.DB) error {
+ if err := gtx.Clauses(clause.Locking{Strength: "UPDATE", Options: "SKIP LOCKED"}).
+ Where("status = ?", NotificationStatusPending).
+ Where("next_attempt_at IS NULL OR next_attempt_at <= ?", time.Now().UTC()).
+ Order("id ASC").
+ Limit(limit).
+ Find(&rows).Error; err != nil {
+ return err
+ }
+ now := time.Now().UTC()
+ for i := range rows {
+ rows[i].Status = NotificationStatusProcessing
+ rows[i].Attempts++
+ rows[i].LockedBy = &leaseOwner
+ rows[i].LockedAt = &now
+ if err := gtx.Model(&rows[i]).Updates(map[string]any{
+ "status": rows[i].Status,
+ "attempts": rows[i].Attempts,
+ "locked_by": rows[i].LockedBy,
+ "locked_at": rows[i].LockedAt,
+ }).Error; err != nil {
+ return err
+ }
+ }
+ return nil
+ })
+}
+
+// MarkSent marks a row as sent and clears the active lease.
+func (db *DB) MarkSent(ctx context.Context, tx *gorm.DB, id uint) error {
+ return db.execWithTx(ctx, tx, func(gtx *gorm.DB) error {
+ res := gtx.Model(&NotificationOutbox{}).
+ Where("id = ?", id).Updates(map[string]any{
+ "status": NotificationStatusSent,
+ "locked_by": nil,
+ "locked_at": nil,
+ "last_error": nil,
+ })
+ if res.Error != nil {
+ return res.Error
+ }
+ if res.RowsAffected == 0 {
+ return ErrNotificationNotFound
+ }
+ return nil
+ })
+}
+
+// getRowByID loads a single notification row by primary key.
+func (db *DB) getRowByID(tx *gorm.DB, id uint) (*NotificationOutbox, error) {
+ q := db.g
+ if tx != nil {
+ q = tx
+ }
+
+ var row NotificationOutbox
+ if err := q.First(&row, id).Error; err != nil {
+ if errors.Is(err, gorm.ErrRecordNotFound) {
+ return nil, ErrNotificationNotFound
+ }
+ return nil, err
+ }
+ return &row, nil
+}
+
+// MarkFailed marks a row as failed or pending with retry metadata.
+// If retries remain, keep status='pending' and set next_attempt_at.
+// Otherwise set status='failed' and last_error.
+func (db *DB) MarkFailed(
+ ctx context.Context, tx *gorm.DB, id uint, errText string, maxAttempts int, backoff func(attempts int) time.Time,
+) error {
+ return db.execWithTx(ctx, tx, func(gtx *gorm.DB) error {
+ row, err := db.getRowByID(gtx, id)
+ if err != nil {
+ return err
+ }
+
+ updates := map[string]any{
+ "last_error": errText,
+ "locked_by": nil,
+ "locked_at": nil,
+ }
+
+ if row.Attempts >= maxAttempts {
+ updates["status"] = NotificationStatusFailed
+ updates["next_attempt_at"] = nil
+ } else {
+ updates["status"] = NotificationStatusPending
+ updates["next_attempt_at"] = backoff(row.Attempts)
+ }
+
+ return gtx.Model(&NotificationOutbox{}).Where("id = ?", id).Updates(updates).Error
+ })
+}
+
+// MarkFailedTerminal fails a row outright, ignoring the remaining attempts. Used for
+// rejections the server will repeat on every retry, such as an unknown recipient.
+func (db *DB) MarkFailedTerminal(ctx context.Context, tx *gorm.DB, id uint, errText string) error {
+ return db.execWithTx(ctx, tx, func(gtx *gorm.DB) error {
+ res := gtx.Model(&NotificationOutbox{}).
+ Where("id = ?", id).Updates(map[string]any{
+ "status": NotificationStatusFailed,
+ "last_error": errText,
+ "next_attempt_at": nil,
+ "locked_by": nil,
+ "locked_at": nil,
+ })
+ if res.Error != nil {
+ return res.Error
+ }
+ if res.RowsAffected == 0 {
+ return ErrNotificationNotFound
+ }
+ return nil
+ })
+}
+
+// RecoverStaleProcessing returns stale processing rows back to pending,
+// or marks them failed if they exhausted all attempts.
+func (db *DB) RecoverStaleProcessing(
+ ctx context.Context, tx *gorm.DB, leaseTimeout time.Duration, maxAttempts int,
+) ([]NotificationOutbox, error) {
+ var rows []NotificationOutbox
+
+ return rows, db.execWithTx(ctx, tx, func(gtx *gorm.DB) error {
+ cutoff := time.Now().UTC().Add(-leaseTimeout)
+ now := time.Now().UTC()
+ if err := gtx.Where("status = ?", NotificationStatusProcessing).
+ Where("locked_at < ?", cutoff).
+ Find(&rows).Error; err != nil {
+ return err
+ }
+ for i := range rows {
+ if rows[i].Attempts >= maxAttempts {
+ rows[i].Status = NotificationStatusFailed
+ rows[i].NextAttemptAt = nil
+ } else {
+ rows[i].Status = NotificationStatusPending
+ rows[i].NextAttemptAt = &now
+ }
+ rows[i].LockedBy = nil
+ rows[i].LockedAt = nil
+
+ if err := gtx.Model(&rows[i]).Updates(map[string]any{
+ "status": rows[i].Status,
+ "next_attempt_at": rows[i].NextAttemptAt,
+ "locked_by": rows[i].LockedBy,
+ "locked_at": rows[i].LockedAt,
+ }).Error; err != nil {
+ return err
+ }
+ }
+ return nil
+ })
+}
+
+// execWithTx runs the callback in a transaction if tx is nil; otherwise it uses the provided tx.
+func (db *DB) execWithTx(ctx context.Context, tx *gorm.DB, fn func(*gorm.DB) error) error {
+ if tx != nil {
+ return fn(tx)
+ }
+
+ return db.g.WithContext(ctx).Transaction(func(gtx *gorm.DB) error {
+ return fn(gtx)
+ })
+}
diff --git a/internal/db/notification_ops.go b/internal/db/notification_ops.go
new file mode 100644
index 0000000..cbae0c6
--- /dev/null
+++ b/internal/db/notification_ops.go
@@ -0,0 +1,112 @@
+package db
+
+import (
+ "context"
+ "time"
+)
+
+// NotificationStats is a snapshot of the outbox queue for the ops interface.
+type NotificationStats struct {
+ Pending int64 `json:"pending"`
+ Processing int64 `json:"processing"`
+ Sent int64 `json:"sent"`
+ Failed int64 `json:"failed"`
+ StaleProcessing int64 `json:"stale_processing"`
+ RetryBacklog int64 `json:"retry_backlog"`
+ OldestPendingAgeSeconds float64 `json:"oldest_pending_age_seconds"`
+}
+
+// GetNotificationStats returns queue depth and health counters in one query.
+// staleThreshold marks processing rows whose lease is older than it as stuck.
+func (db *DB) GetNotificationStats(ctx context.Context, staleThreshold time.Duration) (*NotificationStats, error) {
+ cutoff := time.Now().UTC().Add(-staleThreshold)
+
+ var stats NotificationStats
+ query := `
+SELECT
+ COALESCE(SUM(CASE WHEN status = 'pending' THEN 1 ELSE 0 END), 0) AS pending,
+ COALESCE(SUM(CASE WHEN status = 'processing' THEN 1 ELSE 0 END), 0) AS processing,
+ COALESCE(SUM(CASE WHEN status = 'sent' THEN 1 ELSE 0 END), 0) AS sent,
+ COALESCE(SUM(CASE WHEN status = 'failed' THEN 1 ELSE 0 END), 0) AS failed,
+ COALESCE(SUM(CASE WHEN status = 'processing' AND locked_at < ? THEN 1 ELSE 0 END), 0) AS stale_processing,
+ COALESCE(SUM(CASE WHEN status = 'pending' AND next_attempt_at > now() THEN 1 ELSE 0 END), 0) AS retry_backlog,
+ COALESCE(EXTRACT(EPOCH FROM now() -
+ MIN(CASE WHEN status IN ('pending', 'processing') THEN created_at END)), 0) AS oldest_pending_age_seconds
+FROM notification_outbox`
+
+ if err := db.g.WithContext(ctx).Raw(query, cutoff).Scan(&stats).Error; err != nil {
+ return nil, err
+ }
+ return &stats, nil
+}
+
+// ListNotificationsByStatus returns the most recently updated rows in the given status.
+// Rows stuck in pending with a rising attempt count are the usual symptom of a
+// misconfigured relay, so every status must be reachable, not just failed.
+func (db *DB) ListNotificationsByStatus(ctx context.Context, status string, limit int) ([]NotificationOutbox, error) {
+ var rows []NotificationOutbox
+ err := db.g.WithContext(ctx).
+ Where("status = ?", status).
+ Order("updated_at DESC").
+ Limit(limit).
+ Find(&rows).Error
+ if err != nil {
+ return nil, err
+ }
+ return rows, nil
+}
+
+// EnsureNotificationSchema reports whether the outbox table exists. Migrations are
+// applied out of band, so without this check a stale database would let the app start
+// and only fail on the first maintenance change.
+func (db *DB) EnsureNotificationSchema() error {
+ if !db.g.Migrator().HasTable(&NotificationOutbox{}) {
+ return ErrNotificationSchemaMissing
+ }
+
+ return nil
+}
+
+// RedriveFailed resets failed rows back to pending for another delivery cycle,
+// clearing attempts, error and lease. With no ids it re-drives every failed row.
+func (db *DB) RedriveFailed(ctx context.Context, ids ...uint) (int64, error) {
+ q := db.g.WithContext(ctx).Model(&NotificationOutbox{}).Where("status = ?", NotificationStatusFailed)
+ if len(ids) > 0 {
+ q = q.Where("id IN ?", ids)
+ }
+
+ res := q.Updates(map[string]any{
+ "status": NotificationStatusPending,
+ "attempts": 0,
+ "next_attempt_at": time.Now().UTC(),
+ "last_error": nil,
+ "locked_by": nil,
+ "locked_at": nil,
+ })
+ return res.RowsAffected, res.Error
+}
+
+// DeleteSentBefore removes delivered rows older than the cutoff in batches
+// (retention). Failed rows are kept for audit and re-drive.
+func (db *DB) DeleteSentBefore(ctx context.Context, before time.Time, batchSize int) (int64, error) {
+ var total int64
+ for {
+ if ctx.Err() != nil {
+ return total, ctx.Err()
+ }
+
+ res := db.g.WithContext(ctx).Exec(
+ `DELETE FROM notification_outbox WHERE id IN (
+ SELECT id FROM notification_outbox
+ WHERE status = ? AND updated_at < ?
+ ORDER BY id LIMIT ?)`,
+ NotificationStatusSent, before, batchSize)
+ if res.Error != nil {
+ return total, res.Error
+ }
+ total += res.RowsAffected
+ if res.RowsAffected < int64(batchSize) {
+ return total, nil
+ }
+ }
+}
diff --git a/internal/db/notification_test.go b/internal/db/notification_test.go
new file mode 100644
index 0000000..9e6ddf6
--- /dev/null
+++ b/internal/db/notification_test.go
@@ -0,0 +1,6 @@
+package db
+
+// Storage-layer behavior for notification_outbox (Enqueue, ClaimPending,
+// MarkSent, MarkFailed, RecoverStaleProcessing) is verified against a real
+// Postgres container in tests/notifications_test.go, because FOR UPDATE SKIP
+// LOCKED and transactional lease semantics cannot be exercised with sqlmock.
diff --git a/internal/notification/metrics.go b/internal/notification/metrics.go
new file mode 100644
index 0000000..a40f5fc
--- /dev/null
+++ b/internal/notification/metrics.go
@@ -0,0 +1,164 @@
+package notification
+
+import (
+ "context"
+ "time"
+
+ "github.com/prometheus/client_golang/prometheus"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+)
+
+const metricsNamespace = "notification"
+
+// collectTimeout bounds the scrape-time queue query: Prometheus keeps scraping on a
+// schedule, so an unbounded query against a stalled DB would pile up goroutines.
+const collectTimeout = 5 * time.Second
+
+// Metrics holds the worker-driven counters and histogram. Queue-depth gauges are
+// exposed separately by the DB-backed statsCollector (pulled on scrape).
+//
+// All record* methods are nil-safe so the worker can run without metrics (tests).
+type Metrics struct {
+ sent *prometheus.CounterVec // by kind
+ failed *prometheus.CounterVec // by kind
+ attempts prometheus.Counter
+ staleRecovered prometheus.Counter
+ duration prometheus.Histogram
+}
+
+// NewMetrics builds the notification delivery metrics.
+func NewMetrics() *Metrics {
+ return &Metrics{
+ sent: prometheus.NewCounterVec(prometheus.CounterOpts{
+ Namespace: metricsNamespace, Name: "sent_total",
+ Help: "Total notification emails accepted by the mail server, by kind.",
+ }, []string{"kind"}),
+ failed: prometheus.NewCounterVec(prometheus.CounterOpts{
+ Namespace: metricsNamespace, Name: "failed_total",
+ Help: "Total notification send failures (retryable and terminal), by kind.",
+ }, []string{"kind"}),
+ attempts: prometheus.NewCounter(prometheus.CounterOpts{
+ Namespace: metricsNamespace, Name: "attempts_total",
+ Help: "Total notification delivery attempts.",
+ }),
+ staleRecovered: prometheus.NewCounter(prometheus.CounterOpts{
+ Namespace: metricsNamespace, Name: "stale_recovered_total",
+ Help: "Total processing rows recovered after a lease timeout.",
+ }),
+ duration: prometheus.NewHistogram(prometheus.HistogramOpts{
+ Namespace: metricsNamespace, Name: "delivery_duration_seconds",
+ Help: "Time to render and send one notification.",
+ Buckets: prometheus.DefBuckets,
+ }),
+ }
+}
+
+// MustRegister registers the worker metrics on reg.
+func (m *Metrics) MustRegister(reg prometheus.Registerer) {
+ reg.MustRegister(m.sent, m.failed, m.attempts, m.staleRecovered, m.duration)
+}
+
+func (m *Metrics) recordSent(kind string) {
+ if m == nil {
+ return
+ }
+ m.sent.WithLabelValues(kind).Inc()
+}
+
+func (m *Metrics) recordFailed(kind string) {
+ if m == nil {
+ return
+ }
+ m.failed.WithLabelValues(kind).Inc()
+}
+
+func (m *Metrics) recordAttempt() {
+ if m == nil {
+ return
+ }
+ m.attempts.Inc()
+}
+
+func (m *Metrics) recordStaleRecovered(n int) {
+ if m == nil || n <= 0 {
+ return
+ }
+ m.staleRecovered.Add(float64(n))
+}
+
+func (m *Metrics) observeDuration(d time.Duration) {
+ if m == nil {
+ return
+ }
+ m.duration.Observe(d.Seconds())
+}
+
+// statsCollector emits queue-depth gauges by querying the outbox on each scrape,
+// so they always reflect current state without per-operation bookkeeping.
+type statsCollector struct {
+ db *db.DB
+ staleThreshold time.Duration
+
+ pending *prometheus.Desc
+ processing *prometheus.Desc
+ failed *prometheus.Desc
+ staleProcessing *prometheus.Desc
+ retryBacklog *prometheus.Desc
+ oldestAge *prometheus.Desc
+ errors prometheus.Counter
+}
+
+// NewStatsCollector builds the DB-backed queue-depth collector.
+func NewStatsCollector(database *db.DB, staleThreshold time.Duration) prometheus.Collector {
+ desc := func(name, help string) *prometheus.Desc {
+ return prometheus.NewDesc(metricsNamespace+"_"+name, help, nil, nil)
+ }
+ return &statsCollector{
+ db: database,
+ staleThreshold: staleThreshold,
+ pending: desc("outbox_pending", "Outbox rows waiting to be sent."),
+ processing: desc("outbox_processing", "Outbox rows currently being sent."),
+ failed: desc("outbox_failed", "Outbox rows in the terminal failed state."),
+ staleProcessing: desc("outbox_stale_processing", "Processing rows whose lease has expired."),
+ retryBacklog: desc("outbox_retry_backlog", "Pending rows waiting for a future retry."),
+ oldestAge: desc("outbox_oldest_pending_age_seconds", "Age of the oldest undelivered row."),
+ errors: prometheus.NewCounter(prometheus.CounterOpts{
+ Namespace: metricsNamespace, Name: "collector_errors_total",
+ Help: "Total failures to read outbox queue depth at scrape time.",
+ }),
+ }
+}
+
+func (c *statsCollector) Describe(ch chan<- *prometheus.Desc) {
+ ch <- c.pending
+ ch <- c.processing
+ ch <- c.failed
+ ch <- c.staleProcessing
+ ch <- c.retryBacklog
+ ch <- c.oldestAge
+ ch <- c.errors.Desc()
+}
+
+func (c *statsCollector) Collect(ch chan<- prometheus.Metric) {
+ ctx, cancel := context.WithTimeout(context.Background(), collectTimeout)
+ defer cancel()
+
+ stats, err := c.db.GetNotificationStats(ctx, c.staleThreshold)
+ if err != nil {
+ // Gauges are omitted this round; the counter keeps the failure visible.
+ c.errors.Inc()
+ ch <- c.errors
+ return
+ }
+ g := func(d *prometheus.Desc, v float64) {
+ ch <- prometheus.MustNewConstMetric(d, prometheus.GaugeValue, v)
+ }
+ g(c.pending, float64(stats.Pending))
+ g(c.processing, float64(stats.Processing))
+ g(c.failed, float64(stats.Failed))
+ g(c.staleProcessing, float64(stats.StaleProcessing))
+ g(c.retryBacklog, float64(stats.RetryBacklog))
+ g(c.oldestAge, stats.OldestPendingAgeSeconds)
+ ch <- c.errors
+}
diff --git a/internal/notification/notification.go b/internal/notification/notification.go
new file mode 100644
index 0000000..5299d72
--- /dev/null
+++ b/internal/notification/notification.go
@@ -0,0 +1,170 @@
+package notification
+
+// Package notification builds and delivers maintenance email notifications.
+//
+// It is the "notification core" from docs/notifications/architecture.md: it turns
+// a maintenance change into one outbox row per recipient (resolver), renders those
+// rows into emails (renderer) and sends them over SMTP (sender). Delivery timing,
+// claiming and retries live in the storage layer and the worker.
+
+import (
+ "fmt"
+ "math/rand/v2"
+ "strconv"
+ "strings"
+ "time"
+
+ "github.com/stackmon/otc-status-dashboard/internal/conf"
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/event"
+)
+
+// ActorChecker is the actor recorded for automatic checker transitions.
+const ActorChecker = "checker"
+
+// maxBackoff caps the exponential retry delay (architecture §5).
+const maxBackoff = 2 * time.Hour
+
+// backoffJitter spreads retries by up to ±20%. Rows usually fail together (one relay
+// outage), so without it every retry would hit the recovering server at once.
+const backoffJitter = 0.2
+
+// Config is the parsed, ready-to-use notification configuration.
+// It is derived from conf.Config once at startup so the hot path never re-parses
+// durations, integers or recipient lists.
+type Config struct {
+ Enabled bool
+
+ From string
+ Host string
+ Port int
+ User string
+ Password string
+ TLS bool
+ Timeout time.Duration
+
+ LeaseTimeout time.Duration
+ MaxAttempts int
+ BackoffBase time.Duration
+
+ ReviewSMOD string
+ ReviewOperators []string
+ ReviewAdmins []string
+
+ // AllowedDomains limits the user-supplied contact_email. Empty allows any domain.
+ AllowedDomains []string
+ // ExcludedEmails are dropped from every recipient list.
+ ExcludedEmails []string
+
+ // BaseURL is the web origin used to build maintenance deep links.
+ BaseURL string
+}
+
+// ConfigFromConf parses and validates the notification settings from conf.Config.
+// conf.Validate has already guaranteed the raw values are well-formed when enabled,
+// so parsing here cannot fail for an enabled feature; errors are still surfaced.
+func ConfigFromConf(c *conf.Config) (Config, error) {
+ out := Config{
+ Enabled: c.Notifications.Enabled,
+ From: c.SMTP.From,
+ Host: c.SMTP.Host,
+ User: c.SMTP.User,
+ Password: c.SMTP.Password,
+ TLS: c.SMTP.TLS,
+ ReviewSMOD: strings.TrimSpace(c.Notifications.SmodEmail),
+ ReviewOperators: splitEmails(c.Notifications.EmailsOperators),
+ ReviewAdmins: splitEmails(c.Notifications.EmailsAdmins),
+ AllowedDomains: splitEmails(c.Notifications.AllowedDomains),
+ ExcludedEmails: splitEmails(c.Notifications.ExcludedEmails),
+ BaseURL: strings.TrimRight(c.WebURL, "/"),
+ }
+
+ if !c.Notifications.Enabled {
+ return out, nil
+ }
+
+ port, err := strconv.Atoi(c.SMTP.Port)
+ if err != nil {
+ return Config{}, fmt.Errorf("invalid SD_SMTP_PORT: %w", err)
+ }
+ out.Port = port
+
+ if out.Timeout, err = time.ParseDuration(c.SMTP.Timeout); err != nil {
+ return Config{}, fmt.Errorf("invalid SD_SMTP_TIMEOUT: %w", err)
+ }
+ if out.LeaseTimeout, err = time.ParseDuration(c.Notifications.LeaseTimeout); err != nil {
+ return Config{}, fmt.Errorf("invalid SD_NOTIFICATIONS_LEASE_TIMEOUT: %w", err)
+ }
+ if out.BackoffBase, err = time.ParseDuration(c.Notifications.BackoffInterval); err != nil {
+ return Config{}, fmt.Errorf("invalid SD_NOTIFICATIONS_BACKOFF_INTERVAL: %w", err)
+ }
+ if out.MaxAttempts, err = strconv.Atoi(c.Notifications.MaxAttempts); err != nil {
+ return Config{}, fmt.Errorf("invalid SD_NOTIFICATIONS_MAX_ATTEMPTS: %w", err)
+ }
+
+ return out, nil
+}
+
+// KindForStatus maps a resulting maintenance status to a notification kind
+// (architecture §1 recipient table).
+func KindForStatus(status event.Status) string {
+ switch status {
+ case event.MaintenancePendingReview:
+ return db.NotificationKindPendingReview
+ case event.MaintenanceReviewed:
+ return db.NotificationKindReviewed
+ default:
+ return db.NotificationKindStatusChanged
+ }
+}
+
+// isReviewStatus reports whether the status still needs a human decision and thus
+// notifies the review audience in addition to the creator.
+func isReviewStatus(status event.Status) bool {
+ return status == event.MaintenancePendingReview || status == event.MaintenanceReviewed
+}
+
+// Backoff returns a retry-time function for db.MarkFailed: attempt n becomes
+// eligible again after base*2^(n-1), capped at maxBackoff and spread by jitter
+// (architecture §5).
+func Backoff(base time.Duration) func(attempts int) time.Time {
+ return func(attempts int) time.Time {
+ delay := base
+ for i := 1; i < attempts; i++ {
+ delay *= 2
+ if delay >= maxBackoff {
+ delay = maxBackoff
+ break
+ }
+ }
+ if delay > maxBackoff {
+ delay = maxBackoff
+ }
+
+ return time.Now().UTC().Add(withJitter(delay))
+ }
+}
+
+// withJitter shifts d by a random factor within ±backoffJitter.
+func withJitter(d time.Duration) time.Duration {
+ spread := (rand.Float64()*2 - 1) * backoffJitter //nolint:gosec // scheduling spread, not security
+
+ return time.Duration(float64(d) * (1 + spread))
+}
+
+// splitEmails parses a comma-separated recipient list into normalized addresses.
+func splitEmails(raw string) []string {
+ parts := strings.Split(raw, ",")
+ out := make([]string, 0, len(parts))
+ for _, p := range parts {
+ if e := normalizeEmail(p); e != "" {
+ out = append(out, e)
+ }
+ }
+ return out
+}
+
+// normalizeEmail trims surrounding space and lowercases an address for dedup.
+func normalizeEmail(e string) string {
+ return strings.ToLower(strings.TrimSpace(e))
+}
diff --git a/internal/notification/notification_test.go b/internal/notification/notification_test.go
new file mode 100644
index 0000000..a3b87d0
--- /dev/null
+++ b/internal/notification/notification_test.go
@@ -0,0 +1,95 @@
+package notification
+
+import (
+ "testing"
+ "time"
+
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+
+ "github.com/stackmon/otc-status-dashboard/internal/conf"
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/event"
+)
+
+func TestKindForStatus(t *testing.T) {
+ //nolint:exhaustive // duplicate string constants for Info/Maintenance
+ cases := map[event.Status]string{
+ event.MaintenancePendingReview: db.NotificationKindPendingReview,
+ event.MaintenanceReviewed: db.NotificationKindReviewed,
+ event.MaintenancePlanned: db.NotificationKindStatusChanged,
+ event.MaintenanceInProgress: db.NotificationKindStatusChanged,
+ event.MaintenanceCompleted: db.NotificationKindStatusChanged,
+ event.MaintenanceCancelled: db.NotificationKindStatusChanged,
+ }
+ for status, want := range cases {
+ assert.Equal(t, want, KindForStatus(status), "status %s", status)
+ }
+}
+
+func TestBackoff_ProgressionAndCap(t *testing.T) {
+ base := 5 * time.Minute
+ fn := Backoff(base)
+
+ // Delays are jittered by ±backoffJitter, so assert the window, not an exact point.
+ assertWithinJitter := func(attempts int, want time.Duration) {
+ t.Helper()
+ before := time.Now().UTC()
+ got := fn(attempts).Sub(before)
+ tolerance := time.Duration(float64(want)*backoffJitter) + time.Second
+ assert.InDeltaf(t, float64(want), float64(got), float64(tolerance),
+ "attempt %d: got %s, want %s ±%s", attempts, got, want, tolerance)
+ }
+
+ assertWithinJitter(1, 5*time.Minute)
+ assertWithinJitter(2, 10*time.Minute)
+ assertWithinJitter(3, 20*time.Minute)
+
+ // Large attempt count is capped at maxBackoff (2h).
+ assertWithinJitter(20, maxBackoff)
+}
+
+func TestBackoff_JitterSpreadsRetries(t *testing.T) {
+ fn := Backoff(5 * time.Minute)
+
+ seen := make(map[time.Time]struct{})
+ for range 20 {
+ seen[fn(1)] = struct{}{}
+ }
+
+ // Without jitter every caller would queue the retry at the same instant.
+ assert.Greater(t, len(seen), 1, "jitter must spread simultaneous failures")
+}
+
+func TestConfigFromConf_Disabled(t *testing.T) {
+ c := &conf.Config{}
+ cfg, err := ConfigFromConf(c)
+ require.NoError(t, err)
+ assert.False(t, cfg.Enabled)
+}
+
+func TestConfigFromConf_ParsesEnabled(t *testing.T) {
+ c := &conf.Config{
+ WebURL: "https://status.example.com/",
+ SMTP: conf.SMTPConfig{
+ Host: "smtp.otc", Port: "587", From: "sd@com.com",
+ User: "u", Password: "p", TLS: true, Timeout: "30s",
+ },
+ Notifications: conf.NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "5m",
+ SmodEmail: "support@com.com", EmailsOperators: "ops1@com.com, ops2@com.com",
+ EmailsAdmins: "admin@com.com",
+ },
+ }
+
+ cfg, err := ConfigFromConf(c)
+ require.NoError(t, err)
+ assert.True(t, cfg.Enabled)
+ assert.Equal(t, 587, cfg.Port)
+ assert.Equal(t, 30*time.Second, cfg.Timeout)
+ assert.Equal(t, 60*time.Second, cfg.LeaseTimeout)
+ assert.Equal(t, 5, cfg.MaxAttempts)
+ assert.Equal(t, 5*time.Minute, cfg.BackoffBase)
+ assert.Equal(t, []string{"ops1@com.com", "ops2@com.com"}, cfg.ReviewOperators)
+ assert.Equal(t, "https://status.example.com", cfg.BaseURL, "trailing slash trimmed")
+}
diff --git a/internal/notification/publisher.go b/internal/notification/publisher.go
new file mode 100644
index 0000000..04fe88b
--- /dev/null
+++ b/internal/notification/publisher.go
@@ -0,0 +1,98 @@
+package notification
+
+import (
+ "context"
+ "slices"
+ "strings"
+
+ "gorm.io/gorm"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+)
+
+// Publisher records notification intent: it turns a maintenance Change into outbox
+// rows and enqueues them within the caller's transaction, so the email tasks commit
+// together with the business change (architecture §3).
+type Publisher struct {
+ enabled bool
+ resolver *Resolver
+ db *db.DB
+ allowedDomains []string
+ // notify wakes the delivery worker after a change commits (hot path). Optional.
+ notify func()
+}
+
+// NewPublisher builds a Publisher from the parsed config. When cfg.Enabled is false
+// the publisher is inert and PublishTx is a no-op.
+func NewPublisher(cfg Config, database *db.DB) *Publisher {
+ return &Publisher{
+ enabled: cfg.Enabled,
+ resolver: NewResolver(cfg),
+ db: database,
+ allowedDomains: cfg.AllowedDomains,
+ }
+}
+
+// AllowsDomain reports whether a user-supplied address may receive notifications.
+// It permits everything when the feature is off or no allow-list is configured, so
+// existing installations keep working unchanged.
+func (p *Publisher) AllowsDomain(email string) bool {
+ if p == nil || !p.enabled || len(p.allowedDomains) == 0 {
+ return true
+ }
+
+ at := strings.LastIndex(email, "@")
+ if at < 0 {
+ return false
+ }
+ domain := normalizeEmail(email[at+1:])
+
+ return slices.Contains(p.allowedDomains, domain)
+}
+
+// AllowedDomains lists the configured domains, for error messages.
+func (p *Publisher) AllowedDomains() []string {
+ if p == nil {
+ return nil
+ }
+
+ return p.allowedDomains
+}
+
+// SetNotify wires the post-commit wake-up callback (typically Worker.Notify).
+func (p *Publisher) SetNotify(fn func()) {
+ if p != nil {
+ p.notify = fn
+ }
+}
+
+// Notify signals the delivery worker that new rows may be due. Call it after the
+// business transaction commits. Nil-safe and a no-op when disabled or unwired.
+func (p *Publisher) Notify() {
+ if p == nil || !p.enabled || p.notify == nil {
+ return
+ }
+ p.notify()
+}
+
+// Enabled reports whether notifications should be published. It is nil-safe so
+// handlers can hold a nil *Publisher when the feature is off.
+func (p *Publisher) Enabled() bool {
+ return p != nil && p.enabled
+}
+
+// PublishTx enqueues one outbox row per recipient for the change, using tx so the
+// rows share the business transaction. It is a no-op when disabled or when the
+// change resolves to no recipients.
+func (p *Publisher) PublishTx(ctx context.Context, tx *gorm.DB, ch Change) error {
+ if !p.Enabled() {
+ return nil
+ }
+ rows := p.resolver.BuildRows(ch)
+ for i := range rows {
+ if err := p.db.Enqueue(ctx, tx, rows[i]); err != nil {
+ return err
+ }
+ }
+ return nil
+}
diff --git a/internal/notification/publisher_test.go b/internal/notification/publisher_test.go
new file mode 100644
index 0000000..4265d94
--- /dev/null
+++ b/internal/notification/publisher_test.go
@@ -0,0 +1,46 @@
+package notification
+
+import (
+ "testing"
+
+ "github.com/stretchr/testify/assert"
+)
+
+func TestPublisher_AllowsDomain(t *testing.T) {
+ enabled := NewPublisher(Config{
+ Enabled: true,
+ AllowedDomains: []string{"example.com", "t-systems.com"},
+ }, nil)
+
+ t.Run("allowed domain passes", func(t *testing.T) {
+ assert.True(t, enabled.AllowsDomain("user@example.com"))
+ assert.True(t, enabled.AllowsDomain("User@T-Systems.com"), "case-insensitive")
+ })
+
+ t.Run("foreign domain is rejected", func(t *testing.T) {
+ assert.False(t, enabled.AllowsDomain("user@gmail.com"))
+ })
+
+ t.Run("subdomain is not a match", func(t *testing.T) {
+ assert.False(t, enabled.AllowsDomain("user@mail.example.com"))
+ })
+
+ t.Run("malformed address is rejected", func(t *testing.T) {
+ assert.False(t, enabled.AllowsDomain("not-an-email"))
+ })
+
+ t.Run("empty allow-list permits any domain", func(t *testing.T) {
+ p := NewPublisher(Config{Enabled: true}, nil)
+ assert.True(t, p.AllowsDomain("user@anywhere.org"))
+ })
+
+ t.Run("disabled feature permits any domain", func(t *testing.T) {
+ p := NewPublisher(Config{AllowedDomains: []string{"example.com"}}, nil)
+ assert.True(t, p.AllowsDomain("user@gmail.com"))
+ })
+
+ t.Run("nil publisher permits any domain", func(t *testing.T) {
+ var p *Publisher
+ assert.True(t, p.AllowsDomain("user@gmail.com"))
+ })
+}
diff --git a/internal/notification/renderer.go b/internal/notification/renderer.go
new file mode 100644
index 0000000..bbdd86c
--- /dev/null
+++ b/internal/notification/renderer.go
@@ -0,0 +1,114 @@
+package notification
+
+import (
+ "bytes"
+ "embed"
+ "fmt"
+ "strings"
+ "text/template"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/event"
+)
+
+//go:embed templates/subject.tmpl templates/body.tmpl
+var templateFS embed.FS
+
+// Email is a rendered message ready to send.
+type Email struct {
+ Subject string
+ Body string
+}
+
+// templateData is the strongly-typed view a template renders against, extracted
+// from the outbox row's string payload.
+type templateData struct {
+ IncidentID string
+ Title string
+ OldStatus string
+ NewStatus string
+ Actor string
+ ChangedAt string
+ Link string
+ // State names the subject line state, and Headline the opening sentence. Both are
+ // derived here so the templates stay free of status logic.
+ State string
+ Headline string
+}
+
+// Renderer turns an outbox row into an Email using the embedded templates.
+type Renderer struct {
+ subject *template.Template
+ body *template.Template
+}
+
+// NewRenderer parses the embedded templates once. It fails fast on invalid
+// templates so a bad template never reaches the delivery worker.
+func NewRenderer() (*Renderer, error) {
+ subject, err := template.New("subject.tmpl").ParseFS(templateFS, "templates/subject.tmpl")
+ if err != nil {
+ return nil, fmt.Errorf("parse subject template: %w", err)
+ }
+ body, err := template.New("body.tmpl").ParseFS(templateFS, "templates/body.tmpl")
+ if err != nil {
+ return nil, fmt.Errorf("parse body template: %w", err)
+ }
+ return &Renderer{subject: subject, body: body}, nil
+}
+
+// Render produces the subject and body for one outbox row from its payload.
+func (r *Renderer) Render(row db.NotificationOutbox) (Email, error) {
+ oldStatus := payloadString(row.Payload, "old_status")
+ newStatus := payloadString(row.Payload, "new_status")
+ state, headline := describe(oldStatus, newStatus)
+
+ data := templateData{
+ IncidentID: payloadString(row.Payload, "incident_id"),
+ Title: payloadString(row.Payload, "title"),
+ OldStatus: oldStatus,
+ NewStatus: newStatus,
+ Actor: payloadString(row.Payload, "actor"),
+ ChangedAt: payloadString(row.Payload, "changed_at"),
+ Link: payloadString(row.Payload, "link"),
+ State: state,
+ Headline: headline,
+ }
+
+ var subject bytes.Buffer
+ if err := r.subject.Execute(&subject, data); err != nil {
+ return Email{}, fmt.Errorf("render subject: %w", err)
+ }
+ var body bytes.Buffer
+ if err := r.body.Execute(&body, data); err != nil {
+ return Email{}, fmt.Errorf("render body: %w", err)
+ }
+
+ return Email{
+ Subject: strings.TrimSpace(subject.String()),
+ Body: body.String(),
+ }, nil
+}
+
+// payloadString reads a string field from the JSONB payload, tolerating a missing
+// key or a non-string value (returns "").
+func payloadString(payload map[string]any, key string) string {
+ if v, ok := payload[key].(string); ok {
+ return v
+ }
+ return ""
+}
+
+// describe words the notification: the subject-line state and the opening sentence.
+// An empty old status means the maintenance was just created, which must not read as a
+// status change.
+func describe(oldStatus, newStatus string) (string, string) {
+ if oldStatus != "" {
+ return newStatus, "changed status"
+ }
+
+ if event.Status(newStatus) == event.MaintenancePendingReview {
+ return "awaiting review", "has been submitted for review"
+ }
+
+ return "scheduled", "has been scheduled"
+}
diff --git a/internal/notification/renderer_test.go b/internal/notification/renderer_test.go
new file mode 100644
index 0000000..bd91751
--- /dev/null
+++ b/internal/notification/renderer_test.go
@@ -0,0 +1,97 @@
+package notification
+
+import (
+ "testing"
+
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+)
+
+func TestRender_SubjectAndBodyFromPayload(t *testing.T) {
+ r, err := NewRenderer()
+ require.NoError(t, err)
+
+ row := db.NotificationOutbox{
+ Payload: map[string]any{
+ "incident_id": "42",
+ "title": "DB upgrade",
+ "old_status": "pending_review",
+ "new_status": "reviewed",
+ "actor": "admin-user",
+ "changed_at": "2026-08-26T10:00:00Z",
+ "link": "https://status.example.com/incidents/42",
+ },
+ }
+
+ email, err := r.Render(row)
+ require.NoError(t, err)
+
+ assert.Equal(t, "[Maintenance] DB upgrade — reviewed", email.Subject)
+ assert.Contains(t, email.Body, "event #42")
+ assert.Contains(t, email.Body, "pending_review -> reviewed")
+ assert.Contains(t, email.Body, "admin-user")
+ assert.Contains(t, email.Body, "https://status.example.com/incidents/42")
+}
+
+func TestRender_OmitsArrowWhenNoOldStatus(t *testing.T) {
+ r, err := NewRenderer()
+ require.NoError(t, err)
+
+ row := db.NotificationOutbox{
+ Payload: map[string]any{
+ "title": "New maintenance",
+ "new_status": "pending_review",
+ },
+ }
+
+ email, err := r.Render(row)
+ require.NoError(t, err)
+ assert.Contains(t, email.Body, "Status: pending_review")
+ assert.NotContains(t, email.Body, "->", "no arrow without old status")
+}
+
+func TestRender_CreationReadsAsScheduledNotAsChange(t *testing.T) {
+ r, err := NewRenderer()
+ require.NoError(t, err)
+
+ render := func(newStatus string) Email {
+ t.Helper()
+ email, rerr := r.Render(db.NotificationOutbox{
+ Payload: map[string]any{
+ "title": "DB upgrade", "new_status": newStatus, "actor": "admin-user",
+ },
+ })
+ require.NoError(t, rerr)
+ return email
+ }
+
+ t.Run("operator or admin creates a planned window", func(t *testing.T) {
+ email := render("planned")
+ assert.Equal(t, "[Maintenance] DB upgrade — scheduled", email.Subject)
+ assert.Contains(t, email.Body, "has been scheduled")
+ assert.Contains(t, email.Body, "Created by: admin-user")
+ assert.NotContains(t, email.Body, "changed status")
+ })
+
+ t.Run("creator submits for review", func(t *testing.T) {
+ email := render("pending_review")
+ assert.Equal(t, "[Maintenance] DB upgrade — awaiting review", email.Subject)
+ assert.Contains(t, email.Body, "has been submitted for review")
+ assert.NotContains(t, email.Body, "changed status")
+ })
+
+ t.Run("later transition still reads as a change", func(t *testing.T) {
+ email, rerr := r.Render(db.NotificationOutbox{
+ Payload: map[string]any{
+ "title": "DB upgrade", "old_status": "planned",
+ "new_status": "in_progress", "actor": "checker",
+ },
+ })
+ require.NoError(t, rerr)
+ assert.Equal(t, "[Maintenance] DB upgrade — in_progress", email.Subject)
+ assert.Contains(t, email.Body, "changed status")
+ assert.Contains(t, email.Body, "Changed by: checker")
+ })
+}
diff --git a/internal/notification/resolver.go b/internal/notification/resolver.go
new file mode 100644
index 0000000..0e57abb
--- /dev/null
+++ b/internal/notification/resolver.go
@@ -0,0 +1,145 @@
+package notification
+
+import (
+ "fmt"
+ "time"
+
+ "github.com/google/uuid"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/event"
+)
+
+// Change describes a single committed maintenance change to notify about.
+type Change struct {
+ IncidentID uint
+ Title string
+ OldStatus event.Status
+ NewStatus event.Status
+ // ContactEmail is the maintenance creator address (incident.contact_email).
+ ContactEmail string
+ // Actor is the preferred_username for API changes, or ActorChecker.
+ Actor string
+ // ChangedAt is the change time; defaults to now (UTC) when zero.
+ ChangedAt time.Time
+}
+
+// Resolver turns a maintenance Change into outbox rows using the recipient rules
+// (architecture §1). Review-audience addresses come from configuration; the
+// creator address comes from the change.
+type Resolver struct {
+ smod string
+ operators []string
+ admins []string
+ excluded map[string]struct{}
+ baseURL string
+}
+
+// NewResolver builds a Resolver from the parsed notification config.
+func NewResolver(cfg Config) *Resolver {
+ excluded := make(map[string]struct{}, len(cfg.ExcludedEmails))
+ for _, e := range cfg.ExcludedEmails {
+ excluded[normalizeEmail(e)] = struct{}{}
+ }
+
+ return &Resolver{
+ smod: cfg.ReviewSMOD,
+ operators: cfg.ReviewOperators,
+ admins: cfg.ReviewAdmins,
+ excluded: excluded,
+ baseURL: cfg.BaseURL,
+ }
+}
+
+// Recipients returns the normalized, deduplicated recipient list for the resulting
+// status: review audience (SMOD + operators + admins) for review states, plus the
+// creator for every state.
+func (r *Resolver) Recipients(status event.Status, contactEmail string) []string {
+ var ordered []string
+ seen := make(map[string]struct{})
+
+ add := func(raw string) {
+ e := normalizeEmail(raw)
+ if e == "" {
+ return
+ }
+ // Applied here rather than per source, so an exclusion cannot be bypassed
+ // by passing the address as contact_email.
+ if _, blocked := r.excluded[e]; blocked {
+ return
+ }
+ if _, ok := seen[e]; ok {
+ return
+ }
+ seen[e] = struct{}{}
+ ordered = append(ordered, e)
+ }
+
+ if isReviewStatus(status) {
+ add(r.smod)
+ for _, e := range r.operators {
+ add(e)
+ }
+ for _, e := range r.admins {
+ add(e)
+ }
+ }
+ add(contactEmail)
+
+ return ordered
+}
+
+// BuildRows produces one pending outbox row per recipient for the change, sharing a
+// single generated change_id. It returns nil when there are no recipients.
+func (r *Resolver) BuildRows(ch Change) []db.NotificationOutbox {
+ recipients := r.Recipients(ch.NewStatus, ch.ContactEmail)
+ if len(recipients) == 0 {
+ return nil
+ }
+
+ kind := KindForStatus(ch.NewStatus)
+ changeID := uuid.NewString()
+ changedAt := ch.ChangedAt
+ if changedAt.IsZero() {
+ changedAt = time.Now().UTC()
+ }
+ payload := buildPayload(ch, changedAt, r.link(ch.IncidentID))
+
+ rows := make([]db.NotificationOutbox, 0, len(recipients))
+ for _, rcpt := range recipients {
+ rows = append(rows, db.NotificationOutbox{
+ Kind: kind,
+ IncidentID: ch.IncidentID,
+ Recipient: rcpt,
+ Payload: payload,
+ ChangeID: changeID,
+ DedupKey: DedupKey(changeID, kind, rcpt),
+ Status: db.NotificationStatusPending,
+ })
+ }
+ return rows
+}
+
+// DedupKey builds the unique key change_id : kind : recipient (architecture §4).
+func DedupKey(changeID, kind, recipient string) string {
+ return fmt.Sprintf("%s:%s:%s", changeID, kind, recipient)
+}
+
+// link builds the maintenance deep link from the configured web origin.
+func (r *Resolver) link(incidentID uint) string {
+ return fmt.Sprintf("%s/incidents/%d", r.baseURL, incidentID)
+}
+
+// buildPayload snapshots everything the renderer needs, as strings so a JSONB
+// round-trip never changes types (map[string]any with json serializer).
+func buildPayload(ch Change, changedAt time.Time, link string) map[string]any {
+ return map[string]any{
+ "incident_id": fmt.Sprint(ch.IncidentID),
+ "title": ch.Title,
+ "old_status": string(ch.OldStatus),
+ "new_status": string(ch.NewStatus),
+ "actor": ch.Actor,
+ "changed_at": changedAt.UTC().Format(time.RFC3339),
+ "link": link,
+ }
+}
diff --git a/internal/notification/resolver_test.go b/internal/notification/resolver_test.go
new file mode 100644
index 0000000..274ed33
--- /dev/null
+++ b/internal/notification/resolver_test.go
@@ -0,0 +1,122 @@
+package notification
+
+import (
+ "testing"
+
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/event"
+)
+
+func testResolver() *Resolver {
+ return NewResolver(Config{
+ ReviewSMOD: "support@com.com",
+ ReviewOperators: []string{"ops@com.com"},
+ ReviewAdmins: []string{"admin@com.com"},
+ BaseURL: "https://status.example.com",
+ })
+}
+
+func TestRecipients_ReviewStatusesIncludeAudienceAndCreator(t *testing.T) {
+ r := testResolver()
+
+ for _, status := range []event.Status{event.MaintenancePendingReview, event.MaintenanceReviewed} {
+ got := r.Recipients(status, "creator@com.com")
+ assert.ElementsMatch(t,
+ []string{"support@com.com", "ops@com.com", "admin@com.com", "creator@com.com"},
+ got, "status %s", status)
+ }
+}
+
+func TestRecipients_ExcludedAddressesAreDropped(t *testing.T) {
+ r := NewResolver(Config{
+ ReviewSMOD: "support@com.com",
+ ReviewOperators: []string{"ops@com.com"},
+ ReviewAdmins: []string{"admin@com.com"},
+ ExcludedEmails: []string{"ops@com.com", "noreply@com.com"},
+ })
+
+ got := r.Recipients(event.MaintenancePendingReview, "creator@com.com")
+ assert.ElementsMatch(t, []string{"support@com.com", "admin@com.com", "creator@com.com"}, got)
+
+ // The exclusion must hold even when the address arrives as the creator contact.
+ got = r.Recipients(event.MaintenancePlanned, "NoReply@COM.com")
+ assert.Empty(t, got)
+}
+
+func TestRecipients_LifecycleStatusesCreatorOnly(t *testing.T) {
+ r := testResolver()
+
+ for _, status := range []event.Status{
+ event.MaintenancePlanned, event.MaintenanceInProgress,
+ event.MaintenanceCompleted, event.MaintenanceCancelled,
+ } {
+ got := r.Recipients(status, "creator@com.com")
+ assert.Equal(t, []string{"creator@com.com"}, got, "status %s", status)
+ }
+}
+
+func TestRecipients_NormalizesAndDeduplicates(t *testing.T) {
+ r := NewResolver(Config{
+ ReviewSMOD: "Support@Com.com",
+ ReviewOperators: []string{" ops@com.com "},
+ ReviewAdmins: []string{"support@com.com"}, // duplicate of SMOD after normalize
+ })
+
+ // Creator equals the operator address (different case) -> must appear once.
+ got := r.Recipients(event.MaintenancePendingReview, "OPS@com.com")
+ assert.Equal(t, []string{"support@com.com", "ops@com.com"}, got)
+}
+
+func TestRecipients_EmptyContactEmailForLifecycleYieldsNone(t *testing.T) {
+ r := testResolver()
+ assert.Empty(t, r.Recipients(event.MaintenancePlanned, ""))
+}
+
+func TestDedupKey(t *testing.T) {
+ assert.Equal(t, "abc:pending_review:creator@com.com",
+ DedupKey("abc", db.NotificationKindPendingReview, "creator@com.com"))
+}
+
+func TestBuildRows_OneRowPerRecipientSharedChangeID(t *testing.T) {
+ r := testResolver()
+ ch := Change{
+ IncidentID: 42,
+ Title: "DB upgrade",
+ OldStatus: "",
+ NewStatus: event.MaintenancePendingReview,
+ ContactEmail: "creator@com.com",
+ Actor: "admin-user",
+ }
+
+ rows := r.BuildRows(ch)
+ require.Len(t, rows, 4)
+
+ changeID := rows[0].ChangeID
+ require.NotEmpty(t, changeID)
+ seenRecipients := make(map[string]struct{})
+ for _, row := range rows {
+ assert.Equal(t, changeID, row.ChangeID, "all rows share one change_id")
+ assert.Equal(t, db.NotificationKindPendingReview, row.Kind)
+ assert.Equal(t, uint(42), row.IncidentID)
+ assert.Equal(t, db.NotificationStatusPending, row.Status)
+ assert.Equal(t, DedupKey(changeID, row.Kind, row.Recipient), row.DedupKey)
+ assert.Equal(t, "42", row.Payload["incident_id"])
+ assert.Equal(t, "DB upgrade", row.Payload["title"])
+ assert.Equal(t, "https://status.example.com/incidents/42", row.Payload["link"])
+ seenRecipients[row.Recipient] = struct{}{}
+ }
+ assert.Len(t, seenRecipients, 4, "recipients are unique")
+}
+
+func TestBuildRows_NoRecipientsReturnsNil(t *testing.T) {
+ r := testResolver()
+ rows := r.BuildRows(Change{
+ IncidentID: 7,
+ NewStatus: event.MaintenancePlanned, // lifecycle -> creator only
+ // no contact email
+ })
+ assert.Nil(t, rows)
+}
diff --git a/internal/notification/smtp.go b/internal/notification/smtp.go
new file mode 100644
index 0000000..838046e
--- /dev/null
+++ b/internal/notification/smtp.go
@@ -0,0 +1,103 @@
+package notification
+
+import (
+ "context"
+ "errors"
+ "fmt"
+
+ mail "github.com/wneessen/go-mail"
+)
+
+// SMTP reply codes in [500,600) are permanent rejections per RFC 5321 §4.2.1.
+const (
+ smtpPermanentFrom = 500
+ smtpPermanentTo = 600
+)
+
+// Sender delivers one rendered email to one recipient. It is an interface so the
+// worker and tests can substitute a fake without touching a real SMTP server.
+type Sender interface {
+ Send(ctx context.Context, recipient string, email Email) error
+}
+
+// smtpSender sends mail directly to the configured (OTC) SMTP endpoint using
+// github.com/wneessen/go-mail.
+type smtpSender struct {
+ client *mail.Client
+ from string
+}
+
+// NewSMTPSender builds a reusable SMTP sender from the parsed config.
+func NewSMTPSender(cfg Config) (Sender, error) {
+ opts := []mail.Option{
+ mail.WithPort(cfg.Port),
+ mail.WithTimeout(cfg.Timeout),
+ mail.WithTLSPolicy(tlsPolicy(cfg.TLS)),
+ }
+ if cfg.User != "" {
+ // Auto-discover rather than a fixed mechanism: relays differ in what they
+ // offer, and picking PLAIN blindly fails against a LOGIN-only server.
+ opts = append(opts,
+ mail.WithSMTPAuth(mail.SMTPAuthAutoDiscover),
+ mail.WithUsername(cfg.User),
+ mail.WithPassword(cfg.Password),
+ )
+ } else {
+ // Relay without credentials (e.g. a local catcher) must not negotiate AUTH.
+ opts = append(opts, mail.WithSMTPAuth(mail.SMTPAuthNoAuth))
+ }
+
+ client, err := mail.NewClient(cfg.Host, opts...)
+ if err != nil {
+ return nil, fmt.Errorf("build smtp client: %w", err)
+ }
+
+ return &smtpSender{client: client, from: cfg.From}, nil
+}
+
+// Send composes and delivers a single message. The context bounds the whole
+// dial+send so a hung server cannot exceed the lease.
+func (s *smtpSender) Send(ctx context.Context, recipient string, email Email) error {
+ msg := mail.NewMsg()
+ if err := msg.From(s.from); err != nil {
+ return fmt.Errorf("set from: %w", err)
+ }
+ if err := msg.To(recipient); err != nil {
+ return fmt.Errorf("set recipient: %w", err)
+ }
+ msg.Subject(email.Subject)
+ msg.SetBodyString(mail.TypeTextPlain, email.Body)
+
+ if err := s.client.DialAndSendWithContext(ctx, msg); err != nil {
+ if isPermanent(err) {
+ return fmt.Errorf("send mail: %w: %w", ErrPermanentDelivery, err)
+ }
+ return fmt.Errorf("send mail: %w", err)
+ }
+ return nil
+}
+
+// ErrPermanentDelivery marks a rejection the server will repeat for every retry,
+// such as an unknown recipient. The worker fails these rows immediately.
+var ErrPermanentDelivery = errors.New("permanent delivery failure")
+
+// isPermanent reports whether the relay rejected the message for good. Only a 5xx
+// reply qualifies: transport errors carry no code and may succeed later.
+func isPermanent(err error) bool {
+ var sendErr *mail.SendError
+ if !errors.As(err, &sendErr) || sendErr.IsTemp() {
+ return false
+ }
+
+ code := sendErr.ErrorCode()
+
+ return code >= smtpPermanentFrom && code < smtpPermanentTo
+}
+
+// tlsPolicy selects mandatory TLS when configured, otherwise opportunistic.
+func tlsPolicy(enabled bool) mail.TLSPolicy {
+ if enabled {
+ return mail.TLSMandatory
+ }
+ return mail.TLSOpportunistic
+}
diff --git a/internal/notification/templates/body.tmpl b/internal/notification/templates/body.tmpl
new file mode 100644
index 0000000..86b9d99
--- /dev/null
+++ b/internal/notification/templates/body.tmpl
@@ -0,0 +1,7 @@
+Maintenance "{{.Title}}" (event #{{.IncidentID}}) {{.Headline}}.
+
+{{if .OldStatus}}Status: {{.OldStatus}} -> {{.NewStatus}}{{else}}Status: {{.NewStatus}}{{end}}
+{{if .OldStatus}}Changed by{{else}}Created by{{end}}: {{.Actor}}
+Time (UTC): {{.ChangedAt}}
+
+Details: {{.Link}}
diff --git a/internal/notification/templates/subject.tmpl b/internal/notification/templates/subject.tmpl
new file mode 100644
index 0000000..4f250c2
--- /dev/null
+++ b/internal/notification/templates/subject.tmpl
@@ -0,0 +1 @@
+[Maintenance] {{.Title}} — {{.State}}
diff --git a/internal/notification/worker.go b/internal/notification/worker.go
new file mode 100644
index 0000000..ae827ad
--- /dev/null
+++ b/internal/notification/worker.go
@@ -0,0 +1,218 @@
+package notification
+
+import (
+ "context"
+ "errors"
+ "fmt"
+ "os"
+ "time"
+
+ "go.uber.org/zap"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+)
+
+const (
+ // claimBatchSize is deliberately 1: the lease starts at claim time but sends are
+ // sequential, so a larger batch would let later rows outlive their lease and be
+ // re-delivered by the stale-recovery path.
+ claimBatchSize = 1
+ defaultSweepEvery = 5 * time.Minute
+
+ // retentionAge keeps delivered rows for audit/re-drive, then prunes them so the
+ // outbox (and the ops stat queries over it) stay small. Failed rows are kept.
+ retentionAge = 30 * 24 * time.Hour
+ retentionBatch = 500
+)
+
+// Worker delivers queued outbox rows. On the happy path it is woken by Notify right
+// after a change commits; a low-frequency ticker sweeps for retries and rows orphaned
+// by a crashed pod. Sending happens outside any DB transaction (architecture §5).
+type Worker struct {
+ db *db.DB
+ renderer *Renderer
+ sender Sender
+ log *zap.Logger
+
+ leaseOwner string
+ leaseTimeout time.Duration
+ maxAttempts int
+ smtpTimeout time.Duration
+ backoff func(attempts int) time.Time
+
+ batchSize int
+ sweepEvery time.Duration
+
+ metrics *Metrics
+
+ signal chan struct{}
+}
+
+// NewWorker builds a delivery worker from the parsed config and a sender. metrics
+// may be nil (the record* calls are nil-safe).
+func NewWorker(cfg Config, database *db.DB, sender Sender, log *zap.Logger, metrics *Metrics) (*Worker, error) {
+ renderer, err := NewRenderer()
+ if err != nil {
+ return nil, err
+ }
+ return &Worker{
+ db: database,
+ renderer: renderer,
+ sender: sender,
+ log: log,
+ leaseOwner: leaseOwner(),
+ leaseTimeout: cfg.LeaseTimeout,
+ maxAttempts: cfg.MaxAttempts,
+ smtpTimeout: cfg.Timeout,
+ backoff: Backoff(cfg.BackoffBase),
+ batchSize: claimBatchSize,
+ sweepEvery: defaultSweepEvery,
+ metrics: metrics,
+ signal: make(chan struct{}, 1),
+ }, nil
+}
+
+// Notify wakes the worker after a commit. It never blocks: a pending signal already
+// covers the next drain.
+func (w *Worker) Notify() {
+ select {
+ case w.signal <- struct{}{}:
+ default:
+ }
+}
+
+// Run processes due rows on every signal and on a periodic safety sweep until the
+// context is cancelled. In-flight sends finish before Run returns.
+func (w *Worker) Run(ctx context.Context) {
+ w.log.Info("notification worker started", zap.String("lease_owner", w.leaseOwner))
+ ticker := time.NewTicker(w.sweepEvery)
+ defer ticker.Stop()
+
+ for {
+ select {
+ case <-ctx.Done():
+ w.log.Info("notification worker stopped")
+ return
+ case <-w.signal:
+ w.drainQuietly(ctx)
+ case <-ticker.C:
+ w.drainQuietly(ctx)
+ w.runRetention(ctx)
+ }
+ }
+}
+
+func (w *Worker) drainQuietly(ctx context.Context) {
+ if err := w.Drain(ctx); err != nil && ctx.Err() == nil {
+ w.log.Error("notification drain failed", zap.Error(err))
+ }
+}
+
+// runRetention prunes delivered rows older than retentionAge on the safety sweep.
+func (w *Worker) runRetention(ctx context.Context) {
+ before := time.Now().UTC().Add(-retentionAge)
+ n, err := w.db.DeleteSentBefore(ctx, before, retentionBatch)
+ if err != nil && ctx.Err() == nil {
+ w.log.Error("notification retention failed", zap.Error(err))
+ return
+ }
+ if n > 0 {
+ w.log.Info("notification retention pruned sent rows", zap.Int64("count", n))
+ }
+}
+
+// Drain recovers stale rows, then claims and sends batches until none remain due.
+// It is exported so it can be driven deterministically in tests.
+func (w *Worker) Drain(ctx context.Context) error {
+ recovered, rerr := w.db.RecoverStaleProcessing(ctx, nil, w.leaseTimeout, w.maxAttempts)
+ if rerr != nil {
+ return rerr
+ }
+ w.metrics.recordStaleRecovered(len(recovered))
+
+ for {
+ if ctx.Err() != nil {
+ return ctx.Err()
+ }
+
+ rows, err := w.db.ClaimPending(ctx, nil, w.batchSize, w.leaseOwner, w.leaseTimeout)
+ if err != nil {
+ return err
+ }
+ if len(rows) == 0 {
+ return nil
+ }
+
+ for i := range rows {
+ if ctx.Err() != nil {
+ return ctx.Err()
+ }
+ w.deliver(ctx, rows[i])
+ }
+ }
+}
+
+// deliver sends one claimed row and records the outcome. Failures (including panics)
+// are isolated per row so one bad email cannot stop the batch.
+func (w *Worker) deliver(ctx context.Context, row db.NotificationOutbox) {
+ w.metrics.recordAttempt()
+ start := time.Now()
+ err := w.sendGuarded(ctx, row)
+ w.metrics.observeDuration(time.Since(start))
+ if err != nil {
+ w.log.Warn("notification send failed",
+ zap.Uint("outbox_id", row.ID), zap.Uint("incident_id", row.IncidentID),
+ zap.String("recipient", row.Recipient), zap.Int("attempts", row.Attempts),
+ zap.Bool("permanent", errors.Is(err, ErrPermanentDelivery)),
+ zap.Error(err))
+ w.metrics.recordFailed(row.Kind)
+ w.markFailure(ctx, row.ID, err)
+ return
+ }
+
+ w.metrics.recordSent(row.Kind)
+ if err = w.db.MarkSent(ctx, nil, row.ID); err != nil {
+ w.log.Error("mark sent", zap.Uint("outbox_id", row.ID), zap.Error(err))
+ }
+}
+
+// markFailure records the outcome, skipping the retry schedule for rejections that
+// every further attempt would reproduce.
+func (w *Worker) markFailure(ctx context.Context, id uint, sendErr error) {
+ var err error
+ if errors.Is(sendErr, ErrPermanentDelivery) {
+ err = w.db.MarkFailedTerminal(ctx, nil, id, sendErr.Error())
+ } else {
+ err = w.db.MarkFailed(ctx, nil, id, sendErr.Error(), w.maxAttempts, w.backoff)
+ }
+ if err != nil {
+ w.log.Error("mark failed", zap.Uint("outbox_id", id), zap.Error(err))
+ }
+}
+
+// sendGuarded renders and sends one row inside a recover() guard and an SMTP timeout.
+func (w *Worker) sendGuarded(ctx context.Context, row db.NotificationOutbox) (err error) {
+ defer func() {
+ if r := recover(); r != nil {
+ err = fmt.Errorf("panic sending notification: %v", r)
+ }
+ }()
+
+ email, err := w.renderer.Render(row)
+ if err != nil {
+ return fmt.Errorf("render: %w", err)
+ }
+
+ sendCtx, cancel := context.WithTimeout(ctx, w.smtpTimeout)
+ defer cancel()
+ return w.sender.Send(sendCtx, row.Recipient, email)
+}
+
+// leaseOwner identifies this pod for outbox lease bookkeeping.
+func leaseOwner() string {
+ host, err := os.Hostname()
+ if err != nil || host == "" {
+ host = "pod"
+ }
+ return fmt.Sprintf("%s-%d", host, os.Getpid())
+}
diff --git a/tests/app_wiring_test.go b/tests/app_wiring_test.go
new file mode 100644
index 0000000..7b6fdd5
--- /dev/null
+++ b/tests/app_wiring_test.go
@@ -0,0 +1,49 @@
+package tests
+
+import (
+ "testing"
+
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ "go.uber.org/zap"
+
+ "github.com/stackmon/otc-status-dashboard/internal/app"
+ "github.com/stackmon/otc-status-dashboard/internal/conf"
+)
+
+// baseAppConfig is a fully-formed config (app.New does not call FillDefaults) with
+// notifications off. The OpenAPI spec path is resolved from the tests/ working dir.
+func baseAppConfig() *conf.Config {
+ return &conf.Config{
+ DB: databaseURL,
+ Port: "8000",
+ Hostname: "localhost",
+ WebURL: "https://status.example.com",
+ SecretKeyV1: testHMACSecret,
+ OpenAPISpecPath: "../openapi.yaml",
+ RBAC: conf.RBACConfig{Creators: creatorGroup, Operators: operatorGroup, Admins: adminGroup},
+ }
+}
+
+func TestApp_BootsWithNotificationsDisabled(t *testing.T) {
+ s, err := app.New(baseAppConfig(), zap.NewNop())
+ require.NoError(t, err)
+ t.Cleanup(func() { _ = s.DB.Close() })
+
+ assert.Nil(t, s.NotifyFunc(), "no worker wake-up when notifications are disabled")
+}
+
+func TestApp_BootsWithNotificationsEnabled(t *testing.T) {
+ cfg := baseAppConfig()
+ cfg.SMTP = conf.SMTPConfig{Host: "smtp", Port: "587", From: "sd@com.com", Timeout: "30s"}
+ cfg.Notifications = conf.NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "5m",
+ SmodEmail: "smod@com.com", EmailsOperators: "ops@com.com", EmailsAdmins: "admin@com.com",
+ }
+
+ s, err := app.New(cfg, zap.NewNop())
+ require.NoError(t, err)
+ t.Cleanup(func() { _ = s.DB.Close() })
+
+ assert.NotNil(t, s.NotifyFunc(), "worker wired on the shared pool when enabled")
+}
diff --git a/tests/checker_notifications_test.go b/tests/checker_notifications_test.go
new file mode 100644
index 0000000..a7e5dde
--- /dev/null
+++ b/tests/checker_notifications_test.go
@@ -0,0 +1,86 @@
+package tests
+
+import (
+ "testing"
+
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ "go.uber.org/zap"
+ gormpostgres "gorm.io/driver/postgres"
+ "gorm.io/gorm"
+
+ "github.com/stackmon/otc-status-dashboard/internal/checker"
+ "github.com/stackmon/otc-status-dashboard/internal/conf"
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/event"
+)
+
+func notifCheckerConfig() *conf.Config {
+ return &conf.Config{
+ DB: databaseURL,
+ WebURL: "https://status.example.com",
+ SMTP: conf.SMTPConfig{Host: "smtp", Port: "587", From: "sd@com.com", Timeout: "30s"},
+ Notifications: conf.NotificationsConfig{
+ Enabled: true, LeaseTimeout: "60s", MaxAttempts: "5", BackoffInterval: "5m",
+ SmodEmail: "smod@com.com", EmailsOperators: "ops@com.com", EmailsAdmins: "admin@com.com",
+ },
+ }
+}
+
+func TestChecker_ReviewedToPlanned_EnqueuesStatusChangedToCreator(t *testing.T) {
+ truncateIncidents(t)
+
+ // Use a router WITHOUT a publisher so the create + approval produce no outbox rows;
+ // only the checker transition should enqueue.
+ r := initTestsWithHMAC(t)
+ resp := createEventOK(t, r, maintenanceData(), creatorTokenA) // -> pending_review
+ eventID := resp.Result[0].IncidentID
+ transitionTo(t, r, eventID, event.MaintenanceReviewed, adminToken) // -> reviewed
+
+ g, err := gorm.Open(gormpostgres.New(gormpostgres.Config{DSN: databaseURL}), &gorm.Config{})
+ require.NoError(t, err)
+ sqlDB, err := g.DB()
+ require.NoError(t, err)
+ sqlDB.SetMaxOpenConns(2)
+ t.Cleanup(func() { _ = sqlDB.Close() })
+
+ // No outbox rows yet (publisher was off during API calls).
+ require.Equal(t, int64(0), outboxCount(t, g, eventID))
+
+ chk, err := checker.New(notifCheckerConfig(), zap.NewNop())
+ require.NoError(t, err)
+ t.Cleanup(func() { _ = chk.Close() })
+
+ require.NoError(t, chk.CheckMaintenance()) // reviewed -> planned
+
+ var rows []db.NotificationOutbox
+ require.NoError(t, g.Where("incident_id = ?", eventID).Find(&rows).Error)
+ require.Len(t, rows, 1, "one notification per real transition")
+ assert.Equal(t, db.NotificationKindStatusChanged, rows[0].Kind)
+ assert.Equal(t, "test@example.com", rows[0].Recipient, "planned notifies creator only")
+ assert.Equal(t, "checker", rows[0].Payload["actor"])
+}
+
+func TestChecker_NoTransition_EnqueuesNothing(t *testing.T) {
+ truncateIncidents(t)
+
+ r := initTestsWithHMAC(t)
+ resp := createEventOK(t, r, maintenanceData(), adminToken) // admin -> planned (future start)
+ eventID := resp.Result[0].IncidentID
+
+ g, err := gorm.Open(gormpostgres.New(gormpostgres.Config{DSN: databaseURL}), &gorm.Config{})
+ require.NoError(t, err)
+ sqlDB, err := g.DB()
+ require.NoError(t, err)
+ sqlDB.SetMaxOpenConns(2)
+ t.Cleanup(func() { _ = sqlDB.Close() })
+
+ chk, err := checker.New(notifCheckerConfig(), zap.NewNop())
+ require.NoError(t, err)
+ t.Cleanup(func() { _ = chk.Close() })
+
+ // Planned with a future start date: the checker computes planned again -> no change.
+ require.NoError(t, chk.CheckMaintenance())
+
+ assert.Equal(t, int64(0), outboxCount(t, g, eventID), "no notification without a real transition")
+}
diff --git a/tests/db_tx_test.go b/tests/db_tx_test.go
new file mode 100644
index 0000000..a99083a
--- /dev/null
+++ b/tests/db_tx_test.go
@@ -0,0 +1,122 @@
+package tests
+
+import (
+ "context"
+ "errors"
+ "testing"
+ "time"
+
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ "gorm.io/gorm"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/event"
+)
+
+func TestWithTx_CommitsIncidentAndOutboxAtomically(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+
+ var incID uint
+ err := d.WithTx(ctx, func(tx *gorm.DB) error {
+ id, e := d.SaveIncidentTx(tx, newMaintenanceIncident())
+ if e != nil {
+ return e
+ }
+ incID = id
+ return d.Enqueue(ctx, tx, newOutboxRow(id, "creator@com.com"))
+ })
+ require.NoError(t, err)
+
+ var incCount, outCount int64
+ require.NoError(t, g.Model(&db.Incident{}).Where("id = ?", incID).Count(&incCount).Error)
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).Where("incident_id = ?", incID).Count(&outCount).Error)
+ assert.Equal(t, int64(1), incCount)
+ assert.Equal(t, int64(1), outCount)
+}
+
+func TestWithTx_RollsBackBothOnError(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ sentinel := errors.New("boom")
+
+ var incID uint
+ var dedup string
+ err := d.WithTx(ctx, func(tx *gorm.DB) error {
+ id, e := d.SaveIncidentTx(tx, newMaintenanceIncident())
+ if e != nil {
+ return e
+ }
+ incID = id
+ row := newOutboxRow(id, "creator@com.com")
+ dedup = row.DedupKey
+ if e = d.Enqueue(ctx, tx, row); e != nil {
+ return e
+ }
+ return sentinel // force rollback after both writes
+ })
+ require.ErrorIs(t, err, sentinel)
+
+ var incCount, outCount int64
+ require.NoError(t, g.Model(&db.Incident{}).Where("id = ?", incID).Count(&incCount).Error)
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).Where("dedup_key = ?", dedup).Count(&outCount).Error)
+ assert.Equal(t, int64(0), incCount, "incident rolled back")
+ assert.Equal(t, int64(0), outCount, "no orphan email task")
+}
+
+func TestModifyIncidentTx_SharedTxWithEnqueue(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+
+ incID := seedIncident(t, d)
+ inc, err := d.GetIncident(int(incID))
+ require.NoError(t, err)
+ inc.Status = event.MaintenanceReviewed
+
+ row := newOutboxRow(incID, "creator@com.com")
+ err = d.WithTx(ctx, func(tx *gorm.DB) error {
+ if e := d.ModifyIncidentTx(tx, inc); e != nil {
+ return e
+ }
+ return d.Enqueue(ctx, tx, row)
+ })
+ require.NoError(t, err)
+
+ got, err := d.GetIncident(int(incID))
+ require.NoError(t, err)
+ assert.Equal(t, event.MaintenanceReviewed, got.Status)
+
+ var outCount int64
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).Where("dedup_key = ?", row.DedupKey).Count(&outCount).Error)
+ assert.Equal(t, int64(1), outCount)
+}
+
+func TestModifyEventUpdateTx_UpdatesText(t *testing.T) {
+ d, g := newNotifDB(t)
+
+ incID := seedIncident(t, d)
+ // Seed one status row for the incident.
+ status := db.IncidentStatus{IncidentID: incID, Status: event.MaintenancePendingReview, Text: "original"}
+ require.NoError(t, g.Create(&status).Error)
+
+ updated, err := d.ModifyEventUpdateTx(g, db.IncidentStatus{
+ ID: status.ID, IncidentID: incID, Text: "patched",
+ })
+ require.NoError(t, err)
+ assert.Equal(t, "patched", updated.Text)
+}
+
+// newMaintenanceIncident builds a minimal maintenance incident for tx tests.
+func newMaintenanceIncident() *db.Incident {
+ text := "tx maintenance"
+ start := time.Now().UTC()
+ impact := 0
+ return &db.Incident{
+ Text: &text,
+ StartDate: &start,
+ Impact: &impact,
+ System: false,
+ Type: "maintenance",
+ }
+}
diff --git a/tests/main_test.go b/tests/main_test.go
index 7139a08..3ca1259 100644
--- a/tests/main_test.go
+++ b/tests/main_test.go
@@ -262,7 +262,7 @@ func truncateIncidents(t *testing.T) {
gormDB, err := gorm.Open(gormpostgres.Open(databaseURL), &gorm.Config{})
require.NoError(t, err, "failed to open gorm connection for truncation")
- result := gormDB.Exec("TRUNCATE TABLE incident, incident_status, incident_component_relation RESTART IDENTITY")
+ result := gormDB.Exec("TRUNCATE TABLE incident, incident_status, incident_component_relation, notification_outbox RESTART IDENTITY")
require.NoError(t, result.Error, "failed to truncate incident tables")
sqlDB, err := gormDB.DB()
diff --git a/tests/notification_worker_test.go b/tests/notification_worker_test.go
new file mode 100644
index 0000000..e7891da
--- /dev/null
+++ b/tests/notification_worker_test.go
@@ -0,0 +1,195 @@
+package tests
+
+import (
+ "context"
+ "errors"
+ "fmt"
+ "sync"
+ "testing"
+ "time"
+
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ "go.uber.org/zap"
+ "gorm.io/gorm"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/notification"
+)
+
+// fakeSender records deliveries and can fail or panic for chosen recipients.
+type fakeSender struct {
+ mu sync.Mutex
+ sent []string
+ failFor map[string]bool
+ permanentFor map[string]bool
+ panicFor map[string]bool
+}
+
+func (f *fakeSender) Send(_ context.Context, recipient string, _ notification.Email) error {
+ f.mu.Lock()
+ defer f.mu.Unlock()
+ if f.panicFor[recipient] {
+ panic("smtp exploded for " + recipient)
+ }
+ if f.permanentFor[recipient] {
+ return fmt.Errorf("smtp rejected %s: %w", recipient, notification.ErrPermanentDelivery)
+ }
+ if f.failFor[recipient] {
+ return errors.New("smtp failed for " + recipient)
+ }
+ f.sent = append(f.sent, recipient)
+ return nil
+}
+
+func (f *fakeSender) recipients() []string {
+ f.mu.Lock()
+ defer f.mu.Unlock()
+ out := make([]string, len(f.sent))
+ copy(out, f.sent)
+ return out
+}
+
+func testWorker(t *testing.T, d *db.DB, sender notification.Sender, maxAttempts int) *notification.Worker {
+ t.Helper()
+ w, err := notification.NewWorker(notification.Config{
+ Enabled: true,
+ LeaseTimeout: time.Minute,
+ MaxAttempts: maxAttempts,
+ BackoffBase: 5 * time.Minute,
+ Timeout: 30 * time.Second,
+ }, d, sender, zap.NewNop(), nil)
+ require.NoError(t, err)
+ return w
+}
+
+func fetchByDedup(t *testing.T, g *gorm.DB, dedup string) db.NotificationOutbox {
+ t.Helper()
+ var row db.NotificationOutbox
+ require.NoError(t, g.Where("dedup_key = ?", dedup).First(&row).Error)
+ return row
+}
+
+func TestWorker_DeliversAllPending(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ recipients := []string{"a@com.com", "b@com.com", "c@com.com"}
+ for _, rc := range recipients {
+ require.NoError(t, d.Enqueue(ctx, nil, newOutboxRow(incID, rc)))
+ }
+
+ fake := &fakeSender{}
+ require.NoError(t, testWorker(t, d, fake, 3).Drain(ctx))
+
+ assert.ElementsMatch(t, recipients, fake.recipients())
+
+ var notSent int64
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).
+ Where("incident_id = ? AND status <> ?", incID, db.NotificationStatusSent).
+ Count(¬Sent).Error)
+ assert.Equal(t, int64(0), notSent, "all rows delivered")
+}
+
+func TestWorker_FailedSendRetriesWithBackoff(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ good := newOutboxRow(incID, "good@com.com")
+ bad := newOutboxRow(incID, "bad@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, good))
+ require.NoError(t, d.Enqueue(ctx, nil, bad))
+
+ fake := &fakeSender{failFor: map[string]bool{"bad@com.com": true}}
+ require.NoError(t, testWorker(t, d, fake, 3).Drain(ctx)) // retries remain
+
+ assert.Equal(t, []string{"good@com.com"}, fake.recipients())
+
+ badRow := fetchByDedup(t, g, bad.DedupKey)
+ assert.Equal(t, db.NotificationStatusPending, badRow.Status)
+ assert.Equal(t, 1, badRow.Attempts)
+ require.NotNil(t, badRow.NextAttemptAt)
+ assert.True(t, badRow.NextAttemptAt.After(time.Now().UTC()), "backoff pushed next_attempt_at forward")
+ require.NotNil(t, badRow.LastError)
+
+ assert.Equal(t, db.NotificationStatusSent, fetchByDedup(t, g, good.DedupKey).Status)
+}
+
+func TestWorker_FailedSendMarksFailedAtMaxAttempts(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ bad := newOutboxRow(incID, "bad@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, bad))
+
+ fake := &fakeSender{failFor: map[string]bool{"bad@com.com": true}}
+ require.NoError(t, testWorker(t, d, fake, 1).Drain(ctx)) // no retries left after first attempt
+
+ badRow := fetchByDedup(t, g, bad.DedupKey)
+ assert.Equal(t, db.NotificationStatusFailed, badRow.Status)
+ assert.Nil(t, badRow.NextAttemptAt)
+ require.NotNil(t, badRow.LastError)
+}
+
+func TestWorker_PermanentFailureSkipsRetries(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ rejected := newOutboxRow(incID, "unknown@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, rejected))
+
+ fake := &fakeSender{permanentFor: map[string]bool{"unknown@com.com": true}}
+ // maxAttempts is 5, so a retryable error would leave the row pending.
+ require.NoError(t, testWorker(t, d, fake, 5).Drain(ctx))
+
+ row := fetchByDedup(t, g, rejected.DedupKey)
+ assert.Equal(t, db.NotificationStatusFailed, row.Status, "5xx rejection is terminal")
+ assert.Nil(t, row.NextAttemptAt, "no retry scheduled")
+ require.NotNil(t, row.LastError)
+}
+
+func TestWorker_PanicIsolatedPerRow(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ good := newOutboxRow(incID, "good@com.com")
+ boom := newOutboxRow(incID, "boom@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, good))
+ require.NoError(t, d.Enqueue(ctx, nil, boom))
+
+ fake := &fakeSender{panicFor: map[string]bool{"boom@com.com": true}}
+ // Drain must not propagate the panic.
+ require.NoError(t, testWorker(t, d, fake, 3).Drain(ctx))
+
+ assert.Equal(t, []string{"good@com.com"}, fake.recipients())
+
+ boomRow := fetchByDedup(t, g, boom.DedupKey)
+ assert.Equal(t, db.NotificationStatusPending, boomRow.Status)
+ require.NotNil(t, boomRow.LastError)
+ assert.Contains(t, *boomRow.LastError, "panic")
+}
+
+func TestWorker_DrainTwiceDoesNotResend(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, _ := newNotifDB(t)
+ incID := seedIncident(t, d)
+ require.NoError(t, d.Enqueue(ctx, nil, newOutboxRow(incID, "a@com.com")))
+
+ fake := &fakeSender{}
+ w := testWorker(t, d, fake, 3)
+ require.NoError(t, w.Drain(ctx))
+ require.NoError(t, w.Drain(ctx)) // sent rows must not be re-claimed
+
+ assert.Equal(t, []string{"a@com.com"}, fake.recipients())
+}
diff --git a/tests/notifications_api_test.go b/tests/notifications_api_test.go
new file mode 100644
index 0000000..b041b5f
--- /dev/null
+++ b/tests/notifications_api_test.go
@@ -0,0 +1,133 @@
+package tests
+
+import (
+ "net/http"
+ "testing"
+
+ "github.com/gin-gonic/gin"
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ "go.uber.org/zap"
+ gormpostgres "gorm.io/driver/postgres"
+ "gorm.io/gorm"
+
+ "github.com/stackmon/otc-status-dashboard/internal/api"
+ "github.com/stackmon/otc-status-dashboard/internal/api/auth"
+ apiErrors "github.com/stackmon/otc-status-dashboard/internal/api/errors"
+ "github.com/stackmon/otc-status-dashboard/internal/api/rbac"
+ v2 "github.com/stackmon/otc-status-dashboard/internal/api/v2"
+ "github.com/stackmon/otc-status-dashboard/internal/conf"
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/event"
+ "github.com/stackmon/otc-status-dashboard/internal/notification"
+)
+
+// initNotifRouter builds a maintenance router with a real, enabled notification
+// publisher wired into the create/patch handlers, plus a raw gorm handle to verify
+// the outbox.
+func initNotifRouter(t *testing.T) (*gin.Engine, *gorm.DB) {
+ t.Helper()
+
+ d, err := db.New(&conf.Config{DB: databaseURL})
+ require.NoError(t, err)
+ t.Cleanup(func() { _ = d.Close() })
+
+ g, err := gorm.Open(gormpostgres.New(gormpostgres.Config{DSN: databaseURL}), &gorm.Config{})
+ require.NoError(t, err)
+ sqlDB, err := g.DB()
+ require.NoError(t, err)
+ sqlDB.SetMaxOpenConns(2)
+ t.Cleanup(func() { _ = sqlDB.Close() })
+
+ pub := notification.NewPublisher(notification.Config{
+ Enabled: true,
+ ReviewSMOD: "smod@com.com",
+ ReviewOperators: []string{"ops@com.com"},
+ ReviewAdmins: []string{"admin@com.com"},
+ BaseURL: "https://status.example.com",
+ }, d)
+
+ gin.SetMode(gin.TestMode)
+ r := gin.Default()
+ r.NoRoute(apiErrors.Return404)
+ r.Use(api.ErrorHandle())
+
+ logger, _ := zap.NewDevelopment()
+ prov := &auth.Provider{}
+ rbacSvc := rbac.New(creatorGroup, operatorGroup, adminGroup)
+
+ v2Api := r.Group("v2")
+ v2Api.POST("events",
+ api.AuthenticationMW(prov, logger, testHMACSecret),
+ api.RBACAuthorizationMW(rbacSvc, logger),
+ api.ValidateComponentsMW(d, logger),
+ v2.PostIncidentHandler(d, logger, pub))
+ v2Api.GET("events/:eventID",
+ api.SetJWTClaims(prov, logger, testHMACSecret),
+ api.CheckEventExistenceMW(d, logger),
+ v2.GetIncidentHandler(d, logger, rbacSvc))
+ v2Api.PATCH("events/:eventID",
+ api.AuthenticationMW(prov, logger, testHMACSecret),
+ api.RBACAuthorizationMW(rbacSvc, logger),
+ api.CheckEventExistenceMW(d, logger),
+ v2.PatchIncidentHandler(d, logger, pub))
+
+ return r, g
+}
+
+func outboxRecipients(t *testing.T, g *gorm.DB, incidentID int) []string {
+ t.Helper()
+ var rows []db.NotificationOutbox
+ require.NoError(t, g.Where("incident_id = ?", incidentID).Find(&rows).Error)
+ out := make([]string, 0, len(rows))
+ for i := range rows {
+ out = append(out, rows[i].Recipient)
+ }
+ return out
+}
+
+func outboxCount(t *testing.T, g *gorm.DB, incidentID int) int64 {
+ t.Helper()
+ var n int64
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).Where("incident_id = ?", incidentID).Count(&n).Error)
+ return n
+}
+
+func TestAPI_CreatorCreateMaintenance_EnqueuesReviewAudienceAndCreator(t *testing.T) {
+ truncateIncidents(t)
+ r, g := initNotifRouter(t)
+
+ resp := createEventOK(t, r, maintenanceData(), creatorTokenA)
+ eventID := resp.Result[0].IncidentID
+
+ // Creator submission -> pending_review -> review audience + creator (contact_email).
+ assert.ElementsMatch(t,
+ []string{"smod@com.com", "ops@com.com", "admin@com.com", "test@example.com"},
+ outboxRecipients(t, g, eventID))
+}
+
+func TestAPI_AdminCreateMaintenance_EnqueuesCreatorOnly(t *testing.T) {
+ truncateIncidents(t)
+ r, g := initNotifRouter(t)
+
+ resp := createEventOK(t, r, maintenanceData(), adminToken)
+ eventID := resp.Result[0].IncidentID
+
+ // Admin submission bypasses review -> planned -> creator only.
+ assert.Equal(t, []string{"test@example.com"}, outboxRecipients(t, g, eventID))
+}
+
+func TestAPI_FailedPatchVersionConflict_NoNewOutboxRow(t *testing.T) {
+ truncateIncidents(t)
+ r, g := initNotifRouter(t)
+
+ resp := createEventOK(t, r, maintenanceData(), adminToken)
+ eventID := resp.Result[0].IncidentID
+ before := outboxCount(t, g, eventID)
+
+ // Wrong version -> 409 Conflict -> transaction rolls back, no enqueue.
+ w := patchEvent(t, r, eventID, patchData(event.MaintenanceInProgress, intPtr(999)), adminToken)
+ require.Equal(t, http.StatusConflict, w.Code)
+
+ assert.Equal(t, before, outboxCount(t, g, eventID), "no new outbox row on version conflict")
+}
diff --git a/tests/notifications_e2e_test.go b/tests/notifications_e2e_test.go
new file mode 100644
index 0000000..2ea5bfb
--- /dev/null
+++ b/tests/notifications_e2e_test.go
@@ -0,0 +1,83 @@
+package tests
+
+import (
+ "context"
+ "testing"
+
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ "gorm.io/gorm"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/event"
+)
+
+func outboxRecipientsByKind(t *testing.T, g *gorm.DB, incidentID int, kind string) []string {
+ t.Helper()
+ var rows []db.NotificationOutbox
+ require.NoError(t, g.Where("incident_id = ? AND kind = ?", incidentID, kind).Find(&rows).Error)
+ out := make([]string, 0, len(rows))
+ for i := range rows {
+ out = append(out, rows[i].Recipient)
+ }
+ return out
+}
+
+// TestE2E_CreateMaintenanceDeliversToAllRecipients ties the whole pipeline together:
+// API create -> outbox rows -> worker drain -> sender delivers each recipient.
+func TestE2E_CreateMaintenanceDeliversToAllRecipients(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+
+ r, g := initNotifRouter(t)
+ resp := createEventOK(t, r, maintenanceData(), creatorTokenA) // pending_review
+ eventID := resp.Result[0].IncidentID
+ require.Len(t, outboxRecipients(t, g, eventID), 4)
+
+ d, _ := newNotifDB(t)
+ fake := &fakeSender{}
+ require.NoError(t, testWorker(t, d, fake, 3).Drain(ctx))
+
+ assert.ElementsMatch(t,
+ []string{"smod@com.com", "ops@com.com", "admin@com.com", "test@example.com"},
+ fake.recipients())
+
+ var notSent int64
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).
+ Where("incident_id = ? AND status <> ?", eventID, db.NotificationStatusSent).
+ Count(¬Sent).Error)
+ assert.Equal(t, int64(0), notSent, "every recipient delivered")
+}
+
+// TestE2E_ReviewedTransitionNotifiesReviewAudience covers the `reviewed` kind row set.
+func TestE2E_ReviewedTransitionNotifiesReviewAudience(t *testing.T) {
+ truncateIncidents(t)
+
+ r, g := initNotifRouter(t)
+ resp := createEventOK(t, r, maintenanceData(), creatorTokenA) // pending_review
+ eventID := resp.Result[0].IncidentID
+ transitionTo(t, r, eventID, event.MaintenanceReviewed, adminToken) // -> reviewed
+
+ assert.ElementsMatch(t,
+ []string{"smod@com.com", "ops@com.com", "admin@com.com", "test@example.com"},
+ outboxRecipientsByKind(t, g, eventID, db.NotificationKindReviewed))
+}
+
+// TestE2E_LifecycleTransitionNotifiesCreatorOnly verifies that no lifecycle transition
+// ever reaches the review audience — every status_changed row targets the creator.
+func TestE2E_LifecycleTransitionNotifiesCreatorOnly(t *testing.T) {
+ truncateIncidents(t)
+
+ r, g := initNotifRouter(t)
+ resp := createEventOK(t, r, maintenanceData(), adminToken) // admin -> planned
+ eventID := resp.Result[0].IncidentID
+ transitionTo(t, r, eventID, event.MaintenanceCancelled, adminToken) // planned -> cancelled
+
+ var rows []db.NotificationOutbox
+ require.NoError(t, g.Where("incident_id = ?", eventID).Find(&rows).Error)
+ require.NotEmpty(t, rows)
+ for i := range rows {
+ assert.Equal(t, db.NotificationKindStatusChanged, rows[i].Kind)
+ assert.Equal(t, "test@example.com", rows[i].Recipient, "lifecycle notifies creator only")
+ }
+}
diff --git a/tests/notifications_metrics_test.go b/tests/notifications_metrics_test.go
new file mode 100644
index 0000000..5931505
--- /dev/null
+++ b/tests/notifications_metrics_test.go
@@ -0,0 +1,56 @@
+package tests
+
+import (
+ "context"
+ "net/http"
+ "net/http/httptest"
+ "testing"
+ "time"
+
+ "github.com/prometheus/client_golang/prometheus"
+ "github.com/prometheus/client_golang/prometheus/promhttp"
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ "go.uber.org/zap"
+
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+ "github.com/stackmon/otc-status-dashboard/internal/notification"
+)
+
+func TestMetrics_EndpointExposesNotificationSeries(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ // A failed row feeds the outbox_failed gauge.
+ enqueueWithState(t, d, g, incID, "bad@com.com",
+ map[string]any{"status": db.NotificationStatusFailed, "last_error": "x"})
+
+ // A pending row that the worker will deliver, bumping sent_total.
+ require.NoError(t, d.Enqueue(ctx, nil, newOutboxRow(incID, "ok@com.com")))
+
+ metrics := notification.NewMetrics()
+ reg := prometheus.NewRegistry()
+ metrics.MustRegister(reg)
+ reg.MustRegister(notification.NewStatsCollector(d, time.Minute))
+
+ w, err := notification.NewWorker(notification.Config{
+ Enabled: true, LeaseTimeout: time.Minute, MaxAttempts: 3,
+ BackoffBase: 5 * time.Minute, Timeout: 30 * time.Second,
+ }, d, &fakeSender{}, zap.NewNop(), metrics)
+ require.NoError(t, err)
+ require.NoError(t, w.Drain(ctx))
+
+ rec := httptest.NewRecorder()
+ promhttp.HandlerFor(reg, promhttp.HandlerOpts{}).
+ ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/metrics", nil))
+
+ require.Equal(t, http.StatusOK, rec.Code)
+ body := rec.Body.String()
+ assert.Contains(t, body, "notification_attempts_total")
+ assert.Contains(t, body, "notification_delivery_duration_seconds")
+ assert.Contains(t, body, `notification_sent_total{kind="pending_review"} 1`)
+ assert.Contains(t, body, "notification_outbox_failed 1")
+ assert.Contains(t, body, "notification_outbox_pending")
+}
diff --git a/tests/notifications_ops_test.go b/tests/notifications_ops_test.go
new file mode 100644
index 0000000..646543c
--- /dev/null
+++ b/tests/notifications_ops_test.go
@@ -0,0 +1,298 @@
+package tests
+
+import (
+ "bytes"
+ "context"
+ "encoding/json"
+ "net/http"
+ "net/http/httptest"
+ "testing"
+ "time"
+
+ "github.com/gin-gonic/gin"
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ "go.uber.org/zap"
+ gormpostgres "gorm.io/driver/postgres"
+ "gorm.io/gorm"
+
+ "github.com/stackmon/otc-status-dashboard/internal/api"
+ "github.com/stackmon/otc-status-dashboard/internal/api/auth"
+ apiErrors "github.com/stackmon/otc-status-dashboard/internal/api/errors"
+ "github.com/stackmon/otc-status-dashboard/internal/api/rbac"
+ v2 "github.com/stackmon/otc-status-dashboard/internal/api/v2"
+ "github.com/stackmon/otc-status-dashboard/internal/conf"
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+)
+
+// setOutbox mutates an outbox row by dedup key without touching updated_at
+// (UpdateColumns skips autoUpdateTime), so tests can craft ages and states.
+func setOutbox(t *testing.T, g *gorm.DB, dedup string, cols map[string]any) {
+ t.Helper()
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).Where("dedup_key = ?", dedup).UpdateColumns(cols).Error)
+}
+
+func enqueueWithState(t *testing.T, d *db.DB, g *gorm.DB, incID uint, recipient string, cols map[string]any) string {
+ t.Helper()
+ row := newOutboxRow(incID, recipient)
+ require.NoError(t, d.Enqueue(context.Background(), nil, row))
+ if len(cols) > 0 {
+ setOutbox(t, g, row.DedupKey, cols)
+ }
+ return row.DedupKey
+}
+
+func TestGetNotificationStats(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ enqueueWithState(t, d, g, incID, "p1@com.com", nil) // pending
+ enqueueWithState(t, d, g, incID, "p2@com.com", nil) // pending
+ enqueueWithState(t, d, g, incID, "s1@com.com", map[string]any{"status": db.NotificationStatusSent})
+ enqueueWithState(t, d, g, incID, "f1@com.com", map[string]any{"status": db.NotificationStatusFailed, "last_error": "smtp down"})
+ enqueueWithState(t, d, g, incID, "stale@com.com", map[string]any{
+ "status": db.NotificationStatusProcessing, "locked_at": time.Now().UTC().Add(-5 * time.Minute),
+ })
+
+ stats, err := d.GetNotificationStats(ctx, time.Minute)
+ require.NoError(t, err)
+ assert.Equal(t, int64(2), stats.Pending)
+ assert.Equal(t, int64(1), stats.Processing)
+ assert.Equal(t, int64(1), stats.Sent)
+ assert.Equal(t, int64(1), stats.Failed)
+ assert.Equal(t, int64(1), stats.StaleProcessing)
+ assert.Equal(t, int64(0), stats.RetryBacklog)
+ assert.GreaterOrEqual(t, stats.OldestPendingAgeSeconds, float64(0))
+}
+
+func TestListNotificationsByStatus(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ enqueueWithState(t, d, g, incID, "ok@com.com", nil)
+ enqueueWithState(t, d, g, incID, "bad@com.com", map[string]any{"status": db.NotificationStatusFailed, "last_error": "x"})
+
+ failed, err := d.ListNotificationsByStatus(ctx, db.NotificationStatusFailed, 100)
+ require.NoError(t, err)
+ require.Len(t, failed, 1)
+ assert.Equal(t, "bad@com.com", failed[0].Recipient)
+
+ // Pending rows are the usual symptom of a stuck relay, so they must be listable too.
+ pending, err := d.ListNotificationsByStatus(ctx, db.NotificationStatusPending, 100)
+ require.NoError(t, err)
+ require.Len(t, pending, 1)
+ assert.Equal(t, "ok@com.com", pending[0].Recipient)
+
+ limited, err := d.ListNotificationsByStatus(ctx, db.NotificationStatusPending, 0)
+ require.NoError(t, err)
+ assert.Empty(t, limited)
+}
+
+func TestEnsureNotificationSchema(t *testing.T) {
+ d, _ := newNotifDB(t)
+ require.NoError(t, d.EnsureNotificationSchema(), "migrations are applied in the test DB")
+}
+
+func TestRedriveFailed_AllAndByID(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ failedCols := map[string]any{"status": db.NotificationStatusFailed, "attempts": 3, "last_error": "boom"}
+ a := enqueueWithState(t, d, g, incID, "a@com.com", failedCols)
+ b := enqueueWithState(t, d, g, incID, "b@com.com", failedCols)
+
+ // Re-drive only row a.
+ var rowA db.NotificationOutbox
+ require.NoError(t, g.Where("dedup_key = ?", a).First(&rowA).Error)
+ n, err := d.RedriveFailed(ctx, rowA.ID)
+ require.NoError(t, err)
+ assert.Equal(t, int64(1), n)
+
+ got := fetchByDedup(t, g, a)
+ assert.Equal(t, db.NotificationStatusPending, got.Status)
+ assert.Equal(t, 0, got.Attempts)
+ require.NotNil(t, got.NextAttemptAt)
+ assert.Nil(t, got.LastError)
+ // row b untouched
+ assert.Equal(t, db.NotificationStatusFailed, fetchByDedup(t, g, b).Status)
+
+ // Re-drive the rest (all remaining failed).
+ n, err = d.RedriveFailed(ctx)
+ require.NoError(t, err)
+ assert.Equal(t, int64(1), n)
+ assert.Equal(t, db.NotificationStatusPending, fetchByDedup(t, g, b).Status)
+}
+
+func TestDeleteSentBefore(t *testing.T) {
+ truncateIncidents(t)
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ old := time.Now().UTC().Add(-40 * 24 * time.Hour)
+ enqueueWithState(t, d, g, incID, "oldsent@com.com", map[string]any{"status": db.NotificationStatusSent, "updated_at": old})
+ enqueueWithState(t, d, g, incID, "newsent@com.com", map[string]any{"status": db.NotificationStatusSent})
+ enqueueWithState(t, d, g, incID, "failed@com.com", map[string]any{"status": db.NotificationStatusFailed, "updated_at": old})
+
+ cutoff := time.Now().UTC().Add(-30 * 24 * time.Hour)
+ deleted, err := d.DeleteSentBefore(ctx, cutoff, 500)
+ require.NoError(t, err)
+ assert.Equal(t, int64(1), deleted, "only the old sent row is pruned")
+
+ var remaining int64
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).Where("incident_id = ?", incID).Count(&remaining).Error)
+ assert.Equal(t, int64(2), remaining, "recent sent + failed kept")
+}
+
+// --- API endpoints ---
+
+func initNotifOpsRouter(t *testing.T) (*gin.Engine, *db.DB, *gorm.DB) {
+ t.Helper()
+
+ d, err := db.New(&conf.Config{DB: databaseURL})
+ require.NoError(t, err)
+ t.Cleanup(func() { _ = d.Close() })
+
+ g, err := gorm.Open(gormpostgres.New(gormpostgres.Config{DSN: databaseURL}), &gorm.Config{})
+ require.NoError(t, err)
+ sqlDB, err := g.DB()
+ require.NoError(t, err)
+ sqlDB.SetMaxOpenConns(2)
+ t.Cleanup(func() { _ = sqlDB.Close() })
+
+ gin.SetMode(gin.TestMode)
+ r := gin.Default()
+ r.NoRoute(apiErrors.Return404)
+ r.Use(api.ErrorHandle())
+
+ logger := zap.NewNop()
+ prov := &auth.Provider{}
+ rbacSvc := rbac.New(creatorGroup, operatorGroup, adminGroup)
+
+ v2Api := r.Group("v2")
+ v2Api.GET("notifications/stats",
+ api.AuthenticationMW(prov, logger, testHMACSecret),
+ api.RBACAuthorizationMW(rbacSvc, logger),
+ v2.GetNotificationStatsHandler(d, logger))
+ v2Api.POST("notifications/redrive",
+ api.AuthenticationMW(prov, logger, testHMACSecret),
+ api.RBACAuthorizationMW(rbacSvc, logger),
+ v2.RedriveNotificationsHandler(d, logger))
+ v2Api.GET("notifications/failed",
+ api.AuthenticationMW(prov, logger, testHMACSecret),
+ api.RBACAuthorizationMW(rbacSvc, logger),
+ v2.GetFailedNotificationsHandler(d, logger))
+
+ return r, d, g
+}
+
+func TestAPI_ListNotifications_StatusAndLimit(t *testing.T) {
+ truncateIncidents(t)
+ r, d, g := initNotifOpsRouter(t)
+ incID := seedIncident(t, d)
+ enqueueWithState(t, d, g, incID, "stuck@com.com", map[string]any{"attempts": 3})
+ enqueueWithState(t, d, g, incID, "dead@com.com",
+ map[string]any{"status": db.NotificationStatusFailed, "last_error": "x"})
+
+ list := func(query string) (int, []db.NotificationOutbox) {
+ t.Helper()
+ w := httptest.NewRecorder()
+ req, _ := http.NewRequest(http.MethodGet, "/v2/notifications/failed"+query, nil)
+ req.Header.Set("Authorization", "Bearer "+adminToken)
+ r.ServeHTTP(w, req)
+
+ var body struct {
+ Data []db.NotificationOutbox `json:"data"`
+ }
+ _ = json.Unmarshal(w.Body.Bytes(), &body)
+ return w.Code, body.Data
+ }
+
+ t.Run("defaults to failed", func(t *testing.T) {
+ code, rows := list("")
+ require.Equal(t, http.StatusOK, code)
+ require.Len(t, rows, 1)
+ assert.Equal(t, "dead@com.com", rows[0].Recipient)
+ })
+
+ t.Run("pending rows are reachable", func(t *testing.T) {
+ code, rows := list("?status=pending")
+ require.Equal(t, http.StatusOK, code)
+ require.Len(t, rows, 1)
+ assert.Equal(t, "stuck@com.com", rows[0].Recipient)
+ })
+
+ t.Run("limit caps the page", func(t *testing.T) {
+ code, rows := list("?status=pending&limit=1")
+ require.Equal(t, http.StatusOK, code)
+ assert.Len(t, rows, 1)
+ })
+
+ t.Run("unknown status is rejected", func(t *testing.T) {
+ code, _ := list("?status=bogus")
+ assert.Equal(t, http.StatusBadRequest, code)
+ })
+
+ t.Run("out-of-range limit is rejected", func(t *testing.T) {
+ code, _ := list("?limit=0")
+ assert.Equal(t, http.StatusBadRequest, code)
+ code, _ = list("?limit=100000")
+ assert.Equal(t, http.StatusBadRequest, code)
+ })
+}
+
+func TestAPI_NotificationStats_AdminOK(t *testing.T) {
+ truncateIncidents(t)
+ r, d, g := initNotifOpsRouter(t)
+ incID := seedIncident(t, d)
+ enqueueWithState(t, d, g, incID, "f@com.com", map[string]any{"status": db.NotificationStatusFailed, "last_error": "x"})
+
+ w := httptest.NewRecorder()
+ req, _ := http.NewRequest(http.MethodGet, "/v2/notifications/stats", nil)
+ req.Header.Set("Authorization", "Bearer "+adminToken)
+ r.ServeHTTP(w, req)
+
+ require.Equal(t, http.StatusOK, w.Code)
+ var stats db.NotificationStats
+ require.NoError(t, json.Unmarshal(w.Body.Bytes(), &stats))
+ assert.Equal(t, int64(1), stats.Failed)
+}
+
+func TestAPI_NotificationStats_NonAdminForbidden(t *testing.T) {
+ truncateIncidents(t)
+ r, _, _ := initNotifOpsRouter(t)
+
+ w := httptest.NewRecorder()
+ req, _ := http.NewRequest(http.MethodGet, "/v2/notifications/stats", nil)
+ req.Header.Set("Authorization", "Bearer "+creatorTokenA)
+ r.ServeHTTP(w, req)
+
+ assert.Equal(t, http.StatusForbidden, w.Code)
+}
+
+func TestAPI_RedriveNotifications_Admin(t *testing.T) {
+ truncateIncidents(t)
+ r, d, g := initNotifOpsRouter(t)
+ incID := seedIncident(t, d)
+ dedup := enqueueWithState(t, d, g, incID, "f@com.com",
+ map[string]any{"status": db.NotificationStatusFailed, "attempts": 5, "last_error": "x"})
+
+ w := httptest.NewRecorder()
+ req, _ := http.NewRequest(http.MethodPost, "/v2/notifications/redrive", bytes.NewReader([]byte(`{}`)))
+ req.Header.Set("Authorization", "Bearer "+adminToken)
+ r.ServeHTTP(w, req)
+
+ require.Equal(t, http.StatusOK, w.Code)
+ var resp struct {
+ Redriven int64 `json:"redriven"`
+ }
+ require.NoError(t, json.Unmarshal(w.Body.Bytes(), &resp))
+ assert.Equal(t, int64(1), resp.Redriven)
+ assert.Equal(t, db.NotificationStatusPending, fetchByDedup(t, g, dedup).Status)
+}
diff --git a/tests/notifications_test.go b/tests/notifications_test.go
new file mode 100644
index 0000000..d1b3310
--- /dev/null
+++ b/tests/notifications_test.go
@@ -0,0 +1,310 @@
+package tests
+
+import (
+ "context"
+ "fmt"
+ "testing"
+ "time"
+
+ "github.com/google/uuid"
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ gormpostgres "gorm.io/driver/postgres"
+ "gorm.io/gorm"
+
+ "github.com/stackmon/otc-status-dashboard/internal/conf"
+ "github.com/stackmon/otc-status-dashboard/internal/db"
+)
+
+// newNotifDB returns the DB under test plus a raw gorm handle for seeding and
+// verification against the real Postgres container.
+func newNotifDB(t *testing.T) (*db.DB, *gorm.DB) {
+ t.Helper()
+
+ d, err := db.New(&conf.Config{DB: databaseURL})
+ require.NoError(t, err)
+ t.Cleanup(func() { _ = d.Close() })
+
+ g, err := gorm.Open(gormpostgres.New(gormpostgres.Config{DSN: databaseURL}), &gorm.Config{})
+ require.NoError(t, err)
+ sqlDB, err := g.DB()
+ require.NoError(t, err)
+ sqlDB.SetMaxOpenConns(2)
+ t.Cleanup(func() { _ = sqlDB.Close() })
+
+ return d, g
+}
+
+// seedIncident inserts a minimal maintenance incident to satisfy the outbox FK
+// and returns its id.
+func seedIncident(t *testing.T, d *db.DB) uint {
+ t.Helper()
+
+ text := "notif-test maintenance"
+ start := time.Now().UTC()
+ impact := 0
+ id, err := d.SaveIncident(&db.Incident{
+ Text: &text,
+ StartDate: &start,
+ Impact: &impact,
+ System: false,
+ Type: "maintenance",
+ })
+ require.NoError(t, err)
+ return id
+}
+
+// newOutboxRow builds a pending outbox row with a unique dedup key.
+func newOutboxRow(incidentID uint, recipient string) db.NotificationOutbox {
+ changeID := uuid.NewString()
+ return db.NotificationOutbox{
+ Kind: db.NotificationKindPendingReview,
+ IncidentID: incidentID,
+ Recipient: recipient,
+ Payload: map[string]any{"title": "test"},
+ ChangeID: changeID,
+ DedupKey: fmt.Sprintf("%s:%s:%s", changeID, db.NotificationKindPendingReview, recipient),
+ Status: db.NotificationStatusPending,
+ }
+}
+
+func fetchRow(t *testing.T, g *gorm.DB, id uint) db.NotificationOutbox {
+ t.Helper()
+ var row db.NotificationOutbox
+ require.NoError(t, g.First(&row, id).Error)
+ return row
+}
+
+func TestEnqueue_Success(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "creator@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, row))
+
+ var stored db.NotificationOutbox
+ require.NoError(t, g.Where("dedup_key = ?", row.DedupKey).First(&stored).Error)
+ assert.Equal(t, db.NotificationStatusPending, stored.Status)
+ assert.Equal(t, incID, stored.IncidentID)
+ assert.Equal(t, 0, stored.Attempts)
+ assert.Equal(t, map[string]any{"title": "test"}, stored.Payload)
+ // timestamptz stores UTC; the persisted instant must be recent.
+ assert.WithinDuration(t, time.Now().UTC(), stored.CreatedAt.UTC(), 30*time.Second)
+}
+
+func TestEnqueue_DuplicateDedupKey(t *testing.T) {
+ ctx := context.Background()
+ d, _ := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "creator@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, row))
+
+ err := d.Enqueue(ctx, nil, row)
+ require.ErrorIs(t, err, db.ErrNotificationDuplicate)
+}
+
+func TestEnqueue_MissingDedupKey(t *testing.T) {
+ ctx := context.Background()
+ d, _ := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "creator@com.com")
+ row.DedupKey = ""
+ require.Error(t, d.Enqueue(ctx, nil, row))
+}
+
+func TestClaimPending_MarksProcessing(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "ops@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, row))
+
+ claimed, err := d.ClaimPending(ctx, nil, 10, "pod-1", time.Minute)
+ require.NoError(t, err)
+ require.NotEmpty(t, claimed)
+
+ var target *db.NotificationOutbox
+ for i := range claimed {
+ if claimed[i].DedupKey == row.DedupKey {
+ target = &claimed[i]
+ break
+ }
+ }
+ require.NotNil(t, target, "enqueued row must be claimed")
+ assert.Equal(t, db.NotificationStatusProcessing, target.Status)
+ assert.Equal(t, 1, target.Attempts)
+ require.NotNil(t, target.LockedBy)
+ assert.Equal(t, "pod-1", *target.LockedBy)
+
+ stored := fetchRow(t, g, target.ID)
+ assert.Equal(t, db.NotificationStatusProcessing, stored.Status)
+ assert.Equal(t, 1, stored.Attempts)
+}
+
+func TestClaimPending_DoesNotReclaimProcessing(t *testing.T) {
+ ctx := context.Background()
+ d, _ := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "admin@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, row))
+
+ first, err := d.ClaimPending(ctx, nil, 10, "pod-1", time.Minute)
+ require.NoError(t, err)
+ require.NotEmpty(t, first)
+
+ // A second claim must not return the same row (already processing).
+ second, err := d.ClaimPending(ctx, nil, 10, "pod-2", time.Minute)
+ require.NoError(t, err)
+ for i := range second {
+ assert.NotEqual(t, row.DedupKey, second[i].DedupKey,
+ "row already processing must not be reclaimed")
+ }
+}
+
+func TestMarkSent_UpdatesStatusAndClearsLease(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "creator@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, row))
+ claimed, err := d.ClaimPending(ctx, nil, 10, "pod-1", time.Minute)
+ require.NoError(t, err)
+ id := findClaimedID(t, claimed, row.DedupKey)
+
+ require.NoError(t, d.MarkSent(ctx, nil, id))
+
+ stored := fetchRow(t, g, id)
+ assert.Equal(t, db.NotificationStatusSent, stored.Status)
+ assert.Nil(t, stored.LockedBy)
+ assert.Nil(t, stored.LockedAt)
+ assert.Nil(t, stored.LastError)
+}
+
+func TestMarkSent_NotFound(t *testing.T) {
+ ctx := context.Background()
+ d, _ := newNotifDB(t)
+ err := d.MarkSent(ctx, nil, 0)
+ require.ErrorIs(t, err, db.ErrNotificationNotFound)
+}
+
+func TestMarkFailed_RetryWhenAttemptsRemain(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "creator@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, row))
+ claimed, err := d.ClaimPending(ctx, nil, 10, "pod-1", time.Minute)
+ require.NoError(t, err)
+ id := findClaimedID(t, claimed, row.DedupKey) // attempts is now 1
+
+ retryAt := time.Now().UTC().Add(5 * time.Minute)
+ backoff := func(_ int) time.Time { return retryAt }
+ require.NoError(t, d.MarkFailed(ctx, nil, id, "smtp timeout", 5, backoff))
+
+ stored := fetchRow(t, g, id)
+ assert.Equal(t, db.NotificationStatusPending, stored.Status)
+ require.NotNil(t, stored.NextAttemptAt)
+ require.NotNil(t, stored.LastError)
+ assert.Equal(t, "smtp timeout", *stored.LastError)
+ assert.Nil(t, stored.LockedBy)
+}
+
+func TestMarkFailed_FinalWhenAttemptsExhausted(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "creator@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, row))
+ claimed, err := d.ClaimPending(ctx, nil, 10, "pod-1", time.Minute)
+ require.NoError(t, err)
+ id := findClaimedID(t, claimed, row.DedupKey) // attempts is now 1
+
+ backoff := func(_ int) time.Time { return time.Now().UTC().Add(time.Minute) }
+ // maxAttempts=1, current attempts=1 -> final failure.
+ require.NoError(t, d.MarkFailed(ctx, nil, id, "permanent", 1, backoff))
+
+ stored := fetchRow(t, g, id)
+ assert.Equal(t, db.NotificationStatusFailed, stored.Status)
+ assert.Nil(t, stored.NextAttemptAt)
+ require.NotNil(t, stored.LastError)
+ assert.Equal(t, "permanent", *stored.LastError)
+}
+
+func TestRecoverStaleProcessing_ReturnsToPending(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "creator@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, row))
+ claimed, err := d.ClaimPending(ctx, nil, 10, "pod-1", time.Minute)
+ require.NoError(t, err)
+ id := findClaimedID(t, claimed, row.DedupKey)
+
+ // Simulate a crashed pod: push locked_at far into the past.
+ stale := time.Now().UTC().Add(-10 * time.Minute)
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).Where("id = ?", id).
+ Update("locked_at", stale).Error)
+
+ recovered, err := d.RecoverStaleProcessing(ctx, nil, time.Minute, 5)
+ require.NoError(t, err)
+ require.True(t, containsID(recovered, id), "stale row must be recovered")
+
+ stored := fetchRow(t, g, id)
+ assert.Equal(t, db.NotificationStatusPending, stored.Status)
+ assert.Nil(t, stored.LockedBy)
+ assert.Nil(t, stored.LockedAt)
+ require.NotNil(t, stored.NextAttemptAt)
+}
+
+func TestRecoverStaleProcessing_FinalWhenAttemptsExhausted(t *testing.T) {
+ ctx := context.Background()
+ d, g := newNotifDB(t)
+ incID := seedIncident(t, d)
+
+ row := newOutboxRow(incID, "creator@com.com")
+ require.NoError(t, d.Enqueue(ctx, nil, row))
+ claimed, err := d.ClaimPending(ctx, nil, 10, "pod-1", time.Minute)
+ require.NoError(t, err)
+ id := findClaimedID(t, claimed, row.DedupKey) // attempts is now 1
+
+ stale := time.Now().UTC().Add(-10 * time.Minute)
+ require.NoError(t, g.Model(&db.NotificationOutbox{}).Where("id = ?", id).
+ Update("locked_at", stale).Error)
+
+ // maxAttempts=1 with attempts=1 -> recovery marks it failed.
+ recovered, err := d.RecoverStaleProcessing(ctx, nil, time.Minute, 1)
+ require.NoError(t, err)
+ require.True(t, containsID(recovered, id))
+
+ stored := fetchRow(t, g, id)
+ assert.Equal(t, db.NotificationStatusFailed, stored.Status)
+}
+
+func findClaimedID(t *testing.T, rows []db.NotificationOutbox, dedupKey string) uint {
+ t.Helper()
+ for i := range rows {
+ if rows[i].DedupKey == dedupKey {
+ return rows[i].ID
+ }
+ }
+ require.FailNow(t, "claimed row not found for dedup key: "+dedupKey)
+ return 0
+}
+
+func containsID(rows []db.NotificationOutbox, id uint) bool {
+ for i := range rows {
+ if rows[i].ID == id {
+ return true
+ }
+ }
+ return false
+}
diff --git a/tests/rbac_creation_test.go b/tests/rbac_creation_test.go
index c6c99ab..193a436 100644
--- a/tests/rbac_creation_test.go
+++ b/tests/rbac_creation_test.go
@@ -65,12 +65,12 @@ func TestCreation_IncidentByRoles(t *testing.T) {
func TestCreation_MaintenanceValidation(t *testing.T) {
r := initTestsWithHMAC(t)
- t.Run("missing contact_email rejected", func(t *testing.T) {
+ t.Run("missing contact_email accepted, HMAC token carries no email claim", func(t *testing.T) {
truncateIncidents(t)
data := maintenanceData()
data.ContactEmail = ""
w, _ := createEvent(t, r, data, creatorTokenA)
- assert.Equal(t, http.StatusBadRequest, w.Code)
+ assert.Equal(t, http.StatusOK, w.Code)
})
t.Run("invalid contact_email rejected", func(t *testing.T) {