Skip to main content

headless_lms_server/domain/credit_registration/
health.rs

1//! The alert rules the credit registration dashboard renders.
2//!
3//! An alert carries identifiers and numbers, never prose: the study registry's own error text is
4//! written for an integrator and is not translated, so the frontend renders one key per alert id
5//! with these values interpolated. The thresholds travel with the alerts, not hardcoded twice.
6
7use headless_lms_models::credit_registrations::{
8    CreditRegistrationState, StuckRegistrationCount, StuckThresholds,
9};
10use headless_lms_models::library::credit_registration::materialize::get_unmaterialised_eligible_completions;
11use headless_lms_models::{ModelResult, prelude::*};
12use headless_lms_models::{
13    course_module_suotar_configurations, credit_registration_account_linking_emails,
14    credit_registration_events, credit_registration_phase_state,
15    credit_registration_roster_schedules, credit_registrations,
16    study_registry_student_number_conflicts, suotar_api_calls,
17};
18use utoipa::ToSchema;
19
20use crate::domain::system_health::HealthStatus;
21use chrono::TimeDelta;
22use headless_lms_credit_registration::CreditRegistrationPhase;
23use headless_lms_credit_registration::registry_health::max_study_registry_wait;
24use headless_lms_models::credit_registration_phase_state::CreditRegistrationPhaseState;
25
26/// Within this much of the past, one rejected credential is enough.
27const CREDENTIAL_REJECTION_WINDOW: TimeDelta = TimeDelta::hours(1);
28/// Long enough to hold three failed calls at the longest request timeout.
29const UNREACHABLE_WINDOW: TimeDelta = TimeDelta::hours(4);
30/// Below this the run is a bad minute rather than an outage.
31const UNREACHABLE_CONSECUTIVE_FAILURES: i64 = 3;
32const SERVICE_OUTAGE_WINDOW: TimeDelta = TimeDelta::hours(1);
33/// Below this many items the share below is one bad batch, not a signal.
34const SERVICE_OUTAGE_MIN_ITEMS: i64 = 10;
35const SERVICE_OUTAGE_FAILURE_SHARE_PERCENT: i64 = 30;
36/// The longest `submissionPending` asks verify to wait before polling again.
37const SUOTAR_PENDING_WAIT: TimeDelta = TimeDelta::days(1);
38const STUCK_THRESHOLDS: StuckThresholds = StuckThresholds {
39    stuck_ready_to_submit_secs: 2 * 60 * 60,
40    stuck_submitting_secs: 90 * 60,
41    stuck_awaiting_verification_secs: SUOTAR_PENDING_WAIT.num_seconds() + 2 * 60 * 60,
42    stuck_failed_retryable_secs: 3 * 24 * 60 * 60,
43};
44
45const _: () = assert!(
46    STUCK_THRESHOLDS.stuck_failed_retryable_secs
47        < headless_lms_models::library::credit_registration::backoff::SUBMIT_MAX_RETRY_AGE
48            .num_seconds(),
49    "a row must be considered stuck before backoff gives up retrying it"
50);
51const _: () = assert!(
52    STUCK_THRESHOLDS.stuck_submitting_secs
53        > headless_lms_models::library::credit_registration::backoff::SUBMITTING_RECOVERY_GRACE
54            .num_seconds(),
55    "the stuck threshold must outlast the grace period that lets a submit recover on its own"
56);
57/// Above this many stuck rows the backlog stops being something to look at tomorrow.
58const STUCK_CRITICAL_COUNT: i64 = 50;
59const LINKING_MAIL_WINDOW: TimeDelta = TimeDelta::days(7);
60/// A phase is late once this many of its own intervals have passed without a heartbeat.
61pub(crate) const PHASE_HEARTBEAT_INTERVAL_MULTIPLIER: i32 = 2;
62/// Failures in a row before a phase counts as broken rather than unlucky.
63pub(crate) const PHASE_CONSECUTIVE_FAILURE_LIMIT: i32 = 5;
64/// A phase that owns a nonempty queue and has not succeeded within this many of its own intervals
65/// is running without getting anywhere, which no failure count catches. Never less than its
66/// slowest possible iteration plus [`PHASE_SUCCESS_CALL_MARGIN`], or one slow call would look
67/// like a wedge.
68const PHASE_SUCCESS_INTERVAL_MULTIPLIER: i32 = 10;
69const PHASE_SUCCESS_CALL_MARGIN: TimeDelta = TimeDelta::minutes(10);
70/// The window every "in the last day" rule shares.
71const TERMINAL_WINDOW: TimeDelta = TimeDelta::days(1);
72const PERMANENT_FAILURE_COUNT: i64 = 20;
73const PERMANENT_FAILURE_RATE_PERCENT: i64 = 10;
74/// A reversal is always worth saying; this many at once is an incident.
75const MISREGISTRATION_CRITICAL_COUNT: i64 = 5;
76/// Linking mails one hour may hand over before the volume itself is the problem.
77const LINKING_MAIL_HOURLY_CAP: i64 = 500;
78const LINKING_MAIL_RATE_WINDOW: TimeDelta = TimeDelta::hours(1);
79/// Queued work that makes a day without a single completion mean something.
80const IDLE_QUEUE_DEPTH: i64 = 20;
81/// How long a completion may sit outside the ledger before `materialize` is the suspect rather
82/// than the clock.
83const NEVER_ENTERED_MIN_AGE: TimeDelta = TimeDelta::hours(6);
84/// Bounds the anti-join behind that rule; a bigger backlog reports as this many.
85const NEVER_ENTERED_SAMPLE_LIMIT: i64 = 100;
86const LATENCY_WINDOW: TimeDelta = TimeDelta::days(7);
87/// Under this the registry is quick enough that a doubling says nothing.
88const LATENCY_REGRESSION_FLOOR: TimeDelta = TimeDelta::hours(6);
89const LATENCY_REGRESSION_FACTOR: i64 = 2;
90
91#[derive(Debug, Serialize, Deserialize, PartialEq, Eq, Clone, Copy, ToSchema)]
92#[serde(rename_all = "snake_case")]
93pub enum CreditRegistrationAlertId {
94    CredentialsRejected,
95    StudyRegistryUnreachable,
96    ServiceUnavailable,
97    StuckRegistrations,
98    LinkingMailSendFailed,
99    LinkingMailRateCapExceeded,
100    PhaseHeartbeatStale,
101    PhaseFailing,
102    PermanentFailuresAccumulating,
103    MisregistrationsDetected,
104    CourseConfigurationBroken,
105    PipelineIdle,
106    CompletionsNeverEntered,
107    ConfirmationLatencyRegressed,
108    PipelinePausedGlobally,
109    StudyRegistryStudentNumberConflicts,
110    RosterCourseCodeFailing,
111}
112
113#[derive(Debug, Serialize, Deserialize, PartialEq, Eq, PartialOrd, Ord, Clone, Copy, ToSchema)]
114#[serde(rename_all = "snake_case")]
115pub enum CreditRegistrationAlertSeverity {
116    /// Worth knowing, not worth acting on. Never makes the overall status anything but healthy.
117    Info,
118    Warning,
119    Critical,
120}
121
122#[derive(Debug, Serialize, Deserialize, PartialEq, Clone, ToSchema)]
123pub struct CreditRegistrationAlert {
124    pub id: CreditRegistrationAlertId,
125    pub severity: CreditRegistrationAlertSeverity,
126    /// How many rows, calls or phases the rule found.
127    pub count: i64,
128    /// What `count` is out of, where the rule measured one. Not a threshold: thresholds are the
129    /// same for every evaluation and travel separately.
130    pub total: Option<i64>,
131    /// How far back the rule looked, where it looked back at all. `None` for a rule that reads the
132    /// live state or a threshold rather than a window; without it, two alerts counting the same
133    /// thing over different windows read as a contradiction.
134    pub window_secs: Option<i64>,
135    /// When it last happened, where the rule has an instant to point at.
136    pub at: Option<DateTime<Utc>>,
137    /// An identifier the operator can act on — a phase name, a ledger state, a mail domain. Never a
138    /// sentence, and never anything the study registry wrote.
139    pub subject: Option<String>,
140}
141
142#[derive(Debug, Serialize, Deserialize, PartialEq, Clone, ToSchema)]
143pub struct CreditRegistrationHealth {
144    pub status: HealthStatus,
145    /// Critical first, and a rejected credential first of all: nothing registers until it is fixed.
146    pub alerts: Vec<CreditRegistrationAlert>,
147    /// The only thresholds the frontend reads off the health poll: how long a row may sit in each
148    /// state before it counts as stuck. The other rule constants stay server-side.
149    pub thresholds: StuckThresholds,
150}
151
152pub fn stuck_thresholds() -> StuckThresholds {
153    STUCK_THRESHOLDS
154}
155
156/// A phase counts as late once more than [`PHASE_HEARTBEAT_INTERVAL_MULTIPLIER`] of its own
157/// interval has passed since its last heartbeat. A paused phase is never late: it is not expected
158/// to be heartbeating at all.
159pub(crate) fn is_heartbeat_late(
160    last_heartbeat_at: Option<DateTime<Utc>>,
161    expected_interval_secs: i32,
162    paused_at: Option<DateTime<Utc>>,
163    now: DateTime<Utc>,
164) -> bool {
165    paused_at.is_none()
166        && last_heartbeat_at.is_some_and(|at| {
167            (now - at).num_seconds()
168                > i64::from(expected_interval_secs) * i64::from(PHASE_HEARTBEAT_INTERVAL_MULTIPLIER)
169        })
170}
171
172/// Whether a phase counts as failing: too many failures in a row, or a nonempty queue with no
173/// success for too long, or an iteration hung past that same bound. Never while paused. The one
174/// definition behind both the `PhaseFailing` alert and the Workers tab's `failing` flag.
175///
176/// `depth_of` is the live count of a state; `due_enrolment_checks` as for
177/// [`CreditRegistrationPhase::queue_depth`].
178pub(crate) fn is_phase_failing(
179    phase: &CreditRegistrationPhaseState,
180    now: DateTime<Utc>,
181    depth_of: impl Fn(CreditRegistrationState) -> i64,
182    due_enrolment_checks: i64,
183) -> bool {
184    if phase.paused_at.is_some() {
185        return false;
186    }
187    let known_phase = CreditRegistrationPhase::from_phase_name(&phase.phase);
188    let owns_work =
189        known_phase.is_some_and(|known| known.queue_depth(&depth_of, due_enrolment_checks) > 0);
190    let slowest_iteration_secs = known_phase.map_or(0, |known| {
191        i64::try_from(max_study_registry_wait(known).as_secs()).unwrap_or(i64::MAX)
192            + PHASE_SUCCESS_CALL_MARGIN.num_seconds()
193    });
194    let unproductive_after_secs = (i64::from(phase.expected_interval_secs)
195        * i64::from(PHASE_SUCCESS_INTERVAL_MULTIPLIER))
196    .max(slowest_iteration_secs);
197    let unproductive = owns_work
198        && phase.last_success_at.is_some_and(|last_success_at| {
199            (now - last_success_at).num_seconds() > unproductive_after_secs
200        });
201    // The keep-alive refreshes the heartbeat for as long as an iteration runs, so a hung one
202    // shows only here.
203    let hung = phase.last_run_started_at.is_some_and(|started_at| {
204        phase
205            .last_run_finished_at
206            .is_none_or(|finished_at| finished_at < started_at)
207            && (now - started_at).num_seconds() > unproductive_after_secs
208    });
209    phase.consecutive_failures >= PHASE_CONSECUTIVE_FAILURE_LIMIT || unproductive || hung
210}
211
212/// Runs every rule and ranks what it found.
213///
214/// `stuck` and `depths` are passed in because the caller already reads both aggregates, the two
215/// most expensive reads in the request. `depths` is the live count per state, superseded rows
216/// excluded, as [`credit_registrations::count_by_state`] returns it.
217pub async fn evaluate(
218    conn: &mut PgConnection,
219    stuck: &[StuckRegistrationCount],
220    depths: &[(CreditRegistrationState, i64)],
221) -> ModelResult<CreditRegistrationHealth> {
222    let now = Utc::now();
223    let mut alerts = Vec::new();
224
225    let credentials = suotar_api_calls::count_credential_rejections_since(
226        conn,
227        now - CREDENTIAL_REJECTION_WINDOW,
228    )
229    .await?;
230    if credentials.count > 0 {
231        alerts.push(CreditRegistrationAlert {
232            id: CreditRegistrationAlertId::CredentialsRejected,
233            window_secs: Some(CREDENTIAL_REJECTION_WINDOW.num_seconds()),
234            severity: CreditRegistrationAlertSeverity::Critical,
235            count: credentials.count,
236            total: None,
237            at: credentials.last_at,
238            subject: None,
239        });
240    }
241
242    let unreachable =
243        suotar_api_calls::count_unreachable_run_since(conn, now - UNREACHABLE_WINDOW).await?;
244    if unreachable.count >= UNREACHABLE_CONSECUTIVE_FAILURES {
245        alerts.push(CreditRegistrationAlert {
246            id: CreditRegistrationAlertId::StudyRegistryUnreachable,
247            window_secs: Some(UNREACHABLE_WINDOW.num_seconds()),
248            severity: CreditRegistrationAlertSeverity::Critical,
249            count: unreachable.count,
250            total: None,
251            at: unreachable.last_at,
252            subject: None,
253        });
254    }
255
256    if let Some(alert) = service_outage_alert(conn, now).await? {
257        alerts.push(alert);
258    }
259    if let Some(alert) = stuck_alert(stuck) {
260        alerts.push(alert);
261    }
262    if let Some(alert) = linking_mail_alert(conn, now).await? {
263        alerts.push(alert);
264    }
265    if let Some(alert) = linking_mail_rate_alert(conn, now).await? {
266        alerts.push(alert);
267    }
268    alerts.extend(phase_alerts(conn, now, depths).await?);
269    alerts.extend(terminal_outcome_alerts(conn, now, depths).await?);
270    if let Some(alert) = course_configuration_alert(conn).await? {
271        alerts.push(alert);
272    }
273    if let Some(alert) = never_entered_alert(conn).await? {
274        alerts.push(alert);
275    }
276    if let Some(alert) = latency_regression_alert(conn, now).await? {
277        alerts.push(alert);
278    }
279    if let Some(alert) = study_registry_conflict_alert(conn).await? {
280        alerts.push(alert);
281    }
282    if let Some(alert) = roster_course_code_alert(conn).await? {
283        alerts.push(alert);
284    }
285
286    alerts.sort_by_key(|alert| {
287        (
288            std::cmp::Reverse(alert.severity),
289            alert.id != CreditRegistrationAlertId::CredentialsRejected,
290        )
291    });
292    let status = match alerts.iter().map(|alert| alert.severity).max() {
293        Some(CreditRegistrationAlertSeverity::Critical) => HealthStatus::Error,
294        Some(CreditRegistrationAlertSeverity::Warning) => HealthStatus::Warning,
295        Some(CreditRegistrationAlertSeverity::Info) | None => HealthStatus::Healthy,
296    };
297
298    Ok(CreditRegistrationHealth {
299        status,
300        alerts,
301        thresholds: stuck_thresholds(),
302    })
303}
304
305/// The share of recent items that failed on Suotar or Sisu being unavailable. Our only proxy for
306/// Sisu's uptime, which is why it is a rule of its own rather than part of the request-level one
307/// above.
308async fn service_outage_alert(
309    conn: &mut PgConnection,
310    now: DateTime<Utc>,
311) -> ModelResult<Option<CreditRegistrationAlert>> {
312    let totals =
313        credit_registration_events::count_item_outcomes_since(conn, now - SERVICE_OUTAGE_WINDOW)
314            .await?;
315    if totals.item_count < SERVICE_OUTAGE_MIN_ITEMS
316        || totals.service_unavailable_count * 100
317            < totals.item_count * SERVICE_OUTAGE_FAILURE_SHARE_PERCENT
318    {
319        return Ok(None);
320    }
321    Ok(Some(CreditRegistrationAlert {
322        id: CreditRegistrationAlertId::ServiceUnavailable,
323        window_secs: Some(SERVICE_OUTAGE_WINDOW.num_seconds()),
324        severity: CreditRegistrationAlertSeverity::Critical,
325        count: totals.service_unavailable_count,
326        total: Some(totals.item_count),
327        at: totals.last_service_unavailable_at,
328        subject: None,
329    }))
330}
331
332/// Rows the pipeline should have moved on by now. Terminal states are outside this by construction
333/// rather than by a filter that could be forgotten.
334fn stuck_alert(stuck: &[StuckRegistrationCount]) -> Option<CreditRegistrationAlert> {
335    let total: i64 = stuck.iter().map(|row| row.count).sum();
336    if total == 0 {
337        return None;
338    }
339    let severe: i64 = stuck.iter().map(|row| row.severely_stuck_count).sum();
340    let worst = stuck.iter().max_by_key(|row| row.count);
341    let severity = if total > STUCK_CRITICAL_COUNT || severe > 0 {
342        CreditRegistrationAlertSeverity::Critical
343    } else {
344        CreditRegistrationAlertSeverity::Warning
345    };
346    Some(CreditRegistrationAlert {
347        id: CreditRegistrationAlertId::StuckRegistrations,
348        window_secs: None,
349        severity,
350        count: total,
351        total: None,
352        at: worst.and_then(|row| row.oldest_state_entered_at),
353        subject: worst.map(|row| state_name(row.state)),
354    })
355}
356
357/// Linking mails we could not hand over at all. The recipient domain rides along because an
358/// undeliverable host is the usual cause.
359async fn linking_mail_alert(
360    conn: &mut PgConnection,
361    now: DateTime<Utc>,
362) -> ModelResult<Option<CreditRegistrationAlert>> {
363    let since = now - LINKING_MAIL_WINDOW;
364    let totals =
365        credit_registration_account_linking_emails::get_send_status_totals_since(conn, since, now)
366            .await?;
367    if totals.send_failed == 0 {
368        return Ok(None);
369    }
370    let top_domain = credit_registration_account_linking_emails::get_send_failure_domains_since(
371        conn, since, now,
372    )
373    .await?
374    .into_iter()
375    .next()
376    .map(|row| row.domain);
377    Ok(Some(CreditRegistrationAlert {
378        id: CreditRegistrationAlertId::LinkingMailSendFailed,
379        window_secs: Some(LINKING_MAIL_WINDOW.num_seconds()),
380        severity: CreditRegistrationAlertSeverity::Warning,
381        count: totals.send_failed,
382        total: Some(totals.mails_in_window),
383        at: totals.last_send_failed_at,
384        subject: top_domain,
385    }))
386}
387
388/// The volume guard: how many people we mailed in the last hour against what an hour should hold.
389/// Counts addresses, which is what the per-person caps govern.
390async fn linking_mail_rate_alert(
391    conn: &mut PgConnection,
392    now: DateTime<Utc>,
393) -> ModelResult<Option<CreditRegistrationAlert>> {
394    let sent = credit_registration_account_linking_emails::count_sent_since(
395        conn,
396        now - LINKING_MAIL_RATE_WINDOW,
397    )
398    .await?;
399    if sent <= LINKING_MAIL_HOURLY_CAP {
400        return Ok(None);
401    }
402    let severity = if sent > LINKING_MAIL_HOURLY_CAP * 2 {
403        CreditRegistrationAlertSeverity::Critical
404    } else {
405        CreditRegistrationAlertSeverity::Warning
406    };
407    Ok(Some(CreditRegistrationAlert {
408        id: CreditRegistrationAlertId::LinkingMailRateCapExceeded,
409        window_secs: Some(LINKING_MAIL_RATE_WINDOW.num_seconds()),
410        severity,
411        count: sent,
412        total: Some(LINKING_MAIL_HOURLY_CAP),
413        at: Some(now),
414        subject: None,
415    }))
416}
417
418/// What the phase table says about itself: phases that stopped reporting, and phases that report
419/// but get nowhere.
420///
421/// A phase that has never heartbeated is deliberately outside both: a freshly migrated database has
422/// no heartbeats at all, and that would keep the banner permanently red.
423async fn phase_alerts(
424    conn: &mut PgConnection,
425    now: DateTime<Utc>,
426    depths: &[(CreditRegistrationState, i64)],
427) -> ModelResult<Vec<CreditRegistrationAlert>> {
428    let phases = credit_registration_phase_state::get_all(conn).await?;
429    let due_enrolment_checks = credit_registrations::count_due_enrolment_checks(conn).await?;
430    let mut stale: Vec<&str> = Vec::new();
431    let mut failing: Vec<&str> = Vec::new();
432    let mut paused = 0;
433    let mut last_paused_at = None;
434    for phase in &phases {
435        if let Some(paused_at) = phase.paused_at {
436            paused += 1;
437            last_paused_at = last_paused_at.max(Some(paused_at));
438            continue;
439        }
440        if is_heartbeat_late(
441            phase.last_heartbeat_at,
442            phase.expected_interval_secs,
443            None,
444            now,
445        ) {
446            stale.push(&phase.phase);
447        }
448        if is_phase_failing(
449            phase,
450            now,
451            |state| depth_of(depths, state),
452            due_enrolment_checks,
453        ) {
454            failing.push(&phase.phase);
455        }
456    }
457
458    let mut alerts = Vec::new();
459    if !stale.is_empty() {
460        alerts.push(CreditRegistrationAlert {
461            id: CreditRegistrationAlertId::PhaseHeartbeatStale,
462            window_secs: None,
463            severity: CreditRegistrationAlertSeverity::Critical,
464            count: stale.len() as i64,
465            total: Some(phases.len() as i64),
466            at: None,
467            subject: stale.first().map(|phase| (*phase).to_string()),
468        });
469    }
470    if !failing.is_empty() {
471        alerts.push(CreditRegistrationAlert {
472            id: CreditRegistrationAlertId::PhaseFailing,
473            window_secs: None,
474            severity: CreditRegistrationAlertSeverity::Critical,
475            count: failing.len() as i64,
476            total: Some(phases.len() as i64),
477            at: None,
478            subject: failing.first().map(|phase| (*phase).to_string()),
479        });
480    }
481    if paused > 0 && paused == phases.len() {
482        alerts.push(CreditRegistrationAlert {
483            id: CreditRegistrationAlertId::PipelinePausedGlobally,
484            window_secs: None,
485            severity: CreditRegistrationAlertSeverity::Info,
486            count: paused as i64,
487            total: Some(phases.len() as i64),
488            at: last_paused_at,
489            subject: None,
490        });
491    }
492    Ok(alerts)
493}
494
495/// The three rules read off the last day's terminal outcomes: failures piling up, reversals, and a
496/// pipeline that finished nothing while holding work.
497async fn terminal_outcome_alerts(
498    conn: &mut PgConnection,
499    now: DateTime<Utc>,
500    depths: &[(CreditRegistrationState, i64)],
501) -> ModelResult<Vec<CreditRegistrationAlert>> {
502    let since = now - TERMINAL_WINDOW;
503    let totals = credit_registrations::count_terminal_outcomes_since(conn, since).await?;
504    let mut alerts = Vec::new();
505
506    let rate_broken = totals.total_count >= PERMANENT_FAILURE_COUNT
507        && totals.failed_permanent_count * 100
508            > totals.total_count * PERMANENT_FAILURE_RATE_PERCENT;
509    if totals.failed_permanent_count >= PERMANENT_FAILURE_COUNT || rate_broken {
510        alerts.push(CreditRegistrationAlert {
511            id: CreditRegistrationAlertId::PermanentFailuresAccumulating,
512            window_secs: Some(TERMINAL_WINDOW.num_seconds()),
513            severity: CreditRegistrationAlertSeverity::Warning,
514            count: totals.failed_permanent_count,
515            total: Some(totals.total_count),
516            at: Some(now),
517            subject: None,
518        });
519    }
520
521    let misregistered = credit_registrations::count_entered_state_since(
522        conn,
523        CreditRegistrationState::Misregistered,
524        since,
525    )
526    .await?;
527    if misregistered > 0 {
528        alerts.push(CreditRegistrationAlert {
529            id: CreditRegistrationAlertId::MisregistrationsDetected,
530            window_secs: Some(TERMINAL_WINDOW.num_seconds()),
531            severity: if misregistered >= MISREGISTRATION_CRITICAL_COUNT {
532                CreditRegistrationAlertSeverity::Critical
533            } else {
534                CreditRegistrationAlertSeverity::Warning
535            },
536            count: misregistered,
537            total: None,
538            at: Some(now),
539            subject: None,
540        });
541    }
542
543    let queued = depth_of(depths, CreditRegistrationState::ReadyToSubmit)
544        + depth_of(depths, CreditRegistrationState::AwaitingVerification)
545        + depth_of(depths, CreditRegistrationState::PartiallyRegistered);
546    if totals.total_count == 0 && queued > IDLE_QUEUE_DEPTH {
547        alerts.push(CreditRegistrationAlert {
548            id: CreditRegistrationAlertId::PipelineIdle,
549            window_secs: Some(TERMINAL_WINDOW.num_seconds()),
550            severity: CreditRegistrationAlertSeverity::Warning,
551            count: queued,
552            total: None,
553            at: Some(now),
554            subject: None,
555        });
556    }
557    Ok(alerts)
558}
559
560/// Modules the last configuration check found broken. Never checked is not counted: the Courses tab
561/// renders unknown and broken differently, and so must this.
562async fn course_configuration_alert(
563    conn: &mut PgConnection,
564) -> ModelResult<Option<CreditRegistrationAlert>> {
565    let count =
566        course_module_suotar_configurations::count_modules_failing_config_check(conn).await?;
567    Ok((count > 0).then_some(CreditRegistrationAlert {
568        id: CreditRegistrationAlertId::CourseConfigurationBroken,
569        window_secs: None,
570        severity: CreditRegistrationAlertSeverity::Warning,
571        count,
572        total: None,
573        at: None,
574        subject: None,
575    }))
576}
577
578/// Completions old enough that `materialize` has had every chance and still has no ledger row for
579/// them. Sampled rather than counted, so the anti-join stops early on a large backlog.
580async fn never_entered_alert(
581    conn: &mut PgConnection,
582) -> ModelResult<Option<CreditRegistrationAlert>> {
583    let found = get_unmaterialised_eligible_completions(
584        conn,
585        NEVER_ENTERED_MIN_AGE.num_seconds(),
586        NEVER_ENTERED_SAMPLE_LIMIT,
587    )
588    .await?;
589    if found.is_empty() {
590        return Ok(None);
591    }
592    Ok(Some(CreditRegistrationAlert {
593        id: CreditRegistrationAlertId::CompletionsNeverEntered,
594        window_secs: None,
595        severity: CreditRegistrationAlertSeverity::Warning,
596        count: found.len() as i64,
597        total: Some(NEVER_ENTERED_SAMPLE_LIMIT),
598        at: found.first().map(|row| row.created_at),
599        subject: None,
600    }))
601}
602
603/// How long the study registry is taking to confirm, this week against last. `count` is this
604/// week's p95 in seconds and `total` last week's, so the banner can name both.
605async fn latency_regression_alert(
606    conn: &mut PgConnection,
607    now: DateTime<Utc>,
608) -> ModelResult<Option<CreditRegistrationAlert>> {
609    let window = LATENCY_WINDOW;
610    let current =
611        credit_registrations::get_registration_latency_between(conn, now - window, now).await?;
612    let (Some(current_p95), true) = (current.p95_confirmation_secs, current.registered_count > 0)
613    else {
614        return Ok(None);
615    };
616    if current_p95 < LATENCY_REGRESSION_FLOOR.num_seconds() {
617        return Ok(None);
618    }
619    let previous = credit_registrations::get_registration_latency_between(
620        conn,
621        now - window * 2,
622        now - window,
623    )
624    .await?;
625    let Some(previous_p95) = previous
626        .p95_confirmation_secs
627        .filter(|_| previous.registered_count > 0)
628    else {
629        return Ok(None);
630    };
631    if current_p95 <= previous_p95 * LATENCY_REGRESSION_FACTOR {
632        return Ok(None);
633    }
634    Ok(Some(CreditRegistrationAlert {
635        id: CreditRegistrationAlertId::ConfirmationLatencyRegressed,
636        window_secs: Some(LATENCY_WINDOW.num_seconds()),
637        severity: CreditRegistrationAlertSeverity::Info,
638        count: current_p95,
639        total: Some(previous_p95),
640        at: Some(now),
641        subject: None,
642    }))
643}
644
645/// Accounts the study registry reported a number for that another live link kept us from linking.
646async fn study_registry_conflict_alert(
647    conn: &mut PgConnection,
648) -> ModelResult<Option<CreditRegistrationAlert>> {
649    let count = study_registry_student_number_conflicts::count_unresolved(conn).await?;
650    Ok((count > 0).then_some(CreditRegistrationAlert {
651        id: CreditRegistrationAlertId::StudyRegistryStudentNumberConflicts,
652        window_secs: None,
653        severity: CreditRegistrationAlertSeverity::Warning,
654        count,
655        total: None,
656        at: None,
657        subject: None,
658    }))
659}
660
661/// Course codes whose roster fails even when listed on their own, so enrolment discovery is backing
662/// them off. The worst one rides along, since the usual cause is that one code's configuration.
663async fn roster_course_code_alert(
664    conn: &mut PgConnection,
665) -> ModelResult<Option<CreditRegistrationAlert>> {
666    let failing = credit_registration_roster_schedules::get_failing_codes(conn).await?;
667    let Some(worst) = failing.first() else {
668        return Ok(None);
669    };
670    Ok(Some(CreditRegistrationAlert {
671        id: CreditRegistrationAlertId::RosterCourseCodeFailing,
672        window_secs: None,
673        severity: CreditRegistrationAlertSeverity::Warning,
674        count: i64::try_from(failing.len()).unwrap_or(i64::MAX),
675        total: None,
676        at: worst.last_attempted_at,
677        subject: Some(worst.course_code.clone()),
678    }))
679}
680
681fn depth_of(depths: &[(CreditRegistrationState, i64)], state: CreditRegistrationState) -> i64 {
682    depths
683        .iter()
684        .find(|(row_state, _)| *row_state == state)
685        .map_or(0, |(_, count)| *count)
686}
687
688/// The state's own wire name, taken from its serialisation so the two cannot drift.
689fn state_name(state: CreditRegistrationState) -> String {
690    serde_json::to_value(state)
691        .ok()
692        .and_then(|value| value.as_str().map(str::to_string))
693        .unwrap_or_default()
694}
695
696#[cfg(test)]
697mod tests {
698    use super::*;
699
700    /// The alert's `subject` has to be the same spelling the ledger and the filters use.
701    #[test]
702    fn a_state_names_itself_the_way_the_wire_does() {
703        assert_eq!(
704            state_name(CreditRegistrationState::AwaitingVerification),
705            "awaiting_verification"
706        );
707        for state in CreditRegistrationState::ALL {
708            assert!(!state_name(state).is_empty());
709        }
710    }
711
712    /// Info exists to be shown without turning the page red.
713    #[test]
714    fn severity_ranks_the_way_the_banner_reads_it() {
715        assert!(
716            CreditRegistrationAlertSeverity::Critical > CreditRegistrationAlertSeverity::Warning
717        );
718        assert!(CreditRegistrationAlertSeverity::Warning > CreditRegistrationAlertSeverity::Info);
719    }
720}