First public release
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,3 @@
|
||||
export * from './sla.js'
|
||||
export { toReport, formatSla, describeAvailability, describeTarget } from './report.js'
|
||||
export type { SlaReportOptions } from './report.js'
|
||||
+108
@@ -0,0 +1,108 @@
|
||||
import {
|
||||
type CapabilityReport,
|
||||
type Finding,
|
||||
type FormatOptions,
|
||||
evidence,
|
||||
formatReport as renderReport,
|
||||
} from '@extant2000/evidence-record'
|
||||
import type { SlaReport } from './sla.js'
|
||||
|
||||
export interface SlaReportOptions {
|
||||
/** What the statement is about, e.g. "app.example.com". */
|
||||
subject?: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Say the null cases out loud.
|
||||
*
|
||||
* `uptime null, met=false` was correct and unreadable. Nobody should have to
|
||||
* know that zero samples is not 100% in order to read the line, and nobody
|
||||
* should have to know that a null target is not a missed target. The rigour
|
||||
* was already in the arithmetic; this is the phrasing catching up.
|
||||
*/
|
||||
export function describeAvailability(r: SlaReport): string {
|
||||
if (r.uptime_pct === null) {
|
||||
return 'No samples in this period; availability cannot be computed.'
|
||||
}
|
||||
return `Availability ${r.uptime_pct}% over ${r.total_checks} samples, ${r.downtime_minutes} minutes down.`
|
||||
}
|
||||
|
||||
export function describeTarget(r: SlaReport): string {
|
||||
if (r.target_pct === null) return 'No SLA target applies to this plan.'
|
||||
if (r.uptime_pct === null) {
|
||||
return `A ${r.target_pct}% target applies, but nothing was measured — the target was neither met nor missed.`
|
||||
}
|
||||
return r.met
|
||||
? `Target ${r.target_pct}% was met.`
|
||||
: `Target ${r.target_pct}% was NOT met.`
|
||||
}
|
||||
|
||||
const window = (r: SlaReport) =>
|
||||
r.from && r.to ? `${r.from.slice(0, 10)} → ${r.to.slice(0, 10)}` : 'unbounded period'
|
||||
|
||||
/**
|
||||
* Build the portfolio-standard report for one availability statement.
|
||||
*
|
||||
* Zero samples with a target set produces `not-assessed`, not `fail`. The
|
||||
* underlying `met: false` is preserved because the ported arithmetic is
|
||||
* unchanged, but reporting a MISSED SLA when nothing was measured is an
|
||||
* overclaim in the opposite direction from reporting 100% — and this library
|
||||
* exists to avoid exactly that kind of unearned certainty.
|
||||
*/
|
||||
export function toReport(r: SlaReport, opts: SlaReportOptions = {}): CapabilityReport {
|
||||
const subject = opts.subject ?? 'service'
|
||||
const findings: Finding[] = []
|
||||
|
||||
const measured = evidence.measurement('Availability', r.uptime_pct, r.total_checks, {
|
||||
unit: '%',
|
||||
window: window(r),
|
||||
})
|
||||
|
||||
findings.push({
|
||||
id: 'availability',
|
||||
summary: describeAvailability(r),
|
||||
determination: r.uptime_pct === null
|
||||
? 'not-assessed'
|
||||
: r.target_pct === null
|
||||
? 'not-applicable'
|
||||
: r.met ? 'pass' : 'fail',
|
||||
severity: r.uptime_pct === null ? 'high' : r.met === false ? 'critical' : 'info',
|
||||
detail: describeTarget(r),
|
||||
evidence: r.uptime_pct === null ? [] : [measured],
|
||||
})
|
||||
|
||||
if (r.excluded_checks > 0) {
|
||||
findings.push({
|
||||
id: 'maintenance',
|
||||
summary: `${r.excluded_minutes} minutes of scheduled maintenance excluded (${r.excluded_checks} samples)`,
|
||||
determination: 'not-applicable',
|
||||
severity: 'info',
|
||||
detail: 'Planned downtime inside a declared window is removed from the availability math, as if the probe never ran. A cancelled window does not excuse anything.',
|
||||
evidence: [evidence.measurement('Excluded maintenance', r.excluded_minutes, r.excluded_checks, {
|
||||
unit: ' min',
|
||||
window: window(r),
|
||||
})],
|
||||
})
|
||||
}
|
||||
|
||||
return {
|
||||
capability: 'ProofOfUp',
|
||||
scope: `${subject}, ${window(r)}`,
|
||||
// The sample count IS the examined count. Zero samples therefore forces
|
||||
// not-assessed at the report level too, without a special case.
|
||||
examined: r.total_checks,
|
||||
findings,
|
||||
// Both sentences go in the notes, not only on the finding: with zero
|
||||
// samples the renderer leads with "nothing was assessed" and stops before
|
||||
// the findings, and the reader still has to be told why.
|
||||
notes: [describeAvailability(r), describeTarget(r)],
|
||||
}
|
||||
}
|
||||
|
||||
/** Render an availability statement for a terminal, CI log or customer PDF. */
|
||||
export function formatSla(
|
||||
r: SlaReport,
|
||||
opts: SlaReportOptions & FormatOptions = {},
|
||||
): string {
|
||||
return renderReport(toReport(r, opts), opts)
|
||||
}
|
||||
+164
@@ -0,0 +1,164 @@
|
||||
/**
|
||||
* Availability math for uptime / SLA statements.
|
||||
*
|
||||
* Only the arithmetic lives here. Queries, auth, CSV and printable rendering
|
||||
* stay in the product that calls it.
|
||||
*/
|
||||
|
||||
export interface Check {
|
||||
/** Sample timestamp. */
|
||||
ts: Date | string | number
|
||||
/** Whether the probe succeeded. */
|
||||
ok: boolean
|
||||
}
|
||||
|
||||
export interface MaintenanceWindow {
|
||||
starts_at?: Date | string | number
|
||||
ends_at?: Date | string | number
|
||||
/** Aliases accepted for convenience. */
|
||||
start?: Date | string | number
|
||||
end?: Date | string | number
|
||||
/** A `cancelled` window is ignored — it did not happen. */
|
||||
status?: string
|
||||
}
|
||||
|
||||
export interface SlaOptions {
|
||||
from: Date | string | number
|
||||
to: Date | string | number
|
||||
/** Marketed availability target, e.g. 99.9. `null`/absent = no SLA. */
|
||||
target?: number | null
|
||||
/** Probe cadence in seconds. Bounds each failing sample's charged downtime. */
|
||||
intervalSeconds?: number
|
||||
excludeWindows?: MaintenanceWindow[]
|
||||
}
|
||||
|
||||
export interface SlaReport {
|
||||
from: string | null
|
||||
to: string | null
|
||||
total_checks: number
|
||||
up_checks: number
|
||||
down_checks: number
|
||||
/** `null` when there were no samples — NOT 100. */
|
||||
uptime_pct: number | null
|
||||
downtime_minutes: number
|
||||
excluded_checks: number
|
||||
excluded_minutes: number
|
||||
target_pct: number | null
|
||||
/** `null` when no target is set — "no SLA" is not "SLA met". */
|
||||
met: boolean | null
|
||||
}
|
||||
|
||||
function ms(v: unknown): number {
|
||||
if (v instanceof Date) return v.getTime()
|
||||
return new Date(v as string).getTime()
|
||||
}
|
||||
|
||||
function round1(n: number): number {
|
||||
return Math.round(n * 10) / 10
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute an availability report over already-loaded probe samples.
|
||||
*
|
||||
* Pure: no I/O, no clock. Given the same samples it returns the same numbers,
|
||||
* which is what makes an availability figure auditable evidence rather than a
|
||||
* reading taken at a moment.
|
||||
*
|
||||
* The window is half-open `[from, to)` — includes `from`, excludes `to` — so
|
||||
* adjacent monthly statements never double-count a boundary sample.
|
||||
*/
|
||||
export function slaReport(checks: Check[], opts: SlaOptions): SlaReport {
|
||||
const fromMs = ms(opts.from)
|
||||
const toMs = ms(opts.to)
|
||||
|
||||
const target = opts.target != null && Number.isFinite(Number(opts.target))
|
||||
? Number(opts.target)
|
||||
: null
|
||||
|
||||
const intervalSec = Number.isFinite(Number(opts.intervalSeconds)) && Number(opts.intervalSeconds) > 0
|
||||
? Number(opts.intervalSeconds)
|
||||
: 300
|
||||
const intervalMs = intervalSec * 1000
|
||||
|
||||
const fromIso = Number.isFinite(fromMs) ? new Date(fromMs).toISOString() : null
|
||||
const toIso = Number.isFinite(toMs) ? new Date(toMs).toISOString() : null
|
||||
|
||||
// ── Scheduled-maintenance exclusion ──────────────────────────────────────
|
||||
// Normalize windows into half-open [s, e) intervals clipped to the report
|
||||
// period, dropping cancelled ones. A sample inside such an interval is
|
||||
// removed from the availability math entirely — as if the probe never ran —
|
||||
// so planned downtime never counts against uptime_pct.
|
||||
const raw: Array<[number, number]> = []
|
||||
if (Array.isArray(opts.excludeWindows) && Number.isFinite(fromMs) && Number.isFinite(toMs)) {
|
||||
for (const w of opts.excludeWindows) {
|
||||
if (!w || w.status === 'cancelled') continue
|
||||
const s = ms(w.starts_at ?? w.start)
|
||||
const e = ms(w.ends_at ?? w.end)
|
||||
if (!Number.isFinite(s) || !Number.isFinite(e) || e <= s) continue
|
||||
const cs = Math.max(s, fromMs)
|
||||
const ce = Math.min(e, toMs)
|
||||
if (ce > cs) raw.push([cs, ce])
|
||||
}
|
||||
}
|
||||
|
||||
// Merge overlapping/adjacent intervals so excluded_minutes never
|
||||
// double-counts two windows that overlap.
|
||||
raw.sort((a, b) => a[0] - b[0])
|
||||
const merged: Array<[number, number]> = []
|
||||
for (const iv of raw) {
|
||||
const last = merged[merged.length - 1]
|
||||
if (last && iv[0] <= last[1]) last[1] = Math.max(last[1], iv[1])
|
||||
else merged.push([iv[0], iv[1]])
|
||||
}
|
||||
const isExcluded = (t: number) => merged.some(([s, e]) => t >= s && t < e)
|
||||
const excludedMs = merged.reduce((acc, [s, e]) => acc + (e - s), 0)
|
||||
|
||||
// ── Collect in-window samples ────────────────────────────────────────────
|
||||
const kept: Array<{ t: number, ok: boolean }> = []
|
||||
let excludedChecks = 0
|
||||
if (Array.isArray(checks) && Number.isFinite(fromMs) && Number.isFinite(toMs)) {
|
||||
for (const c of checks) {
|
||||
if (!c || c.ts == null) continue
|
||||
const t = ms(c.ts)
|
||||
if (!Number.isFinite(t) || t < fromMs || t >= toMs) continue
|
||||
if (isExcluded(t)) { excludedChecks++; continue }
|
||||
kept.push({ t, ok: !!c.ok })
|
||||
}
|
||||
}
|
||||
kept.sort((a, b) => a.t - b.t)
|
||||
|
||||
// ── Availability ─────────────────────────────────────────────────────────
|
||||
let up = 0
|
||||
let downtimeMs = 0
|
||||
for (let i = 0; i < kept.length; i++) {
|
||||
const cur = kept[i]!
|
||||
if (cur.ok) { up++; continue }
|
||||
// Charge the gap to the next sample, capped at one probe interval, so a
|
||||
// long reporting gap after a single failure cannot inflate downtime. The
|
||||
// last sample has no successor and is charged exactly one interval.
|
||||
const next = kept[i + 1]
|
||||
const gap = next ? next.t - cur.t : intervalMs
|
||||
downtimeMs += Math.min(Math.max(gap, 0), intervalMs)
|
||||
}
|
||||
|
||||
const total = kept.length
|
||||
// null, not 100: zero samples is no evidence, and an availability report that
|
||||
// answers "100%" when nothing was measured is worse than one that abstains.
|
||||
const pct = total === 0 ? null : round1((up / total) * 100)
|
||||
|
||||
return {
|
||||
from: fromIso,
|
||||
to: toIso,
|
||||
total_checks: total,
|
||||
up_checks: up,
|
||||
down_checks: total - up,
|
||||
uptime_pct: pct,
|
||||
downtime_minutes: round1(downtimeMs / 60000),
|
||||
excluded_checks: excludedChecks,
|
||||
excluded_minutes: round1(excludedMs / 60000),
|
||||
target_pct: target,
|
||||
// null, not false: "no SLA target" and "SLA missed" are different claims,
|
||||
// and a free plan must not produce a statement that reads as a failure.
|
||||
met: target == null ? null : (pct != null && pct >= target),
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user