First public release

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-28 14:56:08 -04:00
co-authored by Claude Opus 5.5
commit d2a07d5988
16 changed files with 2272 additions and 0 deletions
+3
View File
@@ -0,0 +1,3 @@
export * from './sla.js'
export { toReport, formatSla, describeAvailability, describeTarget } from './report.js'
export type { SlaReportOptions } from './report.js'
+108
View File
@@ -0,0 +1,108 @@
import {
type CapabilityReport,
type Finding,
type FormatOptions,
evidence,
formatReport as renderReport,
} from '@extant2000/evidence-record'
import type { SlaReport } from './sla.js'
export interface SlaReportOptions {
/** What the statement is about, e.g. "app.example.com". */
subject?: string
}
/**
* Say the null cases out loud.
*
* `uptime null, met=false` was correct and unreadable. Nobody should have to
* know that zero samples is not 100% in order to read the line, and nobody
* should have to know that a null target is not a missed target. The rigour
* was already in the arithmetic; this is the phrasing catching up.
*/
export function describeAvailability(r: SlaReport): string {
if (r.uptime_pct === null) {
return 'No samples in this period; availability cannot be computed.'
}
return `Availability ${r.uptime_pct}% over ${r.total_checks} samples, ${r.downtime_minutes} minutes down.`
}
export function describeTarget(r: SlaReport): string {
if (r.target_pct === null) return 'No SLA target applies to this plan.'
if (r.uptime_pct === null) {
return `A ${r.target_pct}% target applies, but nothing was measured — the target was neither met nor missed.`
}
return r.met
? `Target ${r.target_pct}% was met.`
: `Target ${r.target_pct}% was NOT met.`
}
const window = (r: SlaReport) =>
r.from && r.to ? `${r.from.slice(0, 10)} → ${r.to.slice(0, 10)}` : 'unbounded period'
/**
* Build the portfolio-standard report for one availability statement.
*
* Zero samples with a target set produces `not-assessed`, not `fail`. The
* underlying `met: false` is preserved because the ported arithmetic is
* unchanged, but reporting a MISSED SLA when nothing was measured is an
* overclaim in the opposite direction from reporting 100% — and this library
* exists to avoid exactly that kind of unearned certainty.
*/
export function toReport(r: SlaReport, opts: SlaReportOptions = {}): CapabilityReport {
const subject = opts.subject ?? 'service'
const findings: Finding[] = []
const measured = evidence.measurement('Availability', r.uptime_pct, r.total_checks, {
unit: '%',
window: window(r),
})
findings.push({
id: 'availability',
summary: describeAvailability(r),
determination: r.uptime_pct === null
? 'not-assessed'
: r.target_pct === null
? 'not-applicable'
: r.met ? 'pass' : 'fail',
severity: r.uptime_pct === null ? 'high' : r.met === false ? 'critical' : 'info',
detail: describeTarget(r),
evidence: r.uptime_pct === null ? [] : [measured],
})
if (r.excluded_checks > 0) {
findings.push({
id: 'maintenance',
summary: `${r.excluded_minutes} minutes of scheduled maintenance excluded (${r.excluded_checks} samples)`,
determination: 'not-applicable',
severity: 'info',
detail: 'Planned downtime inside a declared window is removed from the availability math, as if the probe never ran. A cancelled window does not excuse anything.',
evidence: [evidence.measurement('Excluded maintenance', r.excluded_minutes, r.excluded_checks, {
unit: ' min',
window: window(r),
})],
})
}
return {
capability: 'ProofOfUp',
scope: `${subject}, ${window(r)}`,
// The sample count IS the examined count. Zero samples therefore forces
// not-assessed at the report level too, without a special case.
examined: r.total_checks,
findings,
// Both sentences go in the notes, not only on the finding: with zero
// samples the renderer leads with "nothing was assessed" and stops before
// the findings, and the reader still has to be told why.
notes: [describeAvailability(r), describeTarget(r)],
}
}
/** Render an availability statement for a terminal, CI log or customer PDF. */
export function formatSla(
r: SlaReport,
opts: SlaReportOptions & FormatOptions = {},
): string {
return renderReport(toReport(r, opts), opts)
}
+164
View File
@@ -0,0 +1,164 @@
/**
* Availability math for uptime / SLA statements.
*
* Only the arithmetic lives here. Queries, auth, CSV and printable rendering
* stay in the product that calls it.
*/
export interface Check {
/** Sample timestamp. */
ts: Date | string | number
/** Whether the probe succeeded. */
ok: boolean
}
export interface MaintenanceWindow {
starts_at?: Date | string | number
ends_at?: Date | string | number
/** Aliases accepted for convenience. */
start?: Date | string | number
end?: Date | string | number
/** A `cancelled` window is ignored — it did not happen. */
status?: string
}
export interface SlaOptions {
from: Date | string | number
to: Date | string | number
/** Marketed availability target, e.g. 99.9. `null`/absent = no SLA. */
target?: number | null
/** Probe cadence in seconds. Bounds each failing sample's charged downtime. */
intervalSeconds?: number
excludeWindows?: MaintenanceWindow[]
}
export interface SlaReport {
from: string | null
to: string | null
total_checks: number
up_checks: number
down_checks: number
/** `null` when there were no samples — NOT 100. */
uptime_pct: number | null
downtime_minutes: number
excluded_checks: number
excluded_minutes: number
target_pct: number | null
/** `null` when no target is set — "no SLA" is not "SLA met". */
met: boolean | null
}
function ms(v: unknown): number {
if (v instanceof Date) return v.getTime()
return new Date(v as string).getTime()
}
function round1(n: number): number {
return Math.round(n * 10) / 10
}
/**
* Compute an availability report over already-loaded probe samples.
*
* Pure: no I/O, no clock. Given the same samples it returns the same numbers,
* which is what makes an availability figure auditable evidence rather than a
* reading taken at a moment.
*
* The window is half-open `[from, to)` — includes `from`, excludes `to` — so
* adjacent monthly statements never double-count a boundary sample.
*/
export function slaReport(checks: Check[], opts: SlaOptions): SlaReport {
const fromMs = ms(opts.from)
const toMs = ms(opts.to)
const target = opts.target != null && Number.isFinite(Number(opts.target))
? Number(opts.target)
: null
const intervalSec = Number.isFinite(Number(opts.intervalSeconds)) && Number(opts.intervalSeconds) > 0
? Number(opts.intervalSeconds)
: 300
const intervalMs = intervalSec * 1000
const fromIso = Number.isFinite(fromMs) ? new Date(fromMs).toISOString() : null
const toIso = Number.isFinite(toMs) ? new Date(toMs).toISOString() : null
// ── Scheduled-maintenance exclusion ──────────────────────────────────────
// Normalize windows into half-open [s, e) intervals clipped to the report
// period, dropping cancelled ones. A sample inside such an interval is
// removed from the availability math entirely — as if the probe never ran —
// so planned downtime never counts against uptime_pct.
const raw: Array<[number, number]> = []
if (Array.isArray(opts.excludeWindows) && Number.isFinite(fromMs) && Number.isFinite(toMs)) {
for (const w of opts.excludeWindows) {
if (!w || w.status === 'cancelled') continue
const s = ms(w.starts_at ?? w.start)
const e = ms(w.ends_at ?? w.end)
if (!Number.isFinite(s) || !Number.isFinite(e) || e <= s) continue
const cs = Math.max(s, fromMs)
const ce = Math.min(e, toMs)
if (ce > cs) raw.push([cs, ce])
}
}
// Merge overlapping/adjacent intervals so excluded_minutes never
// double-counts two windows that overlap.
raw.sort((a, b) => a[0] - b[0])
const merged: Array<[number, number]> = []
for (const iv of raw) {
const last = merged[merged.length - 1]
if (last && iv[0] <= last[1]) last[1] = Math.max(last[1], iv[1])
else merged.push([iv[0], iv[1]])
}
const isExcluded = (t: number) => merged.some(([s, e]) => t >= s && t < e)
const excludedMs = merged.reduce((acc, [s, e]) => acc + (e - s), 0)
// ── Collect in-window samples ────────────────────────────────────────────
const kept: Array<{ t: number, ok: boolean }> = []
let excludedChecks = 0
if (Array.isArray(checks) && Number.isFinite(fromMs) && Number.isFinite(toMs)) {
for (const c of checks) {
if (!c || c.ts == null) continue
const t = ms(c.ts)
if (!Number.isFinite(t) || t < fromMs || t >= toMs) continue
if (isExcluded(t)) { excludedChecks++; continue }
kept.push({ t, ok: !!c.ok })
}
}
kept.sort((a, b) => a.t - b.t)
// ── Availability ─────────────────────────────────────────────────────────
let up = 0
let downtimeMs = 0
for (let i = 0; i < kept.length; i++) {
const cur = kept[i]!
if (cur.ok) { up++; continue }
// Charge the gap to the next sample, capped at one probe interval, so a
// long reporting gap after a single failure cannot inflate downtime. The
// last sample has no successor and is charged exactly one interval.
const next = kept[i + 1]
const gap = next ? next.t - cur.t : intervalMs
downtimeMs += Math.min(Math.max(gap, 0), intervalMs)
}
const total = kept.length
// null, not 100: zero samples is no evidence, and an availability report that
// answers "100%" when nothing was measured is worse than one that abstains.
const pct = total === 0 ? null : round1((up / total) * 100)
return {
from: fromIso,
to: toIso,
total_checks: total,
up_checks: up,
down_checks: total - up,
uptime_pct: pct,
downtime_minutes: round1(downtimeMs / 60000),
excluded_checks: excludedChecks,
excluded_minutes: round1(excludedMs / 60000),
target_pct: target,
// null, not false: "no SLA target" and "SLA missed" are different claims,
// and a free plan must not produce a statement that reads as a failure.
met: target == null ? null : (pct != null && pct >= target),
}
}