Request latency
p50, p95 and p99 by hour of day over every timed request in the gateway log, and the band p95 stayed inside; reads latency().
Preview
"use client"
import { ChartCard } from "@/components/ui/chart-card"
import { LineChart } from "@/components/ui/line-chart"
import { LiveIndicator } from "@/components/ui/live-indicator"
import { type LatencyPoint } from "../data"
// The three percentiles this console reads, and the colours they keep.
const SERIES = [
{ key: "p50", label: "p50", color: "chart-3" as const },
{ key: "p95", label: "p95", color: "chart-1" as const },
{ key: "p99", label: "p99", color: "chart-5" as const },
]
export type LatencyChartProps = {
points: LatencyPoint[]
/** How many days of requests the percentiles are taken over. */
days: number
/** The band p95 stayed inside, for the footer. */
band: { low: number; high: number }
}
/** Latency by hour of day, three percentiles. The points arrive computed from the server page. */
export function LatencyChart({ points, days, band }: LatencyChartProps) {
return (
<ChartCard
data-widget="widget-engineering-monitoring-latency-chart"
title="Request latency"
description={`By hour of day, over the last ${days} days of requests`}
height={300}
className="h-full"
actions={<LiveIndicator label="Live" />}
footer={`p95 stayed between ${band.low}ms and ${band.high}ms.`}
>
<LineChart
data={points}
index="hour"
series={SERIES}
height={300}
showYAxis
strokeWidth={1.75}
valueFormatter={(value) => `${Math.round(value)}ms`}
/>
</ChartCard>
)
}Install
$
npx shadcn@latest add @vibra/widget-engineering-monitoring-latency-chartNeeds the @vibra registry in your components.json — set it up once.
Source
"use client"
import { ChartCard } from "@/components/ui/chart-card"
import { LineChart } from "@/components/ui/line-chart"
import { LiveIndicator } from "@/components/ui/live-indicator"
import { type LatencyPoint } from "../data"
// The three percentiles this console reads, and the colours they keep.
const SERIES = [
{ key: "p50", label: "p50", color: "chart-3" as const },
{ key: "p95", label: "p95", color: "chart-1" as const },
{ key: "p99", label: "p99", color: "chart-5" as const },
]
export type LatencyChartProps = {
points: LatencyPoint[]
/** How many days of requests the percentiles are taken over. */
days: number
/** The band p95 stayed inside, for the footer. */
band: { low: number; high: number }
}
/** Latency by hour of day, three percentiles. The points arrive computed from the server page. */
export function LatencyChart({ points, days, band }: LatencyChartProps) {
return (
<ChartCard
data-widget="widget-engineering-monitoring-latency-chart"
title="Request latency"
description={`By hour of day, over the last ${days} days of requests`}
height={300}
className="h-full"
actions={<LiveIndicator label="Live" />}
footer={`p95 stayed between ${band.low}ms and ${band.high}ms.`}
>
<LineChart
data={points}
index="hour"
series={SERIES}
height={300}
showYAxis
strokeWidth={1.75}
valueFormatter={(value) => `${Math.round(value)}ms`}
/>
</ChartCard>
)
}/**
* What this page reads. Services, incidents and the gateway log come straight
* from `db`; latency and the error grid are counted out of the same log lines,
* bucketed by hour. Cluster load is the one thing no entity records, so it
* comes off `seeded("dashboard-03")`. "Now" is `REFERENCE_DATE`.
*/
import { getInitials } from "@/lib/format"
import { db, REFERENCE_DATE, seeded, type Incident, type Member, type Service } from "@/lib/sample-data"
import { type IncidentSeverity } from "@/components/ui/incident-list"
import { type ServiceStatus } from "@/components/ui/system-status"
const SERVICES = db.services.all()
const LOGS = db.logEntries.all()
// Five status words in the data, four on a status page: both kinds of outage
// read as "outage" there.
const STATUS: Record<Service["status"], ServiceStatus> = {
operational: "operational",
degraded: "degraded",
partial_outage: "outage",
major_outage: "outage",
maintenance: "maintenance",
}
/** Every service, with the region it is served from. */
export function services() {
return SERVICES.map((service) => ({
name: service.name,
status: STATUS[service.status],
description: service.region,
}))
}
/** How many regions the services run across, for the page's meta line. */
export function fleetSummary(): string {
const regions = new Set(SERVICES.map((service) => service.region))
return `${SERVICES.length} services across ${regions.size} regions`
}
const DAY_MS = 86_400_000
export type UptimeStatus = "up" | "degraded" | "down"
/** A day is up above 99.5%, degraded down to 97%, and down below that. */
function dayStatus(uptime: number): { status: UptimeStatus; label: string } {
if (uptime >= 99.5) return { status: "up", label: "No incidents" }
if (uptime >= 97) return { status: "degraded", label: "Degraded for part of the day" }
return { status: "down", label: "Down" }
}
const UPTIME_COUNT = 3
/** Ninety days of daily uptime for the services that had the worst of it. */
export function uptime() {
const average = (service: Service) =>
service.uptime90d.reduce((sum, day) => sum + day, 0) / service.uptime90d.length
return [...SERVICES]
.sort((a, b) => average(a) - average(b))
.slice(0, UPTIME_COUNT)
.map((service) => {
const start = REFERENCE_DATE.getTime() - (service.uptime90d.length - 1) * DAY_MS
return {
label: service.name,
// Stated rather than left to UptimeBar's own count: the strip marks
// whole days, and this is the minutes the probes actually missed.
uptime: Math.round(average(service) * 100) / 100,
days: service.uptime90d.map((value, index) => ({
date: new Date(start + index * DAY_MS).toISOString().slice(0, 10),
...dayStatus(value),
})),
}
})
}
/** The `p`th percentile of `values`, nearest-rank; 0 for an empty list. */
function percentile(values: number[], p: number): number {
if (values.length === 0) return 0
const sorted = [...values].sort((a, b) => a - b)
return sorted[Math.min(sorted.length - 1, Math.floor(p * sorted.length))]
}
const HOURS = Array.from({ length: 24 }, (_, hour) => hour)
// Only a line that carries a request id was timed, so those are the ones the
// percentiles are taken over.
const TIMED = LOGS.filter((line) => line.durationMs !== undefined)
export type LatencyPoint = { hour: string; p50: number; p95: number; p99: number }
/** Request latency by hour of day, over every timed line in the log. */
export function latency(): LatencyPoint[] {
return HOURS.map((hour) => {
const durations = TIMED.filter((line) => line.at.getUTCHours() === hour).map(
(line) => line.durationMs as number
)
return {
hour: `${String(hour).padStart(2, "0")}:00`,
p50: percentile(durations, 0.5),
p95: percentile(durations, 0.95),
p99: percentile(durations, 0.99),
}
})
}
/** How many days of log the percentiles and the grid below are taken over. */
export function logWindowDays(): number {
const oldest = Math.min(...LOGS.map((line) => line.at.getTime()))
return Math.round((REFERENCE_DATE.getTime() - oldest) / DAY_MS)
}
/** The band p95 stayed inside, for the chart's footer. */
export function p95Band(): { low: number; high: number } {
const values = latency().map((point) => point.p95)
return { low: Math.min(...values), high: Math.max(...values) }
}
export const ERROR_HOURS = HOURS.map((hour) => String(hour).padStart(2, "0"))
export const ERROR_DAYS = ["Mon", "Tue", "Wed", "Thu", "Fri", "Sat", "Sun"]
/** Errors the log recorded, by weekday and hour. */
export function errorsByHour(): number[][] {
const grid = ERROR_DAYS.map(() => HOURS.map(() => 0))
for (const line of LOGS) {
if (line.level !== "error") continue
// getUTCDay is Sunday-first; the grid reads Monday-first.
grid[(line.at.getUTCDay() + 6) % 7][line.at.getUTCHours()] += 1
}
return grid
}
/** The worst block in the grid, named the way the footer says it. */
export function worstErrorBlock(): { label: string; count: number } {
const grid = errorsByHour()
let best = { label: `${ERROR_DAYS[0]} 00:00`, count: -1 }
grid.forEach((row, day) =>
row.forEach((count, hour) => {
if (count > best.count) best = { label: `${ERROR_DAYS[day]} ${ERROR_HOURS[hour]}:00`, count }
})
)
return best
}
export type ErrorGrid = {
/** The rows, Monday first. */
days: string[]
/** The columns, one an hour of the UTC day. */
hours: string[]
/** Errors a cell, `values[day][hour]`. */
values: number[][]
/** How many days of log the grid is counted over. */
windowDays: number
worst: { label: string; count: number }
}
/** The error grid with what its caption and its footer say, counted here for the heatmap to draw. */
export function errorGrid(): ErrorGrid {
return {
days: ERROR_DAYS,
hours: ERROR_HOURS,
values: errorsByHour(),
windowDays: logWindowDays(),
worst: worstErrorBlock(),
}
}
/** Cluster load right now. No entity records it, so it comes off the block's seed. */
export function resources() {
const rand = seeded("dashboard-03")
return [
{ label: "CPU", value: Math.round(52 + rand() * 18), thresholds: { warning: 70, danger: 90 } },
{ label: "Memory", value: Math.round(70 + rand() * 14), thresholds: { warning: 75, danger: 92 } },
{ label: "Disk", value: Math.round(34 + rand() * 14), thresholds: { warning: 80, danger: 92 } },
]
}
// StatusIndicator's vocabulary: "online" is healthy, "warning" is anything
// degraded or worse.
type RegionStatus = "online" | "warning"
/** Where traffic is served from, worst region first. */
export function regions() {
const names = [...new Set(SERVICES.map((service) => service.region))].sort()
return names
.map((name) => {
const here = SERVICES.filter((service) => service.region === name)
const healthy = here.every((service) => service.status === "operational")
const worst = Math.min(
...here.map((service) => service.uptime90d[service.uptime90d.length - 1])
)
return {
name,
status: (healthy ? "online" : "warning") as RegionStatus,
note: `${here.length} ${here.length === 1 ? "service" : "services"} · ${worst.toFixed(2)}% today`,
}
})
.sort((a, b) => Number(a.status === "online") - Number(b.status === "online"))
}
const SEVERITY: Record<Incident["severity"], IncidentSeverity> = {
sev1: "critical",
sev2: "major",
sev3: "minor",
}
const SERVICE_NAME_BY_ID = new Map(SERVICES.map((service) => [service.id, service.name]))
const INCIDENT_COUNT = 6
/** Open incidents first, then the ones that closed most recently. */
export function incidents() {
const open = (incident: Incident) => incident.status !== "resolved"
return [...db.incidents.all()]
.sort((a, b) => {
if (open(a) !== open(b)) return open(a) ? -1 : 1
return b.startedAt.getTime() - a.startedAt.getTime()
})
.slice(0, INCIDENT_COUNT)
.map((incident) => ({
id: incident.id,
title: incident.title,
severity: SEVERITY[incident.severity],
status: incident.status,
startedAt: incident.startedAt,
resolvedAt: incident.resolvedAt,
affected: incident.serviceIds.map((id) => SERVICE_NAME_BY_ID.get(id) ?? id),
}))
}
const LOG_LINES = 60
/** The tail of the gateway log, oldest of the tail first. */
export function logLines() {
return [...LOGS]
.sort((a, b) => a.at.getTime() - b.at.getTime())
.slice(-LOG_LINES)
.map((line) => ({
id: line.id,
timestamp: line.at,
level: line.level,
source: line.service,
message: line.requestId ? `${line.message} request=${line.requestId}` : line.message,
}))
}
function ownerRow(): Member {
return db.members.all().find((member) => member.role === "owner") ?? db.members.all()[0]
}
/** The person looking at the page: whoever owns this workspace. */
export function currentUser() {
const owner = ownerRow()
return { name: owner.name, email: owner.email, initials: getInitials(owner.name), avatarUrl: owner.avatarUrl }
}
/** The bell's contents: the newest notifications, unread first in the panel. */
export function shellNotifications() {
return db.notifications
.all()
.sort((a, b) => b.at.getTime() - a.at.getTime())
.slice(0, 6)
.map(({ id, title, description, at, read, href }) => ({ id, title, description, at, read, href }))
}Its page
On its page the card sits among the rest of the dashboard and shares its range and its data with them.
From the Monitoring console page