fix: report collected fleet metrics and distinguish unavailable readings

This commit is contained in:
archipelago
2026-10-05 20:31:40 -04:00
parent cfd9a596c0
commit 041f1fa2d3
11 changed files with 338 additions and 97 deletions
+8 -7
View File
@@ -352,20 +352,19 @@ impl RpcHandler {
(Some(u), Some(t)) if t > 0 => {
serde_json::json!((u as f64 / t as f64 * 100.0).round())
}
_ => serde_json::json!(0),
_ => serde_json::Value::Null,
}
};
let apps = state.map(|s| s.apps.as_slice()).unwrap_or(&[]);
let reported_at = state
.map(|s| s.timestamp.clone())
.or_else(|| n.last_seen.clone())
.unwrap_or_else(|| n.added_at.clone());
.or_else(|| n.last_seen.clone());
let mut report = serde_json::json!({
"node_id": n.did,
"node_name": state.and_then(|s| s.node_name.clone()).or_else(|| n.name.clone()),
"uptime_secs": state.and_then(|s| s.uptime_secs).unwrap_or(0),
"cpu_pct": state.and_then(|s| s.cpu_usage_percent).map(|v| v.round()).unwrap_or(0.0),
"uptime_secs": state.and_then(|s| s.uptime_secs),
"cpu_pct": state.and_then(|s| s.cpu_usage_percent).filter(|v| v.is_finite() && (0.0..=100.0).contains(v)).map(|v| v.round()),
"mem_pct": pct(state.and_then(|s| s.mem_used_bytes), state.and_then(|s| s.mem_total_bytes)),
"disk_pct": pct(state.and_then(|s| s.disk_used_bytes), state.and_then(|s| s.disk_total_bytes)),
"container_count": apps.len(),
@@ -561,7 +560,7 @@ fn annotate_fleet_report(report: &mut serde_json::Value) {
let is_online = reported
.map(|dt| {
let age = chrono::Utc::now().signed_duration_since(dt);
age.num_minutes() < 30
age.num_seconds() >= -60 && age.num_seconds() < 1800
})
.unwrap_or(false);
@@ -569,7 +568,9 @@ fn annotate_fleet_report(report: &mut serde_json::Value) {
.map(|dt| {
let age = chrono::Utc::now().signed_duration_since(dt);
let mins = age.num_minutes();
if mins < 1 {
if age.num_seconds() < -60 {
"unknown (clock ahead)".to_string()
} else if mins < 1 {
"just now".to_string()
} else if mins < 60 {
format!("{}m ago", mins)
@@ -572,14 +572,27 @@ impl RpcHandler {
None
};
// Reuse the minute collector instead of running expensive probes for
// every peer. An absent/stalled collector is unknown, never zero load.
let now = chrono::Utc::now().timestamp();
let latest = self.metrics_store.latest().await.filter(|sample| {
(0..=180).contains(&now.saturating_sub(sample.timestamp))
});
let metrics = latest.as_ref().map(|sample| &sample.system);
let uptime = tokio::fs::read_to_string("/proc/uptime")
.await
.ok()
.and_then(|s| s.split_whitespace().next()?.parse::<f64>().ok())
.filter(|v| v.is_finite() && *v >= 0.0)
.map(|v| v as u64);
let state = federation::build_local_state(
apps,
0.0,
0,
0,
0,
0,
0,
metrics.map(|m| m.cpu_percent).filter(|v| v.is_finite() && (0.0..=100.0).contains(v)),
metrics.filter(|m| m.mem_total_bytes > 0).map(|m| m.mem_used_bytes),
metrics.filter(|m| m.mem_total_bytes > 0).map(|m| m.mem_total_bytes),
metrics.filter(|m| m.disk_total_bytes > 0).map(|m| m.disk_used_bytes),
metrics.filter(|m| m.disk_total_bytes > 0).map(|m| m.disk_total_bytes),
uptime,
tor_active,
server_name,
nostr_npub,
@@ -158,3 +158,77 @@ async fn managed_relay_receives_approval_rejection_and_cancellation() {
}
}
}
#[tokio::test]
async fn federation_metrics_are_collected_values_or_unknown_never_placeholders() {
for age in [None, Some(0), Some(181), Some(-120)] {
let tmp = tempfile::tempdir().unwrap();
let mut config = crate::config::Config::default();
config.data_dir = tmp.path().to_path_buf();
let metrics = Arc::new(crate::monitoring::MetricsStore::new());
if let Some(age) = age {
metrics
.push(
serde_json::from_value(serde_json::json!({
"timestamp": chrono::Utc::now().timestamp() - age,
"system": {"cpu_percent": 37.5, "mem_used_bytes": 200,
"mem_total_bytes": 800, "disk_used_bytes": 600,
"disk_total_bytes": 1000, "net_rx_bytes": 0, "net_tx_bytes": 0,
"load_avg_1": 0.0, "load_avg_5": 0.0, "load_avg_15": 0.0},
"containers": [], "rpc_latency_ms": 0.0, "ws_connections": 0
}))
.unwrap(),
)
.await;
}
let handler = crate::api::rpc::RpcHandler::new(
config,
Arc::new(crate::state::StateManager::new()),
metrics,
crate::session::SessionStore::new_for_tests(tmp.path().join("sessions.json")),
None,
None,
)
.await
.unwrap();
let snapshot = handler.handle_federation_get_state().await.unwrap();
if age == Some(0) {
assert_eq!(snapshot["cpu_usage_percent"], 37.5);
assert_eq!(snapshot["mem_used_bytes"], 200);
assert_eq!(snapshot["disk_total_bytes"], 1000);
} else {
for field in [
"cpu_usage_percent",
"mem_used_bytes",
"mem_total_bytes",
"disk_used_bytes",
"disk_total_bytes",
] {
assert!(
snapshot.get(field).is_none_or(|v| v.is_null()),
"{age:?}: {field}"
);
}
}
let peer = serde_json::from_value(serde_json::json!({
"did": "did:key:test", "pubkey": "11".repeat(32), "onion": "test.onion",
"trust_level": "trusted", "added_at": chrono::Utc::now().to_rfc3339(),
"last_state": snapshot
}))
.unwrap();
crate::federation::save_nodes(tmp.path(), &[peer])
.await
.unwrap();
let fleet = handler.handle_telemetry_fleet_status().await.unwrap();
let report = &fleet["nodes"][0];
if age == Some(0) {
assert_eq!(report["cpu_pct"], 38.0);
assert_eq!(report["mem_pct"], 25.0);
assert_eq!(report["disk_pct"], 60.0);
} else {
for field in ["cpu_pct", "mem_pct", "disk_pct"] {
assert!(report[field].is_null(), "{age:?}: {field}");
}
}
}
}
+24 -24
View File
@@ -230,12 +230,12 @@ async fn merge_transitive_peers(
#[allow(clippy::too_many_arguments)]
pub fn build_local_state(
apps: Vec<AppStatus>,
cpu: f64,
mem_used: u64,
mem_total: u64,
disk_used: u64,
disk_total: u64,
uptime: u64,
cpu: Option<f64>,
mem_used: Option<u64>,
mem_total: Option<u64>,
disk_used: Option<u64>,
disk_total: Option<u64>,
uptime: Option<u64>,
tor_active: bool,
server_name: Option<String>,
nostr_npub: Option<String>,
@@ -261,12 +261,12 @@ pub fn build_local_state(
timestamp: chrono::Utc::now().to_rfc3339(),
node_name: server_name,
apps,
cpu_usage_percent: Some(cpu),
mem_used_bytes: Some(mem_used),
mem_total_bytes: Some(mem_total),
disk_used_bytes: Some(disk_used),
disk_total_bytes: Some(disk_total),
uptime_secs: Some(uptime),
cpu_usage_percent: cpu,
mem_used_bytes: mem_used,
mem_total_bytes: mem_total,
disk_used_bytes: disk_used,
disk_total_bytes: disk_total,
uptime_secs: uptime,
tor_active: Some(tor_active),
nostr_npub,
own_fips_npub,
@@ -355,12 +355,12 @@ mod tests {
status: "running".to_string(),
version: Some("0.18".to_string()),
}],
25.5,
2_000_000_000,
8_000_000_000,
100_000_000_000,
500_000_000_000,
3600,
Some(25.5),
Some(2_000_000_000),
Some(8_000_000_000),
Some(100_000_000_000),
Some(500_000_000_000),
Some(3600),
true,
Some("Test Node".to_string()),
None,
@@ -430,12 +430,12 @@ mod tests {
];
let state = build_local_state(
vec![],
0.0,
0,
0,
0,
0,
0,
None,
None,
None,
None,
None,
None,
true,
None,
None,
+40
View File
@@ -0,0 +1,40 @@
# Fleet metric repair — qualification in progress
## Confirmed cause
The federation.get-state handler passed literal zero values for CPU, RAM, disk
and uptime into build_local_state. Peers therefore received a valid signed RPC
response containing fabricated measurements. Fleet also turned absent fields
into zeroes and treated a newly added peer's registration time as a report.
Read-only live inspection on the dev node found fifteen federation reports with
zero resource values, alongside two nonzero collector reports. Yaya has no
Trusted fleet reports; peer (Observer) access is deliberately distinct.
## Candidate change
- Use the existing minute MetricsStore sample, avoiding expensive per-peer probes.
- Samples older than three minutes or ahead of the local clock are unavailable.
- Read uptime from the host uptime source; errors remain unavailable.
- Preserve optional measurements through federation and the Fleet response.
- Display unavailable measurements separately from valid zeroes; average only
valid measurements from reporting online nodes.
- Do not infer contact from when a peer was added. Invalid/future report dates
show unknown status; offline cards state when the last report arrived.
- Keep existing trust restrictions and signed FIPS-preferred federation transport.
This change does not claim FIPS media-stream acceptance or fix stalled peering.
## Qualification
All 1,684 backend tests pass (four ignored), including actual-handler fresh,
absent, stale and future sample coverage. All 11 Fleet helper/component tests
pass, including real-zero versus missing values, malformed/future dates,
ordering, averages and rendered unavailable/offline states. Frontend typecheck
passes. The first component assertion incorrectly expected spaces between
separate elements; it was corrected to inspect those elements. No deployment
or live metric repair acceptance yet.
Required: actual-handler fresh/absent/stale/future samples; full relevant suites;
mobile/desktop browser layout; upgraded sender and receiver comparing local
Monitoring to received Fleet values; restart/reconnect, offline ageing and real
transport evidence. Legacy senders still advertising zeros need the sender fix;
the receiver cannot reliably distinguish a legacy fabricated zero from idle CPU.
+1
View File
@@ -82,6 +82,7 @@
:node-count="fleet.nodes.value.length"
:online-count="fleet.onlineCount.value"
:offline-count="fleet.offlineCount.value"
:unknown-count="fleet.unknownCount.value"
:fleet-health-pct="fleet.fleetHealthPct.value"
:healthy-count="fleet.healthyCount.value"
:avg-cpu="fleet.avgCpu.value"
+12 -12
View File
@@ -17,7 +17,7 @@
<!-- Empty State -->
<div v-if="!nodes.length" class="text-white/40 text-sm py-8 text-center">
No nodes reporting. Ensure telemetry is enabled on beta nodes.
No fleet nodes yet. Connect a trusted node to see its status.
</div>
<div v-else class="grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4">
@@ -32,7 +32,7 @@
<div class="flex items-center gap-2">
<span
class="fleet-status-dot"
:class="isOnline(node.reported_at) ? 'fleet-dot-online' : 'fleet-dot-offline'"
:class="fleetStatus(node.reported_at) === 'unknown' ? 'bg-white/30' : isOnline(node.reported_at) ? 'fleet-dot-online' : 'fleet-dot-offline'"
></span>
<span class="text-sm font-semibold text-white truncate">{{ fleetNodeDisplayName(node) }}</span>
</div>
@@ -49,10 +49,10 @@
<div
class="fleet-bar-fill"
:class="healthBarClass(node.cpu_pct)"
:style="{ width: Math.min(node.cpu_pct, 100) + '%' }"
:style="{ width: Math.min(node.cpu_pct ?? 0, 100) + '%' }"
></div>
</div>
<span class="text-xs text-white/60 w-10 text-right">{{ node.cpu_pct.toFixed(0) }}%</span>
<span class="text-xs text-white/60 w-10 text-right">{{ formatMetric(node.cpu_pct) }}</span>
</div>
<div class="fleet-metric-row">
<span class="text-xs text-white/50">RAM</span>
@@ -60,10 +60,10 @@
<div
class="fleet-bar-fill"
:class="healthBarClass(node.mem_pct)"
:style="{ width: Math.min(node.mem_pct, 100) + '%' }"
:style="{ width: Math.min(node.mem_pct ?? 0, 100) + '%' }"
></div>
</div>
<span class="text-xs text-white/60 w-10 text-right">{{ node.mem_pct.toFixed(0) }}%</span>
<span class="text-xs text-white/60 w-10 text-right">{{ formatMetric(node.mem_pct) }}</span>
</div>
<div class="fleet-metric-row">
<span class="text-xs text-white/50">Disk</span>
@@ -71,10 +71,10 @@
<div
class="fleet-bar-fill"
:class="healthBarClass(node.disk_pct)"
:style="{ width: Math.min(node.disk_pct, 100) + '%' }"
:style="{ width: Math.min(node.disk_pct ?? 0, 100) + '%' }"
></div>
</div>
<span class="text-xs text-white/60 w-10 text-right">{{ node.disk_pct.toFixed(0) }}%</span>
<span class="text-xs text-white/60 w-10 text-right">{{ formatMetric(node.disk_pct) }}</span>
</div>
</div>
@@ -82,9 +82,9 @@
<span>{{ node.running_count }}/{{ node.container_count }} containers</span>
<span>{{ node.federation_peers }} peers</span>
</div>
<div class="flex items-center justify-between text-xs text-white/40 mt-1">
<span>Up {{ formatUptime(node.uptime_secs) }}</span>
<span>{{ timeAgo(node.reported_at) }}</span>
<div class="flex flex-wrap gap-x-3 gap-y-1 items-center justify-between text-xs text-white/40 mt-1">
<span>{{ node.uptime_secs === null ? 'Uptime unavailable' : 'Up ' + formatUptime(node.uptime_secs) }}</span>
<span>{{ fleetStatus(node.reported_at) === 'unknown' ? 'Status unknown' : fleetStatus(node.reported_at) === 'offline' ? 'Offline · last seen ' + timeAgo(node.reported_at) : 'Seen ' + timeAgo(node.reported_at) }}</span>
</div>
</div>
</div>
@@ -94,7 +94,7 @@
<script setup lang="ts">
import {
type FleetNode, type SortOption, SORT_OPTIONS,
isOnline, healthBarClass, formatUptime, timeAgo, fleetNodeDisplayName, fleetNodeSubtitle,
isOnline, fleetStatus, formatMetric, healthBarClass, formatUptime, timeAgo, fleetNodeDisplayName, fleetNodeSubtitle,
} from './useFleetData'
defineProps<{
+13 -11
View File
@@ -6,42 +6,44 @@
<p class="text-xs text-white/40">
<span class="fleet-dot-online"></span> {{ onlineCount }} online
<span class="ml-1 fleet-dot-offline"></span> {{ offlineCount }} offline
<span v-if="unknownCount" class="ml-1">{{ unknownCount }} unknown</span>
</p>
</div>
<div data-controller-container tabindex="0" class="monitoring-stat-card">
<p class="text-xs text-white/50 uppercase tracking-wide">Fleet Health</p>
<p class="text-2xl font-bold text-white">{{ fleetHealthPct }}%</p>
<p class="text-xs text-white/40">{{ healthyCount }}/{{ nodeCount }} no alerts</p>
<p class="text-xs text-white/40">{{ healthyCount }}/{{ nodeCount }} reporting without alerts</p>
</div>
<div data-controller-container tabindex="0" class="monitoring-stat-card">
<p class="text-xs text-white/50 uppercase tracking-wide">Avg CPU</p>
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgCpu)">{{ avgCpu.toFixed(1) }}%</p>
<p class="text-xs text-white/40">across fleet</p>
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgCpu)">{{ formatMetric(avgCpu, 1) }}</p>
<p class="text-xs text-white/40">reporting online nodes</p>
</div>
<div data-controller-container tabindex="0" class="monitoring-stat-card">
<p class="text-xs text-white/50 uppercase tracking-wide">Avg RAM</p>
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgMem)">{{ avgMem.toFixed(1) }}%</p>
<p class="text-xs text-white/40">across fleet</p>
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgMem)">{{ formatMetric(avgMem, 1) }}</p>
<p class="text-xs text-white/40">reporting online nodes</p>
</div>
<div data-controller-container tabindex="0" class="monitoring-stat-card">
<p class="text-xs text-white/50 uppercase tracking-wide">Avg Disk</p>
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgDisk)">{{ avgDisk.toFixed(1) }}%</p>
<p class="text-xs text-white/40">across fleet</p>
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgDisk)">{{ formatMetric(avgDisk, 1) }}</p>
<p class="text-xs text-white/40">reporting online nodes</p>
</div>
</div>
</template>
<script setup lang="ts">
import { healthTextClass } from './useFleetData'
import { healthTextClass, formatMetric } from './useFleetData'
defineProps<{
nodeCount: number
onlineCount: number
offlineCount: number
unknownCount: number
fleetHealthPct: number
healthyCount: number
avgCpu: number
avgMem: number
avgDisk: number
avgCpu: number | null
avgMem: number | null
avgDisk: number | null
}>()
</script>
@@ -0,0 +1,38 @@
import { afterEach, describe, expect, it, vi } from 'vitest'
import { mount } from '@vue/test-utils'
import FleetNodeGrid from '../FleetNodeGrid.vue'
import FleetOverviewCards from '../FleetOverviewCards.vue'
import { normalizeFleetNode } from '../useFleetData'
describe('fleet unavailable readings', () => {
afterEach(() => vi.useRealTimers())
it('shows an unavailable reading differently from a real zero and explains offline status', () => {
vi.useFakeTimers()
vi.setSystemTime(new Date('2026-06-10T12:00:00Z'))
const nodes = [normalizeFleetNode({node_id: 'offline', reported_at: '2026-06-08T12:00:00Z', cpu_pct: 0}), normalizeFleetNode({node_id: 'unknown'})]
const wrapper = mount(FleetNodeGrid, {
props: {nodes, sortedNodes: nodes, sortBy: 'status', selectedNodeId: null},
global: {mocks: {$ver: (v: string) => v}},
})
const cards = wrapper.findAll('.fleet-node-card')
expect(cards[0].text()).toContain('0%')
expect(cards[0].text()).toContain('—')
expect(cards[0].text()).toContain('Offline · last seen 2d ago')
expect(cards[1].text()).toContain('Status unknown')
expect(cards[1].text()).not.toContain('0%')
expect(cards[1].text()).toContain('Uptime unavailable')
expect(wrapper.text()).not.toContain('NaN')
wrapper.unmount()
})
it('renders empty averages without claiming zero load', () => {
const wrapper = mount(FleetOverviewCards, {props: {
nodeCount: 1, onlineCount: 0, offlineCount: 0, unknownCount: 1,
fleetHealthPct: 0, healthyCount: 0, avgCpu: null, avgMem: null, avgDisk: null,
}})
expect(wrapper.text()).toContain('1 unknown')
expect(wrapper.findAll('.monitoring-stat-card').slice(2).map(card => card.findAll('p').map(p => p.text()).join(' '))).toEqual([
'Avg CPU — reporting online nodes', 'Avg RAM — reporting online nodes', 'Avg Disk — reporting online nodes',
])
wrapper.unmount()
})
})
@@ -1,5 +1,9 @@
import { describe, expect, it, vi } from 'vitest'
import { afterEach, describe, expect, it, vi } from 'vitest'
import {
averageMetric,
fleetStatus,
formatMetric,
timeAgo,
fleetNodeDisplayName,
fleetNodeSubtitle,
isOnline,
@@ -31,6 +35,7 @@ function node(id: string, reportedAt: string): FleetNode {
}
describe('fleet data helpers', () => {
afterEach(() => vi.useRealTimers())
it('treats nodes reported within 30 minutes as online', () => {
vi.useFakeTimers()
vi.setSystemTime(new Date('2026-06-10T12:00:00Z'))
@@ -79,9 +84,9 @@ describe('fleet data helpers', () => {
expect(normalized.node_name).toBeNull()
expect(normalized.hostname).toBeNull()
expect(normalized.server_url).toBeNull()
expect(normalized.cpu_pct).toBe(0)
expect(normalized.mem_pct).toBe(0)
expect(normalized.disk_pct).toBe(0)
expect(normalized.cpu_pct).toBeNull()
expect(normalized.mem_pct).toBeNull()
expect(normalized.disk_pct).toBeNull()
expect(normalized.containers).toEqual([])
expect(normalized.recent_alerts).toEqual([])
})
@@ -116,3 +121,47 @@ describe('fleet data helpers', () => {
expect(normalizeNodeHistoryResponse({})).toEqual([])
})
})
describe('fleet missing data and clock boundaries', () => {
afterEach(() => vi.useRealTimers())
it('does not turn malformed, missing or future timestamps into online status', () => {
vi.useFakeTimers()
vi.setSystemTime(new Date('2026-06-10T12:00:00Z'))
for (const value of ['', 'broken', '2027-01-01T00:00:00Z']) {
expect(fleetStatus(value)).toBe('unknown')
expect(isOnline(value)).toBe(false)
expect(timeAgo(value)).toBe('Unknown')
}
expect(fleetStatus('2026-06-10T11:30:00Z')).toBe('offline')
expect(fleetStatus('2026-06-10T12:00:30Z')).toBe('online')
expect(sortFleetNodes([node('unknown', ''), node('future', '2027-01-01T00:00:00Z'), node('old', '2026-06-10T10:00:00Z'), node('new', '2026-06-10T11:59:00Z')], 'status').map(n => n.node_id)).toEqual(['new', 'old', 'unknown', 'future'])
})
it('keeps valid zero separate from unavailable and invalid measurements', () => {
expect(normalizeFleetNode({cpu_pct: 0}).cpu_pct).toBe(0)
for (const value of [null, undefined, NaN, Infinity, -1, 101]) {
const result = normalizeFleetNode({cpu_pct: value, mem_pct: value, disk_pct: value})
expect(result.cpu_pct).toBeNull()
expect(result.mem_pct).toBeNull()
expect(result.disk_pct).toBeNull()
}
expect(formatMetric(null)).toBe('—')
expect(formatMetric(0)).toBe('0%')
expect(formatMetric(25.25, 1)).toBe('25.3%')
})
it('averages only real measurements from reporting online nodes', () => {
vi.useFakeTimers()
vi.setSystemTime(new Date('2026-06-10T12:00:00Z'))
const first = node('first', '2026-06-10T11:59:00Z')
first.cpu_pct = 0
const second = node('second', first.reported_at)
second.cpu_pct = 50
const missing = node('missing', first.reported_at)
missing.cpu_pct = null
const offline = node('offline', '2026-06-01T00:00:00Z')
offline.cpu_pct = 100
expect(averageMetric([first, second, missing, offline], 'cpu_pct')).toBe(25)
expect(averageMetric([missing, offline], 'cpu_pct')).toBeNull()
expect(averageMetric([], 'cpu_pct')).toBeNull()
})
})
+56 -33
View File
@@ -12,11 +12,11 @@ export interface FleetNode {
hostname?: string | null
server_url?: string | null
version: string
uptime_secs: number
uptime_secs: number | null
cpu_cores: number
cpu_pct: number
mem_pct: number
disk_pct: number
cpu_pct: number | null
mem_pct: number | null
disk_pct: number | null
container_count: number
running_count: number
federation_peers: number
@@ -43,7 +43,8 @@ export type SortOption = 'status' | 'last-seen' | 'name'
// --- Utility Functions ---
export function formatUptime(secs: number): string {
export function formatUptime(secs: number | null): string {
if (secs === null) return 'Unavailable'
if (secs < 60) return `${secs}s`
const days = Math.floor(secs / 86400)
const hours = Math.floor((secs % 86400) / 3600)
@@ -56,6 +57,7 @@ export function formatUptime(secs: number): string {
export function timeAgo(dateStr: string): string {
const now = Date.now()
const then = new Date(dateStr).getTime()
if (!Number.isFinite(then) || then > now + 60_000) return 'Unknown'
const diffMs = now - then
if (diffMs < 0) return 'just now'
const diffSecs = Math.floor(diffMs / 1000)
@@ -68,18 +70,34 @@ export function timeAgo(dateStr: string): string {
return `${diffDays}d ago`
}
export function isOnline(reportedAt: string): boolean {
const thirtyMinMs = 30 * 60 * 1000
return Date.now() - new Date(reportedAt).getTime() < thirtyMinMs
export function fleetStatus(reportedAt: string): 'online' | 'offline' | 'unknown' {
const age = Date.now() - new Date(reportedAt).getTime()
if (!Number.isFinite(age) || age < -60_000) return 'unknown'
return age < 30 * 60 * 1000 ? 'online' : 'offline'
}
export function healthBarClass(pct: number): string {
export function isOnline(reportedAt: string): boolean {
return fleetStatus(reportedAt) === 'online'
}
export function formatMetric(value: number | null, digits = 0): string {
return value === null ? '—' : `${value.toFixed(digits)}%`
}
export function averageMetric(nodes: FleetNode[], field: 'cpu_pct' | 'mem_pct' | 'disk_pct'): number | null {
const values = nodes.filter(n => isOnline(n.reported_at)).map(n => n[field]).filter((v): v is number => v !== null && Number.isFinite(v))
return values.length ? values.reduce((sum, value) => sum + value, 0) / values.length : null
}
export function healthBarClass(pct: number | null): string {
if (pct === null) return ''
if (pct >= 85) return 'monitoring-bar-danger'
if (pct >= 60) return 'monitoring-bar-warn'
return 'monitoring-bar-ok'
}
export function healthTextClass(pct: number): string {
export function healthTextClass(pct: number | null): string {
if (pct === null) return ''
if (pct >= 85) return 'fleet-text-danger'
if (pct >= 60) return 'fleet-text-warn'
return ''
@@ -132,6 +150,10 @@ export const SORT_OPTIONS: Array<{ label: string; value: SortOption }> = [
{ label: 'Name', value: 'name' },
]
function reportTime(node: FleetNode): number {
return fleetStatus(node.reported_at) === 'unknown' ? 0 : new Date(node.reported_at).getTime()
}
export function sortFleetNodes(nodes: FleetNode[], sortBy: SortOption): FleetNode[] {
const sorted = [...nodes]
switch (sortBy) {
@@ -140,11 +162,11 @@ export function sortFleetNodes(nodes: FleetNode[], sortBy: SortOption): FleetNod
const aOnline = isOnline(a.reported_at)
const bOnline = isOnline(b.reported_at)
if (aOnline !== bOnline) return aOnline ? -1 : 1
return new Date(b.reported_at).getTime() - new Date(a.reported_at).getTime()
return reportTime(b) - reportTime(a)
})
break
case 'last-seen':
sorted.sort((a, b) => new Date(b.reported_at).getTime() - new Date(a.reported_at).getTime())
sorted.sort((a, b) => reportTime(b) - reportTime(a))
break
case 'name':
sorted.sort((a, b) => fleetNodeDisplayName(a).localeCompare(fleetNodeDisplayName(b)))
@@ -153,6 +175,15 @@ export function sortFleetNodes(nodes: FleetNode[], sortBy: SortOption): FleetNod
return sorted
}
function numberOrNull(value: unknown): number | null {
return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : null
}
function percentageOrNull(value: unknown): number | null {
const number = numberOrNull(value)
return number !== null && number <= 100 ? number : null
}
function numberOrZero(value: unknown): number {
return typeof value === 'number' && Number.isFinite(value) ? value : 0
}
@@ -164,17 +195,17 @@ export function normalizeFleetNode(node: Partial<FleetNode>): FleetNode {
hostname: typeof node.hostname === 'string' ? node.hostname : null,
server_url: typeof node.server_url === 'string' ? node.server_url : null,
version: typeof node.version === 'string' ? node.version : 'unknown',
uptime_secs: numberOrZero(node.uptime_secs),
uptime_secs: numberOrNull(node.uptime_secs),
cpu_cores: numberOrZero(node.cpu_cores),
cpu_pct: numberOrZero(node.cpu_pct),
mem_pct: numberOrZero(node.mem_pct),
disk_pct: numberOrZero(node.disk_pct),
cpu_pct: percentageOrNull(node.cpu_pct),
mem_pct: percentageOrNull(node.mem_pct),
disk_pct: percentageOrNull(node.disk_pct),
container_count: numberOrZero(node.container_count),
running_count: numberOrZero(node.running_count),
federation_peers: numberOrZero(node.federation_peers),
recent_alerts: Array.isArray(node.recent_alerts) ? node.recent_alerts : [],
containers: Array.isArray(node.containers) ? node.containers : [],
reported_at: typeof node.reported_at === 'string' ? node.reported_at : new Date(0).toISOString(),
reported_at: typeof node.reported_at === 'string' ? node.reported_at : '',
}
}
@@ -195,7 +226,7 @@ type FleetCache = {
sortBy: SortOption
}
const FLEET_CACHE_KEY = 'archipelago.fleet.cache.v1'
const FLEET_CACHE_KEY = 'archipelago.fleet.cache.v2'
function readFleetCache(): Partial<FleetCache> {
if (typeof window === 'undefined') return {}
@@ -246,28 +277,20 @@ export function useFleetData() {
// --- Computed ---
const onlineCount = computed(() => nodes.value.filter(n => isOnline(n.reported_at)).length)
const offlineCount = computed(() => nodes.value.length - onlineCount.value)
const healthyCount = computed(() => nodes.value.filter(n => n.recent_alerts.length === 0).length)
const offlineCount = computed(() => nodes.value.filter(n => fleetStatus(n.reported_at) === 'offline').length)
const unknownCount = computed(() => nodes.value.filter(n => fleetStatus(n.reported_at) === 'unknown').length)
const healthyCount = computed(() => nodes.value.filter(n => isOnline(n.reported_at) && n.cpu_pct !== null && n.mem_pct !== null && n.disk_pct !== null && n.recent_alerts.length === 0).length)
const fleetHealthPct = computed(() => {
if (!nodes.value.length) return 0
return Math.round((healthyCount.value / nodes.value.length) * 100)
})
const avgCpu = computed(() => {
if (!nodes.value.length) return 0
return nodes.value.reduce((sum, n) => sum + n.cpu_pct, 0) / nodes.value.length
})
const avgCpu = computed(() => averageMetric(nodes.value, 'cpu_pct'))
const avgMem = computed(() => {
if (!nodes.value.length) return 0
return nodes.value.reduce((sum, n) => sum + n.mem_pct, 0) / nodes.value.length
})
const avgMem = computed(() => averageMetric(nodes.value, 'mem_pct'))
const avgDisk = computed(() => {
if (!nodes.value.length) return 0
return nodes.value.reduce((sum, n) => sum + n.disk_pct, 0) / nodes.value.length
})
const avgDisk = computed(() => averageMetric(nodes.value, 'disk_pct'))
const selectedNode = computed(() => {
if (!selectedNodeId.value) return null
@@ -525,7 +548,7 @@ export function useFleetData() {
loading, refreshing, errorMessage, nodes, fleetAlerts, alertsLoading,
selectedNodeId, selectedNode, nodeHistory, nodeHistoryLoading,
autoRefresh, lastRefreshed, sortBy, chartWidth,
onlineCount, offlineCount, healthyCount, fleetHealthPct,
onlineCount, offlineCount, unknownCount, healthyCount, fleetHealthPct,
avgCpu, avgMem, avgDisk, sortedNodes, allAppIds,
nodeHistoryLabels, nodeHistoryCpuDatasets, nodeHistoryMemDatasets, nodeHistoryDiskDatasets,
refreshAll, selectNode, toggleAutoRefresh, exportFleetData,