fix: report collected fleet metrics and distinguish unavailable readings
This commit is contained in:
@@ -352,20 +352,19 @@ impl RpcHandler {
|
||||
(Some(u), Some(t)) if t > 0 => {
|
||||
serde_json::json!((u as f64 / t as f64 * 100.0).round())
|
||||
}
|
||||
_ => serde_json::json!(0),
|
||||
_ => serde_json::Value::Null,
|
||||
}
|
||||
};
|
||||
let apps = state.map(|s| s.apps.as_slice()).unwrap_or(&[]);
|
||||
let reported_at = state
|
||||
.map(|s| s.timestamp.clone())
|
||||
.or_else(|| n.last_seen.clone())
|
||||
.unwrap_or_else(|| n.added_at.clone());
|
||||
.or_else(|| n.last_seen.clone());
|
||||
|
||||
let mut report = serde_json::json!({
|
||||
"node_id": n.did,
|
||||
"node_name": state.and_then(|s| s.node_name.clone()).or_else(|| n.name.clone()),
|
||||
"uptime_secs": state.and_then(|s| s.uptime_secs).unwrap_or(0),
|
||||
"cpu_pct": state.and_then(|s| s.cpu_usage_percent).map(|v| v.round()).unwrap_or(0.0),
|
||||
"uptime_secs": state.and_then(|s| s.uptime_secs),
|
||||
"cpu_pct": state.and_then(|s| s.cpu_usage_percent).filter(|v| v.is_finite() && (0.0..=100.0).contains(v)).map(|v| v.round()),
|
||||
"mem_pct": pct(state.and_then(|s| s.mem_used_bytes), state.and_then(|s| s.mem_total_bytes)),
|
||||
"disk_pct": pct(state.and_then(|s| s.disk_used_bytes), state.and_then(|s| s.disk_total_bytes)),
|
||||
"container_count": apps.len(),
|
||||
@@ -561,7 +560,7 @@ fn annotate_fleet_report(report: &mut serde_json::Value) {
|
||||
let is_online = reported
|
||||
.map(|dt| {
|
||||
let age = chrono::Utc::now().signed_duration_since(dt);
|
||||
age.num_minutes() < 30
|
||||
age.num_seconds() >= -60 && age.num_seconds() < 1800
|
||||
})
|
||||
.unwrap_or(false);
|
||||
|
||||
@@ -569,7 +568,9 @@ fn annotate_fleet_report(report: &mut serde_json::Value) {
|
||||
.map(|dt| {
|
||||
let age = chrono::Utc::now().signed_duration_since(dt);
|
||||
let mins = age.num_minutes();
|
||||
if mins < 1 {
|
||||
if age.num_seconds() < -60 {
|
||||
"unknown (clock ahead)".to_string()
|
||||
} else if mins < 1 {
|
||||
"just now".to_string()
|
||||
} else if mins < 60 {
|
||||
format!("{}m ago", mins)
|
||||
|
||||
@@ -572,14 +572,27 @@ impl RpcHandler {
|
||||
None
|
||||
};
|
||||
|
||||
// Reuse the minute collector instead of running expensive probes for
|
||||
// every peer. An absent/stalled collector is unknown, never zero load.
|
||||
let now = chrono::Utc::now().timestamp();
|
||||
let latest = self.metrics_store.latest().await.filter(|sample| {
|
||||
(0..=180).contains(&now.saturating_sub(sample.timestamp))
|
||||
});
|
||||
let metrics = latest.as_ref().map(|sample| &sample.system);
|
||||
let uptime = tokio::fs::read_to_string("/proc/uptime")
|
||||
.await
|
||||
.ok()
|
||||
.and_then(|s| s.split_whitespace().next()?.parse::<f64>().ok())
|
||||
.filter(|v| v.is_finite() && *v >= 0.0)
|
||||
.map(|v| v as u64);
|
||||
let state = federation::build_local_state(
|
||||
apps,
|
||||
0.0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
metrics.map(|m| m.cpu_percent).filter(|v| v.is_finite() && (0.0..=100.0).contains(v)),
|
||||
metrics.filter(|m| m.mem_total_bytes > 0).map(|m| m.mem_used_bytes),
|
||||
metrics.filter(|m| m.mem_total_bytes > 0).map(|m| m.mem_total_bytes),
|
||||
metrics.filter(|m| m.disk_total_bytes > 0).map(|m| m.disk_used_bytes),
|
||||
metrics.filter(|m| m.disk_total_bytes > 0).map(|m| m.disk_total_bytes),
|
||||
uptime,
|
||||
tor_active,
|
||||
server_name,
|
||||
nostr_npub,
|
||||
|
||||
@@ -158,3 +158,77 @@ async fn managed_relay_receives_approval_rejection_and_cancellation() {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn federation_metrics_are_collected_values_or_unknown_never_placeholders() {
|
||||
for age in [None, Some(0), Some(181), Some(-120)] {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let mut config = crate::config::Config::default();
|
||||
config.data_dir = tmp.path().to_path_buf();
|
||||
let metrics = Arc::new(crate::monitoring::MetricsStore::new());
|
||||
if let Some(age) = age {
|
||||
metrics
|
||||
.push(
|
||||
serde_json::from_value(serde_json::json!({
|
||||
"timestamp": chrono::Utc::now().timestamp() - age,
|
||||
"system": {"cpu_percent": 37.5, "mem_used_bytes": 200,
|
||||
"mem_total_bytes": 800, "disk_used_bytes": 600,
|
||||
"disk_total_bytes": 1000, "net_rx_bytes": 0, "net_tx_bytes": 0,
|
||||
"load_avg_1": 0.0, "load_avg_5": 0.0, "load_avg_15": 0.0},
|
||||
"containers": [], "rpc_latency_ms": 0.0, "ws_connections": 0
|
||||
}))
|
||||
.unwrap(),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
let handler = crate::api::rpc::RpcHandler::new(
|
||||
config,
|
||||
Arc::new(crate::state::StateManager::new()),
|
||||
metrics,
|
||||
crate::session::SessionStore::new_for_tests(tmp.path().join("sessions.json")),
|
||||
None,
|
||||
None,
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let snapshot = handler.handle_federation_get_state().await.unwrap();
|
||||
if age == Some(0) {
|
||||
assert_eq!(snapshot["cpu_usage_percent"], 37.5);
|
||||
assert_eq!(snapshot["mem_used_bytes"], 200);
|
||||
assert_eq!(snapshot["disk_total_bytes"], 1000);
|
||||
} else {
|
||||
for field in [
|
||||
"cpu_usage_percent",
|
||||
"mem_used_bytes",
|
||||
"mem_total_bytes",
|
||||
"disk_used_bytes",
|
||||
"disk_total_bytes",
|
||||
] {
|
||||
assert!(
|
||||
snapshot.get(field).is_none_or(|v| v.is_null()),
|
||||
"{age:?}: {field}"
|
||||
);
|
||||
}
|
||||
}
|
||||
let peer = serde_json::from_value(serde_json::json!({
|
||||
"did": "did:key:test", "pubkey": "11".repeat(32), "onion": "test.onion",
|
||||
"trust_level": "trusted", "added_at": chrono::Utc::now().to_rfc3339(),
|
||||
"last_state": snapshot
|
||||
}))
|
||||
.unwrap();
|
||||
crate::federation::save_nodes(tmp.path(), &[peer])
|
||||
.await
|
||||
.unwrap();
|
||||
let fleet = handler.handle_telemetry_fleet_status().await.unwrap();
|
||||
let report = &fleet["nodes"][0];
|
||||
if age == Some(0) {
|
||||
assert_eq!(report["cpu_pct"], 38.0);
|
||||
assert_eq!(report["mem_pct"], 25.0);
|
||||
assert_eq!(report["disk_pct"], 60.0);
|
||||
} else {
|
||||
for field in ["cpu_pct", "mem_pct", "disk_pct"] {
|
||||
assert!(report[field].is_null(), "{age:?}: {field}");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -230,12 +230,12 @@ async fn merge_transitive_peers(
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn build_local_state(
|
||||
apps: Vec<AppStatus>,
|
||||
cpu: f64,
|
||||
mem_used: u64,
|
||||
mem_total: u64,
|
||||
disk_used: u64,
|
||||
disk_total: u64,
|
||||
uptime: u64,
|
||||
cpu: Option<f64>,
|
||||
mem_used: Option<u64>,
|
||||
mem_total: Option<u64>,
|
||||
disk_used: Option<u64>,
|
||||
disk_total: Option<u64>,
|
||||
uptime: Option<u64>,
|
||||
tor_active: bool,
|
||||
server_name: Option<String>,
|
||||
nostr_npub: Option<String>,
|
||||
@@ -261,12 +261,12 @@ pub fn build_local_state(
|
||||
timestamp: chrono::Utc::now().to_rfc3339(),
|
||||
node_name: server_name,
|
||||
apps,
|
||||
cpu_usage_percent: Some(cpu),
|
||||
mem_used_bytes: Some(mem_used),
|
||||
mem_total_bytes: Some(mem_total),
|
||||
disk_used_bytes: Some(disk_used),
|
||||
disk_total_bytes: Some(disk_total),
|
||||
uptime_secs: Some(uptime),
|
||||
cpu_usage_percent: cpu,
|
||||
mem_used_bytes: mem_used,
|
||||
mem_total_bytes: mem_total,
|
||||
disk_used_bytes: disk_used,
|
||||
disk_total_bytes: disk_total,
|
||||
uptime_secs: uptime,
|
||||
tor_active: Some(tor_active),
|
||||
nostr_npub,
|
||||
own_fips_npub,
|
||||
@@ -355,12 +355,12 @@ mod tests {
|
||||
status: "running".to_string(),
|
||||
version: Some("0.18".to_string()),
|
||||
}],
|
||||
25.5,
|
||||
2_000_000_000,
|
||||
8_000_000_000,
|
||||
100_000_000_000,
|
||||
500_000_000_000,
|
||||
3600,
|
||||
Some(25.5),
|
||||
Some(2_000_000_000),
|
||||
Some(8_000_000_000),
|
||||
Some(100_000_000_000),
|
||||
Some(500_000_000_000),
|
||||
Some(3600),
|
||||
true,
|
||||
Some("Test Node".to_string()),
|
||||
None,
|
||||
@@ -430,12 +430,12 @@ mod tests {
|
||||
];
|
||||
let state = build_local_state(
|
||||
vec![],
|
||||
0.0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
true,
|
||||
None,
|
||||
None,
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
# Fleet metric repair — qualification in progress
|
||||
|
||||
## Confirmed cause
|
||||
|
||||
The federation.get-state handler passed literal zero values for CPU, RAM, disk
|
||||
and uptime into build_local_state. Peers therefore received a valid signed RPC
|
||||
response containing fabricated measurements. Fleet also turned absent fields
|
||||
into zeroes and treated a newly added peer's registration time as a report.
|
||||
Read-only live inspection on the dev node found fifteen federation reports with
|
||||
zero resource values, alongside two nonzero collector reports. Yaya has no
|
||||
Trusted fleet reports; peer (Observer) access is deliberately distinct.
|
||||
|
||||
## Candidate change
|
||||
|
||||
- Use the existing minute MetricsStore sample, avoiding expensive per-peer probes.
|
||||
- Samples older than three minutes or ahead of the local clock are unavailable.
|
||||
- Read uptime from the host uptime source; errors remain unavailable.
|
||||
- Preserve optional measurements through federation and the Fleet response.
|
||||
- Display unavailable measurements separately from valid zeroes; average only
|
||||
valid measurements from reporting online nodes.
|
||||
- Do not infer contact from when a peer was added. Invalid/future report dates
|
||||
show unknown status; offline cards state when the last report arrived.
|
||||
- Keep existing trust restrictions and signed FIPS-preferred federation transport.
|
||||
This change does not claim FIPS media-stream acceptance or fix stalled peering.
|
||||
|
||||
## Qualification
|
||||
|
||||
All 1,684 backend tests pass (four ignored), including actual-handler fresh,
|
||||
absent, stale and future sample coverage. All 11 Fleet helper/component tests
|
||||
pass, including real-zero versus missing values, malformed/future dates,
|
||||
ordering, averages and rendered unavailable/offline states. Frontend typecheck
|
||||
passes. The first component assertion incorrectly expected spaces between
|
||||
separate elements; it was corrected to inspect those elements. No deployment
|
||||
or live metric repair acceptance yet.
|
||||
|
||||
Required: actual-handler fresh/absent/stale/future samples; full relevant suites;
|
||||
mobile/desktop browser layout; upgraded sender and receiver comparing local
|
||||
Monitoring to received Fleet values; restart/reconnect, offline ageing and real
|
||||
transport evidence. Legacy senders still advertising zeros need the sender fix;
|
||||
the receiver cannot reliably distinguish a legacy fabricated zero from idle CPU.
|
||||
@@ -82,6 +82,7 @@
|
||||
:node-count="fleet.nodes.value.length"
|
||||
:online-count="fleet.onlineCount.value"
|
||||
:offline-count="fleet.offlineCount.value"
|
||||
:unknown-count="fleet.unknownCount.value"
|
||||
:fleet-health-pct="fleet.fleetHealthPct.value"
|
||||
:healthy-count="fleet.healthyCount.value"
|
||||
:avg-cpu="fleet.avgCpu.value"
|
||||
|
||||
@@ -17,7 +17,7 @@
|
||||
|
||||
<!-- Empty State -->
|
||||
<div v-if="!nodes.length" class="text-white/40 text-sm py-8 text-center">
|
||||
No nodes reporting. Ensure telemetry is enabled on beta nodes.
|
||||
No fleet nodes yet. Connect a trusted node to see its status.
|
||||
</div>
|
||||
|
||||
<div v-else class="grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4">
|
||||
@@ -32,7 +32,7 @@
|
||||
<div class="flex items-center gap-2">
|
||||
<span
|
||||
class="fleet-status-dot"
|
||||
:class="isOnline(node.reported_at) ? 'fleet-dot-online' : 'fleet-dot-offline'"
|
||||
:class="fleetStatus(node.reported_at) === 'unknown' ? 'bg-white/30' : isOnline(node.reported_at) ? 'fleet-dot-online' : 'fleet-dot-offline'"
|
||||
></span>
|
||||
<span class="text-sm font-semibold text-white truncate">{{ fleetNodeDisplayName(node) }}</span>
|
||||
</div>
|
||||
@@ -49,10 +49,10 @@
|
||||
<div
|
||||
class="fleet-bar-fill"
|
||||
:class="healthBarClass(node.cpu_pct)"
|
||||
:style="{ width: Math.min(node.cpu_pct, 100) + '%' }"
|
||||
:style="{ width: Math.min(node.cpu_pct ?? 0, 100) + '%' }"
|
||||
></div>
|
||||
</div>
|
||||
<span class="text-xs text-white/60 w-10 text-right">{{ node.cpu_pct.toFixed(0) }}%</span>
|
||||
<span class="text-xs text-white/60 w-10 text-right">{{ formatMetric(node.cpu_pct) }}</span>
|
||||
</div>
|
||||
<div class="fleet-metric-row">
|
||||
<span class="text-xs text-white/50">RAM</span>
|
||||
@@ -60,10 +60,10 @@
|
||||
<div
|
||||
class="fleet-bar-fill"
|
||||
:class="healthBarClass(node.mem_pct)"
|
||||
:style="{ width: Math.min(node.mem_pct, 100) + '%' }"
|
||||
:style="{ width: Math.min(node.mem_pct ?? 0, 100) + '%' }"
|
||||
></div>
|
||||
</div>
|
||||
<span class="text-xs text-white/60 w-10 text-right">{{ node.mem_pct.toFixed(0) }}%</span>
|
||||
<span class="text-xs text-white/60 w-10 text-right">{{ formatMetric(node.mem_pct) }}</span>
|
||||
</div>
|
||||
<div class="fleet-metric-row">
|
||||
<span class="text-xs text-white/50">Disk</span>
|
||||
@@ -71,10 +71,10 @@
|
||||
<div
|
||||
class="fleet-bar-fill"
|
||||
:class="healthBarClass(node.disk_pct)"
|
||||
:style="{ width: Math.min(node.disk_pct, 100) + '%' }"
|
||||
:style="{ width: Math.min(node.disk_pct ?? 0, 100) + '%' }"
|
||||
></div>
|
||||
</div>
|
||||
<span class="text-xs text-white/60 w-10 text-right">{{ node.disk_pct.toFixed(0) }}%</span>
|
||||
<span class="text-xs text-white/60 w-10 text-right">{{ formatMetric(node.disk_pct) }}</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -82,9 +82,9 @@
|
||||
<span>{{ node.running_count }}/{{ node.container_count }} containers</span>
|
||||
<span>{{ node.federation_peers }} peers</span>
|
||||
</div>
|
||||
<div class="flex items-center justify-between text-xs text-white/40 mt-1">
|
||||
<span>Up {{ formatUptime(node.uptime_secs) }}</span>
|
||||
<span>{{ timeAgo(node.reported_at) }}</span>
|
||||
<div class="flex flex-wrap gap-x-3 gap-y-1 items-center justify-between text-xs text-white/40 mt-1">
|
||||
<span>{{ node.uptime_secs === null ? 'Uptime unavailable' : 'Up ' + formatUptime(node.uptime_secs) }}</span>
|
||||
<span>{{ fleetStatus(node.reported_at) === 'unknown' ? 'Status unknown' : fleetStatus(node.reported_at) === 'offline' ? 'Offline · last seen ' + timeAgo(node.reported_at) : 'Seen ' + timeAgo(node.reported_at) }}</span>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
@@ -94,7 +94,7 @@
|
||||
<script setup lang="ts">
|
||||
import {
|
||||
type FleetNode, type SortOption, SORT_OPTIONS,
|
||||
isOnline, healthBarClass, formatUptime, timeAgo, fleetNodeDisplayName, fleetNodeSubtitle,
|
||||
isOnline, fleetStatus, formatMetric, healthBarClass, formatUptime, timeAgo, fleetNodeDisplayName, fleetNodeSubtitle,
|
||||
} from './useFleetData'
|
||||
|
||||
defineProps<{
|
||||
|
||||
@@ -6,42 +6,44 @@
|
||||
<p class="text-xs text-white/40">
|
||||
<span class="fleet-dot-online"></span> {{ onlineCount }} online
|
||||
<span class="ml-1 fleet-dot-offline"></span> {{ offlineCount }} offline
|
||||
<span v-if="unknownCount" class="ml-1">{{ unknownCount }} unknown</span>
|
||||
</p>
|
||||
</div>
|
||||
<div data-controller-container tabindex="0" class="monitoring-stat-card">
|
||||
<p class="text-xs text-white/50 uppercase tracking-wide">Fleet Health</p>
|
||||
<p class="text-2xl font-bold text-white">{{ fleetHealthPct }}%</p>
|
||||
<p class="text-xs text-white/40">{{ healthyCount }}/{{ nodeCount }} no alerts</p>
|
||||
<p class="text-xs text-white/40">{{ healthyCount }}/{{ nodeCount }} reporting without alerts</p>
|
||||
</div>
|
||||
<div data-controller-container tabindex="0" class="monitoring-stat-card">
|
||||
<p class="text-xs text-white/50 uppercase tracking-wide">Avg CPU</p>
|
||||
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgCpu)">{{ avgCpu.toFixed(1) }}%</p>
|
||||
<p class="text-xs text-white/40">across fleet</p>
|
||||
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgCpu)">{{ formatMetric(avgCpu, 1) }}</p>
|
||||
<p class="text-xs text-white/40">reporting online nodes</p>
|
||||
</div>
|
||||
<div data-controller-container tabindex="0" class="monitoring-stat-card">
|
||||
<p class="text-xs text-white/50 uppercase tracking-wide">Avg RAM</p>
|
||||
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgMem)">{{ avgMem.toFixed(1) }}%</p>
|
||||
<p class="text-xs text-white/40">across fleet</p>
|
||||
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgMem)">{{ formatMetric(avgMem, 1) }}</p>
|
||||
<p class="text-xs text-white/40">reporting online nodes</p>
|
||||
</div>
|
||||
<div data-controller-container tabindex="0" class="monitoring-stat-card">
|
||||
<p class="text-xs text-white/50 uppercase tracking-wide">Avg Disk</p>
|
||||
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgDisk)">{{ avgDisk.toFixed(1) }}%</p>
|
||||
<p class="text-xs text-white/40">across fleet</p>
|
||||
<p class="text-2xl font-bold text-white" :class="healthTextClass(avgDisk)">{{ formatMetric(avgDisk, 1) }}</p>
|
||||
<p class="text-xs text-white/40">reporting online nodes</p>
|
||||
</div>
|
||||
</div>
|
||||
</template>
|
||||
|
||||
<script setup lang="ts">
|
||||
import { healthTextClass } from './useFleetData'
|
||||
import { healthTextClass, formatMetric } from './useFleetData'
|
||||
|
||||
defineProps<{
|
||||
nodeCount: number
|
||||
onlineCount: number
|
||||
offlineCount: number
|
||||
unknownCount: number
|
||||
fleetHealthPct: number
|
||||
healthyCount: number
|
||||
avgCpu: number
|
||||
avgMem: number
|
||||
avgDisk: number
|
||||
avgCpu: number | null
|
||||
avgMem: number | null
|
||||
avgDisk: number | null
|
||||
}>()
|
||||
</script>
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { mount } from '@vue/test-utils'
|
||||
import FleetNodeGrid from '../FleetNodeGrid.vue'
|
||||
import FleetOverviewCards from '../FleetOverviewCards.vue'
|
||||
import { normalizeFleetNode } from '../useFleetData'
|
||||
|
||||
describe('fleet unavailable readings', () => {
|
||||
afterEach(() => vi.useRealTimers())
|
||||
it('shows an unavailable reading differently from a real zero and explains offline status', () => {
|
||||
vi.useFakeTimers()
|
||||
vi.setSystemTime(new Date('2026-06-10T12:00:00Z'))
|
||||
const nodes = [normalizeFleetNode({node_id: 'offline', reported_at: '2026-06-08T12:00:00Z', cpu_pct: 0}), normalizeFleetNode({node_id: 'unknown'})]
|
||||
const wrapper = mount(FleetNodeGrid, {
|
||||
props: {nodes, sortedNodes: nodes, sortBy: 'status', selectedNodeId: null},
|
||||
global: {mocks: {$ver: (v: string) => v}},
|
||||
})
|
||||
const cards = wrapper.findAll('.fleet-node-card')
|
||||
expect(cards[0].text()).toContain('0%')
|
||||
expect(cards[0].text()).toContain('—')
|
||||
expect(cards[0].text()).toContain('Offline · last seen 2d ago')
|
||||
expect(cards[1].text()).toContain('Status unknown')
|
||||
expect(cards[1].text()).not.toContain('0%')
|
||||
expect(cards[1].text()).toContain('Uptime unavailable')
|
||||
expect(wrapper.text()).not.toContain('NaN')
|
||||
wrapper.unmount()
|
||||
})
|
||||
it('renders empty averages without claiming zero load', () => {
|
||||
const wrapper = mount(FleetOverviewCards, {props: {
|
||||
nodeCount: 1, onlineCount: 0, offlineCount: 0, unknownCount: 1,
|
||||
fleetHealthPct: 0, healthyCount: 0, avgCpu: null, avgMem: null, avgDisk: null,
|
||||
}})
|
||||
expect(wrapper.text()).toContain('1 unknown')
|
||||
expect(wrapper.findAll('.monitoring-stat-card').slice(2).map(card => card.findAll('p').map(p => p.text()).join(' '))).toEqual([
|
||||
'Avg CPU — reporting online nodes', 'Avg RAM — reporting online nodes', 'Avg Disk — reporting online nodes',
|
||||
])
|
||||
wrapper.unmount()
|
||||
})
|
||||
})
|
||||
@@ -1,5 +1,9 @@
|
||||
import { describe, expect, it, vi } from 'vitest'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import {
|
||||
averageMetric,
|
||||
fleetStatus,
|
||||
formatMetric,
|
||||
timeAgo,
|
||||
fleetNodeDisplayName,
|
||||
fleetNodeSubtitle,
|
||||
isOnline,
|
||||
@@ -31,6 +35,7 @@ function node(id: string, reportedAt: string): FleetNode {
|
||||
}
|
||||
|
||||
describe('fleet data helpers', () => {
|
||||
afterEach(() => vi.useRealTimers())
|
||||
it('treats nodes reported within 30 minutes as online', () => {
|
||||
vi.useFakeTimers()
|
||||
vi.setSystemTime(new Date('2026-06-10T12:00:00Z'))
|
||||
@@ -79,9 +84,9 @@ describe('fleet data helpers', () => {
|
||||
expect(normalized.node_name).toBeNull()
|
||||
expect(normalized.hostname).toBeNull()
|
||||
expect(normalized.server_url).toBeNull()
|
||||
expect(normalized.cpu_pct).toBe(0)
|
||||
expect(normalized.mem_pct).toBe(0)
|
||||
expect(normalized.disk_pct).toBe(0)
|
||||
expect(normalized.cpu_pct).toBeNull()
|
||||
expect(normalized.mem_pct).toBeNull()
|
||||
expect(normalized.disk_pct).toBeNull()
|
||||
expect(normalized.containers).toEqual([])
|
||||
expect(normalized.recent_alerts).toEqual([])
|
||||
})
|
||||
@@ -116,3 +121,47 @@ describe('fleet data helpers', () => {
|
||||
expect(normalizeNodeHistoryResponse({})).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
describe('fleet missing data and clock boundaries', () => {
|
||||
afterEach(() => vi.useRealTimers())
|
||||
it('does not turn malformed, missing or future timestamps into online status', () => {
|
||||
vi.useFakeTimers()
|
||||
vi.setSystemTime(new Date('2026-06-10T12:00:00Z'))
|
||||
for (const value of ['', 'broken', '2027-01-01T00:00:00Z']) {
|
||||
expect(fleetStatus(value)).toBe('unknown')
|
||||
expect(isOnline(value)).toBe(false)
|
||||
expect(timeAgo(value)).toBe('Unknown')
|
||||
}
|
||||
expect(fleetStatus('2026-06-10T11:30:00Z')).toBe('offline')
|
||||
expect(fleetStatus('2026-06-10T12:00:30Z')).toBe('online')
|
||||
expect(sortFleetNodes([node('unknown', ''), node('future', '2027-01-01T00:00:00Z'), node('old', '2026-06-10T10:00:00Z'), node('new', '2026-06-10T11:59:00Z')], 'status').map(n => n.node_id)).toEqual(['new', 'old', 'unknown', 'future'])
|
||||
})
|
||||
it('keeps valid zero separate from unavailable and invalid measurements', () => {
|
||||
expect(normalizeFleetNode({cpu_pct: 0}).cpu_pct).toBe(0)
|
||||
for (const value of [null, undefined, NaN, Infinity, -1, 101]) {
|
||||
const result = normalizeFleetNode({cpu_pct: value, mem_pct: value, disk_pct: value})
|
||||
expect(result.cpu_pct).toBeNull()
|
||||
expect(result.mem_pct).toBeNull()
|
||||
expect(result.disk_pct).toBeNull()
|
||||
}
|
||||
expect(formatMetric(null)).toBe('—')
|
||||
expect(formatMetric(0)).toBe('0%')
|
||||
expect(formatMetric(25.25, 1)).toBe('25.3%')
|
||||
})
|
||||
it('averages only real measurements from reporting online nodes', () => {
|
||||
vi.useFakeTimers()
|
||||
vi.setSystemTime(new Date('2026-06-10T12:00:00Z'))
|
||||
const first = node('first', '2026-06-10T11:59:00Z')
|
||||
first.cpu_pct = 0
|
||||
const second = node('second', first.reported_at)
|
||||
second.cpu_pct = 50
|
||||
const missing = node('missing', first.reported_at)
|
||||
missing.cpu_pct = null
|
||||
const offline = node('offline', '2026-06-01T00:00:00Z')
|
||||
offline.cpu_pct = 100
|
||||
expect(averageMetric([first, second, missing, offline], 'cpu_pct')).toBe(25)
|
||||
expect(averageMetric([missing, offline], 'cpu_pct')).toBeNull()
|
||||
expect(averageMetric([], 'cpu_pct')).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
@@ -12,11 +12,11 @@ export interface FleetNode {
|
||||
hostname?: string | null
|
||||
server_url?: string | null
|
||||
version: string
|
||||
uptime_secs: number
|
||||
uptime_secs: number | null
|
||||
cpu_cores: number
|
||||
cpu_pct: number
|
||||
mem_pct: number
|
||||
disk_pct: number
|
||||
cpu_pct: number | null
|
||||
mem_pct: number | null
|
||||
disk_pct: number | null
|
||||
container_count: number
|
||||
running_count: number
|
||||
federation_peers: number
|
||||
@@ -43,7 +43,8 @@ export type SortOption = 'status' | 'last-seen' | 'name'
|
||||
|
||||
// --- Utility Functions ---
|
||||
|
||||
export function formatUptime(secs: number): string {
|
||||
export function formatUptime(secs: number | null): string {
|
||||
if (secs === null) return 'Unavailable'
|
||||
if (secs < 60) return `${secs}s`
|
||||
const days = Math.floor(secs / 86400)
|
||||
const hours = Math.floor((secs % 86400) / 3600)
|
||||
@@ -56,6 +57,7 @@ export function formatUptime(secs: number): string {
|
||||
export function timeAgo(dateStr: string): string {
|
||||
const now = Date.now()
|
||||
const then = new Date(dateStr).getTime()
|
||||
if (!Number.isFinite(then) || then > now + 60_000) return 'Unknown'
|
||||
const diffMs = now - then
|
||||
if (diffMs < 0) return 'just now'
|
||||
const diffSecs = Math.floor(diffMs / 1000)
|
||||
@@ -68,18 +70,34 @@ export function timeAgo(dateStr: string): string {
|
||||
return `${diffDays}d ago`
|
||||
}
|
||||
|
||||
export function isOnline(reportedAt: string): boolean {
|
||||
const thirtyMinMs = 30 * 60 * 1000
|
||||
return Date.now() - new Date(reportedAt).getTime() < thirtyMinMs
|
||||
export function fleetStatus(reportedAt: string): 'online' | 'offline' | 'unknown' {
|
||||
const age = Date.now() - new Date(reportedAt).getTime()
|
||||
if (!Number.isFinite(age) || age < -60_000) return 'unknown'
|
||||
return age < 30 * 60 * 1000 ? 'online' : 'offline'
|
||||
}
|
||||
|
||||
export function healthBarClass(pct: number): string {
|
||||
export function isOnline(reportedAt: string): boolean {
|
||||
return fleetStatus(reportedAt) === 'online'
|
||||
}
|
||||
|
||||
export function formatMetric(value: number | null, digits = 0): string {
|
||||
return value === null ? '—' : `${value.toFixed(digits)}%`
|
||||
}
|
||||
|
||||
export function averageMetric(nodes: FleetNode[], field: 'cpu_pct' | 'mem_pct' | 'disk_pct'): number | null {
|
||||
const values = nodes.filter(n => isOnline(n.reported_at)).map(n => n[field]).filter((v): v is number => v !== null && Number.isFinite(v))
|
||||
return values.length ? values.reduce((sum, value) => sum + value, 0) / values.length : null
|
||||
}
|
||||
|
||||
export function healthBarClass(pct: number | null): string {
|
||||
if (pct === null) return ''
|
||||
if (pct >= 85) return 'monitoring-bar-danger'
|
||||
if (pct >= 60) return 'monitoring-bar-warn'
|
||||
return 'monitoring-bar-ok'
|
||||
}
|
||||
|
||||
export function healthTextClass(pct: number): string {
|
||||
export function healthTextClass(pct: number | null): string {
|
||||
if (pct === null) return ''
|
||||
if (pct >= 85) return 'fleet-text-danger'
|
||||
if (pct >= 60) return 'fleet-text-warn'
|
||||
return ''
|
||||
@@ -132,6 +150,10 @@ export const SORT_OPTIONS: Array<{ label: string; value: SortOption }> = [
|
||||
{ label: 'Name', value: 'name' },
|
||||
]
|
||||
|
||||
function reportTime(node: FleetNode): number {
|
||||
return fleetStatus(node.reported_at) === 'unknown' ? 0 : new Date(node.reported_at).getTime()
|
||||
}
|
||||
|
||||
export function sortFleetNodes(nodes: FleetNode[], sortBy: SortOption): FleetNode[] {
|
||||
const sorted = [...nodes]
|
||||
switch (sortBy) {
|
||||
@@ -140,11 +162,11 @@ export function sortFleetNodes(nodes: FleetNode[], sortBy: SortOption): FleetNod
|
||||
const aOnline = isOnline(a.reported_at)
|
||||
const bOnline = isOnline(b.reported_at)
|
||||
if (aOnline !== bOnline) return aOnline ? -1 : 1
|
||||
return new Date(b.reported_at).getTime() - new Date(a.reported_at).getTime()
|
||||
return reportTime(b) - reportTime(a)
|
||||
})
|
||||
break
|
||||
case 'last-seen':
|
||||
sorted.sort((a, b) => new Date(b.reported_at).getTime() - new Date(a.reported_at).getTime())
|
||||
sorted.sort((a, b) => reportTime(b) - reportTime(a))
|
||||
break
|
||||
case 'name':
|
||||
sorted.sort((a, b) => fleetNodeDisplayName(a).localeCompare(fleetNodeDisplayName(b)))
|
||||
@@ -153,6 +175,15 @@ export function sortFleetNodes(nodes: FleetNode[], sortBy: SortOption): FleetNod
|
||||
return sorted
|
||||
}
|
||||
|
||||
function numberOrNull(value: unknown): number | null {
|
||||
return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : null
|
||||
}
|
||||
|
||||
function percentageOrNull(value: unknown): number | null {
|
||||
const number = numberOrNull(value)
|
||||
return number !== null && number <= 100 ? number : null
|
||||
}
|
||||
|
||||
function numberOrZero(value: unknown): number {
|
||||
return typeof value === 'number' && Number.isFinite(value) ? value : 0
|
||||
}
|
||||
@@ -164,17 +195,17 @@ export function normalizeFleetNode(node: Partial<FleetNode>): FleetNode {
|
||||
hostname: typeof node.hostname === 'string' ? node.hostname : null,
|
||||
server_url: typeof node.server_url === 'string' ? node.server_url : null,
|
||||
version: typeof node.version === 'string' ? node.version : 'unknown',
|
||||
uptime_secs: numberOrZero(node.uptime_secs),
|
||||
uptime_secs: numberOrNull(node.uptime_secs),
|
||||
cpu_cores: numberOrZero(node.cpu_cores),
|
||||
cpu_pct: numberOrZero(node.cpu_pct),
|
||||
mem_pct: numberOrZero(node.mem_pct),
|
||||
disk_pct: numberOrZero(node.disk_pct),
|
||||
cpu_pct: percentageOrNull(node.cpu_pct),
|
||||
mem_pct: percentageOrNull(node.mem_pct),
|
||||
disk_pct: percentageOrNull(node.disk_pct),
|
||||
container_count: numberOrZero(node.container_count),
|
||||
running_count: numberOrZero(node.running_count),
|
||||
federation_peers: numberOrZero(node.federation_peers),
|
||||
recent_alerts: Array.isArray(node.recent_alerts) ? node.recent_alerts : [],
|
||||
containers: Array.isArray(node.containers) ? node.containers : [],
|
||||
reported_at: typeof node.reported_at === 'string' ? node.reported_at : new Date(0).toISOString(),
|
||||
reported_at: typeof node.reported_at === 'string' ? node.reported_at : '',
|
||||
}
|
||||
}
|
||||
|
||||
@@ -195,7 +226,7 @@ type FleetCache = {
|
||||
sortBy: SortOption
|
||||
}
|
||||
|
||||
const FLEET_CACHE_KEY = 'archipelago.fleet.cache.v1'
|
||||
const FLEET_CACHE_KEY = 'archipelago.fleet.cache.v2'
|
||||
|
||||
function readFleetCache(): Partial<FleetCache> {
|
||||
if (typeof window === 'undefined') return {}
|
||||
@@ -246,28 +277,20 @@ export function useFleetData() {
|
||||
// --- Computed ---
|
||||
|
||||
const onlineCount = computed(() => nodes.value.filter(n => isOnline(n.reported_at)).length)
|
||||
const offlineCount = computed(() => nodes.value.length - onlineCount.value)
|
||||
const healthyCount = computed(() => nodes.value.filter(n => n.recent_alerts.length === 0).length)
|
||||
const offlineCount = computed(() => nodes.value.filter(n => fleetStatus(n.reported_at) === 'offline').length)
|
||||
const unknownCount = computed(() => nodes.value.filter(n => fleetStatus(n.reported_at) === 'unknown').length)
|
||||
const healthyCount = computed(() => nodes.value.filter(n => isOnline(n.reported_at) && n.cpu_pct !== null && n.mem_pct !== null && n.disk_pct !== null && n.recent_alerts.length === 0).length)
|
||||
|
||||
const fleetHealthPct = computed(() => {
|
||||
if (!nodes.value.length) return 0
|
||||
return Math.round((healthyCount.value / nodes.value.length) * 100)
|
||||
})
|
||||
|
||||
const avgCpu = computed(() => {
|
||||
if (!nodes.value.length) return 0
|
||||
return nodes.value.reduce((sum, n) => sum + n.cpu_pct, 0) / nodes.value.length
|
||||
})
|
||||
const avgCpu = computed(() => averageMetric(nodes.value, 'cpu_pct'))
|
||||
|
||||
const avgMem = computed(() => {
|
||||
if (!nodes.value.length) return 0
|
||||
return nodes.value.reduce((sum, n) => sum + n.mem_pct, 0) / nodes.value.length
|
||||
})
|
||||
const avgMem = computed(() => averageMetric(nodes.value, 'mem_pct'))
|
||||
|
||||
const avgDisk = computed(() => {
|
||||
if (!nodes.value.length) return 0
|
||||
return nodes.value.reduce((sum, n) => sum + n.disk_pct, 0) / nodes.value.length
|
||||
})
|
||||
const avgDisk = computed(() => averageMetric(nodes.value, 'disk_pct'))
|
||||
|
||||
const selectedNode = computed(() => {
|
||||
if (!selectedNodeId.value) return null
|
||||
@@ -525,7 +548,7 @@ export function useFleetData() {
|
||||
loading, refreshing, errorMessage, nodes, fleetAlerts, alertsLoading,
|
||||
selectedNodeId, selectedNode, nodeHistory, nodeHistoryLoading,
|
||||
autoRefresh, lastRefreshed, sortBy, chartWidth,
|
||||
onlineCount, offlineCount, healthyCount, fleetHealthPct,
|
||||
onlineCount, offlineCount, unknownCount, healthyCount, fleetHealthPct,
|
||||
avgCpu, avgMem, avgDisk, sortedNodes, allAppIds,
|
||||
nodeHistoryLabels, nodeHistoryCpuDatasets, nodeHistoryMemDatasets, nodeHistoryDiskDatasets,
|
||||
refreshAll, selectNode, toggleAutoRefresh, exportFleetData,
|
||||
|
||||
Reference in New Issue
Block a user