diff --git a/aiui/packages/app/src/__tests__/providerBridge.test.ts b/aiui/packages/app/src/__tests__/providerBridge.test.ts new file mode 100644 index 00000000..b8e874a6 --- /dev/null +++ b/aiui/packages/app/src/__tests__/providerBridge.test.ts @@ -0,0 +1,28 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { archyBridge } from '@/services/archyBridge' +const originalParent = window.parent +const origin = 'https://node.example' +afterEach(() => { archyBridge.destroy(); Object.defineProperty(window, 'parent', { value: originalParent, configurable: true }); vi.restoreAllMocks() }) +describe('trusted provider setup bridge', () => { + it('accepts configuration only from the embedding parent, rejects siblings and other origins', () => { + const parent = { postMessage: vi.fn() } + Object.defineProperty(window, 'parent', { value: parent, configurable: true }) + archyBridge.init(origin) + const listener = vi.fn(); const unsubscribe = archyBridge.onProviderConfigured(listener) + const send = (source: unknown, from: string, provider = 'openai') => window.dispatchEvent(new MessageEvent('message', { source: source as Window, origin: from, data: { type: 'ai:provider-configured', provider, model: 'test-model' } })) + send({}, origin); send(parent, 'https://evil.example'); send(parent, origin, 'arbitrary') + expect(listener).not.toHaveBeenCalled() + send(parent, origin) + expect(listener).toHaveBeenCalledExactlyOnceWith({ provider: 'openai', model: 'test-model' }) + archyBridge.requestAISetup() + expect(parent.postMessage).toHaveBeenLastCalledWith({ type: 'ai:setup-request' }, origin) + unsubscribe() + }) + it('replays the selection when the composer mounts after the handshake', () => { + const parent = { postMessage: vi.fn() }; Object.defineProperty(window, 'parent', { value: parent, configurable: true }) + archyBridge.init(origin) + window.dispatchEvent(new MessageEvent('message', { source: parent as unknown as Window, origin, data: { type: 'ai:provider-configured', provider: 'local' } })) + const listener = vi.fn(); const unsubscribe = archyBridge.onProviderConfigured(listener) + expect(listener).toHaveBeenCalledExactlyOnceWith({ provider: 'local', model: '' }); unsubscribe() + }) +}) diff --git a/aiui/packages/app/src/__tests__/useAI.test.ts b/aiui/packages/app/src/__tests__/useAI.test.ts index 330add7a..dcad6757 100644 --- a/aiui/packages/app/src/__tests__/useAI.test.ts +++ b/aiui/packages/app/src/__tests__/useAI.test.ts @@ -249,6 +249,24 @@ describe('useAI', () => { expect(chatStore.isStreaming).toBe(false) }) + it.each([502, 503, 429, 401])('distinguishes HTTP %s from a missing credential', async (status) => { + globalThis.fetch = vi.fn().mockResolvedValue({ ok: false, status, text: async () => 'Provider request failed' }) + const store = useChatStore(); store.webSearchEnabled = false + const ai = useAI(); ai.needsApiKey.value = false + await ai.sendMessage('test') + expect(ai.needsApiKey.value).toBe(status === 401) + expect(store.isStreaming).toBe(false) + }) + + it('offers funding for a Routstr payment-required response without mislabeling it a key error', async () => { + globalThis.fetch = vi.fn().mockResolvedValue({ ok: false, status: 402, text: async () => JSON.stringify({ error: { message: 'Your spending allowance is exhausted' } }) }) + const store = useChatStore(); store.webSearchEnabled = false + const ai = useAI(); ai.setProvider('routstr'); ai.needsApiKey.value = false; ai.needsFunding.value = false + await ai.sendMessage('test') + expect(ai.needsFunding.value).toBe(true); expect(ai.needsApiKey.value).toBe(false) + expect(store.messages.find(m => m.role === 'assistant')?.content).toContain('spending allowance') + }) + it('handles connection errors gracefully', async () => { globalThis.fetch = vi.fn().mockRejectedValue(new Error('Network failure')) diff --git a/aiui/packages/app/src/components/chat/ChatHeader.vue b/aiui/packages/app/src/components/chat/ChatHeader.vue index d51b3436..7f7c4ef8 100644 --- a/aiui/packages/app/src/components/chat/ChatHeader.vue +++ b/aiui/packages/app/src/components/chat/ChatHeader.vue @@ -123,6 +123,7 @@ :style="modelPickerDropdownStyle" @click.stop > +

{{ provider.name }} @@ -247,6 +248,7 @@ import { useAI } from '@/composables/useAI' import { useContentPanel } from '@/composables/useContentPanel' import { downloadConversation, type ExportFormat } from '@/utils/conversation-export' import { parseImportFile } from '@/utils/conversation-import' +import { archyBridge } from '@/services/archyBridge' import { useComparisonMode } from '@/composables/useComparisonMode' defineProps<{ @@ -332,8 +334,7 @@ const modelDisplayName = computed(() => { }) function selectModel(providerId: string, modelId: string) { - setProvider(providerId as 'routstr' | 'claude' | 'openrouter' | 'mock') - setModel(modelId) + if (setProvider(providerId as Parameters[0])) setModel(modelId) showModelPicker.value = false } diff --git a/aiui/packages/app/src/components/chat/ChatWindow.vue b/aiui/packages/app/src/components/chat/ChatWindow.vue index 0e921682..7b8d2008 100644 --- a/aiui/packages/app/src/components/chat/ChatWindow.vue +++ b/aiui/packages/app/src/components/chat/ChatWindow.vue @@ -186,8 +186,9 @@ defineEmits<{ }>() const chatStore = useChatStore() -const { sendMessage, stopGeneration, editAndResend, regenerateLastResponse, activeModel, needsApiKey } = useAI() +const { sendMessage, stopGeneration, editAndResend, regenerateLastResponse, activeModel, needsApiKey, needsFunding } = useAI() const { updatePanelFromText, panelOpen, panelFilms, panelTitle, activeTab, availableTabs, setActiveTab, enterDesignSystemMode } = useContentPanel() +import { archyBridge } from '@/services/archyBridge' import { useCodeContext } from '@/composables/useCodeContext' import { useVisualViewport } from '@/composables/useVisualViewport' const codeContext = useCodeContext() @@ -206,11 +207,19 @@ const showSettings = ref(false) // without fixing anything). watch(needsApiKey, (needs) => { if (needs) { - showSettings.value = true + if (archyBridge.isInArchy()) archyBridge.requestAISetup() + else showSettings.value = true needsApiKey.value = false } }) +watch(needsFunding, needed => { + if (!needed) return + if (archyBridge.isInArchy()) archyBridge.requestAISetup('funding') + else showSettings.value = true + needsFunding.value = false +}) + // Scroll position memory per conversation const scrollPositions = new Map() diff --git a/aiui/packages/app/src/components/settings/ApiKeyManager.vue b/aiui/packages/app/src/components/settings/ApiKeyManager.vue index 17783960..f091163b 100644 --- a/aiui/packages/app/src/components/settings/ApiKeyManager.vue +++ b/aiui/packages/app/src/components/settings/ApiKeyManager.vue @@ -1,5 +1,9 @@ diff --git a/aiui/packages/app/src/composables/useAI.ts b/aiui/packages/app/src/composables/useAI.ts index a65b922f..f1520a3a 100644 --- a/aiui/packages/app/src/composables/useAI.ts +++ b/aiui/packages/app/src/composables/useAI.ts @@ -13,7 +13,7 @@ import { useCodeContext } from '@/composables/useCodeContext' import { apiFetch } from '@/utils/api-fetch' import { useSettingsStore } from '@/stores/settings' -type Provider = 'routstr' | 'claude' | 'openrouter' | 'mock' +type Provider = 'routstr' | 'claude' | 'openrouter' | 'mock' | 'openai' | 'auto' | 'local' // API paths are relative to the base URL so they work both in dev (/) and Archy (/aiui/) const BASE = import.meta.env.BASE_URL || '/' @@ -120,34 +120,15 @@ Prioritize Podcasting 2.0–friendly platforms: Fountain.fm, Podcast Index, Cast Always include these tags so the UI can render rich cards. Write a brief reason why each is worth checking out. ${librarySection}` -const activeProvider = ref('claude') +const activeProvider = ref(archyBridge.isInArchy() ? 'auto' : 'claude') const activeModel = ref('claude-haiku-4.5') -// One-shot signal a send/regenerate/edit failure looked like a missing or -// invalid API key (or an unreachable proxy) rather than a transient/server -// error — consumed by ChatWindow.vue to auto-open Settings so the user isn't -// left in a dead end with no obvious next step. Deliberately narrow (401/403, -// explicit "api key"/"unauthorized" text, or a connection-level failure to -// reach the proxy at all) so a rate-limited or momentarily-flaky provider -// response does NOT send the user to Settings for a problem Settings can't -// fix. Reset to false by the consumer immediately after acting on it, so it -// behaves as a pulse rather than sticky state (each new failure can re-fire). +// Credentials require setup; network failures and provider outages require retry. const needsApiKey = ref(false) - +const needsFunding = ref(false) function looksLikeMissingApiKey(err: string): boolean { - const lower = err.toLowerCase() - return ( - /\b(401|403)\b/.test(err) || - lower.includes('api key') || - lower.includes('x-api-key') || - lower.includes('unauthorized') || - lower.includes('authentication_error') || - lower.includes('failed to fetch') || - lower.includes('econnrefused') || - lower.includes(' 502') || - lower.includes(' 503') - ) + return /\b(401|403)\b|api[ _-]?key|unauthorized|authentication_error|credential/i.test(err) } // ─── Routstr model catalog (fetched from the node's session-gated proxy) ─── @@ -161,7 +142,7 @@ async function refreshRoutstrModels() { routstrModelsFetched = true try { const res = await apiFetch(ROUTSTR_MODELS_PATH) - if (!res.ok) return + if (!res.ok) { routstrModelsFetched = false; return } const data = await res.json() if (Array.isArray(data?.data)) { routstrModels.value = data.data @@ -170,13 +151,21 @@ async function refreshRoutstrModels() { id: m.id as string, name: (m.name as string) || (m.id as string), })) - } + if (activeProvider.value === 'routstr' && activeModel.value === 'routstr-unavailable' && routstrModels.value[0]) activeModel.value = routstrModels.value[0].id + } else { routstrModelsFetched = false } } catch { routstrModelsFetched = false // allow a retry on the next send/open } } const availableProviders = computed(() => { + if (archyBridge.isInArchy()) return [ + { id: 'local' as Provider, name: 'Local AI', models: [{ id: 'node', name: 'Node configuration' }] }, + { id: 'auto' as Provider, name: 'Node AI', models: [{ id: 'node', name: 'Node configuration' }] }, + { id: 'claude' as Provider, name: 'Claude API', models: [{ id: 'node', name: 'Node configuration' }] }, + { id: 'openai' as Provider, name: 'OpenAI API', models: [{ id: activeProvider.value === 'openai' ? activeModel.value : 'node', name: activeProvider.value === 'openai' ? activeModel.value : 'Configure model' }] }, + { id: 'routstr' as Provider, name: 'Routstr (sats)', models: routstrModels.value.length ? routstrModels.value : [{ id: 'routstr-unavailable', name: 'Models unavailable — retry' }] }, + ] const providers: { id: Provider; name: string; models: { id: string; name: string }[] }[] = [ { id: 'routstr', @@ -187,7 +176,7 @@ const availableProviders = computed(() => { }, { id: 'claude', - name: 'Claude (Max)', + name: 'Claude API', models: [ { id: 'claude-haiku-4.5', name: 'Claude 4.5 Haiku' }, { id: 'claude-sonnet-4', name: 'Claude Sonnet 4' }, @@ -207,24 +196,36 @@ const availableProviders = computed(() => { }) providers.push({ id: 'mock', - name: 'Local (no API)', + name: 'Demo echo', models: [{ id: 'echo', name: 'Echo (mirror input)' }], }) return providers }) function setProvider(provider: Provider) { + if (archyBridge.isInArchy()) { + if (provider === 'routstr' && activeProvider.value === 'routstr') return true + archyBridge.requestAISetup(); return false + } activeProvider.value = provider const p = availableProviders.value.find((pp) => pp.id === provider) if (p && p.models.length > 0) { activeModel.value = p.models[0].id } + return true } function setModel(model: string) { + if (archyBridge.isInArchy() && activeProvider.value !== 'routstr') { archyBridge.requestAISetup(); return } activeModel.value = model } +archyBridge.onProviderConfigured(({ provider, model }) => { + activeProvider.value = provider + activeModel.value = model || (provider === 'routstr' ? routstrModels.value[0]?.id || 'routstr-unavailable' : 'node') + if (provider === 'routstr') void refreshRoutstrModels() +}) + interface ChatMessage { role: 'user' | 'assistant' content: string @@ -448,6 +449,7 @@ async function streamRoutstr( }) const bodyText = await res.text().catch(() => '') + if (res.status === 402) needsFunding.value = true if (!res.ok) { // The node's refusals carry a plain-language error.message (budget not // set, budget spent, wallet can't fund) — surface it verbatim. @@ -974,5 +976,6 @@ export function useAI() { setProvider, setModel, needsApiKey, + needsFunding, } } diff --git a/aiui/packages/app/src/services/archyBridge.ts b/aiui/packages/app/src/services/archyBridge.ts index c4fb3f5f..3cd2e71e 100644 --- a/aiui/packages/app/src/services/archyBridge.ts +++ b/aiui/packages/app/src/services/archyBridge.ts @@ -55,6 +55,10 @@ interface ThemeInfo { type PermissionsCallback = (categories: AIContextCategory[]) => void type ThemeCallback = (theme: ThemeInfo) => void +export interface AIProviderSelection { provider: 'auto' | 'local' | 'claude' | 'openai' | 'routstr'; model: string } +const providerCallbacks = new Set<(selection: AIProviderSelection) => void>() +let currentProvider: AIProviderSelection | null = null + let requestId = 0 const pendingRequests = new Map void @@ -80,12 +84,19 @@ function postToParent(msg: unknown) { function handleMessage(event: MessageEvent) { // Always validate origin — reject if not configured or mismatched - if (!allowedOrigin || event.origin !== allowedOrigin) return + if (!allowedOrigin || event.origin !== allowedOrigin || event.source !== window.parent) return const msg = event.data if (!msg || typeof msg.type !== 'string') return switch (msg.type) { + case 'ai:provider-configured': { + if (!['auto', 'local', 'claude', 'openai', 'routstr'].includes(msg.provider)) break + const selection = { provider: msg.provider as AIProviderSelection['provider'], model: typeof msg.model === 'string' ? msg.model : '' } + currentProvider = selection + for (const callback of providerCallbacks) callback(selection) + break + } case 'context:response': { const pending = pendingRequests.get(msg.id) if (pending) { @@ -223,11 +234,21 @@ export const archyBridge = { } }, + requestAISetup(reason?: 'funding') { postToParent({ type: 'ai:setup-request', ...(reason ? { reason } : {}) }) }, + + onProviderConfigured(callback: (selection: AIProviderSelection) => void) { + providerCallbacks.add(callback) + if (currentProvider) callback(currentProvider) + return () => { providerCallbacks.delete(callback) } + }, + /** Clean up listeners */ destroy() { window.removeEventListener('message', handleMessage) pendingRequests.clear() initialized = false + currentProvider = null + allowedOrigin = null }, /** Check if running inside Archy iframe */ diff --git a/core/archipelago/src/api/rpc/analytics.rs b/core/archipelago/src/api/rpc/analytics.rs index 3532a812..8e6a7e97 100644 --- a/core/archipelago/src/api/rpc/analytics.rs +++ b/core/archipelago/src/api/rpc/analytics.rs @@ -352,20 +352,19 @@ impl RpcHandler { (Some(u), Some(t)) if t > 0 => { serde_json::json!((u as f64 / t as f64 * 100.0).round()) } - _ => serde_json::json!(0), + _ => serde_json::Value::Null, } }; let apps = state.map(|s| s.apps.as_slice()).unwrap_or(&[]); let reported_at = state .map(|s| s.timestamp.clone()) - .or_else(|| n.last_seen.clone()) - .unwrap_or_else(|| n.added_at.clone()); + .or_else(|| n.last_seen.clone()); let mut report = serde_json::json!({ "node_id": n.did, "node_name": state.and_then(|s| s.node_name.clone()).or_else(|| n.name.clone()), - "uptime_secs": state.and_then(|s| s.uptime_secs).unwrap_or(0), - "cpu_pct": state.and_then(|s| s.cpu_usage_percent).map(|v| v.round()).unwrap_or(0.0), + "uptime_secs": state.and_then(|s| s.uptime_secs), + "cpu_pct": state.and_then(|s| s.cpu_usage_percent).filter(|v| v.is_finite() && (0.0..=100.0).contains(v)).map(|v| v.round()), "mem_pct": pct(state.and_then(|s| s.mem_used_bytes), state.and_then(|s| s.mem_total_bytes)), "disk_pct": pct(state.and_then(|s| s.disk_used_bytes), state.and_then(|s| s.disk_total_bytes)), "container_count": apps.len(), @@ -561,7 +560,7 @@ fn annotate_fleet_report(report: &mut serde_json::Value) { let is_online = reported .map(|dt| { let age = chrono::Utc::now().signed_duration_since(dt); - age.num_minutes() < 30 + age.num_seconds() >= -60 && age.num_seconds() < 1800 }) .unwrap_or(false); @@ -569,7 +568,9 @@ fn annotate_fleet_report(report: &mut serde_json::Value) { .map(|dt| { let age = chrono::Utc::now().signed_duration_since(dt); let mins = age.num_minutes(); - if mins < 1 { + if age.num_seconds() < -60 { + "unknown (clock ahead)".to_string() + } else if mins < 1 { "just now".to_string() } else if mins < 60 { format!("{}m ago", mins) diff --git a/core/archipelago/src/api/rpc/federation/handlers.rs b/core/archipelago/src/api/rpc/federation/handlers.rs index c901086d..98465e90 100644 --- a/core/archipelago/src/api/rpc/federation/handlers.rs +++ b/core/archipelago/src/api/rpc/federation/handlers.rs @@ -602,14 +602,39 @@ impl RpcHandler { None }; + // Reuse the minute collector instead of running expensive probes for + // every peer. An absent/stalled collector is unknown, never zero load. + let now = chrono::Utc::now().timestamp(); + let latest = self + .metrics_store + .latest() + .await + .filter(|sample| (0..=180).contains(&now.saturating_sub(sample.timestamp))); + let metrics = latest.as_ref().map(|sample| &sample.system); + let uptime = tokio::fs::read_to_string("/proc/uptime") + .await + .ok() + .and_then(|s| s.split_whitespace().next()?.parse::().ok()) + .filter(|v| v.is_finite() && *v >= 0.0) + .map(|v| v as u64); let state = federation::build_local_state( apps, - 0.0, - 0, - 0, - 0, - 0, - 0, + metrics + .map(|m| m.cpu_percent) + .filter(|v| v.is_finite() && (0.0..=100.0).contains(v)), + metrics + .filter(|m| m.mem_total_bytes > 0) + .map(|m| m.mem_used_bytes), + metrics + .filter(|m| m.mem_total_bytes > 0) + .map(|m| m.mem_total_bytes), + metrics + .filter(|m| m.disk_total_bytes > 0) + .map(|m| m.disk_used_bytes), + metrics + .filter(|m| m.disk_total_bytes > 0) + .map(|m| m.disk_total_bytes), + uptime, tor_active, server_name, nostr_npub, @@ -1254,76 +1279,120 @@ impl RpcHandler { ); } + let reply = self.prepare_peer_approval_reply(&req).await?; + // Persist the operator decision before transport. A relay outage must + // not require another approval or lose the already-authorized reply. + pending::decide(&self.config.data_dir, id, pending::PendingState::Approved).await?; + let delivered = self.deliver_peer_approval_reply(&reply).await?; + Ok(serde_json::json!({ "approved": true, "id": id, "delivery_pending": !delivered })) + } + + async fn prepare_peer_approval_reply( + &self, + req: &pending::PendingPeerRequest, + ) -> Result { + use federation::handshake_delivery::{self, ApprovalReply}; + if let Some(reply) = handshake_delivery::find(&self.config.data_dir, &req.id).await? { + anyhow::ensure!( + reply.recipient == req.from_nostr_pubkey && reply.expected_did == req.from_did, + "Approval recipient changed" + ); + return Ok(reply); + } let (data, _) = self.state_manager.get_snapshot().await; let local_did = identity::did_key_from_pubkey_hex(&data.server_info.pubkey)?; let local_onion = data .server_info .tor_address - .clone() + .as_deref() .ok_or_else(|| anyhow::anyhow!("Tor address not available"))?; - let local_pubkey = data.server_info.pubkey.clone(); - - // Generate a one-shot federation invite. The code embeds OUR onion - // and OUR pubkey, but it leaves this box only inside the NIP-44 - // ciphertext below. - let identity_dir = self.config.data_dir.join("identity"); - let local_fips_npub = identity::fips_npub(&identity_dir).await.unwrap_or(None); - // Discovery/connection-request approvals admit the requester as - // Observer — the invite itself now carries that level, so both - // sides converge on Observer without post-hoc demotion. + let local_fips_npub = identity::fips_npub(&self.config.data_dir.join("identity")) + .await + .unwrap_or(None); let invite_code = federation::create_invite( &self.config.data_dir, &local_did, - &local_onion, - &local_pubkey, + local_onion, + &data.server_info.pubkey, local_fips_npub.as_deref(), TrustLevel::Observer, ) .await?; + handshake_delivery::stage( + &self.config.data_dir, + ApprovalReply { + request_id: req.id.clone(), + recipient: req.from_nostr_pubkey.clone(), + expected_did: req.from_did.clone(), + invite_code, + attempts: 0, + next_attempt: 0, + }, + ) + .await + } - // Pre-add the requester to OUR federation list as Observer so that - // when their `federation.peer-joined` callback arrives over Tor we - // already trust their pubkey enough to accept the join. Their DID - // and pubkey come from the request — we'll cross-check the pubkey - // against the eventual peer-joined signature in the existing - // verification path (handlers.rs line ~365). - if !req.from_did.is_empty() { - // We don't know the requester's onion or ed25519 pubkey yet — - // they'll send those in the federation.peer-joined callback - // after they apply our invite. Until then we can't add a real - // FederatedNode entry. We just store the pending row as - // Approved so the UI shows progress, and trust the existing - // peer-joined handler to admit them as Observer when they call. - // - // Caveat: peer-joined currently hardcodes TrustLevel::Trusted. - // We override that below by demoting on success. - debug!( - requester_did = %req.from_did, - "Approval pending — waiting for federation.peer-joined callback over Tor" - ); - } - - // Encrypt + send the invite over NIP-44 to the requester. - let identity_dir = self.config.data_dir.join("identity"); - nostr_handshake::send_peer_invite( - &identity_dir, - &req.from_nostr_pubkey, - &invite_code, - &self.config.nostr_relays, + async fn deliver_peer_approval_reply( + &self, + reply: &federation::handshake_delivery::ApprovalReply, + ) -> Result { + let Some(claimed) = federation::handshake_delivery::claim( + &self.config.data_dir, + &reply.request_id, + chrono::Utc::now().timestamp(), + ) + .await? + else { + return Ok(false); + }; + let result = nostr_handshake::send_peer_invite( + &self.config.data_dir.join("identity"), + &claimed.recipient, + &claimed.invite_code, + &self.handshake_relays().await, self.config.nostr_tor_proxy.as_deref(), ) - .await?; + .await; + if result.is_err() { + warn!(request_id = %reply.request_id, "Peer approval delivery deferred; durable retry scheduled"); + } + Ok(result.is_ok()) + } - pending::set_state(&self.config.data_dir, id, pending::PendingState::Approved).await?; - info!( - id = %id, - from = %req.from_nostr_pubkey, - "Approved peer request and shipped invite over NIP-44" - ); - Ok(serde_json::json!({ - "approved": true, - "id": id, - })) + /// Recover relay loss and legacy approvals without changing trust or + /// resurrecting a node that the operator explicitly removed. + pub(in crate::api::rpc) async fn retry_peer_approval_replies(&self) -> Result<()> { + let requests = pending::load_pending(&self.config.data_dir).await?; + let nodes = federation::load_nodes(&self.config.data_dir).await?; + let removed = federation::load_removed_dids(&self.config.data_dir).await?; + let cutoff = chrono::Utc::now() - chrono::Duration::days(30); + let mut sent = 0; + for req in requests { + if req.outbound || req.state != pending::PendingState::Approved { + federation::handshake_delivery::remove(&self.config.data_dir, &req.id).await?; + continue; + } + let expired = chrono::DateTime::parse_from_rfc3339(&req.received_at) + .map(|time| time < cutoff) + .unwrap_or(true); + if expired + || removed.contains(&req.from_did) + || nodes.iter().any(|node| node.did == req.from_did) + { + federation::handshake_delivery::remove(&self.config.data_dir, &req.id).await?; + continue; + } + if req.from_did.is_empty() || sent >= 4 { + continue; + } + let reply = self.prepare_peer_approval_reply(&req).await?; + if reply.next_attempt > chrono::Utc::now().timestamp() { + continue; + } + self.deliver_peer_approval_reply(&reply).await?; + sent += 1; + } + Ok(()) } /// federation.reject-request — drop a pending request and, if requested, @@ -1353,19 +1422,19 @@ impl RpcHandler { ); } + pending::decide(&self.config.data_dir, id, pending::PendingState::Rejected).await?; if notify { let identity_dir = self.config.data_dir.join("identity"); let _ = nostr_handshake::send_peer_reject( &identity_dir, &req.from_nostr_pubkey, reason, - &self.config.nostr_relays, + &self.handshake_relays().await, self.config.nostr_tor_proxy.as_deref(), ) .await; } - pending::set_state(&self.config.data_dir, id, pending::PendingState::Rejected).await?; info!(id = %id, from = %req.from_nostr_pubkey, "Rejected peer request"); Ok(serde_json::json!({ "rejected": true, "id": id })) } @@ -1410,7 +1479,7 @@ impl RpcHandler { &identity_dir, &req.from_nostr_pubkey, reason, - &self.config.nostr_relays, + &self.handshake_relays().await, self.config.nostr_tor_proxy.as_deref(), ) .await diff --git a/core/archipelago/src/api/rpc/federation/handshake_tests.rs b/core/archipelago/src/api/rpc/federation/handshake_tests.rs new file mode 100644 index 00000000..c4407d2b --- /dev/null +++ b/core/archipelago/src/api/rpc/federation/handshake_tests.rs @@ -0,0 +1,276 @@ +//! Exercise encrypted replies through a relay configured in the UI only. +use crate::federation::pending::{self, PendingState}; +use futures_util::{SinkExt, StreamExt}; +use nostr_sdk::prelude::{nip44, Event, Keys}; +use std::sync::Arc; +use std::time::Duration; + +#[tokio::test] +async fn managed_relay_receives_approval_rejection_and_cancellation() { + for (operation, accepted) in [ + ("approve", true), + ("reject", true), + ("cancel", true), + ("approve", false), + ("retry", true), + ] { + let tmp = tempfile::tempdir().unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let relay_url = format!("ws://{}", listener.local_addr().unwrap()); + let relay = tokio::spawn(async move { + let (socket, _) = listener.accept().await.unwrap(); + let mut ws = tokio_tungstenite::accept_async(socket).await.unwrap(); + while let Some(Ok(message)) = ws.next().await { + if !message.is_text() { + continue; + } + let value: serde_json::Value = + serde_json::from_str(message.to_text().unwrap()).unwrap(); + if value[0] != "EVENT" { + continue; + } + let event: Event = serde_json::from_value(value[1].clone()).unwrap(); + event.verify().unwrap(); + ws.send(tokio_tungstenite::tungstenite::Message::Text( + serde_json::json!([ + "OK", + event.id.to_hex(), + accepted, + "blocked: fixture rejection" + ]) + .to_string(), + )) + .await + .unwrap(); + return event; + } + panic!("relay closed without a signed event"); + }); + let mut config = crate::config::Config::default(); + config.data_dir = tmp.path().to_path_buf(); + config.nostr_relays.clear(); + config.nostr_tor_proxy = None; + crate::nostr_relays::save_relays( + tmp.path(), + &crate::nostr_relays::RelayStore { + relays: vec![crate::nostr_relays::RelayConfig { + url: relay_url, + enabled: true, + added_at: chrono::Utc::now().to_rfc3339(), + }], + }, + ) + .await + .unwrap(); + let sender = Keys::parse(&"11".repeat(32)).unwrap(); + let recipient = Keys::parse(&"22".repeat(32)).unwrap(); + let identity_dir = tmp.path().join("identity"); + tokio::fs::create_dir_all(&identity_dir).await.unwrap(); + tokio::fs::write(identity_dir.join("nostr_secret"), "11".repeat(32)) + .await + .unwrap(); + tokio::fs::write(identity_dir.join("node_key"), [0x33; 32]) + .await + .unwrap(); + let state = Arc::new(crate::state::StateManager::new()); + state + .mutate_data(|data| { + data.server_info.pubkey = "33".repeat(32); + data.server_info.tor_address = Some(format!("{}.onion", "a".repeat(56))); + }) + .await; + let handler = crate::api::rpc::RpcHandler::new( + config, + state, + Arc::new(crate::monitoring::MetricsStore::new()), + crate::session::SessionStore::new_for_tests(tmp.path().join("sessions.json")), + None, + None, + ) + .await + .unwrap(); + let row = if operation == "cancel" { + pending::insert_outbound( + tmp.path(), + recipient.public_key().to_hex(), + String::new(), + crate::identity::did_key_from_pubkey_hex(&"44".repeat(32)).unwrap(), + None, + None, + ) + .await + .unwrap() + } else { + pending::insert_inbound( + tmp.path(), + recipient.public_key().to_hex(), + String::new(), + crate::identity::did_key_from_pubkey_hex(&"44".repeat(32)).unwrap(), + None, + None, + ) + .await + .unwrap() + .unwrap() + }; + if operation == "retry" { + pending::decide(tmp.path(), &row.id, PendingState::Approved) + .await + .unwrap(); + } + let params = Some(serde_json::json!({"id": row.id, "notify": true})); + let action = async { + match operation { + "approve" => handler.handle_federation_approve_request(params).await, + "reject" => handler.handle_federation_reject_request(params).await, + "retry" => handler + .retry_peer_approval_replies() + .await + .map(|_| serde_json::json!({"ok": true})), + _ => handler.handle_federation_cancel_request(params).await, + } + }; + let outcome = tokio::time::timeout(Duration::from_secs(20), action) + .await + .unwrap(); + assert_eq!(outcome.is_ok(), accepted || operation == "approve"); + let event = tokio::time::timeout(Duration::from_secs(5), relay) + .await + .unwrap() + .unwrap(); + assert_eq!(event.pubkey, sender.public_key()); + let plaintext = + nip44::decrypt(recipient.secret_key(), &event.pubkey, &event.content).unwrap(); + let message: serde_json::Value = serde_json::from_str(&plaintext).unwrap(); + let expected = match operation { + "approve" | "retry" => "peer-invite", + "reject" => "peer-reject", + _ => "peer-cancel", + }; + assert_eq!(message["type"], expected); + let saved = pending::find_by_id(tmp.path(), &row.id).await.unwrap(); + if !accepted { + assert_eq!( + saved.unwrap().state, + if operation == "approve" { + PendingState::Approved + } else { + PendingState::Pending + } + ); + if operation == "approve" { + assert_eq!(outcome.unwrap()["delivery_pending"], true); + let durable = crate::federation::handshake_delivery::find(tmp.path(), &row.id) + .await + .unwrap() + .unwrap(); + assert_eq!(durable.recipient, recipient.public_key().to_hex()); + assert_eq!(durable.attempts, 1); + } + assert!(crate::federation::load_nodes(tmp.path()) + .await + .unwrap() + .is_empty()); + continue; + } + match operation { + "approve" | "retry" => { + assert_eq!(saved.unwrap().state, PendingState::Approved); + let first = crate::federation::handshake_delivery::find(tmp.path(), &row.id) + .await + .unwrap() + .unwrap(); + assert_eq!(first.attempts, 1); + // Concurrent/background polling honors the persisted backoff. + handler.retry_peer_approval_replies().await.unwrap(); + let second = crate::federation::handshake_delivery::find(tmp.path(), &row.id) + .await + .unwrap() + .unwrap(); + assert_eq!(second.attempts, 1); + assert_eq!(first.invite_code, second.invite_code); + let invite = + crate::federation::parse_invite(message["invite_code"].as_str().unwrap()) + .unwrap(); + assert_eq!(invite.trust_level, crate::federation::TrustLevel::Observer); + assert!(!event.content.contains(".onion")); + } + "reject" => assert_eq!(saved.unwrap().state, PendingState::Rejected), + _ => assert!(saved.is_none()), + } + } +} + +#[tokio::test] +async fn federation_metrics_are_collected_values_or_unknown_never_placeholders() { + for age in [None, Some(0), Some(181), Some(-120)] { + let tmp = tempfile::tempdir().unwrap(); + let mut config = crate::config::Config::default(); + config.data_dir = tmp.path().to_path_buf(); + let metrics = Arc::new(crate::monitoring::MetricsStore::new()); + if let Some(age) = age { + metrics + .push( + serde_json::from_value(serde_json::json!({ + "timestamp": chrono::Utc::now().timestamp() - age, + "system": {"cpu_percent": 37.5, "mem_used_bytes": 200, + "mem_total_bytes": 800, "disk_used_bytes": 600, + "disk_total_bytes": 1000, "net_rx_bytes": 0, "net_tx_bytes": 0, + "load_avg_1": 0.0, "load_avg_5": 0.0, "load_avg_15": 0.0}, + "containers": [], "rpc_latency_ms": 0.0, "ws_connections": 0 + })) + .unwrap(), + ) + .await; + } + let handler = crate::api::rpc::RpcHandler::new( + config, + Arc::new(crate::state::StateManager::new()), + metrics, + crate::session::SessionStore::new_for_tests(tmp.path().join("sessions.json")), + None, + None, + ) + .await + .unwrap(); + let snapshot = handler.handle_federation_get_state().await.unwrap(); + if age == Some(0) { + assert_eq!(snapshot["cpu_usage_percent"], 37.5); + assert_eq!(snapshot["mem_used_bytes"], 200); + assert_eq!(snapshot["disk_total_bytes"], 1000); + } else { + for field in [ + "cpu_usage_percent", + "mem_used_bytes", + "mem_total_bytes", + "disk_used_bytes", + "disk_total_bytes", + ] { + assert!( + snapshot.get(field).is_none_or(|v| v.is_null()), + "{age:?}: {field}" + ); + } + } + let peer = serde_json::from_value(serde_json::json!({ + "did": "did:key:test", "pubkey": "11".repeat(32), "onion": "test.onion", + "trust_level": "trusted", "added_at": chrono::Utc::now().to_rfc3339(), + "last_state": snapshot + })) + .unwrap(); + crate::federation::save_nodes(tmp.path(), &[peer]) + .await + .unwrap(); + let fleet = handler.handle_telemetry_fleet_status().await.unwrap(); + let report = &fleet["nodes"][0]; + if age == Some(0) { + assert_eq!(report["cpu_pct"], 38.0); + assert_eq!(report["mem_pct"], 25.0); + assert_eq!(report["disk_pct"], 60.0); + } else { + for field in ["cpu_pct", "mem_pct", "disk_pct"] { + assert!(report[field].is_null(), "{age:?}: {field}"); + } + } + } +} diff --git a/core/archipelago/src/api/rpc/federation/mod.rs b/core/archipelago/src/api/rpc/federation/mod.rs index fbefc121..2dd40a30 100644 --- a/core/archipelago/src/api/rpc/federation/mod.rs +++ b/core/archipelago/src/api/rpc/federation/mod.rs @@ -1,4 +1,6 @@ mod handlers; +#[cfg(test)] +mod handshake_tests; use anyhow::Result; diff --git a/core/archipelago/src/api/rpc/handshake.rs b/core/archipelago/src/api/rpc/handshake.rs index 6399f7a1..2373256b 100644 --- a/core/archipelago/src/api/rpc/handshake.rs +++ b/core/archipelago/src/api/rpc/handshake.rs @@ -276,6 +276,11 @@ impl RpcHandler { } Err(e) => tracing::debug!("background handshake poll failed: {e:#}"), } + if load_discovery_state(&self.config.data_dir).await.enabled { + if let Err(error) = self.retry_peer_approval_replies().await { + tracing::warn!("Peer approval retry could not complete: {error:#}"); + } + } } pub(super) async fn handle_handshake_poll(&self) -> Result { @@ -356,6 +361,16 @@ impl RpcHandler { ); continue; }; + let scoped_invite = match crate::federation::restrict_discovery_invite( + invite_code, + &row.from_did, + ) { + Ok(code) => code, + Err(_) => { + tracing::warn!("Rejected peer invite with mismatched identity"); + continue; + } + }; let row_id = row.id.clone(); let (data, _) = self.state_manager.get_snapshot().await; let local_did = @@ -373,7 +388,7 @@ impl RpcHandler { let local_name = data.server_info.name.clone(); match crate::federation::accept_invite( &self.config.data_dir, - invite_code, + &scoped_invite, &local_did, &local_onion, &local_pubkey, diff --git a/core/archipelago/src/api/rpc/system/handlers.rs b/core/archipelago/src/api/rpc/system/handlers.rs index 9f89721b..c1ec24e6 100644 --- a/core/archipelago/src/api/rpc/system/handlers.rs +++ b/core/archipelago/src/api/rpc/system/handlers.rs @@ -1025,10 +1025,46 @@ impl RpcHandler { .ok_or_else(|| anyhow::anyhow!("Missing key"))?; match key { - "claude_api_key_set" => { - let key_file = self.config.data_dir.join("secrets/claude-api-key"); - let has_key = tokio::fs::metadata(&key_file).await.is_ok(); - Ok(serde_json::json!({ "value": has_key })) + "claude_api_key_set" | "openai_api_key_set" => { + let provider = if key == "claude_api_key_set" { + "claude" + } else { + "openai" + }; + Ok( + serde_json::json!({ "value": crate::settings::model_provider::has_key(&self.config.data_dir, provider).await }), + ) + } + "ai_provider" => { + let settings = + crate::settings::model_provider::ModelProvider::load(&self.config.data_dir) + .await?; + Ok(serde_json::json!({ "value": settings })) + } + "ai_provider_status" => { + let settings = + crate::settings::model_provider::ModelProvider::load(&self.config.data_dir) + .await?; + let local = tokio::time::timeout(std::time::Duration::from_secs(4), async { + let (detected, _) = crate::api::rpc::mesh::assistant::detect_ollama().await; + detected + && crate::assistant::backends::ollama::model_supports_tools( + crate::assistant::backends::ollama::OLLAMA_BASE_URL, + crate::assistant::backends::ollama::OLLAMA_DEFAULT_MODEL, + ) + .await + }); + let (claude, openai, local) = tokio::join!( + crate::settings::model_provider::has_key(&self.config.data_dir, "claude"), + crate::settings::model_provider::has_key(&self.config.data_dir, "openai"), + local, + ); + let budget = crate::assistant::AssistantBudget::load(&self.config.data_dir).await; + Ok(serde_json::json!({ "value": { + "schema": 1, "settings": settings, "claude_configured": claude, + "openai_configured": openai, "local_ready": local.ok(), + "routstr_remaining_sats": budget.remaining_sats(), + }})) } _ => Ok(serde_json::json!({ "value": null })), } @@ -1210,38 +1246,21 @@ impl RpcHandler { let value = params.get("value").and_then(|v| v.as_str()).unwrap_or(""); match key { - "claude_api_key" => { - let secrets_dir = self.config.data_dir.join("secrets"); - tokio::fs::create_dir_all(&secrets_dir) - .await - .context("Failed to create secrets dir")?; - let key_file = secrets_dir.join("claude-api-key"); - - if value.is_empty() { - // Remove key - tokio::fs::remove_file(&key_file).await.ok(); - info!("Claude API key removed"); + "claude_api_key" | "openai_api_key" => { + let provider = if key == "claude_api_key" { + "claude" } else { - // Save key - tokio::fs::write(&key_file, value) - .await - .context("Failed to write API key")?; - #[cfg(unix)] - { - use std::os::unix::fs::PermissionsExt; - std::fs::set_permissions(&key_file, std::fs::Permissions::from_mode(0o600)) - .ok(); - } - info!("Claude API key saved"); - } - - // `secrets/claude-api-key` (above) is deliberately the ONLY - // Claude key ledger on this node (13-02-PLAN.md). A second - // copy used to be written alongside it for a standalone, - // unauthenticated sidecar process on port 3142 — that - // sidecar and its key copy are retired; the session-gated - // Rust daemon reads this one file directly. - + "openai" + }; + crate::settings::model_provider::save_key(&self.config.data_dir, provider, value) + .await?; + info!(provider, "AI provider credential updated"); + Ok(serde_json::json!({ "saved": true })) + } + "ai_provider" => { + let settings: crate::settings::model_provider::ModelProvider = + serde_json::from_str(value).context("Invalid AI provider settings")?; + settings.save(&self.config.data_dir).await?; Ok(serde_json::json!({ "saved": true })) } _ => anyhow::bail!("Unknown setting: {}", key), diff --git a/core/archipelago/src/assistant/backends/mod.rs b/core/archipelago/src/assistant/backends/mod.rs index d8f24057..72f1245c 100644 --- a/core/archipelago/src/assistant/backends/mod.rs +++ b/core/archipelago/src/assistant/backends/mod.rs @@ -11,6 +11,7 @@ use crate::api::rpc::RpcHandler; pub mod claude; pub mod ollama; +pub mod openai; pub mod routstr; #[cfg(test)] pub mod scripted; @@ -39,6 +40,8 @@ pub trait Backend: Send + Sync { pub enum BackendId { Ollama, Claude, + Openai, + Unavailable, /// 13-13: the third D-04 leg. Not currently returned as the "primary" /// id by `select_backend` (mirroring the existing convention that the /// returned id names the primary attempt, not necessarily which leg of @@ -52,11 +55,23 @@ impl std::fmt::Display for BackendId { match self { BackendId::Ollama => write!(f, "ollama"), BackendId::Claude => write!(f, "claude"), + BackendId::Openai => write!(f, "openai"), + BackendId::Unavailable => write!(f, "unavailable"), BackendId::Routstr => write!(f, "routstr"), } } } +struct InvalidProviderSettings; +#[async_trait] +impl Backend for InvalidProviderSettings { + async fn send(&self, _: &str, _: &[ToolDef], _: &[ChatMessage]) -> Result { + anyhow::bail!( + "AI connection settings could not be loaded. Review them before sending a message." + ) + } +} + /// D-04's per-call fallback: try `primary`'s `send()`, and on a transport /// error fall through to `secondary` for that SAME call rather than /// failing the whole turn — a local model that answers earlier turns and @@ -115,6 +130,54 @@ fn ollama_is_selectable(detected: bool, tool_capable: bool) -> bool { /// tools-free degrade. pub async fn select_backend(handler: &RpcHandler) -> (Box, BackendId) { let data_dir = handler.data_dir(); + // An explicit provider is a privacy and billing choice. Never silently + // fall through to another provider if its credentials or network fail. + match crate::settings::model_provider::ModelProvider::load(data_dir).await { + Ok(settings) => match settings.provider { + crate::settings::model_provider::Provider::Openai => { + return ( + Box::new(openai::OpenaiBackend::new( + data_dir.to_path_buf(), + settings.openai_model, + )), + BackendId::Openai, + ) + } + crate::settings::model_provider::Provider::Claude => { + return ( + Box::new(claude::ClaudeBackend::new(data_dir.to_path_buf())), + BackendId::Claude, + ) + } + crate::settings::model_provider::Provider::Local => { + return ( + Box::new(ollama::OllamaBackend::new( + ollama::OLLAMA_BASE_URL.to_string(), + ollama::OLLAMA_DEFAULT_MODEL.to_string(), + )), + BackendId::Ollama, + ) + } + crate::settings::model_provider::Provider::Routstr => { + let budget = crate::assistant::AssistantBudget::load(data_dir).await; + let mints = crate::wallet::ecash::load_accepted_mints(data_dir) + .await + .map(|m| m.mints) + .unwrap_or_default(); + return ( + Box::new(routstr::RoutstrBackend::new( + data_dir.to_path_buf(), + budget.payment_policy(), + mints, + handler.nostr_tor_proxy(), + )), + BackendId::Routstr, + ); + } + crate::settings::model_provider::Provider::Auto => {} + }, + Err(_) => return (Box::new(InvalidProviderSettings), BackendId::Unavailable), + } let (detected, _models) = crate::api::rpc::mesh::assistant::detect_ollama().await; let model = ollama::OLLAMA_DEFAULT_MODEL; let tool_capable = if detected { diff --git a/core/archipelago/src/assistant/backends/openai.rs b/core/archipelago/src/assistant/backends/openai.rs new file mode 100644 index 00000000..04ce733a --- /dev/null +++ b/core/archipelago/src/assistant/backends/openai.rs @@ -0,0 +1,279 @@ +//! Explicit OpenAI API selection using the shared tool loop and egress policy. +//! Keys stay node-side; no redirects, automatic retries, or provider fallback. +use super::{Backend, BackendTurn}; +use crate::assistant::{ + egress::{self, EgressVerdict}, + tools::{ChatMessage, ToolCall, ToolDef}, +}; +use anyhow::{Context, Result}; +use async_trait::async_trait; +use serde_json::{json, Value}; +use std::{path::PathBuf, time::Duration}; + +const URL: &str = "https://api.openai.com/v1/chat/completions"; +const RESPONSE_LIMIT: usize = 2 * 1024 * 1024; +pub struct OpenaiBackend { + data_dir: PathBuf, + model: String, +} +impl OpenaiBackend { + pub fn new(data_dir: PathBuf, model: String) -> Self { + Self { data_dir, model } + } + async fn send_at( + &self, + url: &str, + system: &str, + tools: &[ToolDef], + history: &[ChatMessage], + ) -> Result { + let key = tokio::fs::read_to_string(self.data_dir.join("secrets/openai-api-key")) + .await + .map_err(|_| { + anyhow::anyhow!("OpenAI API key is not configured. Open AI connection settings.") + })?; + anyhow::ensure!(!key.trim().is_empty(), "OpenAI API key is not configured"); + anyhow::ensure!( + !self.model.is_empty(), + "Choose an OpenAI model in AI connection settings" + ); + let mut messages = vec![json!({"role": "system", "content": system})]; + messages.extend(history.iter().flat_map(super::routstr::message_to_wire)); + let mut body = json!({"model": self.model, "messages": messages, "stream": false, + "store": false, "max_completion_tokens": 2048, "n": 1}); + if !tools.is_empty() { + body["tools"] = json!(tools.iter().map(|tool| json!({"type": "function", "function": { + "name": tool.name, "description": tool.description, "parameters": tool.parameters, + }})).collect::>()); + body["parallel_tool_calls"] = json!(false); + } + let context = egress::EgressContext::from_turn( + history, + &tools.iter().map(|tool| tool.name).collect::>(), + &self.data_dir.join("secrets"), + ) + .await; + match egress::screen_outbound(&body.to_string(), &context) { + EgressVerdict::Allow => {} + EgressVerdict::Truncate(value) => { + body = serde_json::from_str(&value) + .context("Could not apply outbound privacy filter")?; + } + EgressVerdict::BlockFallBackLocal => { + crate::assistant::global_counters().note_blocked_egress(); + anyhow::bail!("This message contains private key or recovery material and was not sent to OpenAI"); + } + } + let client = reqwest::Client::builder() + .timeout(Duration::from_secs(180)) + .connect_timeout(Duration::from_secs(15)) + .redirect(reqwest::redirect::Policy::none()) + .build()?; + let mut response = client.post(url).bearer_auth(key.trim()).json(&body).send().await + .map_err(|_| anyhow::anyhow!("OpenAI is temporarily unreachable. Your request was not retried automatically."))?; + if !response.status().is_success() { + anyhow::bail!("{}", error_message(response.status().as_u16())); + } + let mut bytes = Vec::new(); + while let Some(chunk) = response + .chunk() + .await + .context("OpenAI response interrupted")? + { + anyhow::ensure!( + bytes.len().saturating_add(chunk.len()) <= RESPONSE_LIMIT, + "OpenAI response exceeded the size limit" + ); + bytes.extend_from_slice(&chunk); + } + parse_response( + &serde_json::from_slice(&bytes).context("OpenAI returned an invalid response")?, + ) + } +} +#[async_trait] +impl Backend for OpenaiBackend { + async fn send( + &self, + system: &str, + tools: &[ToolDef], + history: &[ChatMessage], + ) -> Result { + self.send_at(URL, system, tools, history).await + } +} +fn error_message(status: u16) -> &'static str { + match status { + 401 | 403 => "OpenAI rejected the API key or project access. Check AI connection settings.", + 404 => "This OpenAI model is unavailable for your account. Choose another model in AI connection settings.", + 429 => "OpenAI usage or rate limit reached. Check your API billing and retry later.", + 500..=599 => "OpenAI is temporarily unavailable. Retry later.", + _ => "OpenAI rejected the request. Check the selected model and retry.", + } +} +fn parse_response(value: &Value) -> Result { + let choice = value["choices"] + .as_array() + .and_then(|items| items.first()) + .context("OpenAI returned no answer")?; + anyhow::ensure!( + choice["finish_reason"] != "length", + "OpenAI reached the response limit. Try a shorter request." + ); + let message = &choice["message"]; + if let Some(calls) = message["tool_calls"] + .as_array() + .filter(|calls| !calls.is_empty()) + { + let mut parsed = Vec::new(); + for call in calls { + anyhow::ensure!( + call["type"] == "function", + "Unsupported OpenAI tool response" + ); + let id = call["id"] + .as_str() + .filter(|id| !id.is_empty()) + .context("Missing OpenAI tool call ID")?; + let name = call["function"]["name"] + .as_str() + .filter(|name| !name.is_empty()) + .context("Missing OpenAI tool name")?; + let arguments: Value = serde_json::from_str( + call["function"]["arguments"] + .as_str() + .context("Invalid OpenAI tool arguments")?, + ) + .context("Invalid OpenAI tool arguments")?; + anyhow::ensure!( + arguments.is_object() + && !parsed.iter().any(|previous: &ToolCall| previous.id == id), + "Invalid OpenAI tool call" + ); + parsed.push(ToolCall { + id: id.into(), + name: name.into(), + arguments, + }); + } + return Ok(BackendTurn::ToolCalls(parsed)); + } + let text = message["content"] + .as_str() + .or_else(|| message["refusal"].as_str()) + .filter(|text| !text.trim().is_empty()) + .context("OpenAI returned no text; check model compatibility")?; + Ok(BackendTurn::Text(text.into())) +} +#[cfg(test)] +mod tests { + use super::*; + use crate::assistant::tools::{Role, ToolResult}; + #[test] + fn parses_text_and_rejects_incomplete_or_malformed_tool_calls() { + assert!( + matches!(parse_response(&json!({"choices":[{"message":{"content":"hello"}}]})).unwrap(), BackendTurn::Text(text) if text == "hello") + ); + let valid = json!({"choices":[{"message":{"tool_calls":[{"id":"call_1","type":"function","function":{"name":"status","arguments":"{\"count\":1}"}}]}}]}); + assert!( + matches!(parse_response(&valid).unwrap(), BackendTurn::ToolCalls(calls) if calls[0].arguments["count"] == 1) + ); + for args in ["{", "null", "[]"] { + let mut invalid = valid.clone(); + invalid["choices"][0]["message"]["tool_calls"][0]["function"]["arguments"] = + json!(args); + assert!(parse_response(&invalid).is_err()); + } + assert!(parse_response( + &json!({"choices":[{"finish_reason":"length","message":{"content":"partial"}}]}) + ) + .is_err()); + assert!(parse_response(&json!({"choices":[]})).is_err()); + } + #[test] + fn errors_distinguish_credentials_limits_and_outages_without_raw_provider_data() { + assert!(error_message(401).contains("API key")); + assert!(error_message(429).contains("limit")); + assert!(!error_message(503).contains("key")); + let wire = super::super::routstr::message_to_wire(&ChatMessage { + role: Role::Tool, + text: None, + tool_calls: vec![], + tool_results: vec![ToolResult { + call_id: "call_1".into(), + content: "result".into(), + is_error: false, + }], + }); + assert_eq!(wire[0]["tool_call_id"], "call_1"); + } + #[tokio::test] + async fn real_http_adapter_sends_private_key_only_in_header_and_never_follows_redirect() { + use hyper::{ + service::{make_service_fn, service_fn}, + Body, Response, Server, + }; + use std::sync::{Arc, Mutex}; + let captured = Arc::new(Mutex::new(Vec::new())); + let capture = captured.clone(); + let server = Server::bind(&([127, 0, 0, 1], 0).into()).serve(make_service_fn(move |_| { + let capture = capture.clone(); + async move { + Ok::<_, hyper::Error>(service_fn(move |request: hyper::Request| { + let capture = capture.clone(); + async move { + let (parts, body) = request.into_parts(); + let body = hyper::body::to_bytes(body).await?; + capture.lock().unwrap().push(( + parts.headers, + serde_json::from_slice::(&body).unwrap(), + )); + Ok::<_, hyper::Error>( + Response::builder() + .status(302) + .header("Location", "/leak") + .body(Body::empty()) + .unwrap(), + ) + } + })) + } + })); + let url = format!("http://{}/v1/chat/completions", server.local_addr()); + let task = tokio::spawn(server); + let dir = tempfile::tempdir().unwrap(); + crate::settings::model_provider::save_key(dir.path(), "openai", "fixture-private-key") + .await + .unwrap(); + let backend = OpenaiBackend::new(dir.path().into(), "test-model".into()); + let history = [ChatMessage { + role: Role::User, + text: Some("Hello".into()), + tool_calls: vec![], + tool_results: vec![], + }]; + assert!(backend + .send_at(&url, "Be helpful", &[], &history) + .await + .is_err()); + let requests = captured.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert_eq!(requests[0].0["authorization"], "Bearer fixture-private-key"); + assert_eq!(requests[0].1["store"], false); + assert_eq!(requests[0].1["max_completion_tokens"], 2048); + assert!(!requests[0].1.to_string().contains("fixture-private-key")); + drop(requests); + let private = [ChatMessage { + role: Role::User, + text: Some("fixture-private-key".into()), + tool_calls: vec![], + tool_results: vec![], + }]; + assert!(backend + .send_at(&url, "Be helpful", &[], &private) + .await + .is_err()); + assert_eq!(captured.lock().unwrap().len(), 1); + task.abort(); + } +} diff --git a/core/archipelago/src/assistant/backends/routstr.rs b/core/archipelago/src/assistant/backends/routstr.rs index 10331833..f751d6f0 100644 --- a/core/archipelago/src/assistant/backends/routstr.rs +++ b/core/archipelago/src/assistant/backends/routstr.rs @@ -295,7 +295,7 @@ fn parse_openai_tool_calls(raw_calls: &[Value]) -> Vec { /// (the wire-format inverse of `parse_openai_tool_calls`), and tool-result /// turns carry `tool_call_id` so each call's id is echoed back exactly — /// the OpenAI-shape contract this adapter's edge is responsible for. -fn message_to_wire(msg: &ChatMessage) -> Vec { +pub(super) fn message_to_wire(msg: &ChatMessage) -> Vec { match msg.role { Role::System => vec![], Role::User => vec![json!({ diff --git a/core/archipelago/src/federation/handshake_delivery.rs b/core/archipelago/src/federation/handshake_delivery.rs new file mode 100644 index 00000000..9be1d309 --- /dev/null +++ b/core/archipelago/src/federation/handshake_delivery.rs @@ -0,0 +1,199 @@ +//! Durable, node-encrypted approval replies. Relay acknowledgement is not peer +//! acceptance: keep retrying the same invite until reciprocal membership exists. +use anyhow::{Context, Result}; +use serde::{Deserialize, Serialize}; +use std::path::Path; +use tokio::{fs, io::AsyncWriteExt}; + +const FILE: &str = "federation/handshake-delivery.enc"; +const DOMAIN: &[u8] = b"archipelago-handshake-delivery-v1"; +static LOCK: tokio::sync::Mutex<()> = tokio::sync::Mutex::const_new(()); + +#[derive(Clone, Serialize, Deserialize)] +pub(crate) struct ApprovalReply { + pub request_id: String, + pub recipient: String, + pub expected_did: String, + pub invite_code: String, + pub attempts: u32, + pub next_attempt: i64, +} + +async fn load(data_dir: &Path) -> Result> { + let bytes = match fs::read(data_dir.join(FILE)).await { + Ok(bytes) => bytes, + Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(Vec::new()), + Err(e) => return Err(e.into()), + }; + let key = crate::storage_crypto::derive_key(data_dir, DOMAIN).await?; + let plaintext = crate::storage_crypto::open(&bytes, &key)?; + serde_json::from_slice(&plaintext) + .context("Invalid handshake delivery store; preserved for recovery") +} + +async fn save(data_dir: &Path, entries: &[ApprovalReply]) -> Result<()> { + let key = crate::storage_crypto::derive_key(data_dir, DOMAIN).await?; + let bytes = crate::storage_crypto::seal(&serde_json::to_vec(entries)?, &key)?; + let path = data_dir.join(FILE); + let parent = path.parent().context("Delivery parent missing")?; + fs::create_dir_all(parent).await?; + let temporary = parent.join(format!(".delivery-{}.tmp", uuid::Uuid::new_v4())); + let result = async { + let mut file = fs::OpenOptions::new() + .create_new(true) + .write(true) + .mode(0o600) + .open(&temporary) + .await?; + file.write_all(&bytes).await?; + file.sync_all().await?; + drop(file); + fs::rename(&temporary, &path).await?; + fs::File::open(parent).await?.sync_all().await?; + Ok::<_, anyhow::Error>(()) + } + .await; + if result.is_err() { + let _ = fs::remove_file(temporary).await; + } + result +} + +pub(crate) async fn find(data_dir: &Path, request_id: &str) -> Result> { + let _guard = LOCK.lock().await; + Ok(load(data_dir) + .await? + .into_iter() + .find(|entry| entry.request_id == request_id)) +} + +pub(crate) async fn stage(data_dir: &Path, reply: ApprovalReply) -> Result { + let _guard = LOCK.lock().await; + let mut entries = load(data_dir).await?; + if let Some(existing) = entries + .iter() + .find(|entry| entry.request_id == reply.request_id) + { + anyhow::ensure!( + existing.recipient == reply.recipient && existing.expected_did == reply.expected_did, + "Approval recipient changed; refusing delivery" + ); + return Ok(existing.clone()); + } + anyhow::ensure!(entries.len() < 1024, "Handshake delivery queue is full"); + entries.push(reply.clone()); + save(data_dir, &entries).await?; + Ok(reply) +} + +/// Claim before sending, including failed sends. Concurrent polls cannot create +/// retry storms; a crash after this write delays but never loses the reply. +pub(crate) async fn claim( + data_dir: &Path, + request_id: &str, + now: i64, +) -> Result> { + let _guard = LOCK.lock().await; + let mut entries = load(data_dir).await?; + let Some(entry) = entries + .iter_mut() + .find(|entry| entry.request_id == request_id) + else { + return Ok(None); + }; + if entry.next_attempt > now { + return Ok(None); + } + entry.attempts = entry.attempts.saturating_add(1); + let delay = 30_i64 + .saturating_mul(1_i64 << entry.attempts.min(7)) + .min(3600); + entry.next_attempt = now.saturating_add(delay); + let claimed = entry.clone(); + save(data_dir, &entries).await?; + Ok(Some(claimed)) +} + +pub(crate) async fn remove(data_dir: &Path, request_id: &str) -> Result<()> { + let _guard = LOCK.lock().await; + let mut entries = load(data_dir).await?; + let before = entries.len(); + entries.retain(|entry| entry.request_id != request_id); + if entries.len() != before { + save(data_dir, &entries).await?; + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + async fn fixture() -> tempfile::TempDir { + let dir = tempfile::tempdir().unwrap(); + fs::create_dir_all(dir.path().join("identity")) + .await + .unwrap(); + fs::write(dir.path().join("identity/node_key"), [7; 32]) + .await + .unwrap(); + dir + } + fn reply() -> ApprovalReply { + ApprovalReply { + request_id: "request-1".into(), + recipient: "recipient".into(), + expected_did: "did:key:peer".into(), + invite_code: "secret-invite".into(), + attempts: 0, + next_attempt: 0, + } + } + #[tokio::test] + async fn encrypted_reply_survives_reload_and_retry_claim_is_exclusive() { + let dir = fixture().await; + stage(dir.path(), reply()).await.unwrap(); + let raw = fs::read(dir.path().join(FILE)).await.unwrap(); + assert!(!raw.windows(13).any(|bytes| bytes == b"secret-invite")); + assert_eq!( + find(dir.path(), "request-1") + .await + .unwrap() + .unwrap() + .invite_code, + "secret-invite" + ); + let (a, b) = tokio::join!( + claim(dir.path(), "request-1", 100), + claim(dir.path(), "request-1", 100) + ); + assert_eq!( + usize::from(a.unwrap().is_some()) + usize::from(b.unwrap().is_some()), + 1 + ); + assert!(claim(dir.path(), "request-1", 159).await.unwrap().is_none()); + assert_eq!( + claim(dir.path(), "request-1", 160) + .await + .unwrap() + .unwrap() + .attempts, + 2 + ); + remove(dir.path(), "request-1").await.unwrap(); + assert!(find(dir.path(), "request-1").await.unwrap().is_none()); + } + #[tokio::test] + async fn corrupt_store_is_not_overwritten_and_recipient_cannot_change() { + let dir = fixture().await; + stage(dir.path(), reply()).await.unwrap(); + let mut other = reply(); + other.recipient = "different-recipient".into(); + assert!(stage(dir.path(), other).await.is_err()); + let path = dir.path().join(FILE); + let mut raw = fs::read(&path).await.unwrap(); + raw[20] ^= 1; + fs::write(&path, &raw).await.unwrap(); + assert!(stage(dir.path(), reply()).await.is_err()); + assert_eq!(fs::read(path).await.unwrap(), raw); + } +} diff --git a/core/archipelago/src/federation/invites.rs b/core/archipelago/src/federation/invites.rs index 63f4bf71..97b7382d 100644 --- a/core/archipelago/src/federation/invites.rs +++ b/core/archipelago/src/federation/invites.rs @@ -134,6 +134,32 @@ pub fn parse_invite(code: &str) -> Result { }) } +/// Bind a Nostr-discovery reply to the node the operator requested, and cap +/// its grant before any local node entry or callback is written. Legacy invites +/// default to Trusted, which must never transiently authorize discovery peers. +pub(crate) fn restrict_discovery_invite(code: &str, expected_did: &str) -> Result { + use base64::Engine; + let parsed = parse_invite(code)?; + anyhow::ensure!( + !expected_did.is_empty() && parsed.did == expected_did, + "Peer invite does not match the requested node" + ); + anyhow::ensure!( + crate::identity::did_key_from_pubkey_hex(&parsed.pubkey)? == parsed.did, + "Peer invite DID does not match its identity key" + ); + let bytes = base64::engine::general_purpose::URL_SAFE_NO_PAD.decode( + code.strip_prefix("fed1:") + .context("Invalid invite prefix")?, + )?; + let mut payload: serde_json::Value = serde_json::from_slice(&bytes)?; + payload["trust"] = serde_json::json!("observer"); + Ok(format!( + "fed1:{}", + base64::engine::general_purpose::URL_SAFE_NO_PAD.encode(serde_json::to_vec(&payload)?) + )) +} + /// Accept an invite: parse code, verify the remote node, add to federation. pub async fn accept_invite( data_dir: &Path, @@ -621,3 +647,35 @@ mod tests { assert_eq!(nodes.len(), 1, "re-accept should not duplicate"); } } + +#[cfg(test)] +mod discovery_invite_scope_tests { + use super::*; + use base64::Engine; + #[test] + fn discovery_reply_binds_identity_and_caps_legacy_trust_before_acceptance() { + let key = "33".repeat(32); + let did = crate::identity::did_key_from_pubkey_hex(&key).unwrap(); + let payload = + serde_json::json!({"did":did,"pubkey":key,"onion":"test.onion","token":"test-token"}); + let code = format!( + "fed1:{}", + base64::engine::general_purpose::URL_SAFE_NO_PAD + .encode(serde_json::to_vec(&payload).unwrap()) + ); + let restricted = restrict_discovery_invite(&code, &did).unwrap(); + let parsed = parse_invite(&restricted).unwrap(); + assert_eq!(parsed.trust_level, TrustLevel::Observer); + assert_eq!(parsed.token, "test-token"); + assert!(restrict_discovery_invite(&code, "did:key:someone-else").is_err()); + assert!(restrict_discovery_invite(&code, "").is_err()); + let mut forged = payload; + forged["pubkey"] = serde_json::json!("44".repeat(32)); + let forged = format!( + "fed1:{}", + base64::engine::general_purpose::URL_SAFE_NO_PAD + .encode(serde_json::to_vec(&forged).unwrap()) + ); + assert!(restrict_discovery_invite(&forged, &did).is_err()); + } +} diff --git a/core/archipelago/src/federation/mod.rs b/core/archipelago/src/federation/mod.rs index 44b781e2..c762a457 100644 --- a/core/archipelago/src/federation/mod.rs +++ b/core/archipelago/src/federation/mod.rs @@ -6,12 +6,14 @@ mod invites; pub mod pending; +pub(crate) mod handshake_delivery; mod storage; mod sync; mod types; // Re-export all public items so `crate::federation::*` continues to work. pub use invites::{accept_invite, create_invite, parse_invite}; +pub(crate) use invites::restrict_discovery_invite; // Crate-internal: used by the periodic federation auto-sync to re-assert // membership to peers that don't list us back (asymmetry self-heal). pub(crate) use invites::notify_join; diff --git a/core/archipelago/src/federation/pending.rs b/core/archipelago/src/federation/pending.rs index dc0d0e96..e10d0963 100644 --- a/core/archipelago/src/federation/pending.rs +++ b/core/archipelago/src/federation/pending.rs @@ -14,6 +14,9 @@ use anyhow::{Context, Result}; use serde::{Deserialize, Serialize}; use std::path::Path; use tokio::fs; +use tokio::io::AsyncWriteExt; + +static PENDING_STORE_LOCK: tokio::sync::Mutex<()> = tokio::sync::Mutex::const_new(()); const PENDING_FILE: &str = "federation/pending_requests.json"; const MAX_PENDING_PER_PUBKEY: usize = 5; @@ -76,11 +79,12 @@ pub async fn load_pending(data_dir: &Path) -> Result> { let content = fs::read_to_string(&path) .await .context("Failed to read pending requests file")?; - let file: PendingRequestsFile = serde_json::from_str(&content).unwrap_or_default(); + let file: PendingRequestsFile = serde_json::from_str(&content) + .context("Invalid pending requests file; preserving existing data")?; Ok(file.requests) } -pub async fn save_pending(data_dir: &Path, requests: &[PendingPeerRequest]) -> Result<()> { +async fn save_pending(data_dir: &Path, requests: &[PendingPeerRequest]) -> Result<()> { let path = data_dir.join(PENDING_FILE); if let Some(parent) = path.parent() { fs::create_dir_all(parent) @@ -92,9 +96,27 @@ pub async fn save_pending(data_dir: &Path, requests: &[PendingPeerRequest]) -> R }; let content = serde_json::to_string_pretty(&file).context("Failed to serialize pending requests")?; - fs::write(&path, content) - .await - .context("Failed to write pending requests file")?; + let parent = path.parent().context("Pending requests parent missing")?; + let temporary = parent.join(format!(".pending-{}.tmp", uuid::Uuid::new_v4())); + let result = async { + let mut file = fs::OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&temporary) + .await?; + file.write_all(content.as_bytes()).await?; + file.sync_all().await?; + drop(file); + fs::rename(&temporary, &path).await?; + fs::File::open(parent).await?.sync_all().await?; + Ok::<_, anyhow::Error>(()) + } + .await; + if result.is_err() { + let _ = fs::remove_file(&temporary).await; + } + result.context("Failed to atomically save pending requests")?; Ok(()) } @@ -102,7 +124,10 @@ pub async fn save_pending(data_dir: &Path, requests: &[PendingPeerRequest]) -> R fn expire_stale(requests: &mut Vec) { let cutoff = chrono::Utc::now() - chrono::Duration::days(PENDING_EXPIRY_DAYS); for r in requests.iter_mut() { - if !matches!(r.state, PendingState::Pending | PendingState::Sent) { + if !matches!( + r.state, + PendingState::Pending | PendingState::Sent | PendingState::Approved + ) { continue; } if let Ok(ts) = chrono::DateTime::parse_from_rfc3339(&r.received_at) { @@ -131,6 +156,7 @@ pub async fn insert_inbound( from_name: Option, message: Option, ) -> Result> { + let _guard = PENDING_STORE_LOCK.lock().await; let mut requests = load_pending(data_dir).await?; expire_stale(&mut requests); @@ -189,6 +215,7 @@ pub async fn insert_outbound( to_name: Option, message: Option, ) -> Result { + let _guard = PENDING_STORE_LOCK.lock().await; let mut requests = load_pending(data_dir).await?; expire_stale(&mut requests); requests.retain(|r| { @@ -218,6 +245,7 @@ pub async fn find_by_id(data_dir: &Path, id: &str) -> Result Result<()> { + let _guard = PENDING_STORE_LOCK.lock().await; let mut requests = load_pending(data_dir).await?; if let Some(r) = requests.iter_mut().find(|r| r.id == id) { r.state = state; @@ -228,10 +256,32 @@ pub async fn set_state(data_dir: &Path, id: &str, state: PendingState) -> Result Ok(()) } +/// Resolve a pending decision once; concurrent approval/rejection cannot +/// overwrite each other after a slow network request. +pub async fn decide(data_dir: &Path, id: &str, decision: PendingState) -> Result<()> { + anyhow::ensure!( + matches!(decision, PendingState::Approved | PendingState::Rejected), + "Invalid pending decision" + ); + let _guard = PENDING_STORE_LOCK.lock().await; + let mut requests = load_pending(data_dir).await?; + let row = requests + .iter_mut() + .find(|row| row.id == id) + .context("Pending request not found")?; + anyhow::ensure!( + !row.outbound && row.state == PendingState::Pending, + "Request has already been decided" + ); + row.state = decision; + save_pending(data_dir, &requests).await +} + /// Remove a pending request entirely. Used when the sender cancels an /// outbound request they initiated and we want it gone (not just marked /// Rejected/Cancelled — those states fill up the UI audit trail). pub async fn delete(data_dir: &Path, id: &str) -> Result<()> { + let _guard = PENDING_STORE_LOCK.lock().await; let mut requests = load_pending(data_dir).await?; let before = requests.len(); requests.retain(|r| r.id != id); @@ -372,3 +422,140 @@ mod tests { assert_eq!(reloaded.state, PendingState::Approved); } } + +#[cfg(test)] +mod persistence_regressions { + use super::*; + + #[tokio::test] + async fn concurrent_requests_are_not_lost() { + let dir = tempfile::tempdir().unwrap(); + let mut tasks = Vec::new(); + for i in 0..24 { + let path = dir.path().to_path_buf(); + tasks.push(tokio::spawn(async move { + insert_inbound( + &path, + format!("key-{i}"), + format!("npub-{i}"), + format!("did:key:{i}"), + None, + None, + ) + .await + .unwrap() + })); + } + for task in tasks { + assert!(task.await.unwrap().is_some()); + } + let rows = load_pending(dir.path()).await.unwrap(); + assert_eq!(rows.len(), 24); + assert_eq!( + rows.iter() + .map(|r| &r.id) + .collect::>() + .len(), + 24 + ); + } + + #[tokio::test] + async fn malformed_store_is_preserved_instead_of_replaced_with_one_request() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join(PENDING_FILE); + fs::create_dir_all(path.parent().unwrap()).await.unwrap(); + let damaged = b"{incomplete existing requests"; + fs::write(&path, damaged).await.unwrap(); + assert!(insert_inbound( + dir.path(), + "key".into(), + "npub".into(), + "did:key:test".into(), + None, + None + ) + .await + .is_err()); + assert_eq!(fs::read(path).await.unwrap(), damaged); + } +} + +#[cfg(test)] +mod decision_regressions { + use super::*; + #[tokio::test] + async fn only_one_concurrent_operator_decision_wins() { + let dir = tempfile::tempdir().unwrap(); + let row = insert_inbound( + dir.path(), + "key".into(), + "npub".into(), + "did:key:peer".into(), + None, + None, + ) + .await + .unwrap() + .unwrap(); + let (approve, reject) = tokio::join!( + decide(dir.path(), &row.id, PendingState::Approved), + decide(dir.path(), &row.id, PendingState::Rejected) + ); + assert_eq!( + usize::from(approve.is_ok()) + usize::from(reject.is_ok()), + 1 + ); + let saved = find_by_id(dir.path(), &row.id).await.unwrap().unwrap(); + assert_eq!( + saved.state, + if approve.is_ok() { + PendingState::Approved + } else { + PendingState::Rejected + } + ); + } + #[tokio::test] + async fn an_expired_approval_does_not_block_a_new_request_forever() { + let dir = tempfile::tempdir().unwrap(); + let mut row = insert_inbound( + dir.path(), + "key".into(), + "npub".into(), + "did:key:peer".into(), + None, + None, + ) + .await + .unwrap() + .unwrap(); + row.state = PendingState::Approved; + row.received_at = (chrono::Utc::now() - chrono::Duration::days(31)).to_rfc3339(); + save_pending(dir.path(), &[row]).await.unwrap(); + let renewed = insert_inbound( + dir.path(), + "key".into(), + "npub".into(), + "did:key:peer".into(), + None, + None, + ) + .await + .unwrap(); + assert!(renewed.is_some()); + let rows = load_pending(dir.path()).await.unwrap(); + assert_eq!( + rows.iter() + .filter(|r| r.state == PendingState::Expired) + .count(), + 1 + ); + assert_eq!( + rows.iter() + .filter(|r| r.state == PendingState::Pending) + .count(), + 1 + ); + } +} diff --git a/core/archipelago/src/federation/sync.rs b/core/archipelago/src/federation/sync.rs index a161a576..209554c7 100644 --- a/core/archipelago/src/federation/sync.rs +++ b/core/archipelago/src/federation/sync.rs @@ -230,12 +230,12 @@ async fn merge_transitive_peers( #[allow(clippy::too_many_arguments)] pub fn build_local_state( apps: Vec, - cpu: f64, - mem_used: u64, - mem_total: u64, - disk_used: u64, - disk_total: u64, - uptime: u64, + cpu: Option, + mem_used: Option, + mem_total: Option, + disk_used: Option, + disk_total: Option, + uptime: Option, tor_active: bool, server_name: Option, nostr_npub: Option, @@ -261,12 +261,12 @@ pub fn build_local_state( timestamp: chrono::Utc::now().to_rfc3339(), node_name: server_name, apps, - cpu_usage_percent: Some(cpu), - mem_used_bytes: Some(mem_used), - mem_total_bytes: Some(mem_total), - disk_used_bytes: Some(disk_used), - disk_total_bytes: Some(disk_total), - uptime_secs: Some(uptime), + cpu_usage_percent: cpu, + mem_used_bytes: mem_used, + mem_total_bytes: mem_total, + disk_used_bytes: disk_used, + disk_total_bytes: disk_total, + uptime_secs: uptime, tor_active: Some(tor_active), nostr_npub, own_fips_npub, @@ -355,12 +355,12 @@ mod tests { status: "running".to_string(), version: Some("0.18".to_string()), }], - 25.5, - 2_000_000_000, - 8_000_000_000, - 100_000_000_000, - 500_000_000_000, - 3600, + Some(25.5), + Some(2_000_000_000), + Some(8_000_000_000), + Some(100_000_000_000), + Some(500_000_000_000), + Some(3600), true, Some("Test Node".to_string()), None, @@ -430,12 +430,12 @@ mod tests { ]; let state = build_local_state( vec![], - 0.0, - 0, - 0, - 0, - 0, - 0, + None, + None, + None, + None, + None, + None, true, None, None, diff --git a/core/archipelago/src/monitoring/collector.rs b/core/archipelago/src/monitoring/collector.rs index 01678161..c85af143 100644 --- a/core/archipelago/src/monitoring/collector.rs +++ b/core/archipelago/src/monitoring/collector.rs @@ -13,11 +13,11 @@ pub async fn collect_snapshot() -> Result { read_loadavg(), ); - let cpu = cpu.unwrap_or(0.0); - let (mem_used, mem_total) = mem.unwrap_or((0, 0)); - let (disk_used, disk_total) = disk.unwrap_or((0, 0)); - let (net_rx, net_tx) = net.unwrap_or((0, 0)); - let (l1, l5, l15) = load.unwrap_or((0.0, 0.0, 0.0)); + let cpu = cpu?; + let (mem_used, mem_total) = mem?; + let (disk_used, disk_total) = disk?; + let (net_rx, net_tx) = net?; + let (l1, l5, l15) = load?; let system = SystemMetrics { cpu_percent: cpu, @@ -120,10 +120,9 @@ async fn read_disk_usage() -> Result<(u64, u64)> { } else { "/" }; - let output = tokio::process::Command::new("df") - .args(["--block-size=1", "--output=used,size", target]) - .output() - .await + let mut command = tokio::process::Command::new("df"); + command.args(["--block-size=1", "--output=used,size", target]); + let output = bounded_output(command, std::time::Duration::from_secs(3)).await .context("Failed to run df")?; if !output.status.success() { @@ -215,12 +214,23 @@ async fn read_network_totals() -> Result<(u64, u64)> { Ok((rx_total, tx_total)) } +/// A wedged runtime or filesystem must not freeze every monitoring snapshot. +/// Dropping a timed-out child kills it, so repeated polls cannot leak processes. +async fn bounded_output( + mut command: tokio::process::Command, + timeout: std::time::Duration, +) -> Result { + command.kill_on_drop(true); + tokio::time::timeout(timeout, command.output()) + .await.context("Metrics subprocess timed out")? + .context("Metrics subprocess failed") +} + /// Get per-container resource stats via `podman stats --no-stream --format json`. async fn read_container_stats() -> Result> { - let output = tokio::process::Command::new("podman") - .args(["stats", "--no-stream", "--format", "json"]) - .output() - .await + let mut command = tokio::process::Command::new("podman"); + command.args(["stats", "--no-stream", "--format", "json"]); + let output = bounded_output(command, std::time::Duration::from_secs(8)).await .context("Failed to run podman stats")?; if !output.status.success() { @@ -391,3 +401,35 @@ mod tests { assert_eq!(parse_bytes_field(&obj, "mem"), Some(268435456)); } } + +#[cfg(test)] +mod subprocess_deadline_tests { + use super::*; + + #[tokio::test] + async fn a_stalled_metrics_command_is_bounded_and_killed() { + let dir = tempfile::tempdir().unwrap(); + let pid_file = dir.path().join("pid"); + let mut command = tokio::process::Command::new("sh"); + command.arg("-c").arg("echo $$ > \"$1\"; exec sleep 30").arg("metrics-test").arg(&pid_file); + let start = std::time::Instant::now(); + let error = bounded_output(command, std::time::Duration::from_millis(500)).await.unwrap_err(); + assert!(error.to_string().contains("timed out")); + assert!(start.elapsed() < std::time::Duration::from_secs(3)); + let pid = tokio::fs::read_to_string(pid_file).await.unwrap(); + for _ in 0..40 { + if !std::path::Path::new(&format!("/proc/{}", pid.trim())).exists() { return; } + tokio::time::sleep(std::time::Duration::from_millis(25)).await; + } + panic!("Timed-out metrics subprocess was not reaped"); + } + + #[tokio::test] + async fn successful_metrics_output_is_preserved() { + let mut command = tokio::process::Command::new("printf"); + command.arg("metrics-ok"); + let output = bounded_output(command, std::time::Duration::from_secs(1)).await.unwrap(); + assert!(output.status.success()); + assert_eq!(output.stdout, b"metrics-ok"); + } +} diff --git a/core/archipelago/src/monitoring/mod.rs b/core/archipelago/src/monitoring/mod.rs index 45baf043..d02c8dbb 100644 --- a/core/archipelago/src/monitoring/mod.rs +++ b/core/archipelago/src/monitoring/mod.rs @@ -14,21 +14,21 @@ use std::path::PathBuf; use std::sync::Arc; use tracing::{debug, warn}; -/// Spawn the background metrics collector (runs every 300 seconds / 5 minutes). +/// Spawn the background metrics collector at the store's one-minute resolution. /// Evaluates alert rules on each snapshot and dispatches notifications. /// Note: health_monitor.rs handles container state polling at 120s intervals. -/// This collector handles system-level metrics (CPU, disk, network) and only -/// calls podman stats every 5 minutes to avoid duplicate subprocess overhead. +/// Runtime commands have deadlines; unavailable container stats cannot hold +/// system readings indefinitely. Missed ticks are skipped, never replayed. pub fn spawn_metrics_collector( store: Arc, state: Option>, data_dir: Option, ) { tokio::spawn(async move { - // Wait 60s for system to stabilize after boot - tokio::time::sleep(std::time::Duration::from_secs(60)).await; + // Start promptly without competing with the very first boot tasks. + tokio::time::sleep(std::time::Duration::from_secs(5)).await; - let mut interval = tokio::time::interval(std::time::Duration::from_secs(300)); + let mut interval = tokio::time::interval(std::time::Duration::from_secs(60)); interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); loop { diff --git a/core/archipelago/src/server.rs b/core/archipelago/src/server.rs index 8af49982..949cfb25 100644 --- a/core/archipelago/src/server.rs +++ b/core/archipelago/src/server.rs @@ -291,15 +291,15 @@ impl Server { ); // Background handshake poll: fetch inbound nostr peer requests every - // 5 minutes instead of only when a user presses the Federation Poll + // 30 seconds instead of only when a user presses the Federation Poll // button (requests used to sit on relays unseen — 2026-07-22). The // handler's own discoverability gate makes this a no-op until the // user opts in. { let rpc = api_handler.rpc_handler().clone(); tokio::spawn(async move { - let mut tick = tokio::time::interval(std::time::Duration::from_secs(300)); - tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + let mut tick = tokio::time::interval(std::time::Duration::from_secs(30)); + tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); loop { tick.tick().await; rpc.background_handshake_poll().await; diff --git a/core/archipelago/src/settings/mod.rs b/core/archipelago/src/settings/mod.rs index 8ffaf648..9c1237e8 100644 --- a/core/archipelago/src/settings/mod.rs +++ b/core/archipelago/src/settings/mod.rs @@ -9,3 +9,5 @@ pub mod session_policy; pub mod transport; pub mod bitcoin_storage; + +pub mod model_provider; diff --git a/core/archipelago/src/settings/model_provider.rs b/core/archipelago/src/settings/model_provider.rs new file mode 100644 index 00000000..f7b33d17 --- /dev/null +++ b/core/archipelago/src/settings/model_provider.rs @@ -0,0 +1,178 @@ +//! Owner-selected chat provider. API keys remain in the node's private secret +//! ledger and are never returned by settings or included in chat context. +use anyhow::{Context, Result}; +use serde::{Deserialize, Serialize}; +use std::path::Path; +use tokio::{fs, io::AsyncWriteExt}; + +#[derive(Clone, Copy, Debug, Default, Deserialize, Serialize, PartialEq, Eq)] +#[serde(rename_all = "snake_case")] +pub enum Provider { + #[default] + Auto, + Claude, + Openai, + Local, + Routstr, +} + +#[derive(Clone, Debug, Default, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct ModelProvider { + #[serde(default)] + pub provider: Provider, + #[serde(default)] + pub openai_model: String, +} + +impl ModelProvider { + pub fn validate(&self) -> Result<()> { + anyhow::ensure!( + !self.openai_model.starts_with("sk-") + && self.openai_model.len() <= 128 + && self + .openai_model + .bytes() + .all(|b| b.is_ascii_alphanumeric() || b"-_.:".contains(&b)), + "Invalid OpenAI model name" + ); + anyhow::ensure!( + self.provider != Provider::Openai || !self.openai_model.is_empty(), + "Choose an OpenAI model before connecting" + ); + Ok(()) + } + pub async fn load(data_dir: &Path) -> Result { + match fs::read(data_dir.join("settings/model-provider.json")).await { + Ok(bytes) => { + let settings: Self = serde_json::from_slice(&bytes) + .context("Invalid AI provider settings; preserved for recovery")?; + settings.validate()?; + Ok(settings) + } + Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(Self::default()), + Err(error) => Err(error.into()), + } + } + pub async fn save(&self, data_dir: &Path) -> Result<()> { + self.validate()?; + write_private( + &data_dir.join("settings/model-provider.json"), + &serde_json::to_vec(self)?, + ) + .await + } +} + +pub fn key_name(provider: &str) -> Result<&'static str> { + match provider { + "claude" => Ok("claude-api-key"), + "openai" => Ok("openai-api-key"), + _ => anyhow::bail!("Unsupported AI provider"), + } +} +pub async fn has_key(data_dir: &Path, provider: &str) -> bool { + let Ok(name) = key_name(provider) else { + return false; + }; + fs::read_to_string(data_dir.join("secrets").join(name)) + .await + .is_ok_and(|key| !key.trim().is_empty()) +} +pub async fn save_key(data_dir: &Path, provider: &str, value: &str) -> Result<()> { + let path = data_dir.join("secrets").join(key_name(provider)?); + let value = value.trim(); + anyhow::ensure!( + value.len() <= 4096 && value.bytes().all(|b| b.is_ascii_graphic()), + "Invalid API key format" + ); + if value.is_empty() { + match fs::remove_file(path).await { + Ok(()) => Ok(()), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(()), + Err(e) => Err(e.into()), + } + } else { + write_private(&path, value.as_bytes()).await + } +} +async fn write_private(path: &Path, bytes: &[u8]) -> Result<()> { + let parent = path.parent().context("Missing settings directory")?; + fs::create_dir_all(parent).await?; + let temporary = parent.join(format!(".provider-{}.tmp", uuid::Uuid::new_v4())); + let result = async { + let mut file = fs::OpenOptions::new() + .create_new(true) + .write(true) + .mode(0o600) + .open(&temporary) + .await?; + file.write_all(bytes).await?; + file.sync_all().await?; + drop(file); + fs::rename(&temporary, path).await?; + fs::File::open(parent).await?.sync_all().await?; + Ok::<_, anyhow::Error>(()) + } + .await; + if result.is_err() { + let _ = fs::remove_file(temporary).await; + } + result +} + +#[cfg(test)] +mod tests { + use super::*; + #[tokio::test] + async fn private_keys_replace_atomically_and_never_enter_public_settings() { + use std::os::unix::fs::PermissionsExt; + let dir = tempfile::tempdir().unwrap(); + assert!(!has_key(dir.path(), "openai").await); + save_key(dir.path(), "openai", "test-key-one") + .await + .unwrap(); + save_key(dir.path(), "openai", "test-key-two") + .await + .unwrap(); + assert!(has_key(dir.path(), "openai").await); + let key_path = dir.path().join("secrets/openai-api-key"); + assert_eq!( + fs::metadata(&key_path).await.unwrap().permissions().mode() & 0o777, + 0o600 + ); + assert_eq!(fs::read_to_string(&key_path).await.unwrap(), "test-key-two"); + let settings = ModelProvider { + provider: Provider::Openai, + openai_model: "test-model".into(), + }; + settings.save(dir.path()).await.unwrap(); + let body = serde_json::to_string(&ModelProvider::load(dir.path()).await.unwrap()).unwrap(); + assert!(!body.contains("test-key")); + assert!(save_key(dir.path(), "../openai", "key").await.is_err()); + assert!(save_key(dir.path(), "openai", "key\nInjected: bad") + .await + .is_err()); + assert_eq!(fs::read_to_string(&key_path).await.unwrap(), "test-key-two"); + save_key(dir.path(), "openai", "").await.unwrap(); + assert!(!has_key(dir.path(), "openai").await); + } + #[tokio::test] + async fn invalid_settings_preserve_existing_configuration() { + let dir = tempfile::tempdir().unwrap(); + ModelProvider::default().save(dir.path()).await.unwrap(); + let invalid = ModelProvider { + provider: Provider::Openai, + openai_model: String::new(), + }; + assert!(invalid.save(dir.path()).await.is_err()); + assert_eq!( + ModelProvider::load(dir.path()).await.unwrap().provider, + Provider::Auto + ); + let path = dir.path().join("settings/model-provider.json"); + fs::write(&path, b"broken").await.unwrap(); + assert!(ModelProvider::load(dir.path()).await.is_err()); + assert_eq!(fs::read(&path).await.unwrap(), b"broken"); + } +} diff --git a/core/archipelago/src/storage_crypto.rs b/core/archipelago/src/storage_crypto.rs index 0cef0971..e3c1120d 100644 --- a/core/archipelago/src/storage_crypto.rs +++ b/core/archipelago/src/storage_crypto.rs @@ -75,11 +75,15 @@ pub fn open(data: &[u8], key: &[u8; 32]) -> Result> { .map_err(|_| anyhow::anyhow!("decryption failed — key mismatch or corruption")) } -/// Heuristic: does this look like legacy plaintext JSON (starts with `{`/`[`)? -/// Encrypted blobs start with a random nonce byte, so a `{`/`[` first byte is a -/// reliable migration signal. +/// Recognize a complete legacy JSON object/array, including leading whitespace. +/// A random nonce can start with `{` or `[`; checking only that byte misclassifies +/// valid ciphertext and can trigger an empty-store migration. Validate the entire +/// document without allocating a second copy of the store's object tree. pub fn is_plaintext_json(raw: &[u8]) -> bool { - matches!(raw.first(), Some(b'{') | Some(b'[')) + matches!( + raw.iter().copied().find(|byte| !byte.is_ascii_whitespace()), + Some(b'{') | Some(b'[') + ) && serde_json::from_slice::(raw).is_ok() } #[cfg(test)] @@ -110,9 +114,44 @@ mod tests { fn detects_plaintext_vs_ciphertext() { assert!(is_plaintext_json(b"{\"a\":1}")); assert!(is_plaintext_json(b"[]")); + assert!(is_plaintext_json(b" \r\n\t{\"messages\": []}\n")); + for invalid in [ + b"{".as_slice(), + b"[", + b"{}trailing", + b"[\xff]", + b"null", + b"\"text\"", + ] { + assert!(!is_plaintext_json(invalid)); + } assert!(!is_plaintext_json(&seal(b"x", &[3u8; 32]).unwrap())); } + #[test] + fn json_prefix_nonce_does_not_trigger_plaintext_migration() { + use chacha20poly1305::aead::{Aead, KeyInit}; + let key = [3u8; 32]; + let plaintext = br#"{"messages":[{"message":"preserve me"}]}"#; + let cipher = chacha20poly1305::ChaCha20Poly1305::new_from_slice(&key).unwrap(); + // Deterministically reproduce both collisions instead of relying on + // OsRng to happen to pick one during a test run. + for prefix in [b'{', b'['] { + let mut nonce = [0xffu8; 12]; + nonce[0] = prefix; + let encrypted = cipher + .encrypt( + chacha20poly1305::aead::generic_array::GenericArray::from_slice(&nonce), + plaintext.as_slice(), + ) + .unwrap(); + let mut envelope = nonce.to_vec(); + envelope.extend(encrypted); + assert!(!is_plaintext_json(&envelope)); + assert_eq!(open(&envelope, &key).unwrap(), plaintext); + } + } + /// KEY-05 regression: a blob written by the pre-migration `seal` must still /// open after the migration. /// diff --git a/docs/aiui-provider-setup-followup.md b/docs/aiui-provider-setup-followup.md new file mode 100644 index 00000000..f3faa7c2 --- /dev/null +++ b/docs/aiui-provider-setup-followup.md @@ -0,0 +1,49 @@ +# AIUI provider setup follow-up + +Status: implementation in progress; not deployed or accepted. + +The app's browser key vault did not configure the node's authoritative Claude +ledger. Embedded chat delegates to the node's tool loop, whose provider selection +also ignored the frontend's Claude/OpenRouter picker. The backend retired the +OpenRouter relay while that picker still offered it. Generic502/503 errors were +classified as missing keys and sent users back to settings. + +Implementation scope: offer setup in trusted dashboard chrome before first use; +keep credentials out of the iframe's chat, prompts, history and browser storage; +use private atomic node credential writes; persist an explicit provider choice; +retain the existing tool permissions and outbound privacy screen. Explicit +provider selection must not silently send a failed request to a different cloud +provider. Routstr funding and allowance remain separate, deliberate actions. + +OpenAI support uses its standard API key, not a presumed Codex subscription key. +For the existing node tool loop, the Chat Completions API retains the same +message/tool-result representation and existing egress checks. This is an +intentional integration choice, not a claim that it is the newer Responses API. +The HTTP adapter has a fixed HTTPS destination, no redirects or retries, a bounded +completion and response, store:false, and errors that do not echo upstream bodies. +The operator supplies the model ID; no inference is issued merely by saving a key. + +Official documentation checked2026-10-06: +- https://developers.openai.com/api/reference/overview (server-side bearer credentials) +- https://developers.openai.com/api/reference/resources/chat/subresources/completions/methods/create + (max_completion_tokens, tool calls, store) +- https://developers.openai.com/api/docs/guides/streaming-responses + (Responses recommendation and distinction from Chat Completions) + +Qualification so far: all 1,244 dashboard tests and 363 AIUI tests pass before +final toolbar placement/funding refinements; latest focused checks pass 15 +dashboard and 24 AIUI tests. Both production bundles build. Chromium390/1440px +verifies first-use setup, private fixture key/model save, no key in localStorage, +retained unsent draft and no page errors. Visual review found an overlapping +mobile setup button; moved setup into the existing model menu. That final layout +still needs rebuilt-browser qualification. No fixture key reached a real provider. + +Explicit local selection now stays local; Claude/OpenAI selection cannot silently +fall through. Routstr selection persists and retains the existing budget checks. +Payment-required responses request the funding view, while temporary502/503 and +rate limits do not claim credentials are missing. Reopening ecash funding reloads +the address, including repeated opens on the same tab. + +Remaining: isolated backend results, final browser/funding/key-error checks, +actual provider compatibility and node deployment. No paid inference tests or new +wallet spending have been performed or authorized by this implementation work. diff --git a/docs/fleet-metrics-followup.md b/docs/fleet-metrics-followup.md new file mode 100644 index 00000000..0ea45fcb --- /dev/null +++ b/docs/fleet-metrics-followup.md @@ -0,0 +1,63 @@ +# Fleet metric repair — qualification in progress + +## Confirmed cause + +The federation.get-state handler passed literal zero values for CPU, RAM, disk +and uptime into build_local_state. Peers therefore received a valid signed RPC +response containing fabricated measurements. Fleet also turned absent fields +into zeroes and treated a newly added peer's registration time as a report. +Read-only live inspection on the dev node found fifteen federation reports with +zero resource values, alongside two nonzero collector reports. Yaya has no +Trusted fleet reports; peer (Observer) access is deliberately distinct. + +## Candidate change + +- Use the existing minute MetricsStore sample, avoiding expensive per-peer probes. +- Samples older than three minutes or ahead of the local clock are unavailable. +- Read uptime from the host uptime source; errors remain unavailable. +- Preserve optional measurements through federation and the Fleet response. +- Display unavailable measurements separately from valid zeroes; average only + valid measurements from reporting online nodes. +- Do not infer contact from when a peer was added. Invalid/future report dates + show unknown status; offline cards state when the last report arrived. +- Keep existing trust restrictions and signed FIPS-preferred federation transport. + This change does not claim FIPS media-stream acceptance or fix stalled peering. + +## Qualification + +All 1,684 backend tests pass (four ignored), including actual-handler fresh, +absent, stale and future sample coverage. All 11 Fleet helper/component tests +pass, including real-zero versus missing values, malformed/future dates, +ordering, averages and rendered unavailable/offline states. Frontend typecheck +passes. The first component assertion incorrectly expected spaces between +separate elements; it was corrected to inspect those elements. No deployment +or live metric repair acceptance yet. + +Required: actual-handler fresh/absent/stale/future samples; full relevant suites; +mobile/desktop browser layout; upgraded sender and receiver comparing local +Monitoring to received Fleet values; restart/reconnect, offline ageing and real +transport evidence. Legacy senders still advertising zeros need the sender fix; +the receiver cannot reliably distinguish a legacy fabricated zero from idle CPU. + +## Dev deployment and collector follow-up + +Dev qualification deployment uses backend041f1fa2 (SHA256 +11b62f697a71b762bf8638063d1a858a68eea6d7268e300f6320c99068efa78a) +and frontend3d0c67eb. Health and unchanged app-container identities/start times +pass. Live federation CPU/memory/disk values exactly match local Monitoring; +Web5 mobile390px/desktop1440px checks pass. Rollback retained on the node at +/var/lib/archipelago/support/followup-20261005-2120. Yaya remains on the prior +backend; no receiver/fleet-wide acceptance is claimed. + +Live testing caught an unbounded podman stats subprocess delaying the first +snapshot, and a300-second collection interval conflicting with180-second Fleet +freshness. The next candidate starts after5seconds and collects once per minute, +with3-second df and8-second podman deadlines and kill-on-drop cleanup. Failed +system reads no longer become invented zero samples. Container-stat failure +leaves container readings unavailable while retaining valid system readings. +The actual subprocess timeout/reaping regression passes; all1,686 backend tests +pass (four ignored). This collector correction is not deployed yet. + +Framework SSH and backend health work, but its stored dashboard session returns +401. Authenticated Framework Monitoring acceptance remains open. No authentication +boundary was bypassed to produce an apparent pass. diff --git a/docs/indeehub-distribution-current-design.md b/docs/indeehub-distribution-current-design.md new file mode 100644 index 00000000..39cb6d33 --- /dev/null +++ b/docs/indeehub-distribution-current-design.md @@ -0,0 +1,73 @@ +# IndeeHub distribution: current implementation design + +Status: design for the post-1.9 follow-up, not implemented/accepted. Earlier +swarm plans describe historical experiments and must not be read as live proof. + +## Standards checked on 2026-10-05 + +- Nostr [NIP-71](https://github.com/nostr-protocol/nips/blob/master/71.md) + defines video metadata, including addressable normal-video kind34235. Use a + stable producer/project identifier for revisions. This does not define paid + viewing rights. Hash/media metadata follows + [NIP-94](https://github.com/nostr-protocol/nips/blob/master/94.md). +- [NIP-98](https://github.com/nostr-protocol/nips/blob/master/98.md) authenticates + HTTP requests; it is not proof of payment or permission to another creator's + project. Keep the existing verified app session and project ownership checks. +- [Cashu NUT-18](https://github.com/cashubtc/nuts/blob/main/18.md) supplies payment + request negotiation. [NUT-04](https://github.com/cashubtc/nuts/blob/main/04.md) + covers mint quotes/issuance; method-specific current specifications and older + deployed mint responses must both be capability-tested. Keep quote identifiers + private to the receiving wallet. Payment settlement must be correlated to the + purchase, never inferred from a change in wallet balance. +- [NUT-19](https://github.com/cashubtc/nuts/blob/main/19.md) provides mint-side + cached responses where supported. It supplements a durable local payment + journal; it does not replace one or make an arbitrary retry safe. +- [LNURL-pay](https://github.com/lnurl/luds/blob/luds/06.md) supports an invoice + handoff. A Lightning address by itself is not a signed settlement receipt. + +## Required implementation contract + +1. Backstage publishes only the selected, owned project. Announcements contain + public metadata, producer identity, node identity, content hashes, current + price/window and accepted-method capabilities. Never publish Cloud paths, + mint quotes, tokens, paid media keys or private management addresses. +2. Each instance's Archipelago source verifies signatures, identity bindings and + monotonic event revisions. Persist discovery so a relay outage does not empty + an existing library. Keep other sources and the current default intact. +3. A purchase binds buyer identity, publisher, project revision, amount, currency, + selected payment method and an unpredictable idempotency identifier. Persist + the quote before requesting payment. Snapshot the offer so later edits cannot + silently change the purchased terms. +4. Reuse file-payment capability negotiation and proven settlement primitives. + Existing file Lightning invoices currently require LND; therefore adding + first-use Lightning-to-Cashu receiving is real work, not a display label. The + receiver must verify a correlated invoice/mint receipt and recover issuance + after a lost response before granting access. Its provisioned ecash address + must map to the actual receiving wallet. No new real payment is authorized by + this design; use isolated fixtures until a bounded payment is approved. +5. A paid purchase creates one durable entitlement. Default demo window proposal: + start at the first successfully authorized media response; persist start and + expiry atomically. Retries, seek and reconnect reuse it without another + payment. Check the entitlement for every media/range/key request. Clock + rollback must not extend an already-started window. +6. Browser/companion playback remains ordinary authenticated local HTTP. Actual + inter-node media bytes use FIPS, with a bound node identity and authenticated + purchase capability. No silent Tor/LAN/iroh fallback for required FIPS media. + Show a recoverable unavailable route without requesting another payment. +7. Range/segment serving streams bounded buffers with backpressure and cancel + propagation. Do not read an entire paid film into a Vec before serving it. + Verify length/hash/revision, reject malformed ranges and path escapes, and + keep cache access subject to the same entitlement. Delivered plaintext cannot + be made impossible to copy; expiry controls subsequent authorized delivery. + +## Qualification that remains required + +Publisher/receiver integration must cover altered metadata, forged identities, +wrong mint, rejected/late/duplicate payment, missing transaction response, restart, +clock change, expired window, revocation, missing FIPS route, seek and disconnect. +Measure real media-byte transport and memory use. Test the actual Backstage, +Archipelago listing, purchase and player on mobile/desktop and companion. + +Use only the operator-designated Yaya Cloud video, preserve the source, and +publish the IndeeHub app image/catalog update at the end of qualification. +One working fixture is not acceptance of the complete distributed flow. diff --git a/docs/indeehub-fips-protocol-review.md b/docs/indeehub-fips-protocol-review.md new file mode 100644 index 00000000..53ceff78 --- /dev/null +++ b/docs/indeehub-fips-protocol-review.md @@ -0,0 +1,82 @@ +# IndeeHub distributed viewing: protocol review + +Status: preliminary design, 2026-10-05. Not implemented or qualified. Reconcile +with actual IndeeHub source before selecting the final event/API contract. + +## Updated constraints + +The older `phase4-streaming-ecash-plan.md` mixes producer content sales with +bandwidth resale and an optional iroh swarm. Its implementation assertions are +historical. The operator now requires FIPS for inter-node media bytes, producer +payments, timed viewing, and an Archipelago catalog across instances. A successful +bandwidth payment alone must not unlock a producer's protected film. + +The old document also describes a fail-open paid-serving path. Audit the current +implementation; never adopt fail-open behavior for paid media or key delivery. +Keep free software updates outside any paid-content gate. + +## Primary specifications checked + +- [NIP-71 video events](https://github.com/nostr-protocol/nips/blob/master/71.md): + defines ordinary and addressable video metadata, including variants. Evaluate + addressable normal-video events for stable film identity and metadata updates. + This is a draft optional specification, not a complete rental/access protocol. +- [NIP-94 file metadata](https://github.com/nostr-protocol/nips/blob/master/94.md): + describes file hashes, MIME types, sizes and locations. Reuse compatible fields + rather than inventing incompatible meanings for standard tags. +- [Blossom BUD-01](https://github.com/hzrd149/blossom/blob/master/buds/01.md): + specifies SHA256-addressed HTTP blob retrieval. Its public cross-origin server + conventions must not be copied onto dashboard/RPC authentication boundaries. + Hash addressing can complement a FIPS-backed gateway; Blossom alone does not + establish payment or timed viewing rights. +- [Cashu NUT-18](https://github.com/cashubtc/nuts/blob/main/18.md): receiver requests + can describe amount, unit, accepted mints and token delivery. Reconcile supported + mint preferences/methods with our deployed wallets, not just the latest schema. +- [NUT-04](https://github.com/cashubtc/nuts/blob/main/04.md) and + [NUT-23](https://github.com/cashubtc/nuts/blob/main/23.md): verify current mint + quote/payment accounting and BOLT11 behavior against supported mints. Keep quote + identifiers private. Successful invoice payment and successful token issuance + are distinct recovery steps; never repeat payment to recover an issuance reply. + +These are source/specification findings. Compatibility with existing clients and +mints remains to be tested. Pin specification revisions when implementing so a +moving document cannot silently change the wire contract. + +## Proposed separation of responsibilities + +1. **Discovery:** publisher-authorized signed metadata; stable film/version ID, + public title/artwork/teaser and explicit supported paid-content extension. + Deduplicate, reconcile updates/deletions and recover missed events. Do not put + private viewing keys, receipts, quotes or wallet credentials on public relays. +2. **Producer offer and settlement:** reuse recipient-capability negotiation from + file purchases. Bind amount, recipient, content version and duration to one + durable purchase ID. Verify settlement at the seller before issuing access. + Support the existing Lightning-to-ecash-address flow without requiring LND. +3. **Viewing entitlement:** a versioned authenticated grant with content, buyer, + validity and replay rules. This is application-specific until interoperability + is demonstrated; do not present it as defined by the metadata/payment NIPs. +4. **Playback gateway:** browser/companion use normal authenticated media requests + to their node. The node obtains protected segments over FIPS and verifies + content integrity and entitlement. Seek/retry/resume reuse the purchase. +5. **Caching:** peers may cache authorized ciphertext. Key delivery and subsequent + segment access remain gated. Already delivered plaintext or keys cannot be + made uncopyable or retroactively revoked; do not promise DRM guarantees. + +No silent media fallback to Tor/LAN/iroh satisfies the operator's FIPS requirement. +If FIPS is unavailable, retain paid ownership and explain retry/recovery rather +than charge again. Separate routing diagnostics from normal playback controls. + +## Decisions and proofs required before publishing the test film + +Identify the exact Yaya Cloud video, preserve its original and obtain the intended +price and viewing-window semantics. Define activation versus expiry, clock skew, +multiple devices, publisher outage, refund policy and content-version replacement. +Use isolated/regtest funds for automated tests; new real payments require a bounded +amount authorization. Publish no unrelated Cloud file. + +Qualification must cover settlement with lost replies, duplicate payment callbacks, +wrong mint/recipient/content, denied keys, expired grants, FIPS outage and recovery, +range/HLS seek, mobile background/resume, source/peer restart and storage recovery. +Verify actual transport and producer balance changes rather than relying on UI +labels. Include fresh/upgrade tests and any required IndeeHub image/catalog update +at the end of the implementation. diff --git a/docs/indeehub-signer-followup.md b/docs/indeehub-signer-followup.md new file mode 100644 index 00000000..c5cd44a1 --- /dev/null +++ b/docs/indeehub-signer-followup.md @@ -0,0 +1,64 @@ +# IndeeHub native signer follow-up + +## Reproduced on Yaya + +The installed app and dashboard have the same provider SHA256 +`529fe82e7c16b51c62678427f565048c3dd93ca28320bd8494c17236ff57bb81`. +A clean mobile Chromium session opens the identity picker. After selecting the +existing profile and Authenticate, the picker disappears but the broker stays +full-screen (`aria-hidden=false`, 390 by 844). Its document has no visible text +or buttons, and the app still shows Sign In. The trace contains signer-ready and +signer-show but no signer-identity or signer-hide through 18 seconds. + +The picker emits an entry from a Vue reactive array. NostrTabSigner passes that +proxy object directly to cross-frame postMessage after hiding the picker. +Structured cloning rejects reactive proxies, interrupting the handoff before +it schedules the broker hide. Reload uses the already-serialized identity from +localStorage, explaining why refresh can appear to repair the problem. + +## Candidate fix + +Send an explicit plain object containing only the selected public identity +fields. The private signer remains on the node; unrelated metadata is excluded. +Consent behavior and the existing green completion animation are unchanged. + +A regression uses a real Vue reactive identity and structuredClone, verifies +that the proxy itself is rejected, the public handoff is cloneable, metadata is +excluded, and the overlay closes. Eleven focused signer/provider tests pass, +and frontend typecheck passes. A browser qualification using candidate dashboard +assets with the actual Yaya app/auth backend is in progress. No deployment or +actual companion acceptance is claimed yet. + +## App source and further review + +The canonical app is the Vite/Vue source in the separate IndeeHub repository, +not the obsolete Next.js Dockerfile under apps/indeedhub. Its existing local +nginx edit and untracked workflow documentation were preserved; new app work +uses a separate worktree. + +Review found additional app-side concerns to test: production network failures +can flip the app into mock mode and fabricate subscribed users, and the header +can discard a valid backend Nostr session when the local signer account has not +been restored. Do not claim these repaired by the broker object-cloning fix. + +## Live candidate qualification and restored-session prompting + +2026-10-05: actual Yaya NIP-98 session exchange returns201 and authenticated +profile returns200 using candidate dashboard and app assets. The overlay hides; +a full refresh reuses the session and profile without another auth exchange. +Evidence: /tmp/archy-indeehub-full-candidate-2.log. This uses headless Chromium +390x844, real cookies/signatures/RPC/API, with only static candidate assets routed +locally. Initial asset-routing attempts hit Chromium private-network checks; +forwarding real API requests in the fixture resolved that test harness failure. +No backend authentication was mocked or bypassed. + +Public-key lookups can inherit transient activation from the identity-picker +click/reload. The provider now avoids interpreting a restored-session hint or +already selected identity as a new account-switch request. Requests still pass +to the authenticated broker; the hint grants no signature permission. Explicit +selectIdentity remains available. Twelve provider/tab-signer regressions pass. + +The combined frontend production check found strict TypeScript nullability +errors in Fleet test array indexing; the fixture now asserts both cards exist +and uses non-null indexing. No production Fleet behavior changed in that repair. +Actual deployment, companion lifecycle and the remaining recovery cases stay open. diff --git a/docs/node-connection-flow-plan.md b/docs/node-connection-flow-plan.md new file mode 100644 index 00000000..eab0a685 --- /dev/null +++ b/docs/node-connection-flow-plan.md @@ -0,0 +1,85 @@ +# Node connection flow plan + +Status: proposal for the post-1.9.0 work. Uses existing components, colors, +spacing, glass cards, typography and motion. No broad navigation redesign has +been deployed. Connection reliability must be qualified before this flow ships. + +## Entry and return paths + +- Web5 always exposes **Connect with Nodes**, including when mobile quick actions + are collapsed. Keep **Connected Nodes** beside the entry or directly below it. +- Cloud peer files links to the same connection flow and retains its return + location. Successful connection returns to that peer's files when appropriate. +- Fleet provides the same connection entry, with an explicit distinction between + connecting to another person's node and linking a node the operator owns. +- Open the route immediately with cached safe summaries or a loading state; + discovery and transport checks run after navigation. Do not await remote calls + before rendering the destination. Cancel obsolete work on navigation away. + +## Connect with Nodes + +Use one page with existing tabs: **Discover**, **Requests**, **Connected**. +On mobile keep tabs in one horizontally scrollable row. Preserve the selected +view, search and scroll position when opening a node and returning. + +Discover shows the existing opt-in Nostr presence results and an explicit invite +entry. Search updates locally; refresh provides immediate progress, timeout and +retry feedback. Distinguish stale advertisements from recently contacted nodes. +A presence event is discovery information, not authorization or proof of reachability. + +Each node has a single clear action: Request connection, View request, or Open +node according to its actual state. Explain what information the request shares. +Avoid duplicate requests on repeated taps or when responses arrive late. + +## Requests and approval + +Display incoming and sent Nostr requests in the same Requests view, with counts +and a readable node identity/name. Incoming requests offer Approve or Reject; +sent requests offer Cancel. Keep completed history available but secondary. + +An approval progresses through distinct states: + +1. Request sent / Awaiting approval. +2. Approved / Connecting — authenticated invitation accepted, join not confirmed. +3. Connected — persisted relationship and authenticated reciprocal confirmation. +4. Connection delayed — show bounded retry and a useful error; retain the approved + operation so restart, lost acknowledgement or transient outage can recover. + +Do not label relay acceptance as peer connection. A retry must reuse the same +logical operation, prevent duplicate peers and retain the operator's trust choice. +Cancellation/rejection delivery failures must be visible rather than reported as +successfully notified. Define recovery for already-approved legacy requests. + +Normal discovery connections grant Observer access. **Link your own nodes** must +be a separate explicit flow with existing ownership/authentication requirements; +being reachable over FIPS never grants Trusted access or remote management rights. + +## Connected nodes and Fleet + +Show actual connection state and last successful authenticated contact. Distinguish +**Offline**, **Connecting**, **Unknown** and **Metrics unavailable**. Last report +age alone does not establish when a node went offline. Future/skewed timestamps +must not make a node permanently online. + +Default ordering: online, connecting, unknown, confirmed offline; stable ordering +within groups. Honor manually selected sorting/filtering and do not disrupt the +user's selection while metrics update. Offline rows show last contact; show an +"offline for" duration only when an observed transition supports it. + +The existing network map uses matching status labels and accessible details; +color alone is insufficient. Node detail keeps Connect/Retry, Files and permitted +management actions together. Do not add duplicate connection mechanisms. + +## Acceptance before deployment + +- Two real nodes: request, approval, reciprocal connection and persisted lists. +- Retry after lost reply, duplicate/reordered events, restart on each side, + unavailable relay, FIPS outage and supported transport recovery. +- Invalid signatures, unsolicited invites, wrong identities, stale/cancelled + requests and blocked peers cannot gain access or elevate trust. +- Desktop and actual companion: first connection, revisit, back navigation, + search, tab switching, refresh, background/resume and interrupted network. +- Measure tap-to-feedback, first usable content, discovery completion and + approval-to-confirmed-connection before and after. Preserve unknown data. +- Operator UAT gives exact nodes, steps and expected states; no extra payment or + wallet/channel changes are needed for connection testing. diff --git a/docs/peering-reliability-followup.md b/docs/peering-reliability-followup.md new file mode 100644 index 00000000..42b8d476 --- /dev/null +++ b/docs/peering-reliability-followup.md @@ -0,0 +1,64 @@ +# Peer requests, delivery, and availability follow-up + +Status: implementation under qualification. Not a claim of reciprocal live-node +acceptance or completion of the post-1.9 backlog. + +## Confirmed failures + +Yaya retained the dev node's approved inbound request while the dev node retained +its outbound Sent request, without reciprocal federation membership. Approval +previously had no durable delivery/retry record. Configured managed relays were +also omitted from reply publication; that separate repair is in 9a041bed. + +The Connected Nodes card cached untimestamped reachability booleans, with cached +results overriding the shared store. A failing RPC was rendered as an offline +route. Nostr requests were absent from its Requests tab and badge. + +## Changes + +- Persist the approval decision and a node-key-encrypted reply before delivery. + Retry the same invite with bounded exponential backoff, at most four eligible + replies per background pass. Relay acknowledgement alone does not remove it; + reciprocal membership does. Removed peers and expired requests are excluded. +- Recover legacy Approved rows through the same supported delivery path. Do not + edit peer files by hand or elevate Observer relationships to Trusted. +- Serialize pending-store mutations and replace its private file atomically. + Malformed storage is preserved and fails explicitly. Conflicting decisions + cannot both win. Approved requests expire after 30 days to permit reconnect. +- Validate the invite's DID and key against the requested identity, normalize + discovery trust to Observer before acceptance and callback. +- Poll in the background every 30 seconds, skipping missed ticks. +- Show Nostr requests in Connected Nodes with a direct link to review them. + Preserve pending rows on failed refresh, and surface partial failures. +- Render independently arriving node lists, limit reachability probes to four, + ignore superseded replies and age timestamped reachability after 90 seconds. + An RPC failure is unknown; an explicit failed reachability check is unreachable. + Report last successful contact without inventing continuous offline duration. +- Place online nodes first, then unknown and unreachable, preserving order within + each group. Fleet describes stale reports as Not reporting. Map labels include + last contact and dashed links indicate no recent contact, not a live route. + +## Evidence so far + +- Original full backend candidate: 1,693 passed, 4 ignored, no failures, through + the isolated runner (`/tmp/archy-peering-full-backend.log`). +- Additional review added recovery-through-real-relay and retry-backoff checks; + all 50 focused federation tests pass, including actual encrypted relay delivery + for legacy Approved rows and suppression of duplicate attempts during backoff. + Final formatted source also passes all 1,693 backend tests (4 ignored). +- Connected Nodes: 9 focused tests pass, including independent rendering, + mixed failed/negative/successful probes, Nostr requests, cache age and clock skew. +- Fleet/request display: 13 focused tests pass. +- Type checking and production UI build pass. Final frontend suite: 1,238 tests + in 153 files pass (`/tmp/archy-peering-full-ui-final.log`). +- Yaya Chromium at 390 and 1440px passes candidate-asset browser checks with + deterministic peer RPC fixtures: availability text/order, approved Nostr requests, + connection navigation and no page errors (`/tmp/archy-peering-browser-candidate-7.log`). + Fixture checks are not evidence of actual reciprocal membership. +- Clean backend artifact at 10d31ae1: SHA256 + `120bd0f51fbceafeceb5557117a442fffc432a7b4d5115da1a163e472f7160da`. + Actual-node deployment and reciprocal membership checks remain required. + +The prior reply-rejection test expected a Pending row. This revision deliberately +supersedes that behavior: the decision remains Approved with delivery pending, +so a relay outage does not undo an operator decision or require reapproval. diff --git a/docs/post-1.9.0-reliability-investigation.md b/docs/post-1.9.0-reliability-investigation.md new file mode 100644 index 00000000..174a0d79 --- /dev/null +++ b/docs/post-1.9.0-reliability-investigation.md @@ -0,0 +1,123 @@ +# Post-1.9.0 reliability investigation + +Status: implementation and qualification in progress. These changes are on +`work/post-190-reliability`, separate from the published 1.9.0-alpha artifacts. +Nothing here constitutes live acceptance or permission to change peer trust. + +## Peering: approved request does not establish a relationship + +Read-only inspection on 2026-10-05 reproduced the operator's report: the receiving +node retains an approved inbound request while the requesting node retains its +outbound request in Sent state; neither has the reciprocal federation entry. +Both have discoverability enabled. FIPS is running on both (different service +names); checking only `fips.service` would incorrectly report one as inactive. + +Source discrepancy: request publication and polling use `handshake_relays()`, +which combines managed relays and configured defaults. Approval, rejection and +cancellation instead use only configuration defaults. Align all three reply +paths with the shared resolver. Add an actual-handler integration test with a +UI-configured local WebSocket relay, signed event verification, recipient-only +NIP-44 decryption, reply types, persisted request states and Observer-only approval. +The isolated test passes: all three reply types reach the managed-only relay, +and a rejected approval remains Pending with no peer added. The complete backend +suite is running; actual-node handshake recovery remains outstanding. + +This discrepancy is not yet a proven complete explanation of the live failure. +Read-only relay queries are being checked. Other source risks requiring separate +qualification include best-effort peer-joined callbacks without durable retry, +five-minute background polling, latest-50-event fetching without a cursor, and +pending-file read/modify/write operations without serialization. Do not manually +mark requests completed or elevate trust to make the UI appear connected. + +## Web5 navigation cleanup + +The Wallet quick-action only toggles a local disconnected flag and refreshes LND +information. Remove it and its independent LND polling; wallet interfaces remain +elsewhere. Rename Find Nodes to Connect with Nodes, including English/Spanish +translations and existing navigation assertions. Four focused component tests +pass. Full frontend tests, type checking and mobile/desktop rendering checks are +still required; no deployment or operator acceptance is claimed. + +## Fleet findings to investigate + +`normalizeFleetNode` substitutes zero for absent metrics, and timestamp age alone +is interpreted as online/offline. Missing values must not be presented as real +zero measurements, and stale reporting must not be treated as a measured offline +transition. Trace telemetry collection and transport before changing presentation. + +## V4V source and demo catalog located + +The existing Gitea repository is `v4v/v4v`, branch `demo-portainer`, at +`3ae171d6b0c728665a860520fe393c0abb772798`. The earlier `lfg2025/v4v` +location does not resolve on that server. Read-only inspection of the actual +Portainer documentation confirms a sanitized demo catalog with 52 entries and +47 playable tracks: 21 bundled demo WAVs and 26 publisher-hosted entries. These +counts describe the documented seed, not a fresh playback acceptance result. + +Its importer is designed to back up existing state, merge by ID and retain +accounts, payment records and edits. Qualify those behaviors before deployment. +The source explicitly keeps private catalog/media archives out of public source +publication. Preserve that boundary when packaging the Yaya-only demo: hiding a +card is not sufficient protection for private media or credentials. Inspect the +existing node's state and exact deployed revision before modifying its stack. + +## Qualification results so far + +Frontend type checking passes. Full suite: 1,221 passed, one unchanged paid-file +case hit its 20-second test timeout under concurrent build/upload load. That +file's six tests all passed when rerun alone. Retain the original failure; do not +rewrite it as an entirely green full-suite run. Four targeted navigation tests +passed separately. Mobile/desktop browser acceptance remains outstanding. + +The first new Rust test compile exposed two fixture-only String/&str mismatches. +Corrected them and added rejected-relay coverage: failed delivery must leave the +request Pending and must not add a peer. Isolated compilation/execution now passes (one integration test covers four +scenarios). No live peering repair or deployment is claimed. + +Public relay queries found no matching reply on the managed relays that completed +the query; some endpoints were unavailable. The default Damus relay requires +authentication for this filter, so its unauthenticated rejection is not evidence +of a missing event. The installed SDK already enables automatic authentication. +Do not infer that the node has the same rejection without authenticated evidence. + +### Completed source-test runs + +The second full frontend run, limited to two workers, passes all1,222 tests across +150 files. Type checking passes. The first full backend run exposed a real +pre-existing bug: encrypted chat/contact nonces starting with `{` or `[` were +misclassified as legacy JSON. This is not dismissed as a flaky test. + +Replace prefix-only classification with full JSON object/array validation using +`IgnoredAny` to avoid building another object tree. Deterministic valid encrypted +fixtures cover both prefixes; existing pre-migration ciphertext compatibility +and tamper/wrong-key tests remain. Full isolated backend rerun passes1,683 tests, +zero failures, four existing ignored. Logs are retained in +`/tmp/archy-followup-backend-suite-2.log` and +`/tmp/archy-followup-ui-suite-2.log`. + +Published1.9.0 artifacts remain unchanged, and their release page discloses this +newly discovered issue. Private snapshots were attempted on all four authorized +nodes before further restarts: only Shorty had an affected message store at the +expected paths; its bytes were verified after backup. No message contents or +wallet data were exported. Actual candidate build/deployment and store reload +acceptance still remain; do not equate source-test success with delivery. + +## Production storage candidate qualification + +The cfd9a596 production binary (SHA256 +3d720c6ddb2622ddef956b5ef0d54ecef64617e4bd16d07327b55ac737d00fe7) +was exercised in the disposable installed-system VM. Independent Python +ChaCha20-Poly1305 fixtures used the guest's key locally, without exporting it. +Encrypted nonces starting with `{` and `[` loaded through the real message RPC, +retained their exact ciphertext, and survived a second service restart. Legacy +JSON with leading whitespace migrated to authenticated ciphertext and survived +another restart with all message fields intact. + +Fixture corrections were required: the first root-owned 0600 file was unreadable +by the service; the next legacy assertion omitted optional fields that serde +normally emits as null. Both initial failures remain in the qualification logs. +The corrected run passed all three data cases. SSH disconnected during final +cleanup, so restoration was completed as a guest systemd job and independently +checked: original 560aa600 backend, active service, and original absent message +store restored. The VM was then shut down. This is not live-fleet deployment or +proof of every corrupt-store recovery path. diff --git a/docs/post-1.9.0-work-backlog.md b/docs/post-1.9.0-work-backlog.md index 3d97dd92..56850faf 100644 --- a/docs/post-1.9.0-work-backlog.md +++ b/docs/post-1.9.0-work-backlog.md @@ -1,8 +1,17 @@ # Work requested after 1.9.0-alpha -Status: queued by the operator on 2026-10-05. Complete the current release first; -these requests do not silently expand its artifact scope. No implementation or -acceptance is claimed by this backlog. +Status: implementation and qualification in progress. 1.9.0-alpha was published +separately; these follow-ups are not in its immutable artifacts. The list below +remains the complete acceptance scope, not a claim that every item is finished. + +Current evidence is recorded in [Fleet metrics](fleet-metrics-followup.md), +[peering reliability](peering-reliability-followup.md), and the +[IndeeHub design review](indeehub-distribution-current-design.md). Monitoring and +the tested signer/dashboard candidate are deployed to dev and Yaya with rollback +backups; actual IndeeHub image publication is deferred until the end as requested. +Framework's authenticated Monitoring check awaits an operator dashboard login. +Streaming, storage-source integrations, AIUI setup, V4V packaging/player, +companion hardware checks and the full Fleet acceptance matrix remain open. ## 1. Distributed IndeeHub publishing and paid viewing diff --git a/neode-ui/public/nostr-provider.js b/neode-ui/public/nostr-provider.js index 804d080d..c27fe2a3 100644 --- a/neode-ui/public/nostr-provider.js +++ b/neode-ui/public/nostr-provider.js @@ -211,7 +211,16 @@ selectedPublicKeyTimer = null; return Promise.resolve(publicKey); } - if (navigator.userActivation && navigator.userActivation.isActive) { + // The picker click itself leaves transient user activation active. A second + // public-key lookup in the same login must not open another picker. + var restoringSession = false; + try { + var sessionHint = sessionStorage.getItem('nostr_token'); + restoringSession = !!sessionHint && sessionHint.indexOf('mock-') !== 0; + } catch (_) {} + // This is only a UI hint: the broker still verifies the node session and + // signing permissions. A restored app token never authorizes a signature. + if (!restoringSession && !selectedIdentity && navigator.userActivation && navigator.userActivation.isActive) { return selectIdentity().then(function () { return getPublicKey(); }); diff --git a/neode-ui/src/components/AIConnectionModal.vue b/neode-ui/src/components/AIConnectionModal.vue new file mode 100644 index 00000000..cdf574f4 --- /dev/null +++ b/neode-ui/src/components/AIConnectionModal.vue @@ -0,0 +1,135 @@ + + + diff --git a/neode-ui/src/components/ReceiveBitcoinModal.vue b/neode-ui/src/components/ReceiveBitcoinModal.vue index 432efedd..12bde4c0 100644 --- a/neode-ui/src/components/ReceiveBitcoinModal.vue +++ b/neode-ui/src/components/ReceiveBitcoinModal.vue @@ -147,6 +147,7 @@ const lightning = useLightningRequired() const props = defineProps<{ show: boolean + initialMethod?: 'lightning' | 'onchain' | 'ecash' | 'ark' /** Optional info banner shown on the on-chain tab (e.g. Zeus channel limits) */ note?: string /** Generate an on-chain address immediately when the modal opens */ @@ -168,7 +169,7 @@ watch(() => props.show, (open) => { // Blank slate on every open: a leftover amount/memo/token or a previous // invoice quietly carrying into a new receive flow is exactly the stale- // state class the operator flagged on the send modal (2026-08-05). - receiveMethod.value = 'onchain' + receiveMethod.value = props.initialMethod ?? 'onchain' invoiceAmount.value = 0 invoiceMemo.value = '' invoiceResult.value = '' @@ -342,8 +343,8 @@ async function pollLnClaims() { onUnmounted(stopLnClaimPoll) // Fetch the address the first time the operator opens the ecash tab. -watch(receiveMethod, (m) => { - if (m === 'ecash' && props.show) { +watch([receiveMethod, () => props.show], ([m, open]) => { + if (m === 'ecash' && open) { lnWatchStartedAt.value = Math.floor(Date.now() / 1000) void loadLnAddress() } diff --git a/neode-ui/src/components/__tests__/AIConnectionModal.test.ts b/neode-ui/src/components/__tests__/AIConnectionModal.test.ts new file mode 100644 index 00000000..9210973f --- /dev/null +++ b/neode-ui/src/components/__tests__/AIConnectionModal.test.ts @@ -0,0 +1,60 @@ +import { mount, flushPromises } from '@vue/test-utils' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import AIConnectionModal from '../AIConnectionModal.vue' +import { rpcClient } from '@/api/rpc-client' +vi.mock('@/api/rpc-client', () => ({ rpcClient: { call: vi.fn() } })) +vi.mock('../ReceiveBitcoinModal.vue', () => ({ default: { props: ['show', 'initialMethod'], template: '

' } })) +vi.mock('@/views/settings/RoutstrBudgetSection.vue', () => ({ default: { template: '
' } })) +const state = () => ({ schema: 1, settings: { provider: 'auto', openai_model: '' }, claude_configured: false, openai_configured: false, local_ready: false, routstr_remaining_sats: 0 }) +function mountModal() { return mount(AIConnectionModal, { props: { show: false }, global: { stubs: { BaseModal: { props: ['show'], template: '
' } } } }) } +function button(w: ReturnType, label: string) { return w.findAll('button').find(b => b.text() === label)! } +beforeEach(() => { vi.clearAllMocks(); vi.mocked(rpcClient.call).mockImplementation(async ({ method }) => method === 'system.settings.get' ? { value: state() } : {}) }) +describe('AI connection setup', () => { + it('detects absent configuration without treating a failed status query as missing keys', async () => { + const w = mountModal() + expect(await (w.vm as any).checkNeeded()).toBe(true) + vi.mocked(rpcClient.call).mockRejectedValueOnce(new Error('offline')) + expect(await (w.vm as any).checkNeeded()).toBe(false) + expect(rpcClient.call).not.toHaveBeenCalledWith(expect.objectContaining({ method: 'system.settings.set' })) + w.unmount() + }) + it('sends a key only to private settings, clears the field, and selects the persisted model', async () => { + const w = mountModal(); await w.setProps({ show: true }); await flushPromises() + await button(w, 'OpenAI API').trigger('click') + await w.get('#ai-connection-key').setValue('test-private-key') + await w.get('#ai-connection-model').setValue('test-chat-model') + await w.get('form').trigger('submit'); await flushPromises() + const writes = vi.mocked(rpcClient.call).mock.calls.map(([r]) => r).filter(r => r.method === 'system.settings.set') + expect(writes.map(r => r.params)).toEqual([{ key: 'openai_api_key', value: 'test-private-key' }, { key: 'ai_provider', value: JSON.stringify({ provider: 'openai', openai_model: 'test-chat-model' }) }]) + expect((w.get('#ai-connection-key').element as HTMLInputElement).value).toBe('') + expect(w.emitted('configured')).toEqual([['openai', 'test-chat-model']]) + expect(JSON.stringify(w.emitted())).not.toContain('test-private-key') + w.unmount() + }) + it('clears unsaved keys when switching provider and closing', async () => { + const w = mountModal(); await w.setProps({ show: true }); await flushPromises() + await button(w, 'Claude API').trigger('click'); await w.get('#ai-connection-key').setValue('unsaved') + await button(w, 'OpenAI API').trigger('click') + expect((w.get('#ai-connection-key').element as HTMLInputElement).value).toBe('') + await w.get('#ai-connection-key').setValue('unsaved-again'); await w.setProps({ show: false }); await w.setProps({ show: true }) + expect((w.get('#ai-connection-key').element as HTMLInputElement).value).toBe('') + expect(w.emitted('configured')).toBeUndefined(); w.unmount() + }) + it('does not reuse an OpenAI model ID when restoring a Routstr connection', async () => { + const value = { ...state(), settings: { provider: 'routstr', openai_model: 'previous-openai-model' } } + vi.mocked(rpcClient.call).mockResolvedValue({ value }) + const w = mountModal(); await (w.vm as any).syncSelection() + expect(w.emitted('configured')).toEqual([['routstr', undefined]]) + w.unmount() + }) + + it('does not authorize Routstr spending from setup when allowance is zero', async () => { + const w = mountModal(); await w.setProps({ show: true }); await flushPromises() + await button(w, 'Routstr · sats').trigger('click'); await flushPromises() + await button(w, 'Use Routstr').trigger('click'); await flushPromises() + expect(w.text()).toContain('Set a spending allowance') + expect(w.emitted('configured')).toBeUndefined() + expect(vi.mocked(rpcClient.call).mock.calls.every(([r]) => !['assistant.budget-set', 'system.settings.set'].includes(r.method))).toBe(true) + w.unmount() + }) +}) diff --git a/neode-ui/src/components/__tests__/ReceiveBitcoinModal.test.ts b/neode-ui/src/components/__tests__/ReceiveBitcoinModal.test.ts index 6c4dbcf0..da3f6f62 100644 --- a/neode-ui/src/components/__tests__/ReceiveBitcoinModal.test.ts +++ b/neode-ui/src/components/__tests__/ReceiveBitcoinModal.test.ts @@ -40,6 +40,17 @@ beforeEach(() => { // unmounts the dialog — but the RPC-eager tab switch is exactly the kind of // path a future change could regress, so it's worth pinning down. describe('ReceiveBitcoinModal — ecash tab click', () => { + it('loads the ecash address on each open when funding starts on the ecash tab', async () => { + vi.mocked(rpcClient.call).mockResolvedValue({ address: 'funding@minibits.cash' } as never) + const wrapper = mount(ReceiveBitcoinModal, { props: { show: false, initialMethod: 'ecash' }, attachTo: document.body }) + await wrapper.setProps({ show: true }); await flushPromises() + expect(document.body.textContent).toContain('funding@minibits.cash') + await wrapper.setProps({ show: false }); await wrapper.setProps({ show: true }); await flushPromises() + expect(document.body.textContent).toContain('funding@minibits.cash') + expect(vi.mocked(rpcClient.call).mock.calls.filter(([r]) => r.method === 'wallet.ecash-lnaddress')).toHaveLength(2) + wrapper.unmount() + }) + it('offers authenticated setup for an unseeded wallet and retries the address after setup', async () => { let active = false vi.mocked(rpcClient.call).mockImplementation(async ({ method }) => { diff --git a/neode-ui/src/components/federation/NetworkMap3D.vue b/neode-ui/src/components/federation/NetworkMap3D.vue index 14945465..39961782 100644 --- a/neode-ui/src/components/federation/NetworkMap3D.vue +++ b/neode-ui/src/components/federation/NetworkMap3D.vue @@ -12,6 +12,7 @@ Trusted Observer Untrusted + Dashed: no recent contact Request
- No nodes reporting. Ensure telemetry is enabled on beta nodes. + No fleet nodes yet. Connect a trusted node to see its status.
@@ -32,7 +32,7 @@
{{ fleetNodeDisplayName(node) }}
@@ -49,10 +49,10 @@
- {{ node.cpu_pct.toFixed(0) }}% + {{ formatMetric(node.cpu_pct) }}
RAM @@ -60,10 +60,10 @@
- {{ node.mem_pct.toFixed(0) }}% + {{ formatMetric(node.mem_pct) }}
Disk @@ -71,10 +71,10 @@
- {{ node.disk_pct.toFixed(0) }}% + {{ formatMetric(node.disk_pct) }} @@ -82,9 +82,9 @@ {{ node.running_count }}/{{ node.container_count }} containers {{ node.federation_peers }} peers -
- Up {{ formatUptime(node.uptime_secs) }} - {{ timeAgo(node.reported_at) }} +
+ {{ node.uptime_secs === null ? 'Uptime unavailable' : 'Up ' + formatUptime(node.uptime_secs) }} + {{ fleetStatus(node.reported_at) === 'unknown' ? 'Status unknown' : fleetStatus(node.reported_at) === 'stale' ? 'Not reporting · last seen ' + timeAgo(node.reported_at) : 'Seen ' + timeAgo(node.reported_at) }}
@@ -94,7 +94,7 @@ diff --git a/neode-ui/src/views/fleet/__tests__/FleetNodeGrid.test.ts b/neode-ui/src/views/fleet/__tests__/FleetNodeGrid.test.ts new file mode 100644 index 00000000..9bbb8dcc --- /dev/null +++ b/neode-ui/src/views/fleet/__tests__/FleetNodeGrid.test.ts @@ -0,0 +1,39 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { mount } from '@vue/test-utils' +import FleetNodeGrid from '../FleetNodeGrid.vue' +import FleetOverviewCards from '../FleetOverviewCards.vue' +import { normalizeFleetNode } from '../useFleetData' + +describe('fleet unavailable readings', () => { + afterEach(() => vi.useRealTimers()) + it('shows an unavailable reading differently from a real zero and explains offline status', () => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-06-10T12:00:00Z')) + const nodes = [normalizeFleetNode({node_id: 'offline', reported_at: '2026-06-08T12:00:00Z', cpu_pct: 0}), normalizeFleetNode({node_id: 'unknown'})] + const wrapper = mount(FleetNodeGrid, { + props: {nodes, sortedNodes: nodes, sortBy: 'status', selectedNodeId: null}, + global: {mocks: {$ver: (v: string) => v}}, + }) + const cards = wrapper.findAll('.fleet-node-card') + expect(cards).toHaveLength(2) + expect(cards[0]!.text()).toContain('0%') + expect(cards[0]!.text()).toContain('—') + expect(cards[0]!.text()).toContain('Not reporting · last seen 2d ago') + expect(cards[1]!.text()).toContain('Status unknown') + expect(cards[1]!.text()).not.toContain('0%') + expect(cards[1]!.text()).toContain('Uptime unavailable') + expect(wrapper.text()).not.toContain('NaN') + wrapper.unmount() + }) + it('renders empty averages without claiming zero load', () => { + const wrapper = mount(FleetOverviewCards, {props: { + nodeCount: 1, onlineCount: 0, offlineCount: 0, unknownCount: 1, + fleetHealthPct: 0, healthyCount: 0, avgCpu: null, avgMem: null, avgDisk: null, + }}) + expect(wrapper.text()).toContain('1 unknown') + expect(wrapper.findAll('.monitoring-stat-card').slice(2).map(card => card.findAll('p').map(p => p.text()).join(' '))).toEqual([ + 'Avg CPU — reporting online nodes', 'Avg RAM — reporting online nodes', 'Avg Disk — reporting online nodes', + ]) + wrapper.unmount() + }) +}) diff --git a/neode-ui/src/views/fleet/__tests__/useFleetData.test.ts b/neode-ui/src/views/fleet/__tests__/useFleetData.test.ts index 52a6c4b0..2936e682 100644 --- a/neode-ui/src/views/fleet/__tests__/useFleetData.test.ts +++ b/neode-ui/src/views/fleet/__tests__/useFleetData.test.ts @@ -1,5 +1,9 @@ -import { describe, expect, it, vi } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' import { + averageMetric, + fleetStatus, + formatMetric, + timeAgo, fleetNodeDisplayName, fleetNodeSubtitle, isOnline, @@ -31,6 +35,7 @@ function node(id: string, reportedAt: string): FleetNode { } describe('fleet data helpers', () => { + afterEach(() => vi.useRealTimers()) it('treats nodes reported within 30 minutes as online', () => { vi.useFakeTimers() vi.setSystemTime(new Date('2026-06-10T12:00:00Z')) @@ -41,7 +46,7 @@ describe('fleet data helpers', () => { vi.useRealTimers() }) - it('sorts status with online nodes first, then latest report', () => { + it('sorts status with online nodes first without shuffling within a status group', () => { vi.useFakeTimers() vi.setSystemTime(new Date('2026-06-10T12:00:00Z')) const nodes = [ @@ -51,8 +56,8 @@ describe('fleet data helpers', () => { ] expect(sortFleetNodes(nodes, 'status').map(n => n.node_id)).toEqual([ - 'online-new', 'online-old', + 'online-new', 'offline', ]) @@ -79,9 +84,9 @@ describe('fleet data helpers', () => { expect(normalized.node_name).toBeNull() expect(normalized.hostname).toBeNull() expect(normalized.server_url).toBeNull() - expect(normalized.cpu_pct).toBe(0) - expect(normalized.mem_pct).toBe(0) - expect(normalized.disk_pct).toBe(0) + expect(normalized.cpu_pct).toBeNull() + expect(normalized.mem_pct).toBeNull() + expect(normalized.disk_pct).toBeNull() expect(normalized.containers).toEqual([]) expect(normalized.recent_alerts).toEqual([]) }) @@ -116,3 +121,47 @@ describe('fleet data helpers', () => { expect(normalizeNodeHistoryResponse({})).toEqual([]) }) }) + + +describe('fleet missing data and clock boundaries', () => { + afterEach(() => vi.useRealTimers()) + it('does not turn malformed, missing or future timestamps into online status', () => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-06-10T12:00:00Z')) + for (const value of ['', 'broken', '2027-01-01T00:00:00Z']) { + expect(fleetStatus(value)).toBe('unknown') + expect(isOnline(value)).toBe(false) + expect(timeAgo(value)).toBe('Unknown') + } + expect(fleetStatus('2026-06-10T11:30:00Z')).toBe('stale') + expect(fleetStatus('2026-06-10T12:00:30Z')).toBe('online') + expect(sortFleetNodes([node('unknown', ''), node('future', '2027-01-01T00:00:00Z'), node('old', '2026-06-10T10:00:00Z'), node('new', '2026-06-10T11:59:00Z')], 'status').map(n => n.node_id)).toEqual(['new', 'unknown', 'future', 'old']) + }) + it('keeps valid zero separate from unavailable and invalid measurements', () => { + expect(normalizeFleetNode({cpu_pct: 0}).cpu_pct).toBe(0) + for (const value of [null, undefined, NaN, Infinity, -1, 101]) { + const result = normalizeFleetNode({cpu_pct: value, mem_pct: value, disk_pct: value}) + expect(result.cpu_pct).toBeNull() + expect(result.mem_pct).toBeNull() + expect(result.disk_pct).toBeNull() + } + expect(formatMetric(null)).toBe('—') + expect(formatMetric(0)).toBe('0%') + expect(formatMetric(25.25, 1)).toBe('25.3%') + }) + it('averages only real measurements from reporting online nodes', () => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-06-10T12:00:00Z')) + const first = node('first', '2026-06-10T11:59:00Z') + first.cpu_pct = 0 + const second = node('second', first.reported_at) + second.cpu_pct = 50 + const missing = node('missing', first.reported_at) + missing.cpu_pct = null + const offline = node('offline', '2026-06-01T00:00:00Z') + offline.cpu_pct = 100 + expect(averageMetric([first, second, missing, offline], 'cpu_pct')).toBe(25) + expect(averageMetric([missing, offline], 'cpu_pct')).toBeNull() + expect(averageMetric([], 'cpu_pct')).toBeNull() + }) +}) diff --git a/neode-ui/src/views/fleet/useFleetData.ts b/neode-ui/src/views/fleet/useFleetData.ts index b60e3349..202bc68e 100644 --- a/neode-ui/src/views/fleet/useFleetData.ts +++ b/neode-ui/src/views/fleet/useFleetData.ts @@ -12,11 +12,11 @@ export interface FleetNode { hostname?: string | null server_url?: string | null version: string - uptime_secs: number + uptime_secs: number | null cpu_cores: number - cpu_pct: number - mem_pct: number - disk_pct: number + cpu_pct: number | null + mem_pct: number | null + disk_pct: number | null container_count: number running_count: number federation_peers: number @@ -43,7 +43,8 @@ export type SortOption = 'status' | 'last-seen' | 'name' // --- Utility Functions --- -export function formatUptime(secs: number): string { +export function formatUptime(secs: number | null): string { + if (secs === null) return 'Unavailable' if (secs < 60) return `${secs}s` const days = Math.floor(secs / 86400) const hours = Math.floor((secs % 86400) / 3600) @@ -56,6 +57,7 @@ export function formatUptime(secs: number): string { export function timeAgo(dateStr: string): string { const now = Date.now() const then = new Date(dateStr).getTime() + if (!Number.isFinite(then) || then > now + 60_000) return 'Unknown' const diffMs = now - then if (diffMs < 0) return 'just now' const diffSecs = Math.floor(diffMs / 1000) @@ -68,18 +70,34 @@ export function timeAgo(dateStr: string): string { return `${diffDays}d ago` } -export function isOnline(reportedAt: string): boolean { - const thirtyMinMs = 30 * 60 * 1000 - return Date.now() - new Date(reportedAt).getTime() < thirtyMinMs +export function fleetStatus(reportedAt: string): 'online' | 'stale' | 'unknown' { + const age = Date.now() - new Date(reportedAt).getTime() + if (!Number.isFinite(age) || age < -60_000) return 'unknown' + return age < 30 * 60 * 1000 ? 'online' : 'stale' } -export function healthBarClass(pct: number): string { +export function isOnline(reportedAt: string): boolean { + return fleetStatus(reportedAt) === 'online' +} + +export function formatMetric(value: number | null, digits = 0): string { + return value === null ? '—' : `${value.toFixed(digits)}%` +} + +export function averageMetric(nodes: FleetNode[], field: 'cpu_pct' | 'mem_pct' | 'disk_pct'): number | null { + const values = nodes.filter(n => isOnline(n.reported_at)).map(n => n[field]).filter((v): v is number => v !== null && Number.isFinite(v)) + return values.length ? values.reduce((sum, value) => sum + value, 0) / values.length : null +} + +export function healthBarClass(pct: number | null): string { + if (pct === null) return '' if (pct >= 85) return 'monitoring-bar-danger' if (pct >= 60) return 'monitoring-bar-warn' return 'monitoring-bar-ok' } -export function healthTextClass(pct: number): string { +export function healthTextClass(pct: number | null): string { + if (pct === null) return '' if (pct >= 85) return 'fleet-text-danger' if (pct >= 60) return 'fleet-text-warn' return '' @@ -132,19 +150,21 @@ export const SORT_OPTIONS: Array<{ label: string; value: SortOption }> = [ { label: 'Name', value: 'name' }, ] +function reportTime(node: FleetNode): number { + return fleetStatus(node.reported_at) === 'unknown' ? 0 : new Date(node.reported_at).getTime() +} + export function sortFleetNodes(nodes: FleetNode[], sortBy: SortOption): FleetNode[] { const sorted = [...nodes] switch (sortBy) { case 'status': sorted.sort((a, b) => { - const aOnline = isOnline(a.reported_at) - const bOnline = isOnline(b.reported_at) - if (aOnline !== bOnline) return aOnline ? -1 : 1 - return new Date(b.reported_at).getTime() - new Date(a.reported_at).getTime() + const rank = { online: 0, unknown: 1, stale: 2 } + return rank[fleetStatus(a.reported_at)] - rank[fleetStatus(b.reported_at)] }) break case 'last-seen': - sorted.sort((a, b) => new Date(b.reported_at).getTime() - new Date(a.reported_at).getTime()) + sorted.sort((a, b) => reportTime(b) - reportTime(a)) break case 'name': sorted.sort((a, b) => fleetNodeDisplayName(a).localeCompare(fleetNodeDisplayName(b))) @@ -153,6 +173,15 @@ export function sortFleetNodes(nodes: FleetNode[], sortBy: SortOption): FleetNod return sorted } +function numberOrNull(value: unknown): number | null { + return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : null +} + +function percentageOrNull(value: unknown): number | null { + const number = numberOrNull(value) + return number !== null && number <= 100 ? number : null +} + function numberOrZero(value: unknown): number { return typeof value === 'number' && Number.isFinite(value) ? value : 0 } @@ -164,17 +193,17 @@ export function normalizeFleetNode(node: Partial): FleetNode { hostname: typeof node.hostname === 'string' ? node.hostname : null, server_url: typeof node.server_url === 'string' ? node.server_url : null, version: typeof node.version === 'string' ? node.version : 'unknown', - uptime_secs: numberOrZero(node.uptime_secs), + uptime_secs: numberOrNull(node.uptime_secs), cpu_cores: numberOrZero(node.cpu_cores), - cpu_pct: numberOrZero(node.cpu_pct), - mem_pct: numberOrZero(node.mem_pct), - disk_pct: numberOrZero(node.disk_pct), + cpu_pct: percentageOrNull(node.cpu_pct), + mem_pct: percentageOrNull(node.mem_pct), + disk_pct: percentageOrNull(node.disk_pct), container_count: numberOrZero(node.container_count), running_count: numberOrZero(node.running_count), federation_peers: numberOrZero(node.federation_peers), recent_alerts: Array.isArray(node.recent_alerts) ? node.recent_alerts : [], containers: Array.isArray(node.containers) ? node.containers : [], - reported_at: typeof node.reported_at === 'string' ? node.reported_at : new Date(0).toISOString(), + reported_at: typeof node.reported_at === 'string' ? node.reported_at : '', } } @@ -195,7 +224,7 @@ type FleetCache = { sortBy: SortOption } -const FLEET_CACHE_KEY = 'archipelago.fleet.cache.v1' +const FLEET_CACHE_KEY = 'archipelago.fleet.cache.v2' function readFleetCache(): Partial { if (typeof window === 'undefined') return {} @@ -246,28 +275,20 @@ export function useFleetData() { // --- Computed --- const onlineCount = computed(() => nodes.value.filter(n => isOnline(n.reported_at)).length) - const offlineCount = computed(() => nodes.value.length - onlineCount.value) - const healthyCount = computed(() => nodes.value.filter(n => n.recent_alerts.length === 0).length) + const offlineCount = computed(() => nodes.value.filter(n => fleetStatus(n.reported_at) === 'stale').length) + const unknownCount = computed(() => nodes.value.filter(n => fleetStatus(n.reported_at) === 'unknown').length) + const healthyCount = computed(() => nodes.value.filter(n => isOnline(n.reported_at) && n.cpu_pct !== null && n.mem_pct !== null && n.disk_pct !== null && n.recent_alerts.length === 0).length) const fleetHealthPct = computed(() => { if (!nodes.value.length) return 0 return Math.round((healthyCount.value / nodes.value.length) * 100) }) - const avgCpu = computed(() => { - if (!nodes.value.length) return 0 - return nodes.value.reduce((sum, n) => sum + n.cpu_pct, 0) / nodes.value.length - }) + const avgCpu = computed(() => averageMetric(nodes.value, 'cpu_pct')) - const avgMem = computed(() => { - if (!nodes.value.length) return 0 - return nodes.value.reduce((sum, n) => sum + n.mem_pct, 0) / nodes.value.length - }) + const avgMem = computed(() => averageMetric(nodes.value, 'mem_pct')) - const avgDisk = computed(() => { - if (!nodes.value.length) return 0 - return nodes.value.reduce((sum, n) => sum + n.disk_pct, 0) / nodes.value.length - }) + const avgDisk = computed(() => averageMetric(nodes.value, 'disk_pct')) const selectedNode = computed(() => { if (!selectedNodeId.value) return null @@ -525,7 +546,7 @@ export function useFleetData() { loading, refreshing, errorMessage, nodes, fleetAlerts, alertsLoading, selectedNodeId, selectedNode, nodeHistory, nodeHistoryLoading, autoRefresh, lastRefreshed, sortBy, chartWidth, - onlineCount, offlineCount, healthyCount, fleetHealthPct, + onlineCount, offlineCount, unknownCount, healthyCount, fleetHealthPct, avgCpu, avgMem, avgDisk, sortedNodes, allAppIds, nodeHistoryLabels, nodeHistoryCpuDatasets, nodeHistoryMemDatasets, nodeHistoryDiskDatasets, refreshAll, selectNode, toggleAutoRefresh, exportFleetData, diff --git a/neode-ui/src/views/web5/Web5.vue b/neode-ui/src/views/web5/Web5.vue index db004627..5ee2ccd6 100644 --- a/neode-ui/src/views/web5/Web5.vue +++ b/neode-ui/src/views/web5/Web5.vue @@ -12,8 +12,6 @@ :dhtDid="dhtDid" :dhtDidCopied="dhtDidCopied" :publishingDht="publishingDht" - :walletConnected="walletConnected" - :connectingWallet="connectingWallet" :nostrRelayStats="nostrRelaysRef?.nostrRelayStats ?? null" :connectedNodesCount="connectedNodesRef?.peers?.length ?? 0" :detectedHwWallets="detectedHwWallets" @@ -23,7 +21,6 @@ @copyDhtDid="copyDhtDid" @refreshDhtDid="refreshDhtDid" @publishDhtDid="publishDhtDid" - @connectWallet="connectWallet" @manageRelays="nostrRelaysRef?.openRelaysModal()" /> @@ -90,7 +87,7 @@ let web5AnimationDone = false diff --git a/neode-ui/src/views/web5/Web5ConnectedNodes.vue b/neode-ui/src/views/web5/Web5ConnectedNodes.vue index e0d8db2b..e8512e2a 100644 --- a/neode-ui/src/views/web5/Web5ConnectedNodes.vue +++ b/neode-ui/src/views/web5/Web5ConnectedNodes.vue @@ -25,7 +25,7 @@ -
+
+

{{ peersError }}

@@ -65,16 +66,16 @@ {{ t('common.loading') }}
-
+

{{ p.name || p.onion || (p.pubkey || '').slice(0, 16) + '...' }}

-

{{ p.onion }}

+

{{ availabilityText(p) }} · {{ contactText(p) }}

+
{{ t('common.loading') }}
-
+
{{ t('web5.noRequests') }}
@@ -231,13 +239,14 @@ diff --git a/neode-ui/src/views/web5/__tests__/Web5ConnectedNodes.test.ts b/neode-ui/src/views/web5/__tests__/Web5ConnectedNodes.test.ts index d4d18d33..994ac5f9 100644 --- a/neode-ui/src/views/web5/__tests__/Web5ConnectedNodes.test.ts +++ b/neode-ui/src/views/web5/__tests__/Web5ConnectedNodes.test.ts @@ -1,8 +1,10 @@ -import { mount } from '@vue/test-utils' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { mount, flushPromises, enableAutoUnmount } from '@vue/test-utils' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import Web5ConnectedNodes from '../Web5ConnectedNodes.vue' import { rpcClient } from '@/api/rpc-client' +enableAutoUnmount(afterEach) + vi.mock('vue-router', () => ({ useRouter: () => ({ push: vi.fn() }), })) @@ -15,6 +17,7 @@ vi.mock('@/api/rpc-client', () => ({ rpcClient: { listPeers: vi.fn(() => new Promise(() => {})), federationListNodes: vi.fn().mockResolvedValue({ nodes: [] }), + federationListPendingRequests: vi.fn().mockResolvedValue({ requests: [] }), checkPeerReachable: vi.fn().mockResolvedValue({ reachable: false }), call: vi.fn(), }, @@ -41,6 +44,7 @@ vi.mock('@/composables/useModalKeyboard', () => ({ describe('Web5ConnectedNodes', () => { beforeEach(() => { vi.clearAllMocks() + sessionStorage.clear() }) it('shows a loading state for empty trusted nodes while peers are loading', async () => { @@ -83,4 +87,43 @@ describe('Web5ConnectedNodes', () => { expect(wrapper.text()).toContain('Please connect') }) + it('shows Nostr requests beside legacy requests without requiring discovery first', async () => { + vi.mocked(rpcClient.call).mockResolvedValue({ requests: [] }) + vi.mocked(rpcClient.federationListPendingRequests).mockResolvedValueOnce({ requests: [{ + id: 'nostr-1', from_did: 'did:key:yaya', from_name: 'Yaya', from_nostr_npub: 'npub1yaya', + from_nostr_pubkey: 'abc', message: null, received_at: new Date().toISOString(), state: 'approved', outbound: false, + }] }) + const wrapper = mount(Web5ConnectedNodes) + await wrapper.vm.loadConnectionRequests() + expect(wrapper.text()).toContain('Yaya') + expect(wrapper.text()).toContain('Approved — connecting') + expect(wrapper.text()).not.toContain('web5.noRequests') + }) + + it('renders direct peers while federation is slow and distinguishes probe failure from unreachable', async () => { + vi.mocked(rpcClient.listPeers).mockResolvedValueOnce({ peers: [ + { onion: 'online.onion', pubkey: 'online', name: 'Available node' }, + { onion: 'failed.onion', pubkey: 'failed', name: 'Query failed' }, + { onion: 'offline.onion', pubkey: 'offline', name: 'No route' }, + ] }) + let finish!: (value: { nodes: [] }) => void + vi.mocked(rpcClient.federationListNodes).mockReturnValueOnce(new Promise(resolve => { finish = resolve })) + vi.mocked(rpcClient.checkPeerReachable).mockImplementation(async onion => { + if (onion === 'failed.onion') throw new Error('unauthorized') + return { onion, reachable: onion === 'online.onion' } + }) + const wrapper = mount(Web5ConnectedNodes) + const load = wrapper.vm.loadPeers() + await flushPromises() + expect(wrapper.text()).toContain('Available node') + expect(wrapper.text()).toContain('Status unknown') + finish({ nodes: [] }) + await load + await flushPromises() + expect(wrapper.text()).toContain('Online · Last seen just now') + expect(wrapper.text()).toContain('Unreachable · No confirmed contact') + expect(wrapper.text()).toContain('Status unknown · No confirmed contact') + expect(wrapper.text()).not.toContain('Offline for') + }) + }) diff --git a/neode-ui/src/views/web5/__tests__/Web5ConnectedNodesScroll.test.ts b/neode-ui/src/views/web5/__tests__/Web5ConnectedNodesScroll.test.ts index cf6ed9ca..79c0dc2a 100644 --- a/neode-ui/src/views/web5/__tests__/Web5ConnectedNodesScroll.test.ts +++ b/neode-ui/src/views/web5/__tests__/Web5ConnectedNodesScroll.test.ts @@ -1,5 +1,5 @@ -import { mount } from '@vue/test-utils' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { mount, enableAutoUnmount } from '@vue/test-utils' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import Web5ConnectedNodes from '../Web5ConnectedNodes.vue' // Structural pin for UIFIX-02: the connected-nodes card's height must track @@ -7,6 +7,8 @@ import Web5ConnectedNodes from '../Web5ConnectedNodes.vue' // child), and the tab panes must scroll inside that height rather than // growing to fit every row. See: +enableAutoUnmount(afterEach) + vi.mock('vue-router', () => ({ useRouter: () => ({ push: vi.fn() }), })) @@ -19,6 +21,7 @@ vi.mock('@/api/rpc-client', () => ({ rpcClient: { listPeers: vi.fn(() => new Promise(() => {})), federationListNodes: vi.fn().mockResolvedValue({ nodes: [] }), + federationListPendingRequests: vi.fn().mockResolvedValue({ requests: [] }), checkPeerReachable: vi.fn().mockResolvedValue({ reachable: false }), call: vi.fn(), }, @@ -45,6 +48,7 @@ vi.mock('@/composables/useModalKeyboard', () => ({ describe('Web5ConnectedNodes scroll contract (UIFIX-02)', () => { beforeEach(() => { vi.clearAllMocks() + sessionStorage.clear() }) it('gives all three tab panes the bounded, sibling-matched scroll contract', () => { diff --git a/neode-ui/src/views/web5/__tests__/Web5Federation.test.ts b/neode-ui/src/views/web5/__tests__/Web5Federation.test.ts index bafbeda2..1c967d29 100644 --- a/neode-ui/src/views/web5/__tests__/Web5Federation.test.ts +++ b/neode-ui/src/views/web5/__tests__/Web5Federation.test.ts @@ -34,13 +34,13 @@ function mountFederation() { } describe('Web5Federation', () => { - it('surfaces Find Nodes and Fleet routes', () => { + it('surfaces Connect with Nodes and Fleet routes', () => { const wrapper = mountFederation() const links = wrapper.findAll('a').map(link => link.attributes('href')) expect(links).toContain('/dashboard/server/federation') expect(links).toContain('/dashboard/fleet') - expect(wrapper.text()).toContain('Find Nodes') + expect(wrapper.text()).toContain('Connect with Nodes') expect(wrapper.text()).toContain('Fleet') }) diff --git a/neode-ui/src/views/web5/__tests__/nodeAvailability.test.ts b/neode-ui/src/views/web5/__tests__/nodeAvailability.test.ts new file mode 100644 index 00000000..102e36a7 --- /dev/null +++ b/neode-ui/src/views/web5/__tests__/nodeAvailability.test.ts @@ -0,0 +1,24 @@ +import { describe, expect, it } from 'vitest' +import { nodeAvailability, lastContactLabel, validContact, REACHABILITY_MAX_AGE, availabilityRank } from '../nodeAvailability' + +describe('node availability observations', () => { + const now = Date.parse('2026-10-06T00:00:00Z') + it('ages cached observations and rejects malformed or future samples', () => { + expect(nodeAvailability({ reachable: true, checkedAt: now }, now)).toBe('online') + expect(nodeAvailability({ reachable: false, checkedAt: now }, now)).toBe('unreachable') + expect(nodeAvailability({ reachable: null, checkedAt: now }, now)).toBe('unknown') + for (const checkedAt of [NaN, now + 120000, now - REACHABILITY_MAX_AGE - 1]) { + expect(nodeAvailability({ reachable: true, checkedAt }, now)).toBe('unknown') + } + expect(nodeAvailability(undefined, now)).toBe('unknown') + }) + it('reports last contact without inventing continuous offline duration', () => { + expect(lastContactLabel(now - 2 * 86400000, now)).toBe('Last seen 2d ago') + for (const value of [null, undefined, '', 'broken', 0, now + 120000]) { + expect(validContact(value, now)).toBeUndefined() + expect(lastContactLabel(value, now)).toBe('No confirmed contact') + } + expect(lastContactLabel(now + 30000, now)).toBe('Last seen just now') + expect(['unreachable', 'online', 'unknown'].sort((a, b) => availabilityRank(a as any) - availabilityRank(b as any))).toEqual(['online', 'unknown', 'unreachable']) + }) +}) diff --git a/neode-ui/src/views/web5/nodeAvailability.ts b/neode-ui/src/views/web5/nodeAvailability.ts new file mode 100644 index 00000000..68bb2453 --- /dev/null +++ b/neode-ui/src/views/web5/nodeAvailability.ts @@ -0,0 +1,25 @@ +/** Reachability is an observation, not proof of a continuous outage. */ +export interface ReachabilitySample { reachable: boolean | null; checkedAt: number; lastSeen?: number } +export type NodeAvailability = 'online' | 'unreachable' | 'unknown' +export const REACHABILITY_MAX_AGE = 90_000 + +export function validContact(value: unknown, now: number): number | undefined { + const timestamp = typeof value === 'number' ? value : typeof value === 'string' ? Date.parse(value) : NaN + return Number.isFinite(timestamp) && timestamp > 0 && timestamp <= now + 60_000 ? timestamp : undefined +} +export function nodeAvailability(sample: ReachabilitySample | undefined, now: number): NodeAvailability { + if (!sample || !Number.isFinite(sample.checkedAt) || now - sample.checkedAt > REACHABILITY_MAX_AGE || sample.checkedAt > now + 60_000) return 'unknown' + return sample.reachable === true ? 'online' : sample.reachable === false ? 'unreachable' : 'unknown' +} +export function lastContactLabel(value: unknown, now: number): string { + const timestamp = validContact(value, now) + if (!timestamp) return 'No confirmed contact' + const seconds = Math.max(0, Math.floor((now - timestamp) / 1000)) + if (seconds < 60) return 'Last seen just now' + if (seconds < 3600) return `Last seen ${Math.floor(seconds / 60)}m ago` + if (seconds < 86400) return `Last seen ${Math.floor(seconds / 3600)}h ago` + return `Last seen ${Math.floor(seconds / 86400)}d ago` +} +export function availabilityRank(status: NodeAvailability): number { + return status === 'online' ? 0 : status === 'unknown' ? 1 : 2 +} diff --git a/neode-ui/src/views/web5/types.ts b/neode-ui/src/views/web5/types.ts index a6af7a48..071df800 100644 --- a/neode-ui/src/views/web5/types.ts +++ b/neode-ui/src/views/web5/types.ts @@ -159,4 +159,4 @@ export interface DwnMessageEntry { export type VisibilityLevel = 'hidden' | 'discoverable' | 'public' -export type Peer = { onion: string; pubkey: string; name?: string; did?: string } +export type Peer = { onion: string; pubkey: string; name?: string; did?: string; last_seen?: string }