fix(fips): P0 uptime fixes — open peer port 5679, allow /blob+/dwn, fix LAN anchor port, un-deaden direct peering, fast-fail budgets

Phase A1 of docs/FIPS-UPTIME-AND-UI-STATE-PLAN.md — the five changes that
made FIPS fall back to Tor even when a FIPS path existed:

- RC0: the fips.d drop-in now opens PEER_PORT 5679 (was 80+8443 only, so
  every hardened node firewalled peers' FIPS dials; 28k drops on .198)
- RC4: /blob/ and /dwn/ added to the peer-path allowlist — mesh file
  sharing and DWN sync were 404 → 100% Tor by construction
- RC2-G2: lan_fips_anchors dials PUBLISHED_UDP_PORT (2121) instead of the
  dead 8668, with a drift-guard test against the rendered daemon config
- RC2-G1: direct LAN peering actually runs now — mDNS TXT advertises the
  FIPS npub, discovery calls set_fips_npub, and the anchor tick hydrates
  npubs from federation storage for peers on older builds
- RC3: FIPS attempt budget is a hard cap (retry no longer doubles it) and
  the 12 hot call sites get explicit fips_timeout fast-fail so Tor keeps
  its full budget (browse-peer, preview, /blob, DWN, node-message,
  rotation notifies)

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
archipelago
2026-07-27 16:42:44 -04:00
co-authored by Claude Fable 5
parent 94b5374f66
commit eb2fc0f37b
12 changed files with 176 additions and 23 deletions
+37 -2
View File
@@ -290,12 +290,12 @@ pub struct ApplyResult {
/// FIPS UDP transport port (matches `transports.udp.bind_addr` in the generated
/// `fips.yaml`). Direct peer links dial this, NOT the HTTP/LAN messaging port.
const FIPS_UDP_PORT: u16 = 8668;
const FIPS_UDP_PORT: u16 = crate::fips::PUBLISHED_UDP_PORT;
/// Build transient seed-anchor entries that dial LAN-discovered federation peers
/// directly over their FIPS UDP transport. For each peer the registry knows both
/// a LAN socket address AND a FIPS npub for, point a `udp` anchor at
/// `<lan-ip>:8668`. This lets co-located federation nodes form a DIRECT FIPS link
/// `<lan-ip>:<FIPS_UDP_PORT>`. This lets co-located federation nodes form a DIRECT FIPS link
/// instead of depending on the global anchor's spanning tree to route between
/// them (the cause of every dial falling back to Tor when the anchor link flaps).
///
@@ -448,4 +448,39 @@ mod tests {
assert_eq!(a.transport, "udp");
assert_eq!(a.label, "");
}
#[test]
fn lan_fips_anchor_port_matches_daemon_bind() {
// Drift guard: direct LAN anchors must dial the UDP port the
// generated fips.yaml actually binds. These were out of sync for
// months (anchors dialed 8668, the daemon bound 2121), making the
// whole direct-peering feature dial a dead port.
let yaml = crate::fips::config::render_config_yaml();
assert!(
yaml.contains(&format!("0.0.0.0:{FIPS_UDP_PORT}")),
"lan_fips_anchors dials :{FIPS_UDP_PORT} but the daemon config binds elsewhere"
);
}
#[test]
fn lan_fips_anchors_builds_direct_entry() {
let peer = crate::transport::PeerRecord {
did: "did:key:zpeer".to_string(),
lan_address: Some("192.168.63.198:5678".to_string()),
fips_npub: Some("npub1peer".to_string()),
..Default::default()
};
let out = lan_fips_anchors(&[peer]);
assert_eq!(out.len(), 1);
assert_eq!(out[0].address, format!("192.168.63.198:{FIPS_UDP_PORT}"));
assert_eq!(out[0].transport, "udp");
// Peers missing either the LAN address or the npub produce nothing.
let no_npub = crate::transport::PeerRecord {
did: "did:key:zother".to_string(),
lan_address: Some("192.168.63.199:5678".to_string()),
..Default::default()
};
assert!(lan_fips_anchors(&[no_npub]).is_empty());
}
}
+15 -5
View File
@@ -240,11 +240,21 @@ pub async fn install(identity_dir: &Path) -> Result<()> {
// framework-pt). Ship the allowance as a fips.d drop-in on every
// install/upgrade so no node ever regresses to a UI-less mesh.
sudo_install_dir("/etc/fips/fips.d").await?;
let dropin = "# Written by archipelago on every daemon config install.\n\
# Allows the web UI + peer API through the fips0\n\
# default-deny inbound baseline (fips.nft).\n\
tcp dport 80 accept\n\
tcp dport 8443 accept\n";
// PEER_PORT (5679) carries ALL federation sync, cloud browse/download,
// mesh envelopes, DWN and invoices. It was missing from this allowlist
// while the comment claimed "web UI + peer API" — so every hardened
// node silently dropped peers' FIPS dials at the firewall and the whole
// fleet fell back to Tor (root-caused live 2026-07-27: 28k drops on
// .198's counter; :5679 answered in 0.35s once the rule was inserted).
let dropin = format!(
"# Written by archipelago on every daemon config install.\n\
# Allows the web UI + peer API through the fips0\n\
# default-deny inbound baseline (fips.nft).\n\
tcp dport 80 accept\n\
tcp dport 8443 accept\n\
tcp dport {peer_port} accept\n",
peer_port = crate::fips::dial::PEER_PORT
);
let nft_stage = std::env::temp_dir().join(format!("fips-webui-{}.nft", std::process::id()));
tokio::fs::write(&nft_stage, dropin)
.await
+44 -8
View File
@@ -447,14 +447,26 @@ impl<'a> PeerRequest<'a> {
}
};
let url = format!("{}{}", base, self.path);
let c = client_with_timeout(self.fips_attempt_timeout());
let budget = self.fips_attempt_timeout();
// With an explicit fast-fail cap, halve the per-attempt client
// timeout so the one-retry path in send_with_retry fits inside the
// budget instead of silently doubling it ("fips_timeout(6s)" used
// to really mean ~12.6s). Without one (long streaming downloads),
// keep the full budget per attempt — the client timeout also
// governs body streaming and must not truncate a real transfer.
let per_attempt = if self.fips_timeout.is_some() {
budget / 2
} else {
budget
};
let c = client_with_timeout(per_attempt);
let mut rb = c.post(&url).json(body);
for (k, v) in &self.headers {
rb = rb.header(*k, v);
}
match send_with_retry(rb).await {
Ok(r) => Ok(Some(r)),
Err(e) => {
match tokio::time::timeout(budget, send_with_retry(rb)).await {
Ok(Ok(r)) => Ok(Some(r)),
Ok(Err(e)) => {
tracing::debug!(
"FIPS POST {} failed after retry: {}, falling back to Tor",
url,
@@ -462,6 +474,14 @@ impl<'a> PeerRequest<'a> {
);
Ok(None)
}
Err(_) => {
tracing::debug!(
"FIPS POST {} exceeded attempt budget {:?}, falling back to Tor",
url,
budget
);
Ok(None)
}
}
}
@@ -480,14 +500,22 @@ impl<'a> PeerRequest<'a> {
}
};
let url = format!("{}{}", base, self.path);
let c = client_with_timeout(self.fips_attempt_timeout());
let budget = self.fips_attempt_timeout();
// Same budget discipline as the POST path: halve per attempt only
// under an explicit fast-fail cap; hard-cap the retry sequence.
let per_attempt = if self.fips_timeout.is_some() {
budget / 2
} else {
budget
};
let c = client_with_timeout(per_attempt);
let mut rb = c.get(&url);
for (k, v) in &self.headers {
rb = rb.header(*k, v);
}
match send_with_retry(rb).await {
Ok(r) => Ok(Some(r)),
Err(e) => {
match tokio::time::timeout(budget, send_with_retry(rb)).await {
Ok(Ok(r)) => Ok(Some(r)),
Ok(Err(e)) => {
tracing::debug!(
"FIPS GET {} failed after retry: {}, falling back to Tor",
url,
@@ -495,6 +523,14 @@ impl<'a> PeerRequest<'a> {
);
Ok(None)
}
Err(_) => {
tracing::debug!(
"FIPS GET {} exceeded attempt budget {:?}, falling back to Tor",
url,
budget
);
Ok(None)
}
}
}