merge: bring main (v1.7.125 + .126 work) into phase-13 branch pre-deploy

63 main commits since the fork point — gate cookie-strip fix, named-volume
create fix, appgate catalog classification, RNode error surfacing — merged
so 13-14/13-15 on-device verification runs against current production code.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
archipelago
2026-08-06 09:33:40 -04:00
co-authored by Claude Fable 5
117 changed files with 5120 additions and 632 deletions
@@ -412,6 +412,8 @@ impl RpcHandler {
"mesh.send-channel" => self.handle_mesh_send_channel(params).await,
"mesh.broadcast" => self.handle_mesh_broadcast().await,
"mesh.reboot-radio" => self.handle_mesh_reboot_radio(params).await,
"mesh.rnode-config" => self.handle_mesh_rnode_config().await,
"mesh.rnode-config-apply" => self.handle_mesh_rnode_config_apply(params).await,
"mesh.configure" => self.handle_mesh_configure(params).await,
"mesh.send-invoice" => self.handle_mesh_send_invoice(params).await,
"mesh.send-coordinate" => self.handle_mesh_send_coordinate(params).await,
@@ -487,6 +489,8 @@ impl RpcHandler {
"system.disk-cleanup" => self.handle_system_disk_cleanup().await,
"system.reboot" => self.handle_system_reboot(params).await,
"system.factory-reset" => self.handle_system_factory_reset(params).await,
"auth.session-policy.get" => self.handle_session_policy_get().await,
"auth.session-policy.set" => self.handle_session_policy_set(params).await,
"system.settings.get" => self.handle_system_settings_get(params).await,
"system.settings.set" => self.handle_system_settings_set(params).await,
"system.kiosk-display.get" => self.handle_system_kiosk_display_get().await,
@@ -192,6 +192,19 @@ impl RpcHandler {
.get("message")
.and_then(|v| v.as_str())
.unwrap_or("Unknown error");
// LND's sweep refusal reads like a debug dump ("insufficient
// input to create sweep tx: input_sum=0 BTC, output_sum=…").
// input_sum=0 with a tiny output means the wallet's coins are
// unconfirmed or below Bitcoin's dust minimum — say that
// (framework-pt sweep of 92 sats, 2026-08-06).
if msg.contains("insufficient input to create sweep tx") {
return Err(anyhow::anyhow!(
"Failed to send: your on-chain balance is too small or still \
unconfirmed to sweep. Bitcoin cannot build a transaction from \
coins below the dust minimum (~546 sats) or from funds that \
have not confirmed yet. (LND: {msg})"
));
}
return Err(anyhow::anyhow!("Failed to send: {}", msg));
}
+107 -2
View File
@@ -104,10 +104,115 @@ impl RpcHandler {
.as_ref()
.ok_or_else(|| anyhow::anyhow!("Mesh service not running. Enable mesh first."))?;
svc.reboot_radio(seconds).await?;
let message = svc.reboot_radio(seconds).await?;
info!(seconds, "Mesh radio reboot requested via RPC");
Ok(serde_json::json!({ "reboot": true, "seconds": seconds }))
Ok(serde_json::json!({ "reboot": true, "seconds": seconds, "message": message }))
}
/// mesh.rnode-config — persisted RF settings + the live radio state
/// (radio-confirmed values) for the LoRa settings panel. `live` is best-
/// effort: null with `live_error` when no Reticulum radio is connected.
pub(in crate::api::rpc) async fn handle_mesh_rnode_config(&self) -> Result<serde_json::Value> {
let settings = mesh::rnode_settings::RNodeRfSettings::load(&self.config.data_dir).await;
let (live, live_error) = match self.mesh_service.read().await.as_ref() {
Some(svc) => match svc.radio_state().await {
Ok(state) => (Some(state), None),
Err(e) => (None, Some(format!("{e:#}"))),
},
None => (None, Some("Mesh service not running".to_string())),
};
Ok(serde_json::json!({
"settings": settings,
"live": live,
"live_error": live_error,
}))
}
/// mesh.rnode-config-apply — validate + persist the RF settings, restart
/// the radio daemon so they take effect, then read back the radio-
/// confirmed values as proof. Returns { applied, live, message }; a
/// failed read-back still reports the persisted settings with a clear
/// message instead of pretending success.
pub(in crate::api::rpc) async fn handle_mesh_rnode_config_apply(
&self,
params: Option<serde_json::Value>,
) -> Result<serde_json::Value> {
let params = params.ok_or_else(|| anyhow::anyhow!("Missing params"))?;
let settings: mesh::rnode_settings::RNodeRfSettings = serde_json::from_value(
params
.get("settings")
.cloned()
.ok_or_else(|| anyhow::anyhow!("Missing 'settings'"))?,
)
.map_err(|e| anyhow::anyhow!("Invalid settings: {e}"))?;
settings.validate()?;
settings.save(&self.config.data_dir).await?;
info!(?settings, "RNode RF settings persisted");
// Restart the radio daemon so the new args apply. No radio connected
// is fine — the settings apply on the next connect.
let service = self.mesh_service.read().await;
let Some(svc) = service.as_ref() else {
return Ok(serde_json::json!({
"applied": false,
"message": "Settings saved. They apply when the mesh service next connects to the radio.",
}));
};
if let Err(e) = svc.reboot_radio(2).await {
return Ok(serde_json::json!({
"applied": false,
"message": format!(
"Settings saved, but the radio daemon restart failed: {e:#}. \
They apply on the next reconnect."
),
}));
}
// Read-back: poll until the respawned daemon reports the radio online
// with our applied values (the respawn re-detects the RNode, ~15s).
let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(45);
let mut last_live = None;
while tokio::time::Instant::now() < deadline {
tokio::time::sleep(std::time::Duration::from_secs(3)).await;
if let Ok(state) = svc.radio_state().await {
let online = state
.get("online")
.and_then(|v| v.as_bool())
.unwrap_or(false);
last_live = Some(state);
if online {
break;
}
}
}
match last_live {
Some(live) => {
let confirmed = live
.get("r_frequency")
.and_then(|v| v.as_u64())
.map(|f| f == settings.frequency)
.unwrap_or(false);
Ok(serde_json::json!({
"applied": true,
"confirmed": confirmed,
"live": live,
"message": if confirmed {
"The radio confirmed it is now using the applied settings."
} else {
"Settings applied and the daemon restarted; the radio has not \
confirmed the new values yet — recheck in a few seconds."
},
}))
}
None => Ok(serde_json::json!({
"applied": true,
"confirmed": false,
"live": null,
"message": "Settings applied and the daemon restarted, but it has not \
reported the radio state yet — recheck in a few seconds.",
})),
}
}
/// mesh.configure — Enable/disable mesh and set device path.
@@ -85,6 +85,38 @@ pub(super) fn sanitize_error_message(msg: &str) -> String {
// them in the first place (ecash send, 2026-07-22).
"Insufficient balance",
"Insufficient funds",
// On-chain send/sweep refusals from LND ("Failed to send: your
// on-chain balance is too small or still unconfirmed to sweep…").
// Masking sent the operator to journalctl again (framework-pt
// sweep, 2026-08-06) — same lesson as the two above.
"Failed to send",
// A frontend newer than the daemon calls methods it doesn't have.
// Masked, this reads as "the feature is broken" instead of "this
// node needs its update" — hit live the moment the .126 LoRa panel
// was deployed ahead of its binary (2026-08-06).
"Unknown method",
// RNode RF settings validation (mesh::rnode_settings::validate) —
// every one names the offending field and its legal range, which is
// the entire point of validating before touching the radio.
"frequency ",
"bandwidth ",
"spreading factor ",
"coding rate ",
"tx power ",
"airtime_limit_short",
"airtime_limit_long",
"port must be an absolute",
"Invalid settings",
"Missing 'settings'",
// Mesh preconditions the operator can act on directly.
"Mesh service not running",
"No mesh device connected",
"Mesh listener not running",
"MeshCore radios have no remote reboot",
"Radio state read-back",
"The radio daemon did not answer",
"The radio did not acknowledge",
"RNode interface is disabled",
// Lightning payment failures carry LND's reason ("invoice expired.
// Valid until …", "no route", …) — the user can act on every one of
// them, and masking sent the operator to journalctl (invoice-expired
+54 -195
View File
@@ -307,19 +307,24 @@ impl RpcHandler {
let deps = self.gate_install_deps(package_id).await?;
check_bitcoin_pruning_compatibility(package_id).await?;
log_optional_dep_info(package_id, &deps);
let repaired_bitcoin_conf =
if matches!(package_id, "bitcoin" | "bitcoin-core" | "bitcoin-knots") {
// Materialise the RPC password file before any install path
// runs. The orchestrator path resolves secret_env from
// /var/lib/archipelago/secrets/bitcoin-rpc-password at start
// time; if the file is missing, bitcoind exits within ms.
// bitcoin_rpc_credentials() generates + persists on first
// call (OnceCell-cached), so this is idempotent.
let _ = crate::bitcoin_rpc::bitcoin_rpc_credentials().await;
ensure_bitcoin_rpc_config().await?
} else {
false
};
if matches!(package_id, "bitcoin" | "bitcoin-core" | "bitcoin-knots") {
// Materialise the RPC password file before any install path
// runs. The orchestrator path resolves secret_env from
// /var/lib/archipelago/secrets/bitcoin-rpc-password at start
// time; if the file is missing, bitcoind exits within ms.
// bitcoin_rpc_credentials() generates + persists on first
// call (OnceCell-cached), so this is idempotent.
let _ = crate::bitcoin_rpc::bitcoin_rpc_credentials().await;
// A stale datadir bitcoin.conf from an older install conflicts
// with the container's -conf=/tmp/rpc.conf launch (see
// apps/bitcoin-core & bitcoin-knots manifest.yml) and makes
// Bitcoin Core refuse to start at all. Clear it before
// (re)install. Unlike the old bind-setting "repair" this was
// replacing, it never requires restarting an already-running
// container — bitcoind doesn't read this file, so removing it
// changes nothing at runtime.
remove_stale_bitcoin_conf().await?;
}
// For orchestrator-managed apps, skip the legacy "container exists →
// adopt + return" probe entirely. The orchestrator's own install path
@@ -389,37 +394,7 @@ impl RpcHandler {
.trim()
.to_string();
if state == "running" && repaired_bitcoin_conf {
info!(
"Restarting existing container {} after bitcoin.conf RPC repair",
package_id
);
let restart_output = tokio::process::Command::new("podman")
.args(["restart", package_id])
.output()
.await
.context(
"Failed to restart existing container after bitcoin.conf repair",
)?;
if !restart_output.status.success() {
let stderr = String::from_utf8_lossy(&restart_output.stderr);
install_log(&format!(
"INSTALL ADOPT FAIL: {} - restart after RPC repair failed: {}",
package_id, stderr
))
.await;
return Err(anyhow::anyhow!(
"Container {} exists but failed to restart after RPC repair: {}",
package_id,
stderr
));
}
let _ = tokio::process::Command::new("podman")
.args(["restart", "archy-bitcoin-ui"])
.output()
.await;
wait_for_adopted_container(package_id, package_id).await?;
} else if state != "running" {
if state != "running" {
// Start the stopped/exited container
info!("Starting existing container {} (was {})", package_id, state);
let start_output = tokio::process::Command::new("podman")
@@ -715,9 +690,13 @@ impl RpcHandler {
}
}
// Pre-install: write config files BEFORE chown (dir is still owned by archipelago user)
// Pre-install: clear a stale datadir bitcoin.conf BEFORE chown (dir is
// still owned by archipelago user). bitcoind is launched with
// -conf=/tmp/rpc.conf (see apps/bitcoin-core & bitcoin-knots
// manifest.yml) and never reads a datadir bitcoin.conf — if one
// exists, Bitcoin Core's own safety check refuses to start at all.
if matches!(package_id, "bitcoin" | "bitcoin-core" | "bitcoin-knots") {
self.write_bitcoin_conf(&rpc_user, &rpc_pass).await?;
remove_stale_bitcoin_conf().await?;
}
if package_id == "lnd" {
@@ -1435,101 +1414,13 @@ impl RpcHandler {
}
}
/// Write bitcoin.conf with rpcauth (salted HMAC hash, no plaintext password).
async fn write_bitcoin_conf(&self, rpc_user: &str, rpc_pass: &str) -> Result<()> {
let bitcoin_dir = "/var/lib/archipelago/bitcoin";
let conf_path = format!("{}/bitcoin.conf", bitcoin_dir);
// Idempotent: once bitcoin-knots (or a prior install) has started,
// the data dir is chowned into the container's user namespace
// (e.g. UID 100100 on the host) with 700 perms — the archipelago
// daemon can no longer stat or write there. Treat any non-NotFound
// error on the conf as "conf already provisioned by the container
// user" and skip. Matches the lnd.conf behavior below.
match tokio::fs::metadata(&conf_path).await {
Ok(_) => {
ensure_bitcoin_rpc_config().await?;
info!("bitcoin.conf already exists, ensured Bitcoin RPC config");
return Ok(());
}
Err(e) if e.kind() == std::io::ErrorKind::NotFound => {}
Err(_) => {
ensure_bitcoin_rpc_config().await?;
info!("bitcoin.conf path inaccessible, ensured Bitcoin RPC config via host helper");
return Ok(());
}
}
use hmac::{Hmac, Mac};
use sha2::Sha256;
// KEY-05: the salt is half of the stored `rpcauth=` credential line, so
// source named and draw guarded.
let mut salt_bytes = [0u8; 16];
crate::entropy::draw_key_bytes(&mut rand::rngs::OsRng, &mut salt_bytes).map_err(|e| {
anyhow::anyhow!("Refusing to build an rpcauth line from degenerate salt entropy: {e}")
})?;
let salt_hex = hex::encode(salt_bytes);
let mut mac = Hmac::<Sha256>::new_from_slice(salt_hex.as_bytes())
.expect("HMAC accepts any key length");
mac.update(rpc_pass.as_bytes());
let hash_hex = hex::encode(mac.finalize().into_bytes());
let rpcauth_line = format!("rpcauth={}:{}${}", rpc_user, salt_hex, hash_hex);
// Default to full archive — operators with 2TB+ drives shouldn't be
// silently pruned down to 550 MB. Users who want a pruned node can
// set `prune=N` in bitcoin.conf themselves after install.
//
// printtoconsole=0: bitcoind already writes debug.log in the datadir
// (self-shrunk on restart); duplicating it to stdout pushed every IBD
// "UpdateTip" line through conmon into journald (>1 GB/day). Deep
// debugging uses /var/lib/archipelago/bitcoin/debug.log.
// rpcbind=0.0.0.0 is REQUIRED inside a container: with rpcallowip set
// but no rpcbind, bitcoind binds RPC to 127.0.0.1 in the container
// netns only — LND / the Bitcoin UI dialing bitcoin-knots:8332 over
// the bridge get connection refused (fresh-install LND crash-loop +
// bitcoin-rpc 502, seen on the 1.7.99 ISO). The port publish stays
// 127.0.0.1-only on the host, so exposure is unchanged.
// Prune sized to the data volume. A full archive needs ~810 GB and
// grows; silently writing an unpruned config onto a small disk fills
// it mid-IBD (framework node 2026-07-14: unpruned mainnet on a 205 GB
// volume). Volumes with real archival headroom (≥1.2 TB) stay full
// archive; smaller ones get prune = 25% of the volume, clamped to
// [550 MB, 100 GB], leaving room for LND/apps sharing the disk.
let prune_line = match bitcoin_data_volume_gb().await {
Some(total_gb) if total_gb > 0 && total_gb < 1200 => {
let prune_mb = ((total_gb as f64 * 0.25 * 1024.0) as u64).clamp(550, 100_000);
info!(
volume_gb = total_gb,
prune_mb, "Data volume below archival size — enabling sized bitcoin prune"
);
format!("prune={}\n", prune_mb)
}
_ => String::new(),
};
let bitcoin_conf = format!(
"\
# rpcauth: salted hash only - no plaintext password in config or CLI\n\
{}\n\
server=1\n\
rpcbind=0.0.0.0\n\
rpcallowip=0.0.0.0/0\n\
listen=1\n\
rpcthreads=16\n\
rpcworkqueue=256\n\
printtoconsole=0\n\
{}",
rpcauth_line, prune_line
);
tokio::fs::create_dir_all(bitcoin_dir)
.await
.context("Failed to create bitcoin data directory")?;
tokio::fs::write(&conf_path, bitcoin_conf)
.await
.context("Failed to write bitcoin.conf")?;
info!("Created bitcoin.conf with rpcauth (no plaintext credentials)");
Ok(())
}
// write_bitcoin_conf removed: bitcoind is launched with -conf=/tmp/rpc.conf
// (see apps/bitcoin-core & bitcoin-knots manifest.yml, commit a597c1d9)
// and never reads a datadir bitcoin.conf. Writing one here created a
// fatal "-conf vs default bitcoin.conf" conflict on every subsequent
// start (Bitcoin Core's own datadir-conflict safety check). See
// `remove_stale_bitcoin_conf` below, which replaces both this and
// `ensure_bitcoin_rpc_config`.
/// Write LND config file with Bitcoin RPC credentials.
async fn write_lnd_conf(&self, rpc_user: &str, rpc_pass: &str) -> Result<()> {
@@ -2624,28 +2515,12 @@ async fn wait_for_adopted_container(package_id: &str, container_name: &str) -> R
))
}
/// Total size (GB) of the filesystem holding the bitcoin data dir, via
/// `df -k`. None when df fails (containers, exotic mounts) — callers treat
/// unknown as "don't prune" to preserve archival defaults on big iron.
async fn bitcoin_data_volume_gb() -> Option<u64> {
let target = if std::path::Path::new("/var/lib/archipelago").exists() {
"/var/lib/archipelago"
} else {
"/"
};
let output = tokio::process::Command::new("df")
.args(["-k", target])
.output()
.await
.ok()?;
if !output.status.success() {
return None;
}
let stdout = String::from_utf8_lossy(&output.stdout);
let line = stdout.lines().nth(1)?;
let kb: u64 = line.split_whitespace().nth(1)?.parse().ok()?;
Some(kb / 1024 / 1024)
}
// bitcoin_data_volume_gb removed with write_bitcoin_conf: it only fed that
// function's volume-aware `prune=` line, which bitcoind never read either
// (see remove_stale_bitcoin_conf). The manifest's shell entrypoint already
// computes DISK_GB_VALUE and hardcodes -prune=550 on small volumes — a
// real volume-aware prune fix belongs there, not in a conf file nothing
// reads. Tracked as follow-up in bitcoin-conf-crash-patch.md.
/// One-shot probe: does bitcoind answer an authenticated getblockchaininfo?
/// Works during IBD (the call answers with progress while syncing). Goes via
@@ -2723,52 +2598,36 @@ async fn wait_for_bitcoin_rpc_gate(package_id: &str) -> Result<()> {
Ok(())
}
async fn ensure_bitcoin_rpc_config() -> Result<bool> {
/// bitcoind reads only `/tmp/rpc.conf` + CLI args at container start (see
/// apps/bitcoin-core & bitcoin-knots manifest.yml, commit a597c1d9) — it
/// never reads a datadir bitcoin.conf. A leftover file from an older install
/// (or a manual edit) makes Bitcoin Core's own datadir-conflict safety check
/// refuse to start ("-conf=... vs default bitcoin.conf"). Remove it — via
/// the same host-privileged path the old writer/repairer used, since the
/// dir may already be chowned into the container's UID namespace by a
/// previous start — instead of "repairing" it into existence.
async fn remove_stale_bitcoin_conf() -> Result<bool> {
let script = r#"
set -eu
conf=/var/lib/archipelago/bitcoin/bitcoin.conf
[ -f "$conf" ] || exit 0
changed=0
tmp=$(mktemp)
awk -F= '
/^(server|txindex|rpcbind|rpcallowip|rpcport|listen|bind|dbcache|rpcthreads|rpcworkqueue)=/ {
if (seen[$1]++) next
}
{ print }
' "$conf" > "$tmp"
if ! cmp -s "$conf" "$tmp"; then
cat "$tmp" > "$conf"
changed=1
fi
rm -f "$tmp"
ensure_line() {
line="$1"
key="${line%%=*}"
if ! grep -q "^${key}=" "$conf"; then
printf '%s\n' "$line" >> "$conf"
changed=1
fi
}
ensure_line server=1
ensure_line rpcbind=0.0.0.0
ensure_line rpcallowip=0.0.0.0/0
ensure_line listen=1
ensure_line rpcthreads=16
ensure_line rpcworkqueue=256
[ "$changed" -eq 0 ] && exit 0
mv "$conf" "$conf.disabled-$(date +%s)"
exit 2
"#;
let status = host_sudo(&["sh", "-lc", script])
.await
.context("ensure bitcoin.conf RPC bind settings")?;
.context("remove stale bitcoin.conf")?;
match status.code() {
Some(0) => Ok(false),
Some(2) => {
install_log("INSTALL REPAIR: bitcoin.conf RPC bind settings added").await;
install_log(
"INSTALL REPAIR: removed stale bitcoin.conf (conflicts with -conf=/tmp/rpc.conf launch)",
)
.await;
Ok(true)
}
_ => Err(anyhow::anyhow!(
"bitcoin.conf RPC repair helper exited with {}",
"bitcoin.conf removal helper exited with {}",
status
)),
}
@@ -1011,6 +1011,59 @@ impl RpcHandler {
}
}
/// auth.session-policy.get — how long a login lasts on this node.
pub(in crate::api::rpc) async fn handle_session_policy_get(&self) -> Result<serde_json::Value> {
let policy = crate::settings::session_policy::load(&self.config.data_dir).await;
Ok(serde_json::json!({
"idle_timeout_secs": policy.idle_timeout_secs,
"absolute_timeout_secs": policy.absolute_timeout_secs,
"reauth_for_funds": policy.reauth_for_funds,
}))
}
/// auth.session-policy.set — change it.
///
/// Values are clamped rather than rejected: the caller learns what was
/// actually stored from the reply, which is friendlier than an error and
/// makes the bounds discoverable. Fields are individually optional so the
/// UI can change one control without having to send the others back.
pub(in crate::api::rpc) async fn handle_session_policy_set(
&self,
params: Option<serde_json::Value>,
) -> Result<serde_json::Value> {
let params = params.unwrap_or(serde_json::json!({}));
let current = crate::settings::session_policy::load(&self.config.data_dir).await;
let policy = crate::settings::session_policy::SessionPolicy {
idle_timeout_secs: params
.get("idle_timeout_secs")
.and_then(|v| v.as_u64())
.unwrap_or(current.idle_timeout_secs),
absolute_timeout_secs: match params.get("absolute_timeout_secs") {
// Explicit null means "no absolute cap", which is different
// from the field being absent (leave it as it is).
Some(serde_json::Value::Null) => None,
Some(v) => v.as_u64().or(current.absolute_timeout_secs),
None => current.absolute_timeout_secs,
},
reauth_for_funds: params
.get("reauth_for_funds")
.and_then(|v| v.as_bool())
.unwrap_or(current.reauth_for_funds),
};
let saved = crate::settings::session_policy::save(&self.config.data_dir, policy).await?;
tracing::info!(
idle = saved.idle_timeout_secs,
absolute = ?saved.absolute_timeout_secs,
reauth_for_funds = saved.reauth_for_funds,
"session policy updated"
);
Ok(serde_json::json!({
"idle_timeout_secs": saved.idle_timeout_secs,
"absolute_timeout_secs": saved.absolute_timeout_secs,
"reauth_for_funds": saved.reauth_for_funds,
}))
}
/// system.settings.set — Write a settings value
pub(in crate::api::rpc) async fn handle_system_settings_set(
&self,
+59 -5
View File
@@ -222,6 +222,19 @@ pub(in crate::api::rpc) async fn regenerate_torrc(config: &ServicesConfig) -> Re
lines.push("# ControlPort disabled for security".to_string());
lines.push(String::new());
// Ports whose manifests declare `auth: gated` forward to the gate's own
// loopback (127.0.0.2, where the app-gate listener binds — see
// `appgate::listener::GATE_TOR_UPSTREAM`) instead of the app's 127.0.0.1.
// Tor carries no session cookie, so an onion pointed at the app is an
// unauthenticated bypass of the gate. Declared-gated ports only: an
// undeclared port keeps today's target, because absence of the field is
// not an instruction (the v1.7.121 incident rule).
let gated_ports: std::collections::HashSet<u16> = crate::appgate::identity::build_port_map()
.gated_ports()
.filter(|g| g.declared)
.map(|g| g.port)
.collect();
for svc in &config.services {
if !svc.enabled {
continue;
@@ -240,7 +253,7 @@ pub(in crate::api::rpc) async fn regenerate_torrc(config: &ServicesConfig) -> Re
lines.push("HiddenServicePort 10009 127.0.0.1:10009".to_string());
}
} else {
lines.push(format!("HiddenServicePort 80 127.0.0.1:{}", svc.local_port));
lines.push(app_hidden_service_port_line(svc.local_port, &gated_ports));
}
lines.push(String::new());
@@ -248,6 +261,24 @@ pub(in crate::api::rpc) async fn regenerate_torrc(config: &ServicesConfig) -> Re
let content = lines.join("\n");
let staging = "/var/lib/archipelago/tor-config/torrc.staged";
write_staged_torrc(&content, staging).await
}
/// The `HiddenServicePort` line for an HTTP app onion. Gated ports forward to
/// the gate's Tor upstream; everything else to the app itself.
fn app_hidden_service_port_line(
local_port: u16,
gated_ports: &std::collections::HashSet<u16>,
) -> String {
let upstream = if gated_ports.contains(&local_port) {
crate::appgate::listener::GATE_TOR_UPSTREAM.to_string()
} else {
"127.0.0.1".to_string()
};
format!("HiddenServicePort 80 {}:{}", upstream, local_port)
}
async fn write_staged_torrc(content: &str, staging: &str) -> Result<()> {
let config_dir = Path::new(staging)
.parent()
.unwrap_or_else(|| Path::new("/var/lib/archipelago/tor-config"));
@@ -256,14 +287,37 @@ pub(in crate::api::rpc) async fn regenerate_torrc(config: &ServicesConfig) -> Re
.await
.context("Failed to write staged torrc")?;
debug!(
"Staged torrc with {} enabled services",
config.services.iter().filter(|s| s.enabled).count()
);
debug!("Staged torrc ({} bytes)", content.len());
Ok(())
}
#[cfg(test)]
mod torrc_tests {
use super::app_hidden_service_port_line;
use std::collections::HashSet;
#[test]
fn gated_port_forwards_to_the_gate_not_the_app() {
let gated: HashSet<u16> = [8082u16].into_iter().collect();
assert_eq!(
app_hidden_service_port_line(8082, &gated),
"HiddenServicePort 80 127.0.0.2:8082"
);
}
#[test]
fn undeclared_port_keeps_the_app_loopback_target() {
// Absence of `auth: gated` is not an instruction — the onion keeps
// pointing at the app, exactly as before this change.
let gated: HashSet<u16> = [8082u16].into_iter().collect();
assert_eq!(
app_hidden_service_port_line(9100, &gated),
"HiddenServicePort 80 127.0.0.1:9100"
);
}
}
// ─── Hostname Sync ───────────────────────────────────────────────
pub(in crate::api::rpc) async fn sync_single_hostname(name: &str, address: &str) {
+2 -1
View File
@@ -75,7 +75,8 @@ pub fn address_caching_dependents(package_id: &str) -> &'static [&'static str] {
/// The package whose lifecycle lock covers `app_id`: the stack package when
/// `app_id` is a member (RPC ops on "mempool" hold the "mempool" lock while
/// they drive archy-mempool-web), otherwise the app itself.
fn owning_package(app_id: &str) -> &str {
/// Also consulted by the reconciler's absent-stack-member recovery.
pub fn owning_package(app_id: &str) -> &str {
const STACKS: &[&str] = &[
"immich",
"indeedhub",
+254 -100
View File
@@ -26,6 +26,20 @@ pub struct GatedPort {
pub app_name: String,
/// Manifest-declared icon path (`metadata.icon`), when present.
pub icon: Option<String>,
/// True only when the manifest says `auth: gated` in so many words.
///
/// The gated set deliberately also carries undeclared Session-default
/// ports (so the gate challenges them wherever it can already stand, and
/// the audit reports them). But everything that CHANGES where traffic
/// goes — the torrc repoint to 127.0.0.2, the FIPS relay stand-down, the
/// Tor-upstream bind — must key on this flag: acting on an undeclared
/// port is the v1.7.121 incident class, whatever the action.
pub declared: bool,
/// Manifest opt-in (`session_passthrough: true` on the port): forward the
/// node session cookie to the app on authorised requests. First-party
/// companion UIs proxy that cookie to the daemon's authenticated
/// endpoints; for every other app the gate strips its own credential.
pub session_passthrough: bool,
}
/// A port deliberately left unauthenticated, and the manifest's stated reason.
@@ -48,6 +62,7 @@ pub struct ExemptPort {
pub struct PortMap {
gated: HashMap<u16, GatedPort>,
exempt: Vec<ExemptPort>,
local: std::collections::HashSet<u16>,
}
impl PortMap {
@@ -64,8 +79,21 @@ impl PortMap {
&self.exempt
}
/// Declared `auth: local` — host-local by intent, so NOTHING may make it
/// externally reachable.
///
/// The gate honours this by keeping its hands off, but it is not the only
/// thing that can publish a port: the FIPS mesh relay bridges the fips0
/// ULA to `127.0.0.1` for a static port list, and it forwarded nbxplorer
/// 32838 — declared `local` and pinned to loopback — to the mesh
/// unauthenticated (archi-dev-box 2026-08-04). Anything that republishes
/// a loopback port must consult this set first.
pub fn is_declared_local(&self, port: u16) -> bool {
self.local.contains(&port)
}
pub fn is_empty(&self) -> bool {
self.gated.is_empty() && self.exempt.is_empty()
self.gated.is_empty() && self.exempt.is_empty() && self.local.is_empty()
}
}
@@ -101,13 +129,42 @@ fn manifest_icon(manifest: &AppManifest) -> Option<String> {
/// Classify every published port across all installed manifests.
///
/// The first directory that yields a manifest for an app id wins, so a node's
/// `/opt/archipelago/apps` copy shadows a repo checkout rather than merging
/// with it — otherwise a stale checked-out manifest could re-open a port the
/// installed one gates.
/// The signed catalog's embedded manifests are consulted FIRST, because they
/// are what the orchestrator actually publishes containers from
/// (origin-wins; see `app_catalog::catalog_manifest_overlay`). Classifying
/// from disk alone made the gate act on policy the node was no longer
/// running: the catalog declared nbxplorer `auth: local` and pinned it to
/// loopback, the stale disk manifest declared nothing, and the gate
/// externally bound a deliberately host-local port (archi-dev-box
/// 2026-08-04).
///
/// After the catalog, the first directory that yields a manifest for an app
/// id wins, so a node's `/opt/archipelago/apps` copy shadows a repo checkout
/// rather than merging with it — otherwise a stale checked-out manifest could
/// re-open a port the installed one gates.
pub fn build_port_map() -> PortMap {
let mut map = PortMap::default();
let mut seen_apps: HashMap<String, PathBuf> = HashMap::new();
let mut seen_apps: std::collections::HashSet<String> = std::collections::HashSet::new();
for (app_id, value) in crate::container::app_catalog::catalog_manifest_values() {
// Ports-only overlay: unlike the install path, classification also
// accepts BUILD-SOURCE manifests. The on-node-built companion UIs
// are exactly the apps whose gate policy (session_passthrough,
// auth: gated) must arrive reliably, and their disk manifests
// proved stale or absent fleet-wide in the v1.7.125 rollout. The
// gate's binds fail safely on conflict with a differently-published
// container, so a fresher catalog can only tighten, never expose.
let Some(manifest) =
crate::container::app_catalog::catalog_manifest_ports_overlay(&app_id, value)
else {
// Unparseable/invalid → the orchestrator falls back to disk for
// this app, so classification must too.
continue;
};
if seen_apps.insert(app_id) {
classify_manifest(&manifest, &mut map);
}
}
for dir in apps_dirs() {
let Ok(entries) = std::fs::read_dir(&dir) else {
@@ -124,100 +181,8 @@ pub fn build_port_map() -> PortMap {
// would have published.
continue;
};
let app_id = manifest.app.id.clone();
if seen_apps.contains_key(&app_id) {
continue;
}
seen_apps.insert(app_id.clone(), path);
let icon = manifest_icon(&manifest);
let app_name = if manifest.app.name.trim().is_empty() {
app_id.clone()
} else {
manifest.app.name.clone()
};
for port in &manifest.app.ports {
let protocol = if port.protocol.is_empty() {
"tcp"
} else {
port.protocol.as_str()
};
match port.auth_policy() {
PortAuth::None => map.exempt.push(ExemptPort {
port: port.host,
app_id: app_id.clone(),
rationale: port
.auth_rationale
.clone()
.unwrap_or_else(|| "(no rationale recorded)".to_string()),
protocol: protocol.to_string(),
}),
// Declared host-local. Not gated and not reported as
// exposed, because it is neither — see PortAuth::Local
// for why this cannot be inferred from `bind`.
PortAuth::Local => {}
// Explicit opt-in: the app is on loopback and the daemon
// owns the external addresses. This is the ONLY way a
// port gets bound by the gate, regardless of `bind`.
PortAuth::Gated => {
map.gated.insert(
port.host,
GatedPort {
port: port.host,
app_id: app_id.clone(),
app_name: app_name.clone(),
icon: icon.clone(),
},
);
}
PortAuth::Session => {
// UDP cannot carry an HTTP challenge. Such a port has
// no business defaulting into the gated set where it
// would look protected without being protectable —
// surface it as an unrationalised exemption instead,
// which is honest and shows up in the audit list.
if protocol != "tcp" {
map.exempt.push(ExemptPort {
port: port.host,
app_id: app_id.clone(),
rationale: format!(
"{protocol} cannot carry an HTTP challenge; declare auth: none \
with a rationale to record why this is safe"
),
protocol: protocol.to_string(),
});
continue;
}
// A loopback publish is skipped, and this is the
// safety property of the whole module: the gate must
// never be the reason a port becomes reachable
// somewhere it was not. `session` is the DEFAULT, so
// it is what every un-migrated manifest carries —
// and a node's installed manifests always lag the
// repo. Binding those externally published Bitcoin
// RPC across the LAN within seconds of deploy
// (archi-dev-box 2026-08-03). Taking over a port is
// opt-in only: `auth: gated`, shipped in the same
// manifest edit as the loopback pin.
if port
.bind
.parse::<std::net::IpAddr>()
.is_ok_and(|ip| ip.is_loopback())
{
continue;
}
map.gated.insert(
port.host,
GatedPort {
port: port.host,
app_id: app_id.clone(),
app_name: app_name.clone(),
icon: icon.clone(),
},
);
}
}
if seen_apps.insert(manifest.app.id.clone()) {
classify_manifest(&manifest, &mut map);
}
}
}
@@ -226,6 +191,111 @@ pub fn build_port_map() -> PortMap {
map
}
/// Classify one manifest's ports into the map. Split from [`build_port_map`]
/// so the catalog-overlay pass and the disk pass cannot diverge.
fn classify_manifest(manifest: &AppManifest, map: &mut PortMap) {
let app_id = manifest.app.id.clone();
let icon = manifest_icon(manifest);
let app_name = if manifest.app.name.trim().is_empty() {
app_id.clone()
} else {
manifest.app.name.clone()
};
for port in &manifest.app.ports {
let protocol = if port.protocol.is_empty() {
"tcp"
} else {
port.protocol.as_str()
};
match port.auth_policy() {
PortAuth::None => map.exempt.push(ExemptPort {
port: port.host,
app_id: app_id.clone(),
rationale: port
.auth_rationale
.clone()
.unwrap_or_else(|| "(no rationale recorded)".to_string()),
protocol: protocol.to_string(),
}),
// Declared host-local. Not gated and not reported as
// exposed, because it is neither — see PortAuth::Local
// for why this cannot be inferred from `bind`. Recorded so
// the mesh relay (and any future republisher) can refuse to
// expose it.
PortAuth::Local => {
map.local.insert(port.host);
}
// Explicit opt-in: the app is on loopback and the daemon
// owns the external addresses. This is the ONLY way a
// port gets bound by the gate, regardless of `bind`.
PortAuth::Gated => {
map.gated.insert(
port.host,
GatedPort {
port: port.host,
app_id: app_id.clone(),
app_name: app_name.clone(),
icon: icon.clone(),
declared: true,
session_passthrough: port.session_passthrough,
},
);
}
PortAuth::Session => {
// UDP cannot carry an HTTP challenge. Such a port has
// no business defaulting into the gated set where it
// would look protected without being protectable —
// surface it as an unrationalised exemption instead,
// which is honest and shows up in the audit list.
if protocol != "tcp" {
map.exempt.push(ExemptPort {
port: port.host,
app_id: app_id.clone(),
rationale: format!(
"{protocol} cannot carry an HTTP challenge; declare auth: none \
with a rationale to record why this is safe"
),
protocol: protocol.to_string(),
});
continue;
}
// A loopback publish is skipped, and this is the
// safety property of the whole module: the gate must
// never be the reason a port becomes reachable
// somewhere it was not. `session` is the DEFAULT, so
// it is what every un-migrated manifest carries —
// and a node's installed manifests always lag the
// repo. Binding those externally published Bitcoin
// RPC across the LAN within seconds of deploy
// (archi-dev-box 2026-08-03). Taking over a port is
// opt-in only: `auth: gated`, shipped in the same
// manifest edit as the loopback pin.
if port
.bind
.parse::<std::net::IpAddr>()
.is_ok_and(|ip| ip.is_loopback())
{
continue;
}
map.gated.insert(
port.host,
GatedPort {
port: port.host,
app_id: app_id.clone(),
app_name: app_name.clone(),
icon: icon.clone(),
declared: false,
// An undeclared port never gets the node session —
// passthrough is an explicit manifest opt-in only.
session_passthrough: false,
},
);
}
}
}
}
#[cfg(test)]
mod tests {
use super::*;
@@ -257,6 +327,90 @@ mod tests {
}
}
fn manifest(yaml: &str) -> AppManifest {
AppManifest::parse(yaml).expect("test manifest must parse")
}
const BASE: &str = r#"
app:
id: testapp
name: Test App
version: "1.0"
container:
image: example.org/testapp:1.0
"#;
/// `auth: gated` is the only classification allowed to redirect traffic —
/// torrc repoints, relay stand-down, and the 127.0.0.2 bind all key on
/// `declared`. An undeclared Session port is challenged and audited but
/// must never be `declared`.
#[test]
fn declared_tracks_the_manifest_not_the_default() {
let mut map = PortMap::default();
classify_manifest(
&manifest(&format!(
"{BASE} ports:\n - host: 8090\n container: 7777\n protocol: tcp\n bind: 127.0.0.1\n auth: gated\n"
)),
&mut map,
);
assert!(map.gated(8090).expect("gated").declared);
let mut map = PortMap::default();
classify_manifest(
&manifest(&format!(
"{BASE} ports:\n - host: 9100\n container: 9100\n protocol: tcp\n"
)),
&mut map,
);
let undeclared = map.gated(9100).expect("session default is challenged");
assert!(
!undeclared.declared,
"an absent auth field must never read as an instruction"
);
}
/// `auth: local` keeps the gate's hands off entirely — the port is
/// neither gated nor exempt-reported — but it IS recorded, so the mesh
/// relay can refuse to republish a deliberately host-local port.
#[test]
fn local_ports_are_untouched_but_recorded() {
let mut map = PortMap::default();
classify_manifest(
&manifest(&format!(
"{BASE} ports:\n - host: 32838\n container: 32838\n protocol: tcp\n bind: 127.0.0.1\n auth: local\n"
)),
&mut map,
);
assert!(map.gated(32838).is_none());
assert!(map.exempt_ports().is_empty());
assert!(
map.is_declared_local(32838),
"the mesh relay needs this to refuse bridging a host-local port"
);
assert!(!map.is_declared_local(3000));
}
/// The real corpus: every port the FIPS relay can bridge must be safe to
/// bridge. A port that is declared `local` (host-local by intent) or
/// declared `gated` (the app gate owns its external addresses) must be
/// withheld by the relay — this asserts the two sets the relay consults
/// actually classify the live manifests, so a future manifest edit that
/// re-opens one is caught here rather than on a node.
#[test]
fn relay_port_list_respects_local_and_gated_declarations() {
let map = build_port_map();
let relay_would_expose: Vec<u16> = crate::fips::app_ports::APP_LAUNCH_PORTS
.iter()
.copied()
.filter(|p| map.is_declared_local(*p))
.collect();
assert!(
!relay_would_expose.is_empty(),
"expected the corpus to contain at least one local port in the relay list \
(32838/8999) if this fails the guard is untested, not unnecessary"
);
}
/// Protocol ports that wallets dial directly must never end up gated —
/// this is the constraint that decided the design (Zeus and electrum
/// clients keep working untouched).
+73 -7
View File
@@ -44,6 +44,15 @@ use tracing::{debug, info, warn};
/// apps are installed while the daemon runs.
const SWEEP_INTERVAL: std::time::Duration = std::time::Duration::from_secs(60);
/// The gate's own loopback address, distinct from the app's `127.0.0.1`.
///
/// Tor cannot present a session cookie, so `HiddenServicePort → 127.0.0.1`
/// reaches the app around the gate. Instead torrc forwards gated ports to
/// this address (`api/rpc/tor`), where the gate — not the app — listens. A
/// second loopback address rather than a second port number, so no app needs
/// a port it did not declare.
pub const GATE_TOR_UPSTREAM: IpAddr = IpAddr::V4(std::net::Ipv4Addr::new(127, 0, 0, 2));
/// A port the gate should own but could not claim, and why.
#[derive(Debug, Clone, serde::Serialize)]
pub struct UnprotectedPort {
@@ -142,8 +151,12 @@ pub async fn run(
mut shutdown_rx: tokio::sync::watch::Receiver<bool>,
) {
// (port, addr) pairs already served, so a sweep does not rebind what it
// already holds.
let mut held: HashMap<(u16, IpAddr), ()> = HashMap::new();
// already holds. The accept-loop handle is kept so a claim can be
// RELEASED when its port leaves the gated set — a catalog refresh
// declaring a port `local`/`none` must make the gate let go without a
// daemon restart, or the stale bind keeps republishing a port the
// catalog just withdrew (nbxplorer 32838, archi-dev-box 2026-08-04).
let mut held: HashMap<(u16, IpAddr), tokio::task::JoinHandle<()>> = HashMap::new();
let mut interval = tokio::time::interval(SWEEP_INTERVAL);
interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
@@ -160,7 +173,7 @@ pub async fn run(
async fn sweep(
gate: &Arc<AppGate>,
status: &Arc<RwLock<GateStatus>>,
held: &mut HashMap<(u16, IpAddr), ()>,
held: &mut HashMap<(u16, IpAddr), tokio::task::JoinHandle<()>>,
shutdown_rx: &tokio::sync::watch::Receiver<bool>,
) {
// Re-read the manifests every sweep rather than trusting the map built
@@ -170,6 +183,23 @@ async fn sweep(
// enforced while serving a brand-new app to anyone who asked.
gate.refresh().await;
let port_map = gate.port_map().await;
// Release claims whose port left the gated set (or whose Tor-upstream
// claim lost its declaration). Aborting the accept loop drops the
// listener, freeing the address for whoever now legitimately owns it —
// the app itself, or nobody.
held.retain(|(port, addr), handle| {
let keep = match port_map.gated(*port) {
None => false,
Some(app) => *addr != GATE_TOR_UPSTREAM || app.declared,
};
if !keep {
handle.abort();
info!(port, %addr, "app gate released a claim: port is no longer gated here");
}
keep
});
let addresses = host_addresses().await;
if addresses.is_empty() {
debug!("app gate: no external addresses yet");
@@ -192,6 +222,10 @@ async fn sweep(
let mut claimed_any = false;
let mut blocked = false;
// External addresses first, then the gate's Tor upstream. 127.0.0.2
// deliberately does NOT count toward `claimed_any`: the warning below
// is about external exposure, and a port whose only claim is the Tor
// loopback is still wide open on the LAN.
for &addr in &addresses {
let key = (app.port, addr);
if held.contains_key(&key) {
@@ -201,19 +235,48 @@ async fn sweep(
}
match TcpListener::bind(SocketAddr::new(addr, app.port)).await {
Ok(listener) => {
held.insert(key, ());
let handle =
spawn_accept_loop(listener, gate.clone(), app.clone(), shutdown_rx.clone());
held.insert(key, handle);
claimed.push((app.port, addr.to_string()));
claimed_any = true;
info!(
port = app.port, %addr, app = %app.app_id,
"app gate claimed an app port"
);
spawn_accept_loop(listener, gate.clone(), app.clone(), shutdown_rx.clone());
}
// Almost always the app itself holding 0.0.0.0:<port>.
Err(_) => blocked = true,
}
}
// The Tor upstream is bound for DECLARED gated ports only: torrc only
// repoints an onion at 127.0.0.2 for a declared port, and standing a
// challenge on an undeclared port's would-be upstream would change
// where its traffic goes on nothing but a default.
if app.declared {
let tor_key = (app.port, GATE_TOR_UPSTREAM);
if held.contains_key(&tor_key) {
claimed.push((app.port, GATE_TOR_UPSTREAM.to_string()));
} else {
match TcpListener::bind(SocketAddr::new(GATE_TOR_UPSTREAM, app.port)).await {
Ok(listener) => {
let handle = spawn_accept_loop(
listener,
gate.clone(),
app.clone(),
shutdown_rx.clone(),
);
held.insert(tor_key, handle);
claimed.push((app.port, GATE_TOR_UPSTREAM.to_string()));
info!(
port = app.port, app = %app.app_id,
"app gate claimed the Tor upstream (127.0.0.2)"
);
}
Err(_) => blocked = true,
}
}
}
if blocked && !claimed_any {
warn!(
@@ -251,12 +314,15 @@ async fn app_is_listening(port: u16) -> bool {
.is_some()
}
/// Returns the accept-loop task handle so the sweep can release the claim
/// (abort → listener drops → address freed) when the port leaves the gated
/// set. In-flight connections finish on their own tasks.
fn spawn_accept_loop(
listener: TcpListener,
gate: Arc<AppGate>,
app: GatedPort,
mut shutdown_rx: tokio::sync::watch::Receiver<bool>,
) {
) -> tokio::task::JoinHandle<()> {
tokio::spawn(async move {
loop {
tokio::select! {
@@ -287,7 +353,7 @@ fn spawn_accept_loop(
_ = shutdown_rx.changed() => break,
}
}
});
})
}
#[cfg(test)]
+407 -55
View File
@@ -136,7 +136,7 @@ impl AppGate {
}
match self.authorize(req.headers(), &app.app_id).await {
Authorization::Allow => proxy_to_app(req, app.port).await,
Authorization::Allow => proxy_to_app(req, app).await,
// 401 rather than a redirect: a redirect to a login page is
// indistinguishable from the app itself redirecting, and machine
// clients would follow it and parse HTML as if it were their API
@@ -154,6 +154,11 @@ impl AppGate {
action: &str,
client_ip: IpAddr,
) -> Response<Body> {
// Assets are GET and pre-auth by nature: the login page cannot
// render its own background or logo without them.
if let Some(name) = action.strip_prefix("asset/") {
return self.serve_asset(name);
}
if req.method() != Method::POST {
return login_page(app, None, StatusCode::OK);
}
@@ -187,6 +192,26 @@ impl AppGate {
}
}
/// Static assets the login page needs, served from the gate's own origin.
///
/// The backgrounds are ~1 MB each, so inlining them as data URIs would
/// bloat every challenge response. Serving them here keeps the page
/// byte-identical to the dashboard's login while the CSP stays tight:
/// `img-src 'self' data:` and nothing else.
fn serve_asset(&self, name: &str) -> Response<Body> {
let Some((bytes, mime)) = read_ui_asset(name) else {
return not_found();
};
Response::builder()
.status(StatusCode::OK)
.header(header::CONTENT_TYPE, mime)
// Immutable art; caching it costs nothing and keeps the login
// instant on a repeat challenge.
.header(header::CACHE_CONTROL, "public, max-age=86400")
.body(Body::from(bytes))
.expect("asset response builds")
}
async fn do_login(&self, app: &GatedPort, form: &Form, client_ip: IpAddr) -> Response<Body> {
let password = field(form, "password").unwrap_or_default();
@@ -350,7 +375,8 @@ fn percent_decode(input: &str) -> String {
}
/// Forward an authorised request to the app on loopback.
async fn proxy_to_app(req: Request<Body>, port: u16) -> Response<Body> {
async fn proxy_to_app(req: Request<Body>, app: &GatedPort) -> Response<Body> {
let port = app.port;
let path_and_query = req
.uri()
.path_and_query()
@@ -364,10 +390,16 @@ async fn proxy_to_app(req: Request<Body>, port: u16) -> Response<Body> {
let (mut parts, body) = req.into_parts();
parts.uri = uri;
// Strip the gate's own credential before it reaches the app: the app has
// no use for the node session and should never be in a position to log,
// echo, or forward it.
parts.headers.remove(header::COOKIE);
// Strip the gate's own credential before it reaches the app the app
// should never be in a position to log, echo, or forward the node
// session. But ONLY the gate's cookies: apps run their own cookie logins
// (vaultwarden, nextcloud, gitea…), and removing the whole header logged
// every one of them out on each request. Companion UIs that proxy the
// daemon's authenticated endpoints opt in to keeping the session via
// `session_passthrough: true` on their gated port.
if !app.session_passthrough {
strip_gate_cookies(&mut parts.headers);
}
parts.headers.remove(header::AUTHORIZATION);
let client = hyper::Client::new();
@@ -377,6 +409,44 @@ async fn proxy_to_app(req: Request<Body>, port: u16) -> Response<Body> {
}
}
/// Cookie names owned by the gate/daemon, never the app's to see.
const GATE_COOKIE_NAMES: &[&str] = &["session", "csrf_token"];
/// Remove the gate's own cookie pairs from the Cookie header, preserving the
/// app's cookies (its login/session/prefs) untouched. Drops the header
/// entirely when nothing remains.
fn strip_gate_cookies(headers: &mut hyper::HeaderMap) {
let Some(cookie) = headers.get(header::COOKIE) else {
return;
};
let Ok(raw) = cookie.to_str() else {
// Not valid UTF-8 — can't safely filter pairs, so fail closed.
headers.remove(header::COOKIE);
return;
};
let kept: Vec<&str> = raw
.split(';')
.map(str::trim)
.filter(|pair| {
let name = pair.split('=').next().unwrap_or("").trim();
!GATE_COOKIE_NAMES.contains(&name)
})
.filter(|pair| !pair.is_empty())
.collect();
if kept.is_empty() {
headers.remove(header::COOKIE);
return;
}
match header::HeaderValue::from_str(&kept.join("; ")) {
Ok(v) => {
headers.insert(header::COOKIE, v);
}
Err(_) => {
headers.remove(header::COOKIE);
}
}
}
fn set_session_cookie(resp: &mut Response<Body>, token: &str) {
// No Domain attribute, so the cookie is host-only. Cookies ignore port,
// which is what makes one sign-in cover the dashboard and every app port
@@ -428,43 +498,161 @@ fn esc(s: &str) -> String {
/// none. Inlined as a data URI rather than linked: the gate is answering on
/// the app's own port, so any asset URL would either hit the unauthenticated
/// app behind it or a different origin the browser may not reach.
/// One stacked layer per background, each delayed so they cross-fade in turn.
fn background_layers() -> String {
let step = LOGIN_BACKGROUNDS.len() as u32 * 9 / LOGIN_BACKGROUNDS.len() as u32;
LOGIN_BACKGROUNDS
.iter()
.enumerate()
.map(|(i, name)| {
format!(
r#"<div class="bg" style="background-image:url('{prefix}asset/{name}');animation-delay:{delay}s"></div>"#,
prefix = GATE_PREFIX,
delay = i as u32 * step,
)
})
.collect()
}
fn icon_markup(app: &GatedPort) -> String {
if let Some(path) = &app.icon {
if let Some(data_uri) = read_icon_data_uri(path) {
return format!(r#"<img class="icon" src="{}" alt="">"#, esc(&data_uri));
}
}
let letter = app
.app_name
.chars()
.next()
.map(|c| c.to_uppercase().to_string())
.unwrap_or_else(|| "?".to_string());
format!(r#"<div class="icon lettermark">{}</div>"#, esc(&letter))
let inner = app
.icon
.as_deref()
.and_then(read_icon_data_uri)
// A manifest that names no icon still gets one: the dashboard already
// ships icons named after the app, so fall back to those before
// giving up. Without this EVERY gated app showed a lettermark,
// because no manifest declares metadata.icon (archi-dev-box,
// 2026-08-05).
.or_else(|| {
icon_candidates(&app.app_id)
.iter()
.find_map(|c| read_icon_data_uri(c))
})
.map(|data_uri| format!(r#"<img class="icon" src="{}" alt="">"#, esc(&data_uri)))
.unwrap_or_else(|| {
let letter = app
.app_name
.chars()
.find(|c| c.is_alphanumeric())
.map(|c| c.to_uppercase().to_string())
.unwrap_or_else(|| "?".to_string());
format!(r#"<div class="icon lettermark">{}</div>"#, esc(&letter))
});
format!(r#"<div class="tile">{inner}</div>"#)
}
/// Icons live with the web UI. Only files under the icon directory are read,
/// and only known image extensions — the path comes from a manifest, which is
/// signed, but treating it as untrusted costs nothing.
fn read_icon_data_uri(icon_path: &str) -> Option<String> {
let name = std::path::Path::new(icon_path).file_name()?.to_str()?;
let mime = match name.rsplit_once('.')?.1.to_ascii_lowercase().as_str() {
"svg" => "image/svg+xml",
"png" => "image/png",
"webp" => "image/webp",
"jpg" | "jpeg" => "image/jpeg",
_ => return None,
/// Icon basenames to try for an app id, best first.
///
/// The shipped icon set is named for the *product*, while app ids carry
/// packaging detail — `filebrowser` vs `file-browser`, `morphos-server` vs
/// `morphos` — and the per-app screens (`lnd-ui`, `bitcoin-ui`, `electrs-ui`)
/// have no icon of their own but obviously belong to the app they front.
/// Resolving those here keeps the mapping in one readable place instead of
/// adding a `metadata.icon` line to every manifest, which would have to be
/// re-signed into the catalog to take effect.
fn icon_candidates(app_id: &str) -> Vec<String> {
let mut out = vec![app_id.to_string()];
let alias = match app_id {
"filebrowser" => Some("file-browser"),
"home-assistant" => Some("homeassistant"),
"morphos-server" => Some("morphos"),
"barkd" => Some("bark"),
"archy-mempool-web" | "mempool-api" => Some("mempool"),
"lnd-ui" | "lightning-stack" => Some("lnd"),
"bitcoin-ui" => Some("bitcoin-core"),
"electrs-ui" => Some("electrumx"),
"fips-ui" | "aiui" | "did-wallet" => Some("archipelago-a"),
"fedimint-gateway" | "fedimint-clientd" => Some("fedimint"),
_ => None,
};
out.extend(alias.map(str::to_string));
// `<app>-ui` / `-server` / `-web` front an app whose icon is the bare name.
for suffix in ["-ui", "-server", "-web"] {
if let Some(base) = app_id.strip_suffix(suffix) {
out.push(base.to_string());
}
}
out
}
/// Backgrounds the login cycles through, matching the dashboard's own
/// `/login` art. Cross-faded by CSS alone — the CSP forbids script, and a
/// rotation that needs JavaScript would not survive it.
const LOGIN_BACKGROUNDS: [&str; 4] = [
"bg-intro.jpg",
"bg-intro-4.webp",
"bg-intro-6.webp",
"bg-intro-3.jpg",
];
/// Assets the gate will serve, by exact name. An allowlist rather than a path
/// join: the name arrives in a URL, and the gate answers before any
/// authentication, so nothing here may be caller-controlled beyond this set.
fn read_ui_asset(name: &str) -> Option<(Vec<u8>, &'static str)> {
let allowed = LOGIN_BACKGROUNDS.contains(&name) || name == "favico-black-v2.svg";
if !allowed {
return None;
}
let mime = icon_mime(name.rsplit_once('.')?.1)?;
for root in [
"/opt/archipelago/web-ui/assets/img/app-icons",
"web/dist/neode-ui/assets/img/app-icons",
"/opt/archipelago/web-ui/assets/img",
"web/dist/neode-ui/assets/img",
"neode-ui/public/assets/img",
"/opt/archipelago/web-ui/assets/icon",
"web/dist/neode-ui/assets/icon",
"neode-ui/public/assets/icon",
] {
let candidate = std::path::Path::new(root).join(name);
if let Ok(bytes) = std::fs::read(&candidate) {
if bytes.len() > 512 * 1024 {
return None;
if let Ok(bytes) = std::fs::read(std::path::Path::new(root).join(name)) {
return Some((bytes, mime));
}
}
None
}
const ICON_ROOTS: [&str; 2] = [
"/opt/archipelago/web-ui/assets/img/app-icons",
"web/dist/neode-ui/assets/img/app-icons",
];
fn icon_mime(ext: &str) -> Option<&'static str> {
match ext.to_ascii_lowercase().as_str() {
"svg" => Some("image/svg+xml"),
"png" => Some("image/png"),
"webp" => Some("image/webp"),
"jpg" | "jpeg" => Some("image/jpeg"),
_ => None,
}
}
/// Read an app icon as a `data:` URI.
///
/// `icon_ref` may be a filename or path with an extension (a manifest's
/// `metadata.icon`), or a bare name such as an app id — in which case the
/// known extensions are tried in turn. Only the file name is used; the
/// directories searched are fixed, so a manifest cannot point the gate at an
/// arbitrary path.
fn read_icon_data_uri(icon_ref: &str) -> Option<String> {
let name = std::path::Path::new(icon_ref).file_name()?.to_str()?;
let candidates: Vec<(String, &str)> = match name.rsplit_once('.') {
Some((_, ext)) => vec![(name.to_string(), icon_mime(ext)?)],
None => ["svg", "png", "webp", "jpg"]
.iter()
.filter_map(|ext| Some((format!("{name}.{ext}"), icon_mime(ext)?)))
.collect(),
};
for (file, mime) in candidates {
for root in ICON_ROOTS {
let candidate = std::path::Path::new(root).join(&file);
if let Ok(bytes) = std::fs::read(&candidate) {
if bytes.len() > 512 * 1024 {
continue;
}
return Some(format!("data:{mime};base64,{}", base64_encode(&bytes)));
}
return Some(format!("data:{mime};base64,{}", base64_encode(&bytes)));
}
}
None
@@ -484,29 +672,91 @@ fn page(title: &str, app: &GatedPort, body: &str, status: StatusCode) -> Respons
<meta name="robots" content="noindex">
<title>{title} {app_name}</title>
<style>
/* The dashboard's own /login, rebuilt in static CSS: the same rotating
intro art, .glass-card panel, .glass-button action and transparent
white-bordered inputs from neode-ui/src/style.css. Written longhand
rather than shared with the SPA because the gate answers before any
bundle exists, and the CSP forbids external stylesheets and script. */
:root {{ color-scheme: dark; }}
* {{ box-sizing: border-box; }}
body {{ margin:0; min-height:100vh; display:grid; place-items:center;
background:#0b0f14; color:#e6edf3; font:16px/1.5 system-ui,-apple-system,Segoe UI,sans-serif; }}
.card {{ width:min(92vw,380px); padding:2rem; background:#121820;
border:1px solid #223; border-radius:14px; text-align:center; }}
.icon {{ width:64px; height:64px; border-radius:14px; margin:0 auto 1rem; display:block; object-fit:cover; }}
.lettermark {{ display:grid; place-items:center; background:#1d2733; font-size:28px; font-weight:600; }}
h1 {{ font-size:1.15rem; margin:0 0 .25rem; }}
p.sub {{ margin:0 0 1.5rem; color:#8b98a5; font-size:.9rem; }}
input {{ width:100%; padding:.7rem .8rem; margin-bottom:.75rem; border-radius:9px;
border:1px solid #2b3947; background:#0d131a; color:#e6edf3; font-size:1rem; }}
input:focus {{ outline:2px solid #3b82f6; outline-offset:1px; }}
button {{ width:100%; padding:.7rem; border:0; border-radius:9px; background:#3b82f6;
color:#fff; font-size:1rem; font-weight:600; cursor:pointer; }}
button:hover {{ background:#2f6fd6; }}
.err {{ background:#3b1519; border:1px solid #7f1d1d; color:#fca5a5;
padding:.6rem .8rem; border-radius:9px; margin-bottom:1rem; font-size:.9rem; }}
html {{ height:100%; }}
body {{ margin:0; color:#fff; background:#05070a; overflow:hidden;
font:16px/1.5 system-ui,-apple-system,"Segoe UI",sans-serif;
/* Fixed to the viewport rather than a tall scrolling page: an on-screen
keyboard then overlays the card instead of scrolling it away, and the
card stays optically centred. min-height:100vh scrolled with the
keyboard on mobile and left the card off-centre (reported 2026-08-05). */
position:fixed; inset:0;
display:grid; place-items:center; padding:1rem;
height:100vh; height:100svh; }}
/* Very short viewports (landscape phone, or a keyboard eating most of it):
allow the card to scroll INSIDE the fixed frame rather than overflow. */
@media (max-height:640px) {{
body {{ align-items:start; overflow-y:auto; padding-top:3rem; }}
}}
/* Rotating backgrounds: each layer holds its image and cross-fades on a
shared cycle, so the art moves the way /login does with no script. */
.bg {{ position:fixed; inset:0; z-index:0; background-size:cover;
background-position:center; opacity:0; animation:bg-cycle {cycle}s infinite; }}
.bg::after {{ content:''; position:absolute; inset:0;
background:linear-gradient(180deg, rgba(0,0,0,.35), rgba(0,0,0,.72)); }}
@keyframes bg-cycle {{
0% {{ opacity:0; }} 4% {{ opacity:1; }}
{hold}% {{ opacity:1; }} {fade}% {{ opacity:0; }} 100% {{ opacity:0; }}
}}
main {{ position:relative; z-index:1; width:min(92vw,28rem); }}
.card {{ padding:2rem; padding-top:3.5rem; position:relative;
background:rgba(0,0,0,.65); backdrop-filter:blur(18px);
-webkit-backdrop-filter:blur(18px); border:1px solid rgba(255,255,255,.18);
border-radius:1rem; box-shadow:0 8px 24px rgba(0,0,0,.45); text-align:center; }}
/* The Archipelago mark, half in and half out of the panel — same placement
and gradient ring as Login.vue. */
.logo {{ position:absolute; top:-2.5rem; left:50%; transform:translateX(-50%);
width:5rem; height:5rem; border-radius:9999px; padding:3px;
background:linear-gradient(135deg, rgba(255,255,255,.6) 0%, rgba(0,0,0,.8) 100%);
box-shadow:0 8px 24px rgba(0,0,0,.5); }}
.logo img {{ width:100%; height:100%; border-radius:9999px; display:block;
background:#000; padding:.5rem; }}
/* The app's own tile, in the My Apps shape: 18px-rounded square on dark
glass with the same inner highlight and drop shadow. */
.tile {{ width:60px; height:60px; border-radius:18px; margin:0 auto .75rem;
background:rgba(0,0,0,.72); box-shadow:0 8px 18px rgba(0,0,0,.38); }}
.tile .icon {{ width:100%; height:100%; border-radius:18px; display:block;
object-fit:cover; border:1px solid rgba(255,255,255,.18);
background:radial-gradient(circle at 35% 28%, rgba(255,255,255,.1), rgba(255,255,255,0) 42%),
linear-gradient(145deg, rgba(22,22,24,.96), rgba(0,0,0,.96));
box-shadow:inset 0 1px 0 rgba(255,255,255,.12), inset 0 -10px 24px rgba(0,0,0,.34); }}
.lettermark {{ display:grid; place-items:center; font-size:1.6rem; font-weight:600;
color:rgba(255,255,255,.9); }}
h1 {{ font-size:1.5rem; font-weight:600; margin:0 0 .4rem;
color:rgba(255,255,255,.96); text-shadow:0 2px 6px rgba(0,0,0,.4); }}
p.sub {{ margin:0 0 1.75rem; color:rgba(255,255,255,.6); font-size:.875rem; }}
input {{ width:100%; padding:.75rem 1rem; margin-bottom:1rem; border-radius:.5rem;
border:1px solid rgba(255,255,255,.2); background:transparent; color:#fff;
font-size:1rem; transition:border-color .2s ease; }}
input::placeholder {{ color:rgba(255,255,255,.4); }}
input:focus {{ outline:none; border-color:rgba(255,255,255,.4);
box-shadow:0 0 0 1px rgba(255,255,255,.2); }}
button {{ width:100%; min-height:44px; padding:.75rem 1.25rem; border:none;
border-radius:.75rem; background:rgba(0,0,0,.6);
backdrop-filter:blur(24px); -webkit-backdrop-filter:blur(24px);
box-shadow:0 8px 24px rgba(0,0,0,.45), inset 0 1px 0 rgba(255,255,255,.22);
color:rgba(255,255,255,.9); font-size:1rem; font-weight:500; cursor:pointer;
transition:background-color .2s ease, transform .3s cubic-bezier(.4,0,.2,1); }}
button:hover {{ background:rgba(0,0,0,.7); }}
button:active {{ transform:translateY(1px); }}
.err {{ background:rgba(239,68,68,.2); border:1px solid rgba(239,68,68,.4);
color:#fecaca; padding:.75rem; border-radius:.5rem; margin-bottom:1rem;
font-size:.875rem; text-align:left; }}
</style></head>
<body><main class="card">{body}</main></body></html>"#,
<body>{backgrounds}<main><div class="card">{body}</div></main></body></html>"#,
title = esc(title),
app_name = esc(&app.app_name),
body = body,
backgrounds = background_layers(),
cycle = LOGIN_BACKGROUNDS.len() as u32 * 9,
hold = 100 / LOGIN_BACKGROUNDS.len() as u32,
fade = 100 / LOGIN_BACKGROUNDS.len() as u32 + 4,
);
Response::builder()
.status(status)
@@ -514,10 +764,18 @@ button:hover {{ background:#2f6fd6; }}
// The gate answers on the app's own port for an unauthenticated
// caller; nothing here should be cached or framed.
.header(header::CACHE_CONTROL, "no-store")
.header("X-Frame-Options", "DENY")
// NOT X-Frame-Options: DENY. My Apps opens an app in an embedded
// frame, so a blanket DENY made every gated app render as "app is
// not responding" the moment the gate challenged it (reported on
// 100.82.34.38, 2026-08-05). frame-ancestors is the modern control
// and can be precise: only pages from this same node may frame the
// login, on any port or scheme, which is exactly the dashboard.
// Anything else — another site embedding it to harvest the node
// password — is still refused.
.header(
"Content-Security-Policy",
"default-src 'none'; img-src data:; style-src 'unsafe-inline'; form-action 'self'",
"default-src 'none'; img-src 'self' data:; style-src 'unsafe-inline'; \
form-action 'self'; frame-ancestors 'self' http://*:* https://*:*",
)
.body(Body::from(html))
.expect("static response builds")
@@ -528,7 +786,8 @@ button:hover {{ background:#2f6fd6; }}
/// password by an unexplained page.
fn login_page(app: &GatedPort, error: Option<&str>, status: StatusCode) -> Response<Body> {
let body = format!(
r#"{icon}
r#"<div class="logo"><img src="{prefix}asset/favico-black-v2.svg" alt="Archipelago"></div>
{icon}
<h1>Sign in to open {name}</h1>
<p class="sub">This app is protected by your node password.</p>
{err}
@@ -578,6 +837,8 @@ mod tests {
app_id: "strfry".to_string(),
app_name: "Strfry Relay".to_string(),
icon: None,
declared: true,
session_passthrough: false,
}
}
@@ -635,11 +896,64 @@ mod tests {
assert!(!html.contains("<img src=x"));
}
/// The challenge must be framable by this node's own dashboard — My Apps
/// opens apps in an embedded frame, and a blanket `X-Frame-Options: DENY`
/// turned every gated app into "app is not responding" (100.82.34.38,
/// 2026-08-05). It must still be uncacheable, and still refuse to be
/// framed by a foreign origin, which `frame-ancestors` expresses and
/// `X-Frame-Options` cannot.
#[test]
fn challenge_pages_are_not_cacheable_or_framable() {
fn challenge_pages_are_uncacheable_and_framable_only_by_this_node() {
let resp = login_page(&app(), None, StatusCode::UNAUTHORIZED);
assert_eq!(resp.headers()[header::CACHE_CONTROL], "no-store");
assert_eq!(resp.headers()["X-Frame-Options"], "DENY");
assert!(
!resp.headers().contains_key("X-Frame-Options"),
"X-Frame-Options cannot express 'my own node on another port' — it \
blocked the dashboard's own frame"
);
let csp = resp.headers()["Content-Security-Policy"].to_str().unwrap();
assert!(csp.contains("frame-ancestors 'self'"));
assert!(csp.contains("form-action 'self'"));
}
/// The login page must render entirely from the gate's own origin: the
/// CSP allows no external host, so a background or logo that 404s leaves
/// a black page rather than the dashboard's art.
#[tokio::test]
async fn login_page_sources_its_art_from_the_gate() {
let resp = login_page(&app(), None, StatusCode::UNAUTHORIZED);
let body = hyper::body::to_bytes(resp.into_body()).await.unwrap();
let html = String::from_utf8_lossy(&body).to_string();
assert!(html.contains(&format!("{GATE_PREFIX}asset/favico-black-v2.svg")));
for name in LOGIN_BACKGROUNDS {
assert!(
html.contains(&format!("{GATE_PREFIX}asset/{name}")),
"background {name} is not referenced"
);
}
// Every referenced asset must be one the gate will actually serve.
// The logo is the sidebar A mark (favico-black-v2.svg) since the
// 2026-08-05 login-page rework — the old wordmark is off the
// allowlist on purpose.
assert!(read_ui_asset("favico-black-v2.svg").is_some() || cfg!(not(debug_assertions)));
}
/// The allowlist is the whole security boundary for asset serving: the
/// name arrives in a URL and is read before any authentication.
#[test]
fn asset_serving_refuses_anything_off_the_allowlist() {
for name in [
"../../../etc/passwd",
"/etc/passwd",
"db.sqlite3",
"manifest.yml",
"",
] {
assert!(
read_ui_asset(name).is_none(),
"{name} must not be servable by the gate"
);
}
}
#[tokio::test]
@@ -681,6 +995,44 @@ mod tests {
);
}
/// The gate must remove ONLY its own cookie pairs: an app's login cookie
/// riding the same header has to survive, or every gated app with its
/// own auth (vaultwarden, nextcloud, gitea) is logged out on each
/// request — the 2026-08-05 companion-UI/"app logged me out" regression.
#[test]
fn strip_gate_cookies_keeps_app_cookies() {
let mut headers = HeaderMap::new();
headers.insert(
header::COOKIE,
"session=abc; vw_session=keepme; csrf_token=def; theme=dark"
.parse()
.unwrap(),
);
strip_gate_cookies(&mut headers);
assert_eq!(
headers.get(header::COOKIE).unwrap().to_str().unwrap(),
"vw_session=keepme; theme=dark"
);
}
#[test]
fn strip_gate_cookies_drops_header_when_only_gate_cookies() {
let mut headers = HeaderMap::new();
headers.insert(
header::COOKIE,
"session=abc; csrf_token=def".parse().unwrap(),
);
strip_gate_cookies(&mut headers);
assert!(headers.get(header::COOKIE).is_none());
}
#[test]
fn strip_gate_cookies_no_header_is_a_noop() {
let mut headers = HeaderMap::new();
strip_gate_cookies(&mut headers);
assert!(headers.get(header::COOKIE).is_none());
}
/// The load-bearing 2FA property: a session still awaiting its TOTP code
/// fails `validate()`, so the gate rejects it without knowing anything
/// about second factors.
+66 -37
View File
@@ -154,9 +154,9 @@ pub async fn ensure_doctor_installed() {
}
match run_bitcoin_rpc_repair().await {
Ok(true) => {
info!("Repaired Bitcoin RPC bind settings; running Bitcoin containers left untouched")
info!("Removed stale bitcoin.conf; running Bitcoin containers left untouched")
}
Ok(false) => debug!("Bitcoin RPC bind settings already usable"),
Ok(false) => debug!("No stale bitcoin.conf found"),
Err(e) => warn!("Bitcoin RPC repair failed (non-fatal): {:#}", e),
}
match run_apps_dir_repair().await {
@@ -621,52 +621,30 @@ exit 2
}
async fn run_bitcoin_rpc_repair() -> Result<bool> {
// Older installs can have a container-owned bitcoin.conf with only rpcauth
// and printtoconsole. Repair it at startup so OTA fixes existing nodes
// without a manual uninstall/reinstall. Bind/port stay in the container
// command line to avoid duplicate RPC endpoint definitions.
// bitcoind is launched with -conf=/tmp/rpc.conf and never reads a
// datadir bitcoin.conf (apps/bitcoin-core & bitcoin-knots manifest.yml,
// commit a597c1d9 — bind/port live only on the container command line).
// A leftover file from an older install makes Bitcoin Core's own
// datadir-conflict safety check refuse to start on every subsequent
// start. Remove it instead of "repairing" it into existence — this
// previously wrote server=/rpcbind=/rpcallowip=/listen= into the file,
// which is exactly what caused the conflict.
let script = r#"
set -eu
conf=/var/lib/archipelago/bitcoin/bitcoin.conf
[ -f "$conf" ] || exit 0
changed=0
ensure_line() {
line="$1"
key="${line%%=*}"
if ! grep -q "^${key}=" "$conf"; then
printf '%s\n' "$line" >> "$conf"
changed=1
fi
}
ensure_line server=1
# rpcbind=0.0.0.0 is required inside the container: with rpcallowip set but
# no rpcbind, bitcoind binds RPC to the container's loopback only and every
# dial over the container network (LND, bitcoin-ui) is refused the fresh-
# install "LND took 5 attempts" / bitcoin-rpc 502 failure (host publish stays
# 127.0.0.1-only, so exposure is unchanged).
ensure_line rpcbind=0.0.0.0
ensure_line rpcallowip=0.0.0.0/0
ensure_line listen=1
# Log-volume fix: printtoconsole=1 duplicated every log line (incl. per-block
# IBD "UpdateTip" spam) into journald via conmon on top of the datadir
# debug.log bitcoind already writes. Console off; debug.log stays (bitcoind
# self-shrinks it on restart).
if grep -q '^printtoconsole=1' "$conf"; then
sed -i 's/^printtoconsole=1$/printtoconsole=0/' "$conf"
changed=1
fi
[ "$changed" -eq 0 ] && exit 0
mv "$conf" "$conf.disabled-$(date +%s)"
exit 2
"#;
let status = host_sudo(&["sh", "-lc", script])
.await
.context("repair bitcoin.conf RPC bind settings")?;
.context("remove stale bitcoin.conf RPC bind settings")?;
match status.code() {
Some(0) => Ok(false),
// Do not restart Bitcoin from bootstrap. During IBD, an automatic
// restart can cost hours of progress. The repaired file is only a
// fallback for future starts; current containers keep their command-line
// RPC args until an operator or update intentionally restarts them.
// restart can cost hours of progress. Removing the stale file is
// only a fallback for future starts; current containers keep their
// command-line RPC args regardless.
Some(2) => Ok(true),
_ => {
warn!("Bitcoin RPC repair helper exited with {}", status);
@@ -1293,3 +1271,54 @@ mod tests {
assert_ne!(outcome, PodmanHealOutcome::Healthy);
}
}
/// Repair this node's own systemd restart policy.
///
/// The in-process updater replaces the binary and then asks systemd to
/// restart the service, treating `Restart=always` on the unit as its second
/// net if that request is ever lost. On austin-sapien (2026-08-05) the unit
/// was an old one carrying `Restart=on-failure`: the daemon exited cleanly
/// (status 0), systemd read that as success, and the node sat dead for over
/// two hours after a routine update — "server starting" in the UI, with
/// nothing to start it.
///
/// A node cannot be relied on to fix this via `self-update.sh` (which does
/// refresh units) because the in-process update path never runs it. So the
/// daemon checks its own unit at boot: any node that starts even once ends
/// up with a policy that survives the next update. Deliberately narrow —
/// only the `Restart=` line is touched, so local edits elsewhere in the unit
/// are preserved.
pub async fn ensure_restart_policy() {
const UNIT: &str = "/etc/systemd/system/archipelago.service";
let Ok(body) = fs::read_to_string(UNIT).await else {
return; // not a systemd install (container, dev box) — nothing to do
};
if !body.lines().any(|l| {
let l = l.trim();
l.starts_with("Restart=") && l != "Restart=always"
}) {
return; // already correct, or no Restart= line to repair
}
let patched: String = body
.lines()
.map(|l| {
if l.trim().starts_with("Restart=") && l.trim() != "Restart=always" {
"Restart=always"
} else {
l
}
})
.collect::<Vec<_>>()
.join("\n");
match write_root_if_needed(UNIT, &patched).await {
Ok(true) => {
tracing::warn!(
"repaired archipelago.service Restart= policy to always — this node would \
have stayed dead after an in-process update"
);
let _ = host_sudo(&["systemctl", "daemon-reload"]).await;
}
Ok(false) => {}
Err(e) => tracing::warn!(error = %e, "could not repair archipelago.service restart policy"),
}
}
@@ -216,6 +216,73 @@ pub fn catalog_manifest_values() -> Vec<(String, serde_json::Value)> {
.collect()
}
/// A catalog-embedded manifest as the node actually applies it: parsed,
/// id-checked, validated, and image-only (build-source manifests defer to
/// disk). `None` = the caller must fall back to the disk manifest.
///
/// Shared between the orchestrator's load overlay and the app gate's port
/// classification so both answer "which manifest governs this app?" from the
/// same origin. They diverged once — the orchestrator published containers
/// from the catalog while the gate classified from stale disk manifests, and
/// the gate externally bound a port the catalog had declared `auth: local`
/// (nbxplorer 32838, archi-dev-box 2026-08-04).
pub fn catalog_manifest_overlay(
app_id: &str,
value: serde_json::Value,
) -> Option<archipelago_container::manifest::AppManifest> {
let m: archipelago_container::manifest::AppManifest = match serde_json::from_value(value) {
Ok(m) => m,
Err(e) => {
tracing::warn!(app = %app_id, error = %e,
"skipping unparseable catalog manifest; using disk fallback");
return None;
}
};
if m.app.id != app_id {
tracing::warn!(catalog_id = %app_id, manifest_id = %m.app.id,
"skipping catalog manifest: embedded app id mismatches catalog key");
return None;
}
if let Err(e) = m.validate() {
tracing::warn!(app = %app_id, error = %e,
"skipping invalid catalog manifest; using disk fallback");
return None;
}
if m.app.container.build.is_some() {
tracing::debug!(app = %app_id,
"catalog manifest has a build source; deferring to disk (phase 1 = image-only)");
return None;
}
Some(m)
}
/// Like [`catalog_manifest_overlay`] but WITHOUT the build-source refusal —
/// for PORT CLASSIFICATION only, never for install/orchestration.
///
/// The on-node-built companion UIs (lnd-ui, bitcoin-ui, electrs-ui, fips-ui)
/// are exactly the apps whose port policy (auth/bind/session_passthrough)
/// must reach the gate reliably, yet their build sources made the overlay
/// defer to DISK manifests — whose only delivery paths (frontend runtime
/// payload, per-node repo copies) proved stale or absent across the fleet in
/// the v1.7.125 rollout: nodes served ungated UIs or 401-dead panels until
/// hand-fixed. The signed catalog is fresher and operator-signed; and the
/// gate's address binds fail safely on conflict with a container that
/// publishes differently (logged as CANNOT PROTECT), so classifying from the
/// catalog cannot open anything the running container hasn't already opened.
pub fn catalog_manifest_ports_overlay(
app_id: &str,
value: serde_json::Value,
) -> Option<archipelago_container::manifest::AppManifest> {
let m: archipelago_container::manifest::AppManifest = serde_json::from_value(value).ok()?;
if m.app.id != app_id {
return None;
}
if m.validate().is_err() {
return None;
}
Some(m)
}
/// The catalog's default/latest version string for an app (the top-level
/// `version` field), if covered. Used to decide whether an install-time
/// selection should pin (older) or track-latest (default).
+1 -1
View File
@@ -293,6 +293,6 @@ mod tests {
// Lock in the core shape so a bad template edit doesn't ship.
assert!(TEMPLATE.contains("proxy_pass http://127.0.0.1:8332/"));
assert!(TEMPLATE.contains("location /bitcoin-rpc/"));
assert!(TEMPLATE.contains("listen 8334"));
assert!(TEMPLATE.contains("listen 127.0.0.1:8334"));
}
}
@@ -1,5 +1,12 @@
server {
listen 8334;
# Loopback ONLY. This container is host-networked, so this nginx binds the
# HOST's address directly — `listen 8334;` meant every interface, and the
# app gate could never stand in front of it (there is no podman publish to
# pin, and the manifest declared no port, so the gate neither protected it
# nor reported it — it served this page to anyone who asked, on LAN,
# Tailscale and the mesh alike). Binding loopback lets the daemon claim the
# external addresses and authenticate them; see appgate::listener.
listen 127.0.0.1:8334;
server_name _;
root /usr/share/nginx/html;
index index.html;
@@ -214,10 +214,59 @@ pub async fn install_one(spec: &CompanionSpec) -> Result<()> {
}
// Start is idempotent — if already running, systemctl returns 0.
quadlet::enable_now(&unit.service_name()).await?;
// A rebuilt image does NOT reach a container that is already running.
// `ensure_image_present` rebuilds in place under the same tag, so the unit
// body is byte-identical, `write_if_changed` reports no change, and
// `enable_now` is a no-op on a running service — the container keeps the
// old layers indefinitely. That is exactly how archi-dev-box kept serving
// the LND, FIPS, Electrs and Guardian screens on 0.0.0.0 after v1.7.123
// rebuilt every one of those images to bind loopback: the images were
// correct on disk and the running containers were three days old
// (2026-08-05). Compare image IDs and restart when they diverge.
if let Some(running) = container_image_id(spec.name).await {
if let Some(built) = image_id(&image).await {
if running != built {
info!(
companion = spec.name,
"running container uses a stale image; restarting onto the rebuilt one"
);
quadlet::restart_service(&unit.service_name()).await?;
}
}
}
info!(companion = spec.name, "companion started");
Ok(())
}
/// Image ID a container is actually running, or `None` when it does not exist.
async fn container_image_id(name: &str) -> Option<String> {
let out = tokio::process::Command::new("podman")
.args(["inspect", name, "--format", "{{.Image}}"])
.output()
.await
.ok()?;
if !out.status.success() {
return None;
}
let id = String::from_utf8_lossy(&out.stdout).trim().to_string();
(!id.is_empty()).then_some(id)
}
/// Current ID behind an image reference, or `None` when absent.
async fn image_id(image_ref: &str) -> Option<String> {
let out = tokio::process::Command::new("podman")
.args(["image", "inspect", image_ref, "--format", "{{.Id}}"])
.output()
.await
.ok()?;
if !out.status.success() {
return None;
}
let id = String::from_utf8_lossy(&out.stdout).trim().to_string();
(!id.is_empty()).then_some(id)
}
/// Build companion image locally if a Dockerfile exists, otherwise
/// pull from the lfg2025 registry. Returns the image ref the quadlet
/// should reference (`localhost/<base>:latest` for build, registry
@@ -104,6 +104,32 @@ fn dependency_manifests_required_by_active_apps<'a>(
required
}
/// Whether `app_id` is a member of a known multi-container stack that has at
/// least one OTHER member with a live container (any state). A live sibling
/// proves the stack is installed on this node, so an absent member is a hole
/// to repair — while a stack with no containers at all stays untouched
/// (uninstalled, or never installed here). Sibling app ids resolve to
/// container names through the loaded-manifest map when available (immich's
/// `immich-postgres` app id runs as container `immich_postgres`), falling
/// back to the id itself.
fn absent_stack_member_with_live_sibling(
app_id: &str,
present_containers: &HashSet<String>,
container_name_by_app_id: &std::collections::HashMap<String, String>,
) -> bool {
let stack = crate::app_ops::owning_package(app_id);
let members = crate::app_ops::stack_member_app_ids(stack);
members.iter().any(|member| {
*member != app_id
&& present_containers.contains(
container_name_by_app_id
.get(*member)
.map(String::as_str)
.unwrap_or(member),
)
})
}
fn manifest_dependency_app_ids(manifest: &AppManifest) -> Vec<String> {
manifest
.app
@@ -246,10 +272,10 @@ fn build_fingerprint_stamp_path(data_dir: &Path, tag: &str) -> PathBuf {
}
async fn chown_for_rootless_container(uid_gid: &str, path: &str) -> Result<()> {
let uid = uid_gid
let (uid, gid) = uid_gid
.split_once(':')
.and_then(|(uid, _)| uid.parse::<u32>().ok())
.unwrap_or(0);
.map(|(u, g)| (u.parse::<u32>().unwrap_or(0), g.parse::<u32>().unwrap_or(0)))
.unwrap_or((0, 0));
if uid > 0 && uid < 100_000 {
let output = tokio::process::Command::new("podman")
@@ -262,9 +288,22 @@ async fn chown_for_rootless_container(uid_gid: &str, path: &str) -> Result<()> {
}
}
let status = host_sudo(&["chown", "-R", uid_gid, path])
// Host-side fallback. A CONTAINER-namespace id must be translated into
// the subuid range first: `sudo chown 999` writes literal host uid 999,
// which maps to nobody inside the userns — the app then can't open its
// own files while the chown reported success (botfights SQLITE_CANTOPEN
// crash-loop, framework-pt 2026-08-06). Container uid N (N>=1) lives at
// subuid_base + N - 1; the fleet provisions base 100000. uid 0 and
// already-mapped ids (>=100000) pass through untouched.
let host_uid_gid = if uid > 0 && uid < 100_000 {
let map = |id: u32| if id == 0 { 1000 } else { 100_000 + id - 1 };
format!("{}:{}", map(uid), map(gid))
} else {
uid_gid.to_string()
};
let status = host_sudo(&["chown", "-R", &host_uid_gid, path])
.await
.with_context(|| format!("sudo chown -R {uid_gid} {path}"))?;
.with_context(|| format!("sudo chown -R {host_uid_gid} {path}"))?;
if status.success() {
return Ok(());
}
@@ -595,10 +634,20 @@ async fn wait_for_manifest_host_ports(
/// `podman inspect --format '{{json .HostConfig.PortBindings}}'` emits, e.g.
/// `{"8080/tcp":[{"HostIp":"","HostPort":"18080"}]}`. Returns true only when a
/// manifest container-port is positively published to a *different* host port
/// than the manifest now asks for. Absence of a binding is deliberately NOT
/// treated as drift here — that case is handled by the host-port repair/restart
/// path and by host-networked apps that publish nothing — so we never trigger a
/// destructive recreate on a false positive.
/// than the manifest now asks for — or, when the manifest DECLARES a bind
/// address, to a different host address. Absence of a binding is deliberately
/// NOT treated as drift here — that case is handled by the host-port
/// repair/restart path and by host-networked apps that publish nothing — so we
/// never trigger a destructive recreate on a false positive.
///
/// The bind comparison is what lets a node self-heal after a catalog refresh
/// pins an app to loopback for the app gate: a legacy (pre-quadlet) container
/// still publishing `0.0.0.0:P` against a manifest that now declares
/// `bind: 127.0.0.1` is recreated to the declared state, exactly as
/// `package.update` would. An EMPTY manifest bind means "no instruction" and
/// never fires this — recreating a loopback-published container to wildcard on
/// silence is precisely the v1.7.121 incident class (Bitcoin RPC republished
/// on the LAN).
fn host_port_bindings_drifted(
port_bindings_json: &str,
manifest_ports: &[archipelago_container::manifest::PortMapping],
@@ -626,10 +675,26 @@ fn host_port_bindings_drifted(
}
let expected = port.host.to_string();
let matches_expected = bindings.iter().any(|b| {
b.get("HostPort")
let host_port_ok = b
.get("HostPort")
.and_then(|h| h.as_str())
.map(|h| h == expected)
.unwrap_or(false)
.unwrap_or(false);
if !host_port_ok {
return false;
}
// Only a DECLARED bind participates; podman reports a wildcard
// publish as "" or "0.0.0.0".
if port.bind.is_empty() {
return true;
}
let actual_ip = b.get("HostIp").and_then(|h| h.as_str()).unwrap_or("");
let actual = if actual_ip.is_empty() {
"0.0.0.0"
} else {
actual_ip
};
actual == port.bind
});
if !matches_expected {
return true;
@@ -1157,30 +1222,7 @@ struct LoadedManifest {
/// source (build contexts aren't registry-distributed yet — phase 1 is
/// image-only). See `docs/registry-manifest-design.md`.
fn catalog_manifest_to_overlay(app_id: &str, value: serde_json::Value) -> Option<AppManifest> {
let m: AppManifest = match serde_json::from_value(value) {
Ok(m) => m,
Err(e) => {
tracing::warn!(app = %app_id, error = %e,
"skipping unparseable catalog manifest; using disk fallback");
return None;
}
};
if m.app.id != app_id {
tracing::warn!(catalog_id = %app_id, manifest_id = %m.app.id,
"skipping catalog manifest: embedded app id mismatches catalog key");
return None;
}
if let Err(e) = m.validate() {
tracing::warn!(app = %app_id, error = %e,
"skipping invalid catalog manifest; using disk fallback");
return None;
}
if m.app.container.build.is_some() {
tracing::debug!(app = %app_id,
"catalog manifest has a build source; deferring to disk (phase 1 = image-only)");
return None;
}
Some(m)
crate::container::app_catalog::catalog_manifest_overlay(app_id, value)
}
struct OrchestratorState {
@@ -1651,13 +1693,16 @@ impl ProdContainerOrchestrator {
// app whose container vanished (e.g. a wedged teardown cleared by a
// reboot) instead of leaving it down. See the immich .198 incident.
let was_running = crate::crash_recovery::load_last_running_names(&self.data_dir).await;
let manifests: Vec<LoadedManifest> = {
let (manifests, container_name_by_app_id): (
Vec<LoadedManifest>,
std::collections::HashMap<String, String>,
) = {
let state = self.state.read().await;
let dependency_required = dependency_manifests_required_by_active_apps(
state.manifests.values().map(|lm| &lm.manifest),
&user_stopped,
);
state
let filtered = state
.manifests
.iter()
.filter(|(app_id, _)| !state.disabled.contains(*app_id))
@@ -1667,8 +1712,25 @@ impl ProdContainerOrchestrator {
&& !user_stopped.contains(&compute_container_name(&lm.manifest)))
})
.map(|(_, lm)| lm.clone())
.collect()
.collect();
// Unfiltered id→container-name map for the absent-stack-member
// recovery below: a sibling may be excluded from this pass (e.g.
// user-stopped) yet its live container still proves the stack is
// installed.
let names = state
.manifests
.iter()
.map(|(id, lm)| (id.clone(), compute_container_name(&lm.manifest)))
.collect();
(filtered, names)
};
// Live container names (any state), for the same recovery check.
let present_containers: std::collections::HashSet<String> = self
.runtime
.list_containers()
.await
.map(|cs| cs.into_iter().map(|c| c.name).collect())
.unwrap_or_default();
let mut report = ReconcileReport::default();
let disk_gb = self.disk_gb().await;
// Register every candidate before the (sequential, possibly slow)
@@ -1735,7 +1797,20 @@ impl ProdContainerOrchestrator {
Ok(ReconcileAction::Left(reason))
if mode == ReconcileMode::ExistingOnly
&& reason == "absent"
&& was_running.contains(&compute_container_name(&lm.manifest)) =>
&& (was_running.contains(&compute_container_name(&lm.manifest))
// Absent STACK MEMBER whose siblings have live
// containers: the stack is installed, so the
// missing member is a hole, not a choice. The
// was_running snapshot ages out after a few daemon
// restarts, which left indeedhub-minio/-postgres
// permanently absent on .38 (2026-08-06) — nginx
// down on `host not found in upstream "minio"`
// with nothing ever recreating the members.
|| absent_stack_member_with_live_sibling(
&app_id,
&present_containers,
&container_name_by_app_id,
)) =>
{
tracing::warn!(
app_id = %app_id,
@@ -1751,7 +1826,10 @@ impl ProdContainerOrchestrator {
}
Ok(action) => report.record(&app_id, action),
Err(e) => {
tracing::error!(app_id = %app_id, error = %e, "reconcile failed");
// `{:#}` prints the whole anyhow chain — `%e` alone showed
// only the outer context ("create_container X") and hid
// the actual libpod error for days.
tracing::error!(app_id = %app_id, error = %format!("{e:#}"), "reconcile failed");
report.failures.push((app_id, e.to_string()));
}
}
@@ -4437,6 +4515,7 @@ mod tests {
bind: String::new(),
auth: None,
auth_rationale: None,
session_passthrough: false,
}
}
@@ -4444,6 +4523,61 @@ mod tests {
items.iter().map(|s| s.to_string()).collect()
}
/// The .38 indeedhub incident class: an absent stack member must be
/// recovered when its siblings have live containers (the stack is
/// installed), and left alone when the whole stack is gone or the app
/// is not a stack member at all.
#[test]
fn absent_stack_member_recovery_requires_a_live_sibling() {
let present: HashSet<String> = ["indeedhub-redis", "indeedhub-relay", "indeedhub"]
.iter()
.map(|s| s.to_string())
.collect();
let names = std::collections::HashMap::new();
// Missing members of a stack with live siblings → recover.
assert!(absent_stack_member_with_live_sibling(
"indeedhub-minio",
&present,
&names
));
assert!(absent_stack_member_with_live_sibling(
"indeedhub-postgres",
&present,
&names
));
// Whole stack absent → NOT recovered (uninstalled stays uninstalled).
let empty = HashSet::new();
assert!(!absent_stack_member_with_live_sibling(
"indeedhub-minio",
&empty,
&names
));
// Non-stack app → never.
assert!(!absent_stack_member_with_live_sibling(
"vaultwarden",
&present,
&names
));
// An app's OWN container being present proves nothing about siblings.
let only_self: HashSet<String> = std::iter::once("indeedhub-minio".to_string()).collect();
assert!(!absent_stack_member_with_live_sibling(
"indeedhub-minio",
&only_self,
&names
));
// App-id → container-name mapping is honoured (immich_postgres runs
// under an underscore name while its app id is hyphenated).
let mut mapped = std::collections::HashMap::new();
mapped.insert("immich-postgres".to_string(), "immich_postgres".to_string());
let immich_present: HashSet<String> =
std::iter::once("immich_postgres".to_string()).collect();
assert!(absent_stack_member_with_live_sibling(
"immich-redis",
&immich_present,
&mapped
));
}
#[test]
fn command_drift_tolerates_quadlet_entrypoint_split() {
// Quadlet writes Entrypoint=sh + Exec=-lc "<script>", so podman
@@ -4569,6 +4703,76 @@ mod tests {
));
}
fn bound_port(
host: u16,
container: u16,
bind: &str,
) -> archipelago_container::manifest::PortMapping {
archipelago_container::manifest::PortMapping {
bind: bind.to_string(),
..port(host, container)
}
}
#[test]
fn bind_drift_detected_when_declared_loopback_but_published_wildcard() {
// The legacy-container case: a pre-quadlet container still publishes
// 0.0.0.0 while the catalog-delivered manifest pins the app to
// loopback for the app gate. Must recreate, or the port stays open on
// every interface and the gate can never claim it.
for wildcard in [r#""""#, r#""0.0.0.0""#] {
let bindings = format!(r#"{{"80/tcp":[{{"HostIp":{wildcard},"HostPort":"8082"}}]}}"#);
assert!(host_port_bindings_drifted(
&bindings,
&[bound_port(8082, 80, "127.0.0.1")]
));
}
}
#[test]
fn no_bind_drift_when_declared_loopback_and_published_loopback() {
let bindings = r#"{"80/tcp":[{"HostIp":"127.0.0.1","HostPort":"8082"}]}"#;
assert!(!host_port_bindings_drifted(
bindings,
&[bound_port(8082, 80, "127.0.0.1")]
));
}
#[test]
fn no_bind_drift_on_undeclared_bind() {
// Silence is not consent (v1.7.121 incident class): an EMPTY manifest
// bind must never recreate a loopback-published container to
// wildcard — that is how Bitcoin's RPC got republished on the LAN.
let bindings = r#"{"8332/tcp":[{"HostIp":"127.0.0.1","HostPort":"8332"}]}"#;
assert!(!host_port_bindings_drifted(bindings, &[port(8332, 8332)]));
}
#[test]
fn multi_bind_publish_satisfies_each_declared_entry() {
// Same host/container pair listed twice (loopback + archy-net
// gateway): both declared binds are present in the actual publish.
let bindings = r#"{"8332/tcp":[
{"HostIp":"127.0.0.1","HostPort":"8332"},
{"HostIp":"10.89.0.1","HostPort":"8332"}
]}"#;
assert!(!host_port_bindings_drifted(
bindings,
&[
bound_port(8332, 8332, "127.0.0.1"),
bound_port(8332, 8332, "10.89.0.1")
]
));
// And a wildcard-only publish drifts BOTH declared entries.
let wildcard = r#"{"8332/tcp":[{"HostIp":"","HostPort":"8332"}]}"#;
assert!(host_port_bindings_drifted(
wildcard,
&[
bound_port(8332, 8332, "127.0.0.1"),
bound_port(8332, 8332, "10.89.0.1")
]
));
}
#[test]
fn missing_secret_error_names_the_secret() {
use archipelago_container::manifest::SecretsProvider;
+2 -2
View File
@@ -7,6 +7,6 @@
pub const APP_LAUNCH_PORTS: &[u16] = &[
2283, 2342, 3000, 3001, 3002, 4080, 5180, 7778, 8080, 8081, 8082, 8083, 8084, 8085, 8087, 8088,
8089, 8090, 8096, 8123, 8175, 8176, 8240, 8334, 8888, 8999, 9000, 9100, 10380, 11434, 18081,
18083, 23000, 32838, 50002,
8089, 8090, 8096, 8123, 8175, 8176, 8240, 8334, 8336, 8888, 8999, 9000, 9100, 10380, 11434,
18081, 18083, 23000, 32838, 50002,
];
+205
View File
@@ -0,0 +1,205 @@
//! Last-known-good FIPS peer endpoints (A3.10).
//!
//! The LAN direct-peering tick (`anchors::lan_fips_anchors`) only helps peers
//! we can currently see on the LAN. When a federation peer's LAN path is gone
//! (renumbered network, remote site, mDNS blackout) the only route left is the
//! anchor spanning tree — the exact hairpin RC2 calls out. But if we were EVER
//! connected to that peer directly, the daemon knew a working endpoint for it
//! (`fipsctl show peers` → `transport_addr`/`transport_type`, which covers
//! LAN, Tailscale, and WAN endpoints alike). This module persists those
//! npub-keyed endpoints and re-offers them as dial candidates when the live
//! paths disappear: LAN → last-known-good → anchor tree.
//!
//! Persisted at `<data_dir>/fips-endpoints.json`. Entries are refreshed every
//! time the peer is seen connected and dropped after `RETENTION` without a
//! sighting, so a peer that genuinely moved doesn't get dialed at a stale
//! address forever ( `fipsctl connect` to a dead address is harmless but not
//! free).
use std::collections::HashMap;
use std::path::Path;
use std::time::{SystemTime, UNIX_EPOCH};
use anyhow::Result;
use serde::{Deserialize, Serialize};
use tokio::fs;
use super::anchors::SeedAnchor;
const FILE_NAME: &str = "fips-endpoints.json";
/// Forget endpoints not seen connected for this long (seconds) — 30 days.
const RETENTION_SECS: u64 = 30 * 24 * 60 * 60;
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
pub struct KnownEndpoint {
/// "ip:port" as reported by the daemon (`transport_addr`).
pub address: String,
/// "udp" | "tcp" (`transport_type`).
pub transport: String,
/// Unix seconds of the last time this peer was seen connected here.
pub last_ok_unix: u64,
}
/// A currently-connected peer as parsed from `fipsctl show peers`.
#[derive(Debug, Clone)]
pub struct ConnectedPeer {
pub npub: String,
pub address: String,
pub transport: String,
}
fn now_unix() -> u64 {
SystemTime::now()
.duration_since(UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0)
}
pub async fn load(data_dir: &Path) -> HashMap<String, KnownEndpoint> {
let path = data_dir.join(FILE_NAME);
match fs::read(&path).await {
Ok(bytes) => serde_json::from_slice(&bytes).unwrap_or_default(),
Err(_) => HashMap::new(),
}
}
async fn save(data_dir: &Path, map: &HashMap<String, KnownEndpoint>) -> Result<()> {
let path = data_dir.join(FILE_NAME);
let tmp = data_dir.join(format!("{FILE_NAME}.tmp"));
fs::write(&tmp, serde_json::to_vec_pretty(map)?).await?;
fs::rename(&tmp, &path).await?;
Ok(())
}
/// Merge the currently-connected peers into the store (refreshing their
/// timestamps), prune expired entries, persist, and return the updated map.
/// Persistence failures are non-fatal — the in-memory result is still
/// returned so this tick's fallback logic works.
pub async fn record_connected(
data_dir: &Path,
connected: &[ConnectedPeer],
) -> HashMap<String, KnownEndpoint> {
let mut map = load(data_dir).await;
let now = now_unix();
let before = map.clone();
for p in connected {
if p.npub.is_empty() || p.address.is_empty() {
continue;
}
map.insert(
p.npub.clone(),
KnownEndpoint {
address: p.address.clone(),
transport: p.transport.clone(),
last_ok_unix: now,
},
);
}
map.retain(|_, e| now.saturating_sub(e.last_ok_unix) <= RETENTION_SECS);
if map != before {
if let Err(e) = save(data_dir, &map).await {
tracing::debug!("fips endpoint store save failed (non-fatal): {e}");
}
}
map
}
/// Build fallback anchors for federation peers whose live paths are gone:
/// every `wanted_npub` that is neither currently connected nor covered by a
/// live LAN direct entry, but has a last-known-good endpoint, becomes a dial
/// candidate. `fipsctl connect` is idempotent and failure-tolerant, so a
/// stale candidate costs one failed dial, bounded by apply()'s per-connect
/// timeout.
pub fn fallback_anchors(
known: &HashMap<String, KnownEndpoint>,
wanted_npubs: &[String],
connected_npubs: &[String],
lan_direct: &[SeedAnchor],
) -> Vec<SeedAnchor> {
let mut out = Vec::new();
for npub in wanted_npubs {
if connected_npubs.iter().any(|c| c == npub) {
continue;
}
if lan_direct.iter().any(|a| &a.npub == npub) {
continue;
}
if let Some(e) = known.get(npub) {
out.push(SeedAnchor {
npub: npub.clone(),
address: e.address.clone(),
transport: e.transport.clone(),
label: "last-known-good endpoint (direct FIPS)".to_string(),
});
}
}
out
}
#[cfg(test)]
mod tests {
use super::*;
fn ep(addr: &str) -> KnownEndpoint {
KnownEndpoint {
address: addr.to_string(),
transport: "udp".to_string(),
last_ok_unix: now_unix(),
}
}
#[tokio::test]
async fn record_and_reload_roundtrip() {
let dir = tempfile::tempdir().unwrap();
let connected = vec![ConnectedPeer {
npub: "npub1aaa".into(),
address: "100.114.134.21:2121".into(),
transport: "udp".into(),
}];
let map = record_connected(dir.path(), &connected).await;
assert_eq!(map["npub1aaa"].address, "100.114.134.21:2121");
let reloaded = load(dir.path()).await;
assert_eq!(reloaded, map);
}
#[tokio::test]
async fn expired_entries_are_pruned_on_record() {
let dir = tempfile::tempdir().unwrap();
let mut stale = HashMap::new();
stale.insert(
"npub1old".to_string(),
KnownEndpoint {
address: "10.0.0.1:2121".into(),
transport: "udp".into(),
last_ok_unix: now_unix() - RETENTION_SECS - 60,
},
);
save(dir.path(), &stale).await.unwrap();
let map = record_connected(dir.path(), &[]).await;
assert!(map.is_empty());
}
#[test]
fn fallback_skips_connected_and_lan_covered_peers() {
let mut known = HashMap::new();
known.insert("npub1gone".to_string(), ep("100.1.2.3:2121"));
known.insert("npub1conn".to_string(), ep("100.1.2.4:2121"));
known.insert("npub1lan".to_string(), ep("100.1.2.5:2121"));
let wanted: Vec<String> = ["npub1gone", "npub1conn", "npub1lan", "npub1never"]
.iter()
.map(|s| s.to_string())
.collect();
let connected = vec!["npub1conn".to_string()];
let lan = vec![SeedAnchor {
npub: "npub1lan".into(),
address: "192.168.63.198:2121".into(),
transport: "udp".into(),
label: "LAN".into(),
}];
let out = fallback_anchors(&known, &wanted, &connected, &lan);
assert_eq!(out.len(), 1);
assert_eq!(out[0].npub, "npub1gone");
assert_eq!(out[0].address, "100.1.2.3:2121");
// npub1never has no stored endpoint → nothing to dial.
}
}
+1
View File
@@ -29,6 +29,7 @@ pub mod anchors;
pub mod app_ports;
pub mod config;
pub mod dial;
pub mod endpoints;
pub mod iface;
pub mod service;
pub mod telemetry;
+46
View File
@@ -227,6 +227,52 @@ pub async fn peer_connectivity_summary(anchor_candidates: &[String]) -> (u32, bo
(authenticated_peer_count, anchor_connected)
}
/// Currently-connected peers with their live endpoints, from
/// `fipsctl show peers` (`transport_addr`/`transport_type`). Feeds the
/// last-known-good endpoint store (A3.10); empty on any failure.
pub async fn connected_peer_endpoints() -> Vec<crate::fips::endpoints::ConnectedPeer> {
let peers_json = match Command::new("sudo")
.args(["-n", "fipsctl", "show", "peers"])
.output()
.await
{
Ok(o) if o.status.success() => o.stdout,
_ => return Vec::new(),
};
let parsed: serde_json::Value = match serde_json::from_slice(&peers_json) {
Ok(v) => v,
Err(_) => return Vec::new(),
};
parsed
.get("peers")
.and_then(|p| p.as_array())
.map(|peers| {
peers
.iter()
.filter(|p| {
p.get("connectivity")
.and_then(|c| c.as_str())
.map(|s| s == "connected")
.unwrap_or(false)
})
.filter_map(|p| {
let npub = p.get("npub").and_then(|n| n.as_str())?;
let address = p.get("transport_addr").and_then(|a| a.as_str())?;
let transport = p
.get("transport_type")
.and_then(|t| t.as_str())
.unwrap_or("udp");
Some(crate::fips::endpoints::ConnectedPeer {
npub: npub.to_string(),
address: address.to_string(),
transport: transport.to_string(),
})
})
.collect()
})
.unwrap_or_default()
}
/// Read the upstream daemon's public key at `/etc/fips/fips.pub` and return
/// it as a bech32 npub. Returns `Ok(None)` if the file doesn't exist — used
/// as a fallback on legacy/dev nodes where no seed-derived key exists.
+5
View File
@@ -411,6 +411,11 @@ async fn main() -> Result<()> {
// flags) on already-deployed nodes via OTA; no-op if the kiosk isn't installed.
tokio::spawn(bootstrap::ensure_kiosk_hardened());
// Repair our own restart policy before anything else can need it: a node
// whose unit still says Restart=on-failure stays dead after the next
// in-process update, because the daemon exits cleanly to be restarted.
tokio::spawn(bootstrap::ensure_restart_policy());
// HDMI audio: install the PipeWire stack + audio-router daemon on kiosk
// nodes (older ISOs shipped no audio stack; the router also heals the
// boot-time ELD race that leaves HDMI silently unavailable).
+13 -1
View File
@@ -148,9 +148,21 @@ pub enum MeshCommand {
},
SendAdvert,
/// Reboot the locally-connected radio firmware to recover a wedged /
/// RX-deaf radio. Meshtastic-only; meshcore ignores it.
/// RX-deaf radio. Meshtastic: firmware reboot command. Reticulum: the
/// sidecar daemon is restarted (radio re-detected + reconfigured).
/// MeshCore: unsupported, and says so. `reply` (when present) carries
/// the real outcome to the RPC caller — the buttons used to be
/// fire-and-forget `warn!`s, i.e. no feedback ever reached the UI
/// (operator, 2026-08-06).
RebootRadio {
seconds: i64,
reply: Option<tokio::sync::oneshot::Sender<Result<String, String>>>,
},
/// Query the live RNode radio state (Reticulum-only): the sidecar's
/// radio-confirmed parameters, for the LoRa settings panel's current
/// values + apply read-back.
QueryRadioState {
reply: tokio::sync::oneshot::Sender<Result<serde_json::Value, String>>,
},
/// Re-fetch contact list from the radio device.
RefreshContacts,
+45 -11
View File
@@ -165,13 +165,41 @@ impl MeshRadioDevice {
}
}
async fn reboot(&mut self, seconds: i64) -> Result<()> {
async fn reboot(&mut self, seconds: i64) -> Result<String> {
match self {
// Meshcore/Reticulum have no equivalent local-admin reboot in our
// driver; the RX-deaf recovery this targets is Meshtastic-specific.
Self::Meshcore(_) => Ok(()),
Self::Meshtastic(device) => device.reboot(seconds).await,
Self::Reticulum(_) => Ok(()),
// No remote reboot in the MeshCore serial protocol — say so
// instead of silently reporting success (the old `Ok(())` here
// is why the button "did nothing" for the operator).
Self::Meshcore(_) => {
anyhow::bail!("MeshCore radios have no remote reboot — power-cycle the device")
}
Self::Meshtastic(device) => {
device.reboot(seconds).await?;
Ok(format!(
"Radio firmware reboots in {seconds}s and reconnects automatically"
))
}
// Restarting the sidecar drops the serial port, re-detects the
// RNode and reapplies the RF config — the closest thing to a
// reboot the RNS stack has, and exactly what an operator wants
// after changing settings or on a wedged radio.
Self::Reticulum(device) => {
device.restart_daemon().await?;
Ok("Radio daemon restarting — the RNode re-detects and reconnects in about 15 seconds".to_string())
}
}
}
/// Live RNode radio state — Reticulum-only (see ReticulumLink::query_radio_state).
async fn radio_state(&mut self) -> Result<serde_json::Value> {
match self {
Self::Meshcore(_) | Self::Meshtastic(_) => {
anyhow::bail!("Radio state read-back is only available for Reticulum RNode devices")
}
Self::Reticulum(device) => device
.query_radio_state(std::time::Duration::from_secs(5))
.await
.ok_or_else(|| anyhow::anyhow!("The radio daemon did not answer the state query")),
}
}
@@ -1549,12 +1577,18 @@ async fn handle_send_command(
warn!("Failed to send NodeInfo advert: {}", e);
}
}
MeshCommand::RebootRadio { seconds } => {
if let Err(e) = device.reboot(seconds).await {
warn!("Failed to reboot radio: {}", e);
} else {
info!(seconds, "Radio reboot command sent to device");
MeshCommand::RebootRadio { seconds, reply } => {
let outcome = device.reboot(seconds).await;
match &outcome {
Err(e) => warn!("Failed to reboot radio: {}", e),
Ok(_) => info!(seconds, "Radio reboot command sent to device"),
}
if let Some(reply) = reply {
let _ = reply.send(outcome.map_err(|e| format!("{e:#}")));
}
}
MeshCommand::QueryRadioState { reply } => {
let _ = reply.send(device.radio_state().await.map_err(|e| format!("{e:#}")));
}
MeshCommand::RefreshContacts => {
refresh_contacts(device, state).await;
+37 -3
View File
@@ -16,6 +16,7 @@ pub mod outbox;
pub mod protocol;
pub mod ratchet;
pub mod reticulum;
pub mod rnode_settings;
pub mod scheduler;
pub mod serial;
pub mod session;
@@ -2123,20 +2124,53 @@ impl MeshService {
/// RX-deaf radio (one that has stopped hearing the mesh while still able to
/// transmit). The device reconnects via the listener's reboot→reconnect
/// loop. `seconds` is the firmware reboot delay.
pub async fn reboot_radio(&self, seconds: i64) -> Result<()> {
pub async fn reboot_radio(&self, seconds: i64) -> Result<String> {
let status = self.state.status.read().await;
if !status.device_connected {
anyhow::bail!("No mesh device connected. Check USB connection.");
}
drop(status);
let (tx, rx) = tokio::sync::oneshot::channel();
self.state
.send_cmd(listener::MeshCommand::RebootRadio { seconds })
.send_cmd(listener::MeshCommand::RebootRadio {
seconds,
reply: Some(tx),
})
.await
.map_err(|_| anyhow::anyhow!("Mesh listener not running"))?;
// The real outcome, not fire-and-forget: the UI shows this string
// (or the error) instead of pretending success.
let outcome = tokio::time::timeout(std::time::Duration::from_secs(15), rx)
.await
.map_err(|_| anyhow::anyhow!("The radio did not acknowledge the reboot in time"))?
.map_err(|_| anyhow::anyhow!("Mesh session ended before the reboot completed"))?;
let message = outcome.map_err(|e| anyhow::anyhow!(e))?;
info!(seconds, "Mesh radio reboot triggered");
Ok(())
Ok(message)
}
/// Live RNode radio state (Reticulum-only): the sidecar's view of the
/// interface including the radio-confirmed r_* parameters. The LoRa
/// settings panel's source for "what is the device actually running".
pub async fn radio_state(&self) -> Result<serde_json::Value> {
let status = self.state.status.read().await;
if !status.device_connected {
anyhow::bail!("No mesh device connected. Check USB connection.");
}
drop(status);
let (tx, rx) = tokio::sync::oneshot::channel();
self.state
.send_cmd(listener::MeshCommand::QueryRadioState { reply: tx })
.await
.map_err(|_| anyhow::anyhow!("Mesh listener not running"))?;
let state = tokio::time::timeout(std::time::Duration::from_secs(10), rx)
.await
.map_err(|_| anyhow::anyhow!("The radio daemon did not answer the state query"))?
.map_err(|_| anyhow::anyhow!("Mesh session ended before the state query completed"))?;
state.map_err(|e| anyhow::anyhow!(e))
}
/// Current mesh-AI assistant settings (issue #50).
+86
View File
@@ -176,6 +176,7 @@ fn daemon_command(
archy_x25519_pubkey_hex: Option<&str>,
display_name: Option<&str>,
enable_transport: bool,
rf: Option<&super::rnode_settings::RNodeRfSettings>,
) -> Command {
let (program, script) = daemon_program();
let mut cmd = Command::new(program);
@@ -189,6 +190,24 @@ fn daemon_command(
match iface {
ReticulumInterface::Serial(path) => {
cmd.arg("--serial-port").arg(path);
// Operator-editable RF parameters (.126 LoRa panel). Passed
// explicitly on every spawn so the sidecar's argparse defaults
// stop being the silent source of truth. `rf` is None only for
// non-serial interfaces, where these have no meaning.
if let Some(rf) = rf {
cmd.arg("--frequency").arg(rf.frequency.to_string());
cmd.arg("--bandwidth").arg(rf.bandwidth.to_string());
cmd.arg("--txpower").arg(rf.txpower.to_string());
cmd.arg("--spreadingfactor")
.arg(rf.spreading_factor.to_string());
cmd.arg("--codingrate").arg(rf.coding_rate.to_string());
if let Some(pct) = rf.airtime_limit_short {
cmd.arg("--airtime-limit-short").arg(pct.to_string());
}
if let Some(pct) = rf.airtime_limit_long {
cmd.arg("--airtime-limit-long").arg(pct.to_string());
}
}
}
ReticulumInterface::TcpServer(bind) => {
cmd.arg("--tcp-listen").arg(bind);
@@ -318,6 +337,10 @@ pub struct ReticulumLink {
/// down and the outer reconnect loop respawns the daemon — without this
/// a dead daemon was invisible until the 30-minute RX-stall watchdog.
daemon_gone: bool,
/// Latest `radio_state` event from the sidecar (the live RNodeInterface
/// values, radio-confirmed `r_*` included). Refreshed by
/// [`Self::query_radio_state`]; the .126 LoRa panel's read-back source.
last_radio_state: Option<Value>,
}
impl ReticulumLink {
@@ -344,6 +367,16 @@ impl ReticulumLink {
our_x25519_pubkey_hex: Option<&str>,
display_name: Option<&str>,
) -> Result<Self> {
let rf = super::rnode_settings::RNodeRfSettings::load(data_dir).await;
if !rf.enabled {
anyhow::bail!(
"RNode interface is disabled in the LoRa settings — enable it to connect"
);
}
// Operator port override wins over the auto-detected path (.126 LoRa
// panel). The probe below still gates: a wrong override fails with
// the detect error instead of a silent dead transport.
let path = rf.port.as_deref().unwrap_or(path);
probe_rnode(path)
.await
.context("RNode KISS detect failed")?;
@@ -454,6 +487,15 @@ impl ReticulumLink {
}
let enable_transport = daemon_supports_enable_transport().await;
// Operator RF settings ride every serial spawn; loaded here (not by
// callers) so a settings apply only needs a transport restart to take
// effect. Non-serial interfaces carry no RF.
let rf = match iface {
ReticulumInterface::Serial(_) => {
Some(super::rnode_settings::RNodeRfSettings::load(data_dir).await)
}
_ => None,
};
let mut cmd = daemon_command(
&socket_path,
&iface,
@@ -462,6 +504,7 @@ impl ReticulumLink {
our_x25519_pubkey_hex,
display_name,
enable_transport,
rf.as_ref(),
);
cmd.env("TMPDIR", &tmp_dir);
let child = cmd
@@ -534,6 +577,7 @@ impl ReticulumLink {
inbound: std::collections::VecDeque::new(),
resource_id_counter: 0,
daemon_gone: false,
last_radio_state: None,
};
link.load_persisted_peers();
Ok(link)
@@ -896,8 +940,50 @@ impl ReticulumLink {
}
}
/// Restart the sidecar daemon: ask it to shut down cleanly and mark the
/// link dead so the session loop tears down and the outer reconnect loop
/// respawns it — re-detecting the RNode and reapplying the RF config
/// from the (possibly just-edited) persisted settings. This IS the
/// "reboot device" semantic for Reticulum radios, and the apply step of
/// the .126 LoRa settings panel.
pub async fn restart_daemon(&mut self) -> Result<()> {
// Best-effort clean shutdown (lets PyInstaller clear its _MEI dir);
// the SIGTERM path in Drop/terminate covers an already-dead socket.
let _ = self.send_rpc(serde_json::json!({"cmd": "shutdown"})).await;
self.daemon_gone = true;
Ok(())
}
/// Ask the sidecar for the live RNode state and wait briefly for the
/// reply event. Returns the freshest `radio_state` payload, or `None`
/// when the daemon didn't answer in time (dead daemon, no radio build).
pub async fn query_radio_state(&mut self, timeout: Duration) -> Option<Value> {
self.last_radio_state = None;
if self
.send_rpc(serde_json::json!({"cmd": "radio_state"}))
.await
.is_err()
{
return None;
}
let deadline = tokio::time::Instant::now() + timeout;
loop {
self.drain_events().await;
if let Some(state) = &self.last_radio_state {
return Some(state.clone());
}
if self.daemon_gone || tokio::time::Instant::now() >= deadline {
return None;
}
tokio::time::sleep(Duration::from_millis(50)).await;
}
}
fn handle_event(&mut self, ev: Value) {
match ev.get("event").and_then(Value::as_str) {
Some("radio_state") => {
self.last_radio_state = Some(ev);
}
Some("announce") => {
let Some(hash) = ev
.get("dest_hash")
+360
View File
@@ -0,0 +1,360 @@
//! Persisted RNode LoRa RF settings — the operator-editable half of the
//! Reticulum transport (.126 LoRa settings panel).
//!
//! The reticulum sidecar (reticulum-daemon) writes the RNS config from its
//! CLI args at every spawn; before this module those args were never passed,
//! so every node ran the sidecar's argparse defaults and nothing was
//! operator-editable. These settings persist at
//! `<data_dir>/rnode-rf-settings.json`, feed `daemon_command` as explicit
//! args, and the panel confirms application via the sidecar's `radio_state`
//! read-back (the radio-confirmed `r_*` values, not the requested ones).
//!
//! An absent file yields [`RNodeRfSettings::default`], which matches the
//! sidecar's historical argparse defaults exactly — deploying this changes
//! nothing until the operator edits something.
use anyhow::{bail, Result};
use serde::{Deserialize, Serialize};
use std::path::Path;
const SETTINGS_FILE: &str = "rnode-rf-settings.json";
/// Validation bounds mirror RNS `RNodeInterface.py` (`validate_firmware` /
/// the constructor checks) — NOT guessed: frequency 1371020 MHz, sf 512,
/// cr 58, txpower 022 dBm, airtime locks 0100 %.
const FREQ_MIN_HZ: u64 = 137_000_000;
const FREQ_MAX_HZ: u64 = 1_020_000_000;
/// The discrete bandwidths RNode firmware accepts (Hz).
const VALID_BANDWIDTHS: &[u64] = &[
7_800, 10_400, 15_600, 20_800, 31_250, 41_700, 62_500, 125_000, 250_000, 500_000,
];
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
pub struct RNodeRfSettings {
/// Interface on/off. `false` keeps the daemon from opening the radio at
/// all (the mesh service skips the serial transport).
#[serde(default = "default_true")]
pub enabled: bool,
/// Serial device override (e.g. `/dev/ttyACM0`). `None` = auto-detect,
/// which is what every node did before this existed.
#[serde(default)]
pub port: Option<String>,
#[serde(default = "default_frequency")]
pub frequency: u64,
#[serde(default = "default_bandwidth")]
pub bandwidth: u64,
#[serde(default = "default_spreading_factor")]
pub spreading_factor: u8,
#[serde(default = "default_coding_rate")]
pub coding_rate: u8,
#[serde(default = "default_txpower")]
pub txpower: u8,
/// Short-window airtime duty-cycle lock, percent (EU868: 25). `None` =
/// no software lock (RNS default).
#[serde(default)]
pub airtime_limit_short: Option<f64>,
/// Long-window airtime duty-cycle lock, percent (EU868: 10).
#[serde(default)]
pub airtime_limit_long: Option<f64>,
}
fn default_true() -> bool {
true
}
fn default_frequency() -> u64 {
869_525_000
}
fn default_bandwidth() -> u64 {
125_000
}
fn default_spreading_factor() -> u8 {
8
}
fn default_coding_rate() -> u8 {
5
}
fn default_txpower() -> u8 {
17
}
impl Default for RNodeRfSettings {
fn default() -> Self {
Self {
enabled: true,
port: None,
frequency: default_frequency(),
bandwidth: default_bandwidth(),
spreading_factor: default_spreading_factor(),
coding_rate: default_coding_rate(),
txpower: default_txpower(),
airtime_limit_short: None,
airtime_limit_long: None,
}
}
}
impl RNodeRfSettings {
pub fn validate(&self) -> Result<()> {
if !(FREQ_MIN_HZ..=FREQ_MAX_HZ).contains(&self.frequency) {
bail!(
"frequency {} Hz is outside the RNode range ({}{} Hz)",
self.frequency,
FREQ_MIN_HZ,
FREQ_MAX_HZ
);
}
if !VALID_BANDWIDTHS.contains(&self.bandwidth) {
bail!(
"bandwidth {} Hz is not an RNode bandwidth (valid: {:?})",
self.bandwidth,
VALID_BANDWIDTHS
);
}
if !(5..=12).contains(&self.spreading_factor) {
bail!("spreading factor {} is outside 512", self.spreading_factor);
}
if !(5..=8).contains(&self.coding_rate) {
bail!("coding rate {} is outside 58", self.coding_rate);
}
if self.txpower > 22 {
bail!("tx power {} dBm is above the 22 dBm RNode maximum", self.txpower);
}
for (label, v) in [
("airtime_limit_short", self.airtime_limit_short),
("airtime_limit_long", self.airtime_limit_long),
] {
if let Some(pct) = v {
if !(0.0..=100.0).contains(&pct) || !pct.is_finite() {
bail!("{label} {pct} is not a percentage (0100)");
}
}
}
if let Some(port) = &self.port {
// Same shape the flasher accepts: an absolute device node. Keeps
// shell-metacharacter garbage out of the sidecar's argv.
if !port.starts_with("/dev/")
|| port
.chars()
.any(|c| !(c.is_ascii_alphanumeric() || c == '/' || c == '_' || c == '-' || c == '.'))
{
bail!("port must be an absolute /dev device path");
}
}
Ok(())
}
pub async fn load(data_dir: &Path) -> Self {
let path = data_dir.join(SETTINGS_FILE);
match tokio::fs::read_to_string(&path).await {
Ok(raw) => match serde_json::from_str::<Self>(&raw) {
Ok(s) => s,
Err(e) => {
tracing::warn!(error = %e, "rnode-rf-settings.json unparseable — using defaults");
Self::default()
}
},
// First run after the update: no settings file yet. ADOPT the
// node's existing effective RF config rather than imposing
// defaults — the operator's standing requirement is that the
// update changes NO device's applied settings. For archy-managed
// radios the sidecar config equals our defaults anyway; this
// covers any node whose RNS config diverged (hand edits,
// hand-run rnsd).
Err(_) => {
let adopted = Self::adopt_existing_rns_config().await;
if let Some(adopted) = adopted {
tracing::info!(
settings = ?adopted,
"adopted existing RNS RNode config as initial RF settings"
);
if let Err(e) = adopted.save(data_dir).await {
tracing::warn!(error = %e, "could not persist adopted RF settings");
}
adopted
} else {
Self::default()
}
}
}
}
/// Parse the RNodeInterface section out of an existing RNS config file
/// (the sidecar's `~/.archy-reticulum/config`, else a hand-run rnsd's
/// `~/.reticulum/config`). Returns `None` when neither exists or no
/// RNodeInterface section is found. Unparseable/absent fields keep the
/// default (which equals the sidecar's historical argparse default).
async fn adopt_existing_rns_config() -> Option<Self> {
let home = std::env::var("HOME").ok()?;
for candidate in [
format!("{home}/.archy-reticulum/config"),
format!("{home}/.reticulum/config"),
] {
let Ok(raw) = tokio::fs::read_to_string(&candidate).await else {
continue;
};
if let Some(s) = Self::parse_rnode_section(&raw) {
return Some(s);
}
}
None
}
/// Extract RNode parameters from RNS config text. Scoped to the block
/// after a `type = RNodeInterface` line so TCP interface options can
/// never bleed in; stops at the next `[[...]]` section header.
fn parse_rnode_section(raw: &str) -> Option<Self> {
let mut in_rnode = false;
let mut seen_any = false;
let mut s = Self::default();
for line in raw.lines() {
let line = line.trim();
if line.starts_with("[[") {
if in_rnode {
break; // next interface section — RNode block ended
}
continue;
}
let Some((key, value)) = line.split_once('=') else {
continue;
};
let (key, value) = (key.trim(), value.trim());
if key == "type" {
in_rnode = value == "RNodeInterface";
continue;
}
if !in_rnode {
continue;
}
seen_any = true;
match key {
"enabled" | "interface_enabled" => {
s.enabled = matches!(value.to_ascii_lowercase().as_str(), "yes" | "true" | "on")
}
"port" => s.port = Some(value.to_string()),
"frequency" => s.frequency = value.parse().unwrap_or(s.frequency),
"bandwidth" => s.bandwidth = value.parse().unwrap_or(s.bandwidth),
"txpower" => s.txpower = value.parse().unwrap_or(s.txpower),
"spreadingfactor" => {
s.spreading_factor = value.parse().unwrap_or(s.spreading_factor)
}
"codingrate" => s.coding_rate = value.parse().unwrap_or(s.coding_rate),
"airtime_limit_short" => s.airtime_limit_short = value.parse().ok(),
"airtime_limit_long" => s.airtime_limit_long = value.parse().ok(),
_ => {}
}
}
(in_rnode || seen_any).then_some(s)
}
pub async fn save(&self, data_dir: &Path) -> Result<()> {
self.validate()?;
let path = data_dir.join(SETTINGS_FILE);
let tmp = path.with_extension("json.tmp");
let raw = serde_json::to_string_pretty(self)?;
tokio::fs::write(&tmp, raw).await?;
tokio::fs::rename(&tmp, &path).await?;
Ok(())
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn defaults_match_the_sidecar_argparse_defaults() {
// reticulum_daemon.py: --frequency 869525000 --bandwidth 125000
// --txpower 17 --spreadingfactor 8 --codingrate 5, no airtime locks.
let d = RNodeRfSettings::default();
assert_eq!(d.frequency, 869_525_000);
assert_eq!(d.bandwidth, 125_000);
assert_eq!(d.txpower, 17);
assert_eq!(d.spreading_factor, 8);
assert_eq!(d.coding_rate, 5);
assert!(d.airtime_limit_short.is_none() && d.airtime_limit_long.is_none());
assert!(d.enabled && d.port.is_none());
d.validate().unwrap();
}
#[test]
fn operator_portugal_config_validates() {
// The operator's real device config (2026-08-06).
let s = RNodeRfSettings {
enabled: true,
port: Some("/dev/ttyACM0".into()),
frequency: 869_462_500,
bandwidth: 125_000,
spreading_factor: 8,
coding_rate: 5,
txpower: 14,
airtime_limit_short: Some(25.0),
airtime_limit_long: Some(10.0),
};
s.validate().unwrap();
}
#[test]
fn adoption_preserves_the_operator_portugal_config_exactly() {
// The operator's literal RNS config (2026-08-06). The update must
// adopt these values verbatim — changing a node's applied RF
// settings is forbidden.
let raw = "\
[reticulum]
enable_transport = yes
[interfaces]
[[RNode LoRa Portugal]]
type = RNodeInterface
interface_enabled = true
port = /dev/ttyACM0
frequency = 869462500
bandwidth = 125000
spreadingfactor = 8
codingrate = 5
txpower = 14
airtime_limit_short = 25
airtime_limit_long = 10
";
let s = RNodeRfSettings::parse_rnode_section(raw).expect("section found");
assert!(s.enabled);
assert_eq!(s.port.as_deref(), Some("/dev/ttyACM0"));
assert_eq!(s.frequency, 869_462_500);
assert_eq!(s.bandwidth, 125_000);
assert_eq!(s.spreading_factor, 8);
assert_eq!(s.coding_rate, 5);
assert_eq!(s.txpower, 14);
assert_eq!(s.airtime_limit_short, Some(25.0));
assert_eq!(s.airtime_limit_long, Some(10.0));
s.validate().unwrap();
}
#[test]
fn adoption_ignores_non_rnode_sections_and_absent_config() {
let tcp_only = "\
[interfaces]
[[Reticulum TCP Server]]
type = TCPServerInterface
listen_ip = 127.0.0.1
listen_port = 4242
";
assert!(RNodeRfSettings::parse_rnode_section(tcp_only).is_none());
assert!(RNodeRfSettings::parse_rnode_section("").is_none());
}
#[test]
fn out_of_range_values_are_rejected() {
let base = RNodeRfSettings::default();
for bad in [
RNodeRfSettings { frequency: 100, ..base.clone() },
RNodeRfSettings { bandwidth: 123_456, ..base.clone() },
RNodeRfSettings { spreading_factor: 4, ..base.clone() },
RNodeRfSettings { coding_rate: 9, ..base.clone() },
RNodeRfSettings { txpower: 23, ..base.clone() },
RNodeRfSettings { airtime_limit_short: Some(180.0), ..base.clone() },
RNodeRfSettings { port: Some("ttyACM0".into()), ..base.clone() },
RNodeRfSettings { port: Some("/dev/tty; rm -rf /".into()), ..base.clone() },
] {
assert!(bad.validate().is_err(), "{bad:?} should fail validation");
}
}
}
+74 -5
View File
@@ -847,6 +847,39 @@ impl Server {
if !direct.is_empty() {
let _ = crate::fips::anchors::apply(&direct).await;
}
// A3.10 — endpoint fallback for direct peering. Record
// where currently-connected peers actually are (their
// transport_addr covers LAN, Tailscale, and WAN alike),
// then re-dial the last-known-good endpoint of every
// federation peer whose live paths are gone: not
// connected now, no LAN direct entry this tick. Escala-
// tion order is LAN → last-known-good → anchor tree;
// a stale candidate costs one bounded failed dial.
let connected = crate::fips::service::connected_peer_endpoints().await;
let known =
crate::fips::endpoints::record_connected(&data_dir, &connected).await;
let wanted: Vec<String> = reg
.all_peers()
.await
.iter()
.filter_map(|p| p.fips_npub.clone())
.collect();
let connected_npubs: Vec<String> =
connected.iter().map(|c| c.npub.clone()).collect();
let fallback = crate::fips::endpoints::fallback_anchors(
&known,
&wanted,
&connected_npubs,
&direct,
);
if !fallback.is_empty() {
tracing::info!(
count = fallback.len(),
"dialing last-known-good endpoints for disconnected federation peers"
);
let _ = crate::fips::anchors::apply(&fallback).await;
}
}
let next = if daemon_restarting && fast_retries < MAX_FAST_RETRIES {
@@ -1145,16 +1178,52 @@ fn fips_app_relay_addr(ip: std::net::Ipv6Addr, port: u16) -> SocketAddr {
/// without a daemon restart. Each relay binds to the fips0 ULA only and
/// forwards raw TCP to the same port on IPv4 loopback.
async fn app_port_v6_relay_loop(mut shutdown_rx: tokio::sync::watch::Receiver<bool>) {
use std::collections::HashSet;
let mut bridged: HashSet<u16> = HashSet::new();
use std::collections::HashMap;
let mut bridged: HashMap<u16, tokio::task::JoinHandle<()>> = HashMap::new();
let mut interval = tokio::time::interval(std::time::Duration::from_secs(60));
interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
loop {
tokio::select! {
_ = interval.tick() => {
let Some(fips_ip) = crate::fips::iface::fips0_ula() else { continue };
// This relay is a raw unauthenticated forward from the mesh to
// the app's loopback, so it must refuse two classes of port:
//
// * `auth: gated` — the app gate owns the fips0 ULA for these,
// and bridging one would bypass the login page. Which of the
// two won the bind used to be a race.
// * `auth: local` — host-local BY INTENT. Bridging one makes a
// port reachable from the whole mesh that was deliberately
// never externally reachable: nbxplorer 32838 answered HTTP
// 200 over the mesh with no credential (archi-dev-box
// 2026-08-04) purely because it appeared in the static port
// list below.
//
// Undeclared ports keep today's behaviour — silence is not an
// instruction in either direction, and this relay predates the
// declarations.
let port_map = crate::appgate::identity::build_port_map();
let gate_owned: std::collections::HashSet<u16> = port_map
.gated_ports()
.filter(|g| g.declared)
.map(|g| g.port)
.collect();
for &port in crate::fips::app_ports::APP_LAUNCH_PORTS {
if bridged.contains(&port) {
let withhold = if gate_owned.contains(&port) {
Some("port is now gate-owned")
} else if port_map.is_declared_local(port) {
Some("port is declared auth: local (host-local by intent)")
} else {
None
};
if let Some(reason) = withhold {
if let Some(handle) = bridged.remove(&port) {
handle.abort();
info!(port, reason, "v6 relay released a bridge");
}
continue;
}
if bridged.contains_key(&port) {
continue;
}
// ONLY bridge a port that a running app already answers on
@@ -1181,10 +1250,9 @@ async fn app_port_v6_relay_loop(mut shutdown_rx: tokio::sync::watch::Receiver<bo
// EADDRINUSE = fipsd or another process already answers
// on this mesh address/port, so stay out of the way.
let Ok(listener) = bind_v6_only(addr) else { continue };
bridged.insert(port);
debug!("v6 relay bridging [{fips_ip}]:{port} -> 127.0.0.1:{port}");
let mut rx = shutdown_rx.clone();
tokio::spawn(async move {
let handle = tokio::spawn(async move {
loop {
tokio::select! {
accepted = listener.accept() => {
@@ -1205,6 +1273,7 @@ async fn app_port_v6_relay_loop(mut shutdown_rx: tokio::sync::watch::Receiver<bo
}
}
});
bridged.insert(port, handle);
}
}
_ = shutdown_rx.changed() => return,
+39 -4
View File
@@ -40,12 +40,19 @@ struct Session {
created_at: SystemTime,
last_activity: SystemTime,
session_type: SessionType,
/// What kind of screen this login came from. A TV on the wall must not
/// be signed out for sitting still — nobody is there to type a password
/// back in — while a browser must be.
device_class: crate::settings::session_policy::DeviceClass,
}
#[derive(Clone)]
pub struct SessionStore {
sessions: Arc<RwLock<HashMap<[u8; 32], Session>>>,
persist_path: PathBuf,
/// Where the session policy lives. Held rather than looked up globally
/// so tests can point at a temp dir.
data_dir: PathBuf,
}
/// On-disk representation of a persisted session (only Full sessions, no TOTP secrets).
@@ -67,6 +74,7 @@ impl SessionStore {
Self {
sessions: Arc::new(RwLock::new(sessions)),
persist_path,
data_dir: PathBuf::from("/var/lib/archipelago"),
}
}
@@ -75,9 +83,17 @@ impl SessionStore {
/// machine's real /var/lib/archipelago/sessions.json.
#[cfg(test)]
pub fn new_for_tests(persist_path: PathBuf) -> Self {
// data_dir shares the temp path's parent so a test that writes a
// policy file is honoured, and one that doesn't gets the defaults
// rather than the dev machine's real configuration.
let data_dir = persist_path
.parent()
.map(PathBuf::from)
.unwrap_or_else(|| PathBuf::from("."));
Self {
sessions: Arc::new(RwLock::new(HashMap::new())),
persist_path,
data_dir,
}
}
@@ -120,6 +136,7 @@ impl SessionStore {
created_at,
last_activity,
session_type: SessionType::Full,
device_class: crate::settings::session_policy::DeviceClass::Browser,
},
);
}
@@ -160,6 +177,7 @@ impl SessionStore {
created_at: now,
last_activity: now,
session_type: SessionType::Full,
device_class: crate::settings::session_policy::DeviceClass::Browser,
};
let mut sessions = self.sessions.write().await;
@@ -184,6 +202,10 @@ impl SessionStore {
totp_secret,
attempts: 0,
},
// A half-finished login is always treated as a browser: it lives
// for PENDING_SESSION_TTL either way, and a kiosk exemption on a
// session that has not passed 2FA yet would be the wrong default.
device_class: crate::settings::session_policy::DeviceClass::Browser,
};
self.sessions.write().await.insert(hash, session);
token
@@ -192,19 +214,23 @@ impl SessionStore {
/// Validate a full session token. Returns true if the session exists and hasn't expired.
/// Updates last_activity on successful validation (inactivity-based expiry).
pub async fn validate(&self, token: &str) -> bool {
let policy = self.policy().await;
let hash = hash_token(token);
let mut sessions = self.sessions.write().await;
if let Some(session) = sessions.get_mut(&hash) {
if !matches!(session.session_type, SessionType::Full) {
return false;
}
if session
let idle = session
.last_activity
.elapsed()
.unwrap_or_default()
.as_secs()
>= FULL_SESSION_TTL
{
.as_secs();
let age = session.created_at.elapsed().unwrap_or_default().as_secs();
// Both limits, not just idleness: the dashboard polls, so an
// idle timeout alone would never fire on an open tab. The
// absolute cap is what actually guarantees a login ends.
if policy.is_expired(session.device_class, age, idle) {
sessions.remove(&hash);
return false;
}
@@ -215,6 +241,13 @@ impl SessionStore {
}
}
/// The operator's session policy, re-read from disk rather than cached
/// for the process lifetime so a change in Settings takes effect on the
/// next request instead of the next restart.
pub async fn policy(&self) -> crate::settings::session_policy::SessionPolicy {
crate::settings::session_policy::load(&self.data_dir).await
}
/// Get the TOTP secret from a pending session. Returns None if not a valid pending session.
/// Increments the attempt counter.
pub async fn get_pending_secret(&self, token: &str) -> Option<Vec<u8>> {
@@ -259,6 +292,7 @@ impl SessionStore {
created_at: now,
last_activity: now,
session_type: SessionType::Full,
device_class: crate::settings::session_policy::DeviceClass::Browser,
},
);
Self::save_to_disk(&sessions, &self.persist_path).await;
@@ -300,6 +334,7 @@ impl SessionStore {
created_at: now,
last_activity: now,
session_type: SessionType::Full,
device_class: crate::settings::session_policy::DeviceClass::Browser,
},
);
Self::save_to_disk(&sessions, &self.persist_path).await;
+1
View File
@@ -4,4 +4,5 @@
//! call sites (deep in the transport / RPC / ingest stacks) don't need
//! to thread a data_dir or Arc through the entire call graph.
pub mod session_policy;
pub mod transport;
@@ -0,0 +1,200 @@
//! How long a login lasts, and who gets to say so.
//!
//! # Why this is configurable rather than a constant
//!
//! There is no single correct session lifetime. The same node can be a
//! wall-mounted TV in a living room that must never ask for a password
//! mid-film, and a wallet holding real funds where PCI DSS-style guidance
//! says fifteen minutes. Both are legitimate; the operator knows which one
//! this node is and we do not.
//!
//! # The two tokens
//!
//! * **Session token** — short-lived, refreshed silently on every
//! authenticated request. This is what the browser sends; if it leaks, it
//! is useful only until [`SessionPolicy::idle_timeout_secs`] of silence.
//! * **Login (remember) token** — long-lived, and its *only* power is to
//! mint a fresh session token. Kept separate so raising the convenience
//! knob does not put a 30-day bearer credential on every request.
//!
//! Raising the idle timeout therefore does not weaken the credential that
//! actually travels; it only changes how long a quiet tab stays usable.
//!
//! # Why an absolute cap exists at all
//!
//! Idle timeout alone can be defeated by any page that polls — the
//! dashboard polls constantly, so an idle timeout would never fire while a
//! tab is open. The absolute cap is what guarantees a login eventually
//! ends, which is the property an auditor actually asks about.
use serde::{Deserialize, Serialize};
use std::path::Path;
const FILE_PATH: &str = "settings/session_policy.json";
/// Bounds. A setting that can be made meaningless is not a setting, and one
/// that can lock the operator out of their own node is a footgun.
const MIN_IDLE_SECS: u64 = 60;
const MAX_IDLE_SECS: u64 = 90 * 24 * 3600;
const MIN_ABSOLUTE_SECS: u64 = 300;
const MAX_ABSOLUTE_SECS: u64 = 365 * 24 * 3600;
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "kebab-case")]
pub enum DeviceClass {
/// Ordinary browser on a phone or laptop. Policy applies as configured.
Browser,
/// A screen nobody logs into — a wall-mounted dashboard or TV. Being
/// signed out mid-view is the failure mode here, not a stale session:
/// the device is physically in the home, and there is no keyboard to
/// re-authenticate with. Exempt from the idle timeout, still subject to
/// the absolute cap so a stolen box does not stay authenticated forever.
Kiosk,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
pub struct SessionPolicy {
/// Silence after which a session token stops validating.
pub idle_timeout_secs: u64,
/// Hard ceiling from login, regardless of activity. `None` = no cap.
pub absolute_timeout_secs: Option<u64>,
/// Re-prompt for the password before actions that move money, however
/// fresh the session is. Independent of the timeouts on purpose: it is
/// the control that matters when funds are involved, and it costs the
/// operator nothing the rest of the time.
pub reauth_for_funds: bool,
}
impl Default for SessionPolicy {
fn default() -> Self {
Self {
// A day of silence, matching the previous hard-coded constant so
// existing nodes see no behaviour change until someone chooses.
idle_timeout_secs: 86_400,
// 30 days, aligned with the login token's own lifetime: a
// session that outlived the token which could refresh it would
// be an oddity.
absolute_timeout_secs: Some(30 * 24 * 3600),
reauth_for_funds: true,
}
}
}
impl SessionPolicy {
/// Clamp to the supported range. Applied on load as well as on save, so
/// a hand-edited file cannot disable expiry by writing `0`.
pub fn sanitized(mut self) -> Self {
self.idle_timeout_secs = self.idle_timeout_secs.clamp(MIN_IDLE_SECS, MAX_IDLE_SECS);
self.absolute_timeout_secs = self
.absolute_timeout_secs
.map(|v| v.clamp(MIN_ABSOLUTE_SECS, MAX_ABSOLUTE_SECS))
// An absolute cap below the idle timeout would expire sessions
// while they are still active, which reads as random logouts.
.map(|v| v.max(self.idle_timeout_secs));
self
}
/// Idle timeout for a given device, or `None` when idleness is not a
/// reason to expire (kiosk screens).
pub fn idle_timeout_for(&self, class: DeviceClass) -> Option<u64> {
match class {
DeviceClass::Browser => Some(self.idle_timeout_secs),
DeviceClass::Kiosk => None,
}
}
/// Has a session expired? `age` is time since login, `idle` since last
/// use. Both are checked because either alone is insufficient: idle
/// never fires on a polling dashboard, and absolute alone leaves a
/// forgotten tab usable for a month.
pub fn is_expired(&self, class: DeviceClass, age_secs: u64, idle_secs: u64) -> bool {
if let Some(limit) = self.absolute_timeout_secs {
if age_secs >= limit {
return true;
}
}
match self.idle_timeout_for(class) {
Some(limit) => idle_secs >= limit,
None => false,
}
}
}
pub async fn load(data_dir: &Path) -> SessionPolicy {
let path = data_dir.join(FILE_PATH);
match tokio::fs::read(&path).await {
Ok(bytes) => serde_json::from_slice::<SessionPolicy>(&bytes)
.map(SessionPolicy::sanitized)
.unwrap_or_else(|e| {
tracing::warn!(error = %e, "session policy unreadable; using defaults");
SessionPolicy::default()
}),
Err(_) => SessionPolicy::default(),
}
}
pub async fn save(data_dir: &Path, policy: SessionPolicy) -> anyhow::Result<SessionPolicy> {
let policy = policy.sanitized();
let path = data_dir.join(FILE_PATH);
if let Some(parent) = path.parent() {
tokio::fs::create_dir_all(parent).await?;
}
let tmp = path.with_extension("json.tmp");
tokio::fs::write(&tmp, serde_json::to_vec_pretty(&policy)?).await?;
tokio::fs::rename(&tmp, &path).await?;
Ok(policy)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn defaults_match_the_previous_hardcoded_behaviour() {
let p = SessionPolicy::default();
assert_eq!(p.idle_timeout_secs, 86_400);
assert!(p.reauth_for_funds);
}
#[test]
fn expiry_cannot_be_disabled_by_hand_editing_the_file() {
let p = SessionPolicy {
idle_timeout_secs: 0,
absolute_timeout_secs: Some(0),
reauth_for_funds: false,
}
.sanitized();
assert!(p.idle_timeout_secs >= MIN_IDLE_SECS);
assert!(p.absolute_timeout_secs.unwrap() >= MIN_ABSOLUTE_SECS);
}
#[test]
fn absolute_cap_is_never_shorter_than_idle() {
// Otherwise a session dies while actively in use, which the operator
// experiences as being logged out at random.
let p = SessionPolicy {
idle_timeout_secs: 7 * 24 * 3600,
absolute_timeout_secs: Some(3600),
reauth_for_funds: true,
}
.sanitized();
assert_eq!(p.absolute_timeout_secs.unwrap(), p.idle_timeout_secs);
}
#[test]
fn a_kiosk_never_expires_from_idleness_but_still_has_a_ceiling() {
let p = SessionPolicy::default();
let a_week = 7 * 24 * 3600;
assert!(!p.is_expired(DeviceClass::Kiosk, 60, a_week));
assert!(p.is_expired(DeviceClass::Browser, 60, a_week));
// The absolute cap still applies to the TV.
assert!(p.is_expired(DeviceClass::Kiosk, 31 * 24 * 3600, 0));
}
#[test]
fn a_polling_dashboard_still_eventually_expires() {
// idle never grows because the page polls; only the cap saves us.
let p = SessionPolicy::default();
assert!(p.is_expired(DeviceClass::Browser, 30 * 24 * 3600, 0));
}
}
+18 -3
View File
@@ -16,13 +16,28 @@ use ed25519_dalek::VerifyingKey;
/// Hex of the pinned Ed25519 release-root public key (32 bytes / 64 hex chars).
///
/// Pinned 2026-07-02 from the release-root signing ceremony
/// (signer did:key:z6MkkidEnEpo6qHMCNSZoNKWtvQvxq3whnaME9wGgEFhq7ur). The
/// ROTATED 2026-08-04 to did:key:z6Mkfu5LT8d4DjETtrkATvHh9Dvcbnr7zBCUwfau8Sw7DLWT.
///
/// The previous root (z6Mkkid…q7ur, pinned 2026-07-02) was exposed in a chat
/// transcript and is treated as compromised.
///
/// Rotation is ORDERING-CRITICAL. Nodes pin the OLD key, so the release that
/// carries this change must itself be signed with the OLD key — that is the
/// only signature a node running the previous binary will accept. Only the
/// release AFTER it may be signed with the new key. Signing the rotation
/// release with the new key makes every node reject it and ends OTA
/// fleet-wide, recoverable only by touching each node by hand.
///
/// Verified before pinning: this hex and the did:key above are the same
/// keypair (the did:key encodes exactly these 32 bytes), checked with a
/// decoder round-tripped against the previous known-good pair. An earlier
/// candidate hex was rejected because it did not match the stated DID.
/// The
/// corresponding mnemonic is held offline by the publisher — see
/// `docs/workstream-b-signing-runbook.md`. Regenerate/verify with:
/// `RELEASE_MASTER_MNEMONIC=… archipelago ceremony pubkey`.
pub const RELEASE_ROOT_PUBKEY_HEX: Option<&str> =
Some("5d15cbee8a108f7dd288c02d29a1d9d71f198acc99186aad8008b4f28d469951");
Some("1578adccf137024159dd936f44a56e8869ac7775785962f7e92e2faf2c034418");
const ENV_OVERRIDE: &str = "ARCHY_RELEASE_ROOT_PUBKEY";
+48 -13
View File
@@ -74,7 +74,20 @@ fn is_newer(candidate: &str, current: &str) -> bool {
}
}
/// Primary OTA origin. Named host over TLS rather than the bare IP it used
/// to be: the IP pinned the fleet to one machine and one plaintext port, so
/// moving or fronting the origin meant an OTA to change where OTAs come
/// from — the one update you cannot ship if the origin is unreachable. The
/// signature is what establishes trust (see `trust::anchor`), not the
/// transport, but HTTPS also stops a network observer seeing which version
/// a node runs.
const DEFAULT_UPDATE_MANIFEST_URL: &str =
"https://source.archipelago-foundation.org/lfg2025/archy/raw/branch/main/releases/manifest.json";
/// The previous IP-based origin, kept as an automatic fallback so a node
/// whose DNS or TLS is broken still updates. Dropped from the mirror list
/// once the fleet has moved.
const LEGACY_UPDATE_MANIFEST_URL: &str =
"http://146.59.87.168:3000/lfg2025/archy/raw/branch/main/releases/manifest.json";
const UPDATE_STATE_FILE: &str = "update_state.json";
const UPDATE_MIRRORS_FILE: &str = "update-mirrors.json";
@@ -113,10 +126,19 @@ fn mirrors_path(data_dir: &Path) -> std::path::PathBuf {
}
fn default_mirrors() -> Vec<UpdateMirror> {
vec![UpdateMirror {
url: DEFAULT_UPDATE_MANIFEST_URL.to_string(),
label: "Server 1 (OVH)".to_string(),
}]
vec![
UpdateMirror {
url: DEFAULT_UPDATE_MANIFEST_URL.to_string(),
label: "Archipelago Foundation".to_string(),
},
// Fallback, tried only if the named origin fails: a node whose DNS
// or clock is wrong (both break TLS) must still be able to update
// itself, and the signature check is what makes either source safe.
UpdateMirror {
url: LEGACY_UPDATE_MANIFEST_URL.to_string(),
label: "Direct (fallback)".to_string(),
},
]
}
/// Load the operator-configured mirror list. Returns defaults if the
@@ -186,15 +208,18 @@ fn force_ovh_update_primary(list: &mut Vec<UpdateMirror>) {
}
for mirror in list.iter_mut() {
if mirror.url == DEFAULT_UPDATE_MANIFEST_URL {
mirror.label = "Server 1 (OVH)".to_string();
mirror.label = "Archipelago Foundation".to_string();
} else if mirror.url == LEGACY_UPDATE_MANIFEST_URL {
mirror.label = "Direct (fallback)".to_string();
}
}
list.sort_by_key(|m| {
if m.url == DEFAULT_UPDATE_MANIFEST_URL {
0
} else {
1
}
// Named origin first, its IP fallback second, anything the operator
// added after that. Ordering matters: the list is tried in order, so a
// stale entry sitting first costs a timeout on every check.
list.sort_by_key(|m| match m.url.as_str() {
u if u == DEFAULT_UPDATE_MANIFEST_URL => 0,
u if u == LEGACY_UPDATE_MANIFEST_URL => 1,
_ => 2,
});
}
@@ -2373,8 +2398,18 @@ mod tests {
async fn test_load_mirrors_returns_defaults_when_absent() {
let dir = tempfile::tempdir().unwrap();
let list = load_mirrors(dir.path()).await.unwrap();
assert_eq!(list.len(), 1);
assert!(list[0].url.contains("146.59.87.168"));
// The named origin leads, its IP fallback follows. A node with broken
// DNS or a wrong clock (both break TLS) must still have a way to
// update; the signature is what makes either source trustworthy.
assert_eq!(list.len(), 2);
assert!(
list[0]
.url
.starts_with("https://source.archipelago-foundation.org/"),
"the named origin must be primary, got {}",
list[0].url
);
assert!(list[1].url.contains("146.59.87.168"));
assert!(
!list.iter().any(|m| m.url.contains("git.tx1138.com")),
"tx1138 was retired as a release server and must not be a default mirror"
+12 -2
View File
@@ -1040,6 +1040,12 @@ pub async fn receive_token(data_dir: &Path, token_str: &str) -> Result<u64> {
let mut wallet = load_wallet(data_dir).await?;
let mut received_total = 0u64;
// MintClient translates the mint's NUT error code into plain language and
// puts it at the top of the error chain (see `mint_error` in
// mint_client.rs); `{}` surfaces that, `{:#}` keeps the raw status/body
// for the log. Remember the last one so a total failure can tell the user
// *why* instead of just "nothing was received".
let mut last_reason: Option<String> = None;
// Swap proofs at each mint
for entry in &token.token {
@@ -1051,14 +1057,18 @@ pub async fn receive_token(data_dir: &Path, token_str: &str) -> Result<u64> {
received_total += amount;
}
Err(e) => {
warn!("Failed to swap proofs from mint {}: {}", entry.mint, e);
warn!("Failed to swap proofs from mint {}: {:#}", entry.mint, e);
last_reason = Some(e.to_string());
// Continue with other mints if any
}
}
}
if received_total == 0 {
anyhow::bail!("Failed to receive any proofs from token");
match last_reason {
Some(reason) => anyhow::bail!("Could not receive this ecash: {}", reason),
None => anyhow::bail!("Failed to receive any proofs from token"),
}
}
wallet.record_tx(
+71 -5
View File
@@ -59,6 +59,72 @@ pub struct MintResult {
pub proofs: Vec<Proof>,
}
/// Translate a Cashu NUT "transaction validation" error code into plain
/// language a wallet user can act on. Mints respond to a rejected request
/// with `{"code": N, "detail": "..."}`; `detail` is implementation-defined
/// free text, but `code` is the stable identifier from the spec
/// (https://github.com/cashubtc/nuts/blob/main/error_codes.md). Covers the
/// 10001-11017 "proof/transaction validation" range plus the 12001-12003
/// keyset codes shared by NUT-02/03/04/05 — the codes a swap/melt/mint call
/// can actually hit. Returns `None` for anything else (e.g. Lightning/quote
/// codes in the 20000s) so the caller falls back to the mint's own `detail`.
fn describe_mint_error_code(code: i64) -> Option<&'static str> {
Some(match code {
10001 => "The mint rejected these coins as invalid.",
11001 => "This ecash has already been redeemed — it can't be claimed twice.",
11002 => "This ecash is already being redeemed elsewhere — try again in a moment.",
11003 => "The mint already issued new coins for this exact request — there's nothing left to redeem.",
11004 => "This request is still being processed by the mint — try again in a moment.",
11005 => "The token's amounts don't add up (inputs don't match outputs) — it may be corrupt.",
11006 => "That amount is outside the range this mint allows.",
11007 => "This token contains duplicate coins — it may be corrupt or already used.",
11008 => "The mint rejected this as a duplicate request.",
11009 | 11010 => "This token mixes incompatible currency units — the mint rejected it.",
11011 => "That Lightning invoice has no amount, which isn't supported here.",
11012 => "The amount requested doesn't match the Lightning invoice.",
11013 => "The mint doesn't support this currency unit.",
11014 | 11015 => "This token has too many coins for the mint to process in one request.",
11016 => "Duplicate quote IDs were sent in this request.",
11017 => "Too many items were sent in a single request.",
12001 => "The mint no longer recognizes the keyset that signed this token.",
12002 => "The mint's signing key for this token is inactive.",
12003 => "The mint's signing key for this token has expired.",
_ => return None,
})
}
/// Parse a mint's error body (`{"code": N, "detail": "..."}`) and pick the
/// best user-facing message: the plain-language translation when we know the
/// code, otherwise the mint's own `detail` text, otherwise the raw body.
fn describe_mint_error_body(status: reqwest::StatusCode, body: &str) -> String {
let parsed: Option<serde_json::Value> = serde_json::from_str(body).ok();
let code = parsed
.as_ref()
.and_then(|v| v.get("code"))
.and_then(|c| c.as_i64());
let detail = parsed
.as_ref()
.and_then(|v| v.get("detail"))
.and_then(|d| d.as_str());
if let Some(friendly) = code.and_then(describe_mint_error_code) {
return friendly.to_string();
}
match detail {
Some(d) if !d.is_empty() => d.to_string(),
_ => format!("mint returned {} with no further detail", status),
}
}
/// Build the error for a failed mint HTTP call: `op` + status + raw body as
/// the technical cause (visible via `{:#}` in logs), with the plain-language
/// translation layered on top via `.context()` so `{}` — what reaches the
/// wallet user — shows something actionable instead of raw mint JSON.
fn mint_error(op: &str, status: reqwest::StatusCode, body: &str) -> anyhow::Error {
let friendly = describe_mint_error_body(status, body);
anyhow::anyhow!("{} failed ({}): {}", op, status, body).context(friendly)
}
/// HTTP client for a single Cashu mint.
pub struct MintClient {
url: String,
@@ -146,7 +212,7 @@ impl MintClient {
if !res.status().is_success() {
let status = res.status();
let body = res.text().await.unwrap_or_default();
anyhow::bail!("Mint quote failed ({}): {}", status, body);
return Err(mint_error("Mint quote", status, &body));
}
res.json().await.context("Failed to parse mint quote")
@@ -212,7 +278,7 @@ impl MintClient {
if !res.status().is_success() {
let status = res.status();
let body = res.text().await.unwrap_or_default();
anyhow::bail!("Mint tokens failed ({}): {}", status, body);
return Err(mint_error("Minting tokens", status, &body));
}
let body: serde_json::Value = res.json().await.context("Failed to parse mint response")?;
@@ -266,7 +332,7 @@ impl MintClient {
if !res.status().is_success() {
let status = res.status();
let body = res.text().await.unwrap_or_default();
anyhow::bail!("Melt quote failed ({}): {}", status, body);
return Err(mint_error("Melt quote", status, &body));
}
res.json().await.context("Failed to parse melt quote")
@@ -293,7 +359,7 @@ impl MintClient {
if !res.status().is_success() {
let status = res.status();
let body = res.text().await.unwrap_or_default();
anyhow::bail!("Melt failed ({}): {}", status, body);
return Err(mint_error("Melt", status, &body));
}
res.json().await.context("Failed to parse melt response")
@@ -337,7 +403,7 @@ impl MintClient {
if !res.status().is_success() {
let status = res.status();
let body = res.text().await.unwrap_or_default();
anyhow::bail!("Swap failed ({}): {}", status, body);
return Err(mint_error("Swap", status, &body));
}
let body: serde_json::Value = res.json().await.context("Failed to parse swap response")?;