Compare commits

..

No commits in common. "5799c3711121f4a04635d2b1c720ce9089ff3f12" and "8213b0aee391fcb02068ae9bc2839e19a4198462" have entirely different histories.

19 changed files with 20 additions and 1747 deletions

50
core/Cargo.lock generated
View File

@ -84,15 +84,6 @@ version = "1.0.100"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a23eb6b1614318a8071c9b2521f36b424b2c83db5eb3a0fead4a6c0809af6e61"
[[package]]
name = "arbitrary"
version = "1.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c3d036a3c4ab069c7b410a2ce876bd74808d2d0888a82667669f8e783a898bf1"
dependencies = [
"derive_arbitrary",
]
[[package]]
name = "arc-swap"
version = "1.9.1"
@ -169,7 +160,6 @@ dependencies = [
"uuid",
"zbase32",
"zeroize",
"zip",
]
[[package]]
@ -1184,17 +1174,6 @@ version = "0.5.8"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c"
[[package]]
name = "derive_arbitrary"
version = "1.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1e567bd82dcff979e4b03460c307b3cdc9e96fde3d73bed1496d2bc75d9dd62a"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.114",
]
[[package]]
name = "derive_builder"
version = "0.20.2"
@ -6915,37 +6894,8 @@ dependencies = [
"syn 2.0.114",
]
[[package]]
name = "zip"
version = "2.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fabe6324e908f85a1c52063ce7aa26b68dcb7eb6dbc83a2d148403c9bc3eba50"
dependencies = [
"arbitrary",
"crc32fast",
"crossbeam-utils",
"displaydoc",
"flate2",
"indexmap",
"memchr",
"thiserror 2.0.18",
"zopfli",
]
[[package]]
name = "zmij"
version = "1.0.16"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dfcd145825aace48cff44a8844de64bf75feec3080e0aa5cdbde72961ae51a65"
[[package]]
name = "zopfli"
version = "0.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f05cd8797d63865425ff89b5c4a48804f35ba0ce8d125800027ad6017d2b5249"
dependencies = [
"bumpalo",
"crc32fast",
"log",
"simd-adler32",
]

View File

@ -105,10 +105,6 @@ bytes = "1"
# Mesh networking (Meshcore serial protocol over USB LoRa radios)
serial2-tokio = "0.1"
# LoRa radio firmware flashing: Meshtastic ships per-board images inside a
# per-platform release zip (see mesh/flash.rs).
zip = { version = "2", default-features = false, features = ["deflate"] }
# Double Ratchet key derivation (Phase 3: encrypted mesh messaging)
hkdf = "0.12.4"

View File

@ -387,10 +387,6 @@ impl RpcHandler {
// Mesh networking (Meshcore LoRa)
"mesh.status" => self.handle_mesh_status().await,
"mesh.probe-device" => self.handle_mesh_probe_device(params).await,
"mesh.flash-list-firmware" => self.handle_mesh_flash_list_firmware(params).await,
"mesh.flash-device" => self.handle_mesh_flash_device(params).await,
"mesh.flash-status" => self.handle_mesh_flash_status().await,
"mesh.flash-cancel" => self.handle_mesh_flash_cancel().await,
"mesh.peers" => self.handle_mesh_peers().await,
"mesh.messages" => self.handle_mesh_messages(params).await,
"mesh.debug-dump" => self.handle_mesh_debug_dump().await,

View File

@ -1,132 +0,0 @@
use super::super::RpcHandler;
use crate::mesh;
use crate::mesh::flash::{self, FlashBoard, FlashJobStatus};
use crate::mesh::types::DeviceType;
use anyhow::Result;
fn parse_family(s: &str) -> Result<DeviceType> {
match s.trim().to_lowercase().as_str() {
"meshcore" => Ok(DeviceType::Meshcore),
"meshtastic" => Ok(DeviceType::Meshtastic),
"reticulum" | "rnode" => Ok(DeviceType::Reticulum),
other => anyhow::bail!("Unknown firmware family: {other} (expected meshcore|meshtastic|reticulum)"),
}
}
fn parse_board(s: &str) -> Result<FlashBoard> {
match s.trim().to_lowercase().as_str() {
"heltec-v3" | "heltec_v3" | "heltecv3" => Ok(FlashBoard::HeltecV3),
"heltec-v4" | "heltec_v4" | "heltecv4" => Ok(FlashBoard::HeltecV4),
other => anyhow::bail!("Unknown board: {other} (expected heltec-v3|heltec-v4)"),
}
}
impl RpcHandler {
/// mesh.flash-list-firmware — resolve the available firmware version(s)
/// for a given family. v1 only ever surfaces "latest".
pub(in crate::api::rpc) async fn handle_mesh_flash_list_firmware(
&self,
params: Option<serde_json::Value>,
) -> Result<serde_json::Value> {
let family = params
.as_ref()
.and_then(|p| p.get("family"))
.and_then(|v| v.as_str())
.ok_or_else(|| anyhow::anyhow!("Missing family"))?;
let family = parse_family(family)?;
let versions = flash::list_firmware(family).await?;
Ok(serde_json::json!({ "versions": versions }))
}
/// mesh.flash-device — erase and reflash a detected LoRa radio with the
/// latest firmware for the given family, defaulting to a full chip
/// erase before write. `board` is optional: if the port's USB vid:pid
/// unambiguously resolves to a known board, that's used; otherwise the
/// caller must supply it explicitly (see `flash::resolve_flash_board`'s
/// doc comment on why we refuse to guess).
pub(in crate::api::rpc) async fn handle_mesh_flash_device(
&self,
params: Option<serde_json::Value>,
) -> Result<serde_json::Value> {
let path = params
.as_ref()
.and_then(|p| p.get("path"))
.and_then(|v| v.as_str())
.ok_or_else(|| anyhow::anyhow!("Missing path"))?
.to_string();
let family = params
.as_ref()
.and_then(|p| p.get("family"))
.and_then(|v| v.as_str())
.ok_or_else(|| anyhow::anyhow!("Missing family"))?;
let family = parse_family(family)?;
let detected = mesh::detect_devices().await;
anyhow::ensure!(
detected.iter().any(|d| d == &path),
"{path} is not a detected mesh-radio candidate port"
);
let board = match params
.as_ref()
.and_then(|p| p.get("board"))
.and_then(|v| v.as_str())
{
Some(explicit) => parse_board(explicit)?,
None => {
let info = mesh::detect_devices_info()
.await
.into_iter()
.find(|d| d.path == path);
info.as_ref()
.and_then(flash::resolve_flash_board)
.ok_or_else(|| {
anyhow::anyhow!(
"Could not auto-detect the board on {path} — specify board explicitly"
)
})?
}
};
flash::start_flash_job(
&self.flash_job,
&self.mesh_service_arc(),
self.config.data_dir.clone(),
path,
board,
family,
)
.await?;
Ok(serde_json::json!({ "started": true }))
}
/// mesh.flash-status — poll the current (or most recent) flash job.
pub(in crate::api::rpc) async fn handle_mesh_flash_status(&self) -> Result<serde_json::Value> {
let job = self.flash_job.read().await;
match job.as_ref() {
Some(j) => {
let status: FlashJobStatus = j.snapshot().await;
let mut value = serde_json::to_value(&status)?;
if let Some(obj) = value.as_object_mut() {
obj.insert("active".into(), (!status.done).into());
}
Ok(value)
}
None => Ok(serde_json::json!({ "active": false })),
}
}
/// mesh.flash-cancel — best-effort; only honored before erase/write has
/// started (see `FlashJob::cancel`'s doc comment).
pub(in crate::api::rpc) async fn handle_mesh_flash_cancel(&self) -> Result<serde_json::Value> {
let job = self.flash_job.read().await;
match job.as_ref() {
Some(j) => {
j.cancel().await?;
Ok(serde_json::json!({ "cancelled": true }))
}
None => anyhow::bail!("No flash job in progress"),
}
}
}

View File

@ -1,6 +1,5 @@
mod assistant;
mod bitcoin_ops;
mod flash;
mod messaging;
mod safety;
mod status;

View File

@ -88,9 +88,6 @@ pub struct RpcHandler {
endpoint_rate_limiter: EndpointRateLimiter,
response_cache: ResponseCache,
mesh_service: Arc<tokio::sync::RwLock<Option<crate::mesh::MeshService>>>,
/// LoRa radio firmware-flash job state, sibling to `mesh_service` — one
/// job at a time, since flashing needs exclusive access to the port.
flash_job: crate::mesh::flash::FlashJobHandle,
transport_router: Arc<tokio::sync::RwLock<Option<Arc<crate::transport::TransportRouter>>>>,
/// Shared content-addressed blob store. Set by ApiHandler after construction
/// so mesh.send-content / mesh.fetch-content RPCs can reach it without a
@ -162,7 +159,6 @@ impl RpcHandler {
endpoint_rate_limiter,
response_cache: ResponseCache::new(5),
mesh_service: Arc::new(tokio::sync::RwLock::new(None)),
flash_job: crate::mesh::flash::new_job_handle(),
transport_router: Arc::new(tokio::sync::RwLock::new(None)),
blob_store: Arc::new(tokio::sync::RwLock::new(None)),
self_pubkey_hex: Arc::new(tokio::sync::RwLock::new(None)),

View File

@ -1,930 +0,0 @@
// WIP mesh/transport protocol — suppress dead code warnings
#![allow(dead_code)]
//! Firmware flashing for LoRa mesh radios — Heltec V3/V4 in v1, across all
//! three firmware families the mesh module already knows how to detect (see
//! `mesh::types::DeviceType`). Firmware is always fetched from upstream at
//! flash time (never bundled/pinned in the repo), and every flash defaults
//! to a full chip erase before write.
//!
//! MeshCore and Meshtastic are flashed the same way: download a released
//! image, `esptool erase_flash`, then `esptool write_flash 0x0 <image>`.
//! Reticulum/RNode is different: `archy-rnodeconf --autoinstall` owns the
//! whole fetch+erase+flash+EEPROM-bootstrap sequence itself (confirmed live
//! via `archy-rnodeconf --help` — there is no raw esptool path exposed for
//! this family, so we deliberately don't resolve a firmware URL ourselves
//! for Reticulum; rnodeconf already knows how).
use super::serial::DetectedDeviceInfo;
use super::types::DeviceType;
use super::MeshService;
use anyhow::{Context, Result};
use regex::Regex;
use serde::Serialize;
use std::path::{Path, PathBuf};
use std::process::Stdio;
use std::sync::{Arc, OnceLock};
use tokio::io::{AsyncBufReadExt, AsyncWriteExt, BufReader};
use tokio::process::Command;
use tokio::sync::RwLock;
use tracing::{info, warn};
/// Boards supported for v1. Both are ESP32-S3 (a single `--chip esp32s3`
/// esptool target covers both), but ship different USB identities and
/// different per-board firmware assets upstream.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
#[serde(rename_all = "lowercase")]
pub enum FlashBoard {
HeltecV3,
HeltecV4,
}
impl FlashBoard {
/// Meshtastic's board id (matches the release manifest's `board` field
/// and its per-board asset naming, e.g. `firmware-heltec-v3-<ver>.factory.bin`).
fn meshtastic_id(self) -> &'static str {
match self {
Self::HeltecV3 => "heltec-v3",
Self::HeltecV4 => "heltec-v4",
}
}
}
/// Map a detected USB vid:pid to a known flashable board, using the same
/// table as `image-recipe/configs/99-mesh-radio.rules`. CP2102 (10c4:ea60)
/// is confirmed there as Heltec V3's USB-UART bridge chip, and is safe to
/// auto-match since that vid:pid is bridge-chip-specific.
///
/// Heltec V4 is NOT auto-matchable and deliberately has no entry here: it
/// was confirmed live (real hardware, 2026-07-23) to use the ESP32-S3's
/// built-in native-USB JTAG/serial peripheral, reporting vid:pid 303a:1001
/// with product string "USB JTAG/serial debug unit" — that descriptor is
/// baked into the chip's ROM and is IDENTICAL across every ESP32-S3 board
/// with native USB enabled, not just Heltec V4. Adding `303a:1001 =>
/// HeltecV4` here would silently misidentify any other native-USB ESP32-S3
/// board (a T3-S3, a bare devkit, etc.) as a V4 and risk writing the wrong
/// board's image. Callers (the RPC layer / frontend) must let the user pick
/// the board manually whenever this returns `None`.
pub fn resolve_flash_board(info: &DetectedDeviceInfo) -> Option<FlashBoard> {
match (info.vid.as_deref(), info.pid.as_deref()) {
(Some("10c4"), Some("ea60")) => Some(FlashBoard::HeltecV3),
_ => None,
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
#[serde(rename_all = "lowercase")]
pub enum FlashStage {
Downloading,
Erasing,
Writing,
Autoinstalling,
Done,
Failed,
}
#[derive(Debug, Clone, Serialize)]
pub struct FlashJobStatus {
pub board: FlashBoard,
pub family: DeviceType,
pub path: String,
pub stage: FlashStage,
pub percent: Option<u8>,
pub log_tail: Vec<String>,
pub done: bool,
pub error: Option<String>,
}
const LOG_TAIL_MAX: usize = 200;
/// How long to wait after a successful flash before resuming the mesh
/// listener, so the board finishes its own post-flash boot/reset before we
/// start opening the port (which itself toggles DTR/RTS) again.
const POST_FLASH_SETTLE_DELAY: std::time::Duration = std::time::Duration::from_secs(5);
/// Absolute ceiling on a whole flash job (download + erase + write, or
/// autoinstall), regardless of what it's doing internally. Last-resort
/// safety net so a hang anywhere can't wedge the single-flash-job guard
/// forever — generous enough to never trigger on a legitimately slow
/// multi-hundred-MB transfer.
const MAX_JOB_DURATION: std::time::Duration = std::time::Duration::from_secs(15 * 60);
/// How long to wait for MeshService::stop() to release the serial port
/// before giving up. Confirmed live 2026-07-23: the listener's own
/// reconnect/multi-candidate-probe loop doesn't check its shutdown signal
/// between candidates, so stop() can take a while (or, if the loop is
/// wedged, never return) — 20s comfortably covers a normal handshake-probe
/// cycle without leaving a flash request hanging indefinitely if the
/// listener genuinely won't let go.
const STOP_LISTENER_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
/// Live state for the one flash job that can run at a time. A single global
/// slot is sufficient because flashing needs exclusive serial access to the
/// one port being flashed — there is no meaningful concept of two concurrent
/// flash jobs on this node.
pub struct FlashJob {
status: RwLock<FlashJobStatus>,
/// Set once the background task is spawned. Only used while `stage` is
/// still `Downloading` — an interrupted erase/write can leave the chip
/// in a worse state than either finished or unstarted, so cancellation
/// is refused once erase begins (see `cancel()`).
abort_handle: RwLock<Option<tokio::task::AbortHandle>>,
}
impl FlashJob {
fn new(board: FlashBoard, family: DeviceType, path: String) -> Arc<Self> {
Arc::new(Self {
abort_handle: RwLock::new(None),
status: RwLock::new(FlashJobStatus {
board,
family,
path,
stage: FlashStage::Downloading,
percent: None,
log_tail: Vec::new(),
done: false,
error: None,
}),
})
}
pub async fn snapshot(&self) -> FlashJobStatus {
self.status.read().await.clone()
}
async fn set_stage(&self, stage: FlashStage) {
let mut s = self.status.write().await;
s.stage = stage;
s.percent = None;
}
async fn set_percent(&self, percent: u8) {
self.status.write().await.percent = Some(percent.min(100));
}
async fn push_log(&self, line: impl Into<String>) {
let mut s = self.status.write().await;
s.log_tail.push(line.into());
let overflow = s.log_tail.len().saturating_sub(LOG_TAIL_MAX);
if overflow > 0 {
s.log_tail.drain(0..overflow);
}
}
async fn fail(&self, err: &anyhow::Error) {
let mut s = self.status.write().await;
s.stage = FlashStage::Failed;
s.error = Some(format!("{err:#}"));
s.done = true;
}
async fn finish(&self) {
let mut s = self.status.write().await;
s.stage = FlashStage::Done;
s.done = true;
}
/// Best-effort cancel: only honored before erase/write/autoinstall has
/// started (i.e. still in `Downloading`). Once a stage that touches the
/// chip begins, this refuses — interrupting an erase or write can leave
/// the flash in a state worse than either finished or unstarted.
pub async fn cancel(&self) -> Result<()> {
let mut s = self.status.write().await;
if s.done {
anyhow::bail!("Flash job already finished");
}
if s.stage != FlashStage::Downloading {
anyhow::bail!(
"Cannot cancel once {:?} has started — let it finish or fail on its own",
s.stage
);
}
if let Some(handle) = self.abort_handle.write().await.take() {
handle.abort();
}
s.stage = FlashStage::Failed;
s.error = Some("Cancelled by user".to_string());
s.done = true;
Ok(())
}
}
/// Shared handle held by `RpcHandler`, sibling to `mesh_service`.
pub type FlashJobHandle = Arc<RwLock<Option<Arc<FlashJob>>>>;
pub fn new_job_handle() -> FlashJobHandle {
Arc::new(RwLock::new(None))
}
fn firmware_cache_dir(data_dir: &Path) -> PathBuf {
data_dir.join("mesh").join("firmware-cache")
}
/// No blanket `.timeout()` here on purpose: reqwest's request timeout covers
/// the *entire* request including streaming the response body, which would
/// kill a legitimate large download partway through (Meshtastic's esp32s3
/// zip is ~170MB) — not just a hung connection. `download_to_file` instead
/// applies a per-chunk stall timeout, and metadata calls (small JSON
/// responses) get their own short timeout at the call site.
fn github_client() -> Result<reqwest::Client> {
reqwest::Client::builder()
.user_agent("archipelago-mesh-flash")
.connect_timeout(std::time::Duration::from_secs(10))
.build()
.context("Failed to build HTTP client")
}
/// Applied per-chunk while streaming a firmware download — if the transfer
/// stalls (no bytes for this long) it's treated as a failure, but a slow
/// download that's still making progress is never killed just for taking a
/// while.
const DOWNLOAD_STALL_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(30);
/// Applied to metadata calls (GitHub release JSON) — these are small
/// responses with no reason to ever take this long.
const METADATA_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
/// Resolve what firmware is available for a board+family. v1 only ever
/// offers "latest" — MeshCore/Meshtastic latest GitHub release, or, for
/// Reticulum, "latest" meaning "whatever archy-rnodeconf --autoinstall
/// resolves on its own" (it does its own version checking upstream).
pub async fn list_firmware(family: DeviceType) -> Result<Vec<String>> {
match family {
DeviceType::Reticulum => Ok(vec!["latest".to_string()]),
DeviceType::Meshtastic => {
let client = github_client()?;
let release: GithubRelease = client
.get("https://api.github.com/repos/meshtastic/firmware/releases/latest")
.send()
.await
.context("Fetching Meshtastic release list")?
.error_for_status()
.context("Meshtastic releases API error")?
.json()
.await
.context("Parsing Meshtastic release JSON")?;
Ok(vec![release.tag_name])
}
DeviceType::Meshcore => {
let client = github_client()?;
let release: GithubRelease = client
.get("https://api.github.com/repos/meshcore-dev/MeshCore/releases/latest")
.send()
.await
.context("Fetching MeshCore release list")?
.error_for_status()
.context("MeshCore releases API error")?
.json()
.await
.context("Parsing MeshCore release JSON")?;
Ok(vec![release.tag_name])
}
DeviceType::Unknown => anyhow::bail!("Pick a firmware family before listing versions"),
}
}
#[derive(serde::Deserialize)]
struct GithubAsset {
name: String,
browser_download_url: String,
}
#[derive(serde::Deserialize)]
struct GithubRelease {
tag_name: String,
assets: Vec<GithubAsset>,
}
/// Start a flash job in the background. Returns as soon as the job has been
/// registered and the listener released — callers poll `FlashJobHandle` via
/// `mesh.flash-status` for progress. Only one job may be in flight at a time.
pub async fn start_flash_job(
handle: &FlashJobHandle,
mesh_service: &Arc<RwLock<Option<MeshService>>>,
data_dir: PathBuf,
path: String,
board: FlashBoard,
family: DeviceType,
) -> Result<()> {
{
let existing = handle.read().await;
if let Some(job) = existing.as_ref() {
if !job.snapshot().await.done {
anyhow::bail!("A firmware flash is already in progress on this node");
}
}
}
let job = FlashJob::new(board, family, path.clone());
*handle.write().await = Some(Arc::clone(&job));
let bg_job = Arc::clone(&job);
let bg_service = Arc::clone(mesh_service);
let task = tokio::spawn(async move {
// esptool/archy-rnodeconf need exclusive serial access — release
// the listener's hold on the port before touching it. This USED
// TO run synchronously in start_flash_job before the job was even
// spawned, blocking the RPC call itself on s.stop().await — a real
// 2026-07-23 incident: the mesh listener was mid a multi-candidate
// reconnect/probe sequence that doesn't check its shutdown signal
// between candidates, so stop() never returned. The HTTP request
// timed out client-side ("Operation failed"), while the job
// (already inserted into `handle`) was permanently wedged — nothing
// had been spawned yet to ever mark it done, so every later flash
// attempt failed with "already in progress" until a full restart.
// Now this runs inside the spawned task with its own bounded
// timeout, so the RPC call always returns immediately regardless,
// and a slow-to-stop listener fails the job cleanly instead of
// hanging everything downstream of it forever.
let stop_result = tokio::time::timeout(STOP_LISTENER_TIMEOUT, async {
let mut svc = bg_service.write().await;
if let Some(s) = svc.as_mut() {
s.stop().await;
}
})
.await;
if stop_result.is_err() {
let err = anyhow::anyhow!(
"Mesh listener did not release the serial port within {}s — it may still be mid a reconnect attempt. Try again once mesh.status shows the device idle, or restart the archipelago service if this persists.",
STOP_LISTENER_TIMEOUT.as_secs()
);
bg_job.push_log(format!("ERROR: {err:#}")).await;
bg_job.fail(&err).await;
return;
}
// Outer ceiling on top of run_flash's own internal timeouts —
// belt-and-suspenders so that no future hang (network, subprocess,
// anything) can ever wedge the single-flash-job guard permanently
// again the way a stuck download did on 2026-07-23 (every
// subsequent mesh.flash-device call failed with "already in
// progress" until the service was restarted). Generous enough that
// a legitimately slow multi-hundred-MB transfer still completes.
let result = match tokio::time::timeout(
MAX_JOB_DURATION,
run_flash(board, family, &data_dir, &path, &bg_job),
)
.await
{
Ok(inner) => inner,
Err(_) => Err(anyhow::anyhow!(
"Flash job exceeded the {}-minute ceiling — aborted",
MAX_JOB_DURATION.as_secs() / 60
)),
};
let succeeded = result.is_ok();
match &result {
Ok(()) => {
bg_job.push_log("Flash completed successfully".to_string()).await;
bg_job.finish().await;
info!(path = %path, board = ?board, family = %family, "LoRa firmware flash succeeded");
}
Err(e) => {
// {:#} (alternate Display) walks the full anyhow context
// chain — plain {} / %e only prints the outermost .context()
// message, which made a real 2026-07-23 esptool failure
// undiagnosable from journalctl alone (just "esptool
// erase_flash failed", no actual esptool stderr).
warn!(path = %path, error = %format!("{e:#}"), "LoRa firmware flash failed");
bg_job.push_log(format!("ERROR: {e:#}")).await;
bg_job.fail(e).await;
}
}
// The board's firmware may now differ from whatever was pinned
// before — clear the pin either way so a later reconnect's strict
// auto-detect order picks up reality instead of getting wedged
// trying the old protocol first.
if let Ok(mut config) = super::load_config(&data_dir).await {
config.device_kind = None;
if let Err(e) = super::save_config(&data_dir, &config).await {
warn!(error = %e, "Failed to clear device_kind pin after flash");
}
}
if !succeeded {
// Deliberately do NOT auto-restart the listener here. A failed
// flash means we can't vouch for the board's state — reopening
// the port immediately (esptool/rnodeconf's own reset sequence
// plus our open() toggling DTR/RTS again right after) risks
// hammering a marginal device with reconnect attempts. Confirmed
// live 2026-07-23: exactly this sequence left a real Heltec V3
// boot-looping for 5+ minutes after a failed flash. Leave mesh
// stopped; the user reconnects explicitly via the hot-swap
// modal/Mesh page once they've confirmed the board is alive.
warn!(
path = %path,
"Leaving mesh listener stopped after failed flash — reconnect manually once the board is confirmed responsive"
);
return;
}
// On success, give the board a moment to finish booting after the
// flash tool's own reset sequence before we start hammering it with
// connection attempts — same reasoning as above, just the
// lower-risk (successful-flash) side of it.
tokio::time::sleep(POST_FLASH_SETTLE_DELAY).await;
let mut svc = bg_service.write().await;
if let Some(s) = svc.as_mut() {
match super::load_config(&data_dir).await {
Ok(config) => {
// Only resume if mesh is actually still enabled per the
// CURRENT persisted config — confirmed live 2026-07-23:
// unconditionally forcing a restart here, regardless of
// `enabled`, overrode a user's own concurrent "disable
// mesh" toggle and left the listener running while
// config said disabled. That inconsistent state is what
// made a later legitimate "Keep As Is" click (which
// correctly tries to start on a false→true transition)
// fail with "already running" — the listener had already
// been force-started behind the config's back.
let should_run = config.enabled;
if let Err(e) = s.configure(config).await {
warn!(error = %e, "Failed to resume mesh listener after flash");
}
if should_run {
if let Err(e) = s.start() {
warn!(error = %e, "Failed to restart mesh listener after flash");
}
}
}
Err(e) => warn!(error = %e, "Failed to load mesh config after flash"),
}
}
});
*job.abort_handle.write().await = Some(task.abort_handle());
Ok(())
}
async fn run_flash(
board: FlashBoard,
family: DeviceType,
data_dir: &Path,
path: &str,
job: &Arc<FlashJob>,
) -> Result<()> {
match family {
DeviceType::Meshtastic | DeviceType::Meshcore => {
let image = fetch_esptool_image(board, family, data_dir, job).await?;
esptool_erase_and_write(path, &image, job).await
}
DeviceType::Reticulum => {
let lora_region = super::load_config(data_dir)
.await
.ok()
.and_then(|c| c.lora_region);
rnodeconf_autoinstall(path, board, lora_region.as_deref(), job).await
}
DeviceType::Unknown => anyhow::bail!("Pick a firmware family before flashing"),
}
}
// ─── MeshCore / Meshtastic: esptool ─────────────────────────────────────
async fn fetch_esptool_image(
board: FlashBoard,
family: DeviceType,
data_dir: &Path,
job: &Arc<FlashJob>,
) -> Result<PathBuf> {
let cache = firmware_cache_dir(data_dir);
tokio::fs::create_dir_all(&cache)
.await
.context("Creating firmware cache dir")?;
let client = github_client()?;
match family {
DeviceType::Meshtastic => fetch_meshtastic_image(&client, board, &cache, job).await,
DeviceType::Meshcore => fetch_meshcore_image(&client, board, &cache, job).await,
_ => anyhow::bail!("{family} is not flashed via esptool"),
}
}
async fn fetch_meshtastic_image(
client: &reqwest::Client,
board: FlashBoard,
cache: &Path,
job: &Arc<FlashJob>,
) -> Result<PathBuf> {
let release: GithubRelease = client
.get("https://api.github.com/repos/meshtastic/firmware/releases/latest")
.timeout(METADATA_TIMEOUT)
.send()
.await
.context("Fetching Meshtastic release list")?
.error_for_status()
.context("Meshtastic releases API error")?
.json()
.await
.context("Parsing Meshtastic release JSON")?;
// Meshtastic bundles all esp32s3 boards' images inside one per-platform
// zip rather than shipping per-board top-level assets — both Heltec V3
// and V4 are esp32s3, so this is the right zip for both (confirmed live
// against v2.7.26.54e0d8d).
let zip_asset = release
.assets
.iter()
.find(|a| a.name.starts_with("firmware-esp32s3-") && a.name.ends_with(".zip"))
.ok_or_else(|| anyhow::anyhow!("No esp32s3 firmware zip in latest Meshtastic release"))?;
let version = zip_asset
.name
.strip_prefix("firmware-esp32s3-")
.and_then(|s| s.strip_suffix(".zip"))
.ok_or_else(|| anyhow::anyhow!("Unexpected Meshtastic asset name: {}", zip_asset.name))?
.to_string();
let zip_path = cache.join(&zip_asset.name);
if tokio::fs::metadata(&zip_path).await.is_err() {
download_to_file(client, &zip_asset.browser_download_url, &zip_path, job).await?;
} else {
job.push_log(format!("Using cached {}", zip_asset.name)).await;
}
// "*.factory.bin" is Meshtastic's full merged image (bootloader +
// partition table + app) meant to be written at offset 0x0 on a freshly
// erased chip — confirmed by inspecting the real zip's contents, as
// opposed to the plain "*.bin" OTA-update image which assumes an
// existing bootloader/partition table already on the chip.
let entry_name = format!(
"firmware-{}-{}.factory.bin",
board.meshtastic_id(),
version
);
let out_path = cache.join(&entry_name);
if tokio::fs::metadata(&out_path).await.is_ok() {
return Ok(out_path);
}
job.push_log(format!(
"Extracting {entry_name} from {}",
zip_asset.name
))
.await;
let zip_path_owned = zip_path.clone();
let entry_name_owned = entry_name.clone();
let out_path_owned = out_path.clone();
tokio::task::spawn_blocking(move || -> Result<()> {
let file = std::fs::File::open(&zip_path_owned).context("Opening downloaded firmware zip")?;
let mut archive = zip::ZipArchive::new(file).context("Reading firmware zip")?;
let mut entry = archive
.by_name(&entry_name_owned)
.with_context(|| format!("{entry_name_owned} not found in firmware zip"))?;
let mut out =
std::fs::File::create(&out_path_owned).context("Creating extracted firmware file")?;
std::io::copy(&mut entry, &mut out).context("Extracting firmware image")?;
Ok(())
})
.await
.context("Firmware extraction task panicked")??;
Ok(out_path)
}
async fn fetch_meshcore_image(
client: &reqwest::Client,
board: FlashBoard,
cache: &Path,
job: &Arc<FlashJob>,
) -> Result<PathBuf> {
let release: GithubRelease = client
.get("https://api.github.com/repos/meshcore-dev/MeshCore/releases/latest")
.timeout(METADATA_TIMEOUT)
.send()
.await
.context("Fetching MeshCore release list")?
.error_for_status()
.context("MeshCore releases API error")?
.json()
.await
.context("Parsing MeshCore release JSON")?;
// Upstream's casing differs between boards (Heltec_v3_... vs
// heltec_v4_...) — match case-insensitively on the exact per-board
// substring so V4 isn't accidentally matched by "heltec_v4_tft_..."
// variants (there's a "_tft_" in between, so a straight substring match
// on "heltec_v4_companion_radio_usb" is already safe).
let needle = match board {
FlashBoard::HeltecV3 => "heltec_v3_companion_radio_usb",
FlashBoard::HeltecV4 => "heltec_v4_companion_radio_usb",
};
let asset = release
.assets
.iter()
.find(|a| {
let lower = a.name.to_lowercase();
lower.contains(needle) && lower.ends_with("-merged.bin")
})
.ok_or_else(|| {
anyhow::anyhow!("No matching MeshCore image in release {}", release.tag_name)
})?;
let out_path = cache.join(&asset.name);
if tokio::fs::metadata(&out_path).await.is_ok() {
job.push_log(format!("Using cached {}", asset.name)).await;
return Ok(out_path);
}
download_to_file(client, &asset.browser_download_url, &out_path, job).await?;
Ok(out_path)
}
async fn download_to_file(
client: &reqwest::Client,
url: &str,
dest: &Path,
job: &Arc<FlashJob>,
) -> Result<()> {
job.set_stage(FlashStage::Downloading).await;
// Bound only the wait for the response to *start* (headers) — NOT a
// request-level `.timeout()`, which would cap the whole body transfer
// again (the bug this replaced: a blanket 30s client timeout killed
// large downloads mid-stream). If the server never responds at all,
// this is what stops the job from hanging forever; the per-chunk stall
// timeout below is what guards the body once streaming starts. Without
// this, a server that accepts the TCP connection but never sends
// headers back hangs this call indefinitely — confirmed live
// 2026-07-23: a stuck `.send()` here wedged the single-flash-job guard
// for good, permanently blocking every subsequent flash attempt with
// "already in progress" until the service was restarted.
let resp = tokio::time::timeout(METADATA_TIMEOUT, client.get(url).send())
.await
.context("Firmware download server did not respond")?
.context("Starting firmware download")?
.error_for_status()
.context("Firmware download returned an error status")?;
let total = resp.content_length();
let tmp = dest.with_extension("part");
let mut file = tokio::fs::File::create(&tmp)
.await
.context("Creating firmware download file")?;
let mut stream = resp.bytes_stream();
let mut downloaded: u64 = 0;
use futures_util::StreamExt;
loop {
let next = tokio::time::timeout(DOWNLOAD_STALL_TIMEOUT, stream.next())
.await
.context("Firmware download stalled")?;
let Some(chunk) = next else { break };
let chunk = chunk.context("Reading firmware download stream")?;
file.write_all(&chunk)
.await
.context("Writing firmware download")?;
downloaded += chunk.len() as u64;
if let Some(total) = total {
if total > 0 {
job.set_percent(((downloaded.saturating_mul(100)) / total) as u8)
.await;
}
}
}
file.flush().await.ok();
tokio::fs::rename(&tmp, dest)
.await
.context("Finalizing firmware download")?;
job.push_log(format!(
"Downloaded {} ({downloaded} bytes)",
dest.display()
))
.await;
Ok(())
}
/// Both Heltec V3 and V4 are ESP32-S3 boards.
const ESPTOOL_CHIP: &str = "esp32s3";
/// esptool's auto-reset-into-bootloader handshake (toggling DTR/RTS in a
/// specific timed pattern) is well-known to be flaky on some CP2102/CH340
/// board+adapter combinations — esptool's own docs recommend retrying at a
/// lower baud rate when this happens. Rather than fail the whole job on the
/// first hiccup, retry once at a conservative baud before giving up.
const ESPTOOL_FALLBACK_BAUD: &str = "115200";
/// `write_flash --erase-all` erases the whole chip before writing, in one
/// esptool invocation. This needs the esp32s3 stub flasher loaded (see
/// esptool_global_args' doc comment) — without it, --erase-all hits the
/// exact same ROM limitation a standalone `erase_flash` does ("ESP32-S3 ROM
/// does not support function erase_flash", confirmed live 2026-07-23), since
/// esptool's --erase-all is implemented as the same full-chip-erase command,
/// not a per-sector loop.
async fn esptool_erase_and_write(path: &str, image: &Path, job: &Arc<FlashJob>) -> Result<()> {
job.set_stage(FlashStage::Writing).await;
let image_str = image.to_string_lossy().to_string();
esptool_with_retry(
path,
&["write_flash", "--erase-all", "0x0", &image_str],
job,
)
.await
.context("esptool write_flash failed")?;
Ok(())
}
/// esptool's global flags (--chip/--port/--baud) MUST precede the subcommand
/// token (erase_flash/write_flash/...) — confirmed live 2026-07-23:
/// appending `--baud 115200` after the subcommand on the retry path
/// produced "esptool: error: unrecognized arguments: --baud 115200" every
/// time, so the fallback-baud retry never actually got a chance to run.
/// Building global args separately from subcommand args keeps this correct
/// by construction instead of relying on call-site ordering.
///
/// Normal stub-loader mode (no --no-stub) needs the esp32s3 stub flasher
/// blob at /usr/lib/python3/dist-packages/esptool/targets/stub_flasher/
/// stub_flasher_32s3.json — Debian's `esptool` package (4.7.0+dfsg-0.1)
/// ships without it (stripped for DFSG compliance: the prebuilt blob has no
/// buildable-from-source path Debian could verify), so scripts/self-update.sh
/// fetches the exact same file from the matching upstream esptool release
/// tag and installs it alongside the apt package (see the esptool install
/// step there). --no-stub (talk directly to the ROM bootloader, skip the
/// stub) was tried first and works for connecting, but the ROM bootloader
/// doesn't implement a full-chip-erase opcode at all — only the stub does —
/// so --no-stub broke our "always erase before write" default outright
/// rather than just being slower. Restoring the real stub file is the
/// correct fix, not routing around its absence.
fn esptool_global_args<'a>(path: &'a str, baud: Option<&'a str>) -> Vec<&'a str> {
let mut args = vec!["--chip", ESPTOOL_CHIP, "--port", path];
if let Some(b) = baud {
args.push("--baud");
args.push(b);
}
args
}
async fn esptool_with_retry(path: &str, subcommand: &[&str], job: &Arc<FlashJob>) -> Result<()> {
let mut cmd = Command::new("esptool");
cmd.args(esptool_global_args(path, None));
cmd.args(subcommand);
match run_streamed(cmd, None, job).await {
Ok(()) => Ok(()),
Err(first_err) => {
job.push_log(format!(
"First attempt failed ({first_err:#}); retrying once at {ESPTOOL_FALLBACK_BAUD} baud"
))
.await;
let mut retry = Command::new("esptool");
retry.args(esptool_global_args(path, Some(ESPTOOL_FALLBACK_BAUD)));
retry.args(subcommand);
run_streamed(retry, None, job)
.await
.context(format!("retry also failed (first attempt: {first_err:#})"))
}
}
}
// ─── Reticulum/RNode: archy-rnodeconf ───────────────────────────────────
fn rnodeconf_bin() -> String {
std::env::var("ARCHY_RNODECONF_BIN")
.unwrap_or_else(|_| "/usr/local/bin/archy-rnodeconf".to_string())
}
/// `--autoinstall`'s "which board is this" step is interactive by design —
/// confirmed live against a real Heltec V4 (2026-07-23): even with a board
/// given on the command line, rnodeconf can't always tell V3 from V4 apart
/// (their bootstrap-time USB identity is often generic, same root cause as
/// `resolve_flash_board`'s doc comment), so it always asks. The full prompt
/// sequence observed for a Heltec board that already has *some* RNode
/// firmware installed (the common case — a truly blank chip likely skips
/// straight to the same "Device Selection" menu):
/// 1. numbered device-type menu → answer with the menu number
/// 2. "Hit enter to continue" → answer with a blank line
/// 3. numbered band menu → answer with the menu number
/// 4. "Is the above correct? [y/N]" → answer "y"
/// Feeding all four answers up front (rather than watching stdout for each
/// prompt text) works because the menu is always asked in this fixed order
/// for every board that needs (re)provisioning — verified by driving it
/// through an unprovisioned real V4 end-to-end (erase → flash → EEPROM
/// bootstrap → "Device signature validated" on the next probe).
fn rnodeconf_device_menu_number(board: FlashBoard) -> &'static str {
match board {
FlashBoard::HeltecV3 => "8",
FlashBoard::HeltecV4 => "9",
}
}
/// rnodeconf's band choice is a coarse RF-frontend bootstrap parameter
/// (868/915/923 MHz), not the final operating frequency — that's still
/// configured later via the daemon's interface config, same as today. This
/// is a best-effort mapping from the node's persisted Meshtastic-style
/// region code (see `mesh::meshtastic::region_name_to_code`) down to
/// rnodeconf's 3-way menu; regions with no exact 868/923 match fall back to
/// 915 MHz as the broadest-compatibility default.
fn rnodeconf_band_menu_number(lora_region: Option<&str>) -> &'static str {
match lora_region.map(|s| s.trim().to_uppercase()) {
Some(r) if r.contains("868") => "1",
Some(r) if r.contains("923") => "3",
_ => "2",
}
}
/// `--autoinstall` fetches, erases, flashes, and bootstraps the EEPROM for
/// a detected board as one atomic step (confirmed via `archy-rnodeconf
/// --help` AND a real end-to-end flash on real hardware) — this is the
/// RNode-side equivalent of our "always erase before write" default, since
/// autoinstall doesn't try to preserve any existing on-device state.
async fn rnodeconf_autoinstall(
path: &str,
board: FlashBoard,
lora_region: Option<&str>,
job: &Arc<FlashJob>,
) -> Result<()> {
job.set_stage(FlashStage::Autoinstalling).await;
let bin = rnodeconf_bin();
let mut cmd = if Path::new(&bin).exists() {
Command::new(bin)
} else {
// Dev fallback if only a plain venv/system rnodeconf is on PATH.
Command::new("rnodeconf")
};
cmd.args(["--autoinstall", path]);
let stdin = format!(
"{}\n\n{}\ny\n",
rnodeconf_device_menu_number(board),
rnodeconf_band_menu_number(lora_region)
);
run_streamed(cmd, Some(stdin.into_bytes()), job)
.await
.context("archy-rnodeconf --autoinstall failed")
}
// ─── Subprocess streaming ────────────────────────────────────────────────
fn percent_regex() -> &'static Regex {
static RE: OnceLock<Regex> = OnceLock::new();
RE.get_or_init(|| Regex::new(r"\((\d{1,3})\s*%\)").expect("valid regex"))
}
async fn run_streamed(mut cmd: Command, stdin: Option<Vec<u8>>, job: &Arc<FlashJob>) -> Result<()> {
cmd.stdout(Stdio::piped());
cmd.stderr(Stdio::piped());
if stdin.is_some() {
cmd.stdin(Stdio::piped());
}
// Deliberately NOT kill_on_drop: an interrupted erase/write can leave
// the chip in a worse state than either finished or unstarted (see the
// cancellation-safety note in mesh flashing docs). The job is expected
// to run to completion or fail on its own.
let mut child = cmd.spawn().context("Failed to start subprocess")?;
if let Some(bytes) = stdin {
if let Some(mut child_stdin) = child.stdin.take() {
child_stdin
.write_all(&bytes)
.await
.context("Writing to subprocess stdin")?;
}
}
let mut tasks = Vec::new();
if let Some(stdout) = child.stdout.take() {
let job = Arc::clone(job);
tasks.push(tokio::spawn(async move {
let mut lines = BufReader::new(stdout).lines();
while let Ok(Some(line)) = lines.next_line().await {
if let Some(cap) = percent_regex().captures(&line) {
if let Ok(pct) = cap[1].parse::<u8>() {
job.set_percent(pct).await;
}
}
job.push_log(line).await;
}
}));
}
if let Some(stderr) = child.stderr.take() {
let job = Arc::clone(job);
tasks.push(tokio::spawn(async move {
let mut lines = BufReader::new(stderr).lines();
while let Ok(Some(line)) = lines.next_line().await {
job.push_log(line).await;
}
}));
}
let status = child.wait().await.context("Waiting for subprocess")?;
for t in tasks {
let _ = t.await;
}
if !status.success() {
// Exit status alone isn't diagnosable — the actual esptool/rnodeconf
// stderr (already captured into job.log_tail by the reader tasks
// above) is what actually explains a failure. Confirmed live
// 2026-07-23: a bare "Command exited with exit status: 1" told us
// nothing when esptool's real error was sitting in the log tail the
// whole time, only visible via the UI's live poll, not journald.
let tail: Vec<String> = job
.snapshot()
.await
.log_tail
.iter()
.rev()
.take(10)
.rev()
.cloned()
.collect();
anyhow::bail!("Command exited with {status}\n{}", tail.join("\n"));
}
Ok(())
}

View File

@ -87,18 +87,6 @@ const RECONNECT_DELAY_INIT: Duration = Duration::from_secs(5);
/// Maximum reconnect delay (cap for exponential backoff).
const RECONNECT_DELAY_MAX: Duration = Duration::from_secs(60);
/// Minimum time a session must run before we trust it enough to reset
/// backoff to the minimum. Without this gate, a device that connects then
/// fails again within a couple of seconds (e.g. mid-boot-loop) never backs
/// off — every retry immediately re-opens the port, which toggles DTR/RTS
/// (resets many ESP32 boards' MCU on native-USB and CP2102/CH340
/// auto-reset-circuit boards alike), turning a device that's merely
/// unstable into a self-sustaining boot loop that outlasts whatever
/// triggered the original instability. Confirmed live 2026-07-23: a Heltec
/// V3 stuck retrying every ~5-15s for 5+ minutes after a failed firmware
/// flash left it in a marginal state.
const STABLE_SESSION_THRESHOLD: Duration = Duration::from_secs(20);
/// Number of consecutive write failures before we consider the device dead
/// and trigger a reconnection cycle.
const MAX_CONSECUTIVE_WRITE_FAILURES: u32 = 3;
@ -562,7 +550,6 @@ pub fn spawn_mesh_listener(
return;
}
let session_start = std::time::Instant::now();
match session::run_mesh_session(
&state,
&data_dir,
@ -585,14 +572,13 @@ pub fn spawn_mesh_listener(
{
Ok(()) => {
info!("Mesh session ended cleanly");
// Only trust a session that actually ran for a while —
// see STABLE_SESSION_THRESHOLD's doc comment.
if session_start.elapsed() >= STABLE_SESSION_THRESHOLD {
reconnect_delay = RECONNECT_DELAY_INIT;
}
// Session was established before ending — reset backoff
reconnect_delay = RECONNECT_DELAY_INIT;
}
Err(e) => {
if session_start.elapsed() >= STABLE_SESSION_THRESHOLD {
// Check if session was ever connected (vs failed to open)
let was_connected = state.status.read().await.device_connected;
if was_connected {
reconnect_delay = RECONNECT_DELAY_INIT;
}
error!("Mesh session error: {} (retry in {:?})", e, reconnect_delay);

View File

@ -214,20 +214,11 @@ impl MeshtasticDevice {
path
))?;
// See probe_rnode() in reticulum.rs for why: ESP32-S3 native-USB
// boards (and CP2102/CH340-bridged boards wired for Arduino-style
// auto-reset) reset on a DTR/RTS transition, so deassert both and
// settle before the handshake below. 300ms is nowhere near a real
// firmware boot time (LoRa radio init alone can take longer) —
// confirmed live 2026-07-23: with every one of Reticulum/Meshcore/
// Meshtastic's open() doing this same reset, a single auto-detect
// cycle trying multiple protocols in sequence kept re-resetting the
// board before it ever finished booting from the PREVIOUS attempt's
// reset, on both a Heltec V3 and V4, regardless of firmware family —
// a self-sustaining "never finishes booting" loop with a boot-time
// root cause hiding behind what looked like a per-protocol failure.
// boards reset on a DTR/RTS transition, so deassert both and settle
// before the handshake below.
let _ = port.set_dtr(false);
let _ = port.set_rts(false);
tokio::time::sleep(Duration::from_millis(2000)).await;
tokio::time::sleep(Duration::from_millis(300)).await;
info!(path = %path, baud = BAUD_RATE, "Opened Meshtastic serial port");
Ok(Self {

View File

@ -8,7 +8,6 @@
pub mod alerts;
pub mod bitcoin_relay;
pub mod crypto;
pub mod flash;
pub mod listener;
pub mod meshtastic;
pub mod message_types;
@ -39,14 +38,6 @@ use tokio::sync::watch;
use tracing::{error, info, warn};
const MESH_CONFIG_FILE: &str = "mesh-config.json";
/// How long `MeshService::stop()` waits for the listener task to notice its
/// shutdown signal and exit gracefully before force-aborting it. See
/// `stop()`'s doc comment for the real incident this guards against: without
/// a hard abort fallback, a slow-to-notice listener could be left running
/// forever, orphaned, racing a later independently-started listener on the
/// same serial port.
const LISTENER_SHUTDOWN_TIMEOUT: Duration = Duration::from_secs(15);
const MESH_IGNORED_RADIO_FILE: &str = "mesh-ignored-radio-contacts.json";
const MESH_CONTACTS_FILE: &str = "mesh-contacts.json";
@ -743,18 +734,10 @@ impl MeshService {
self.server_name = name;
}
/// Start the background mesh listener. Idempotent: if the listener is
/// already running, this is a harmless no-op rather than an error —
/// confirmed live 2026-07-23, a real race between the flash job's own
/// post-flash restart and a concurrent user "Keep As Is" click (both
/// legitimately trying to ensure the listener is running) surfaced this
/// as a user-facing "Mesh listener already running" RPC error. Ensuring
/// the listener is running is the intent every caller actually has;
/// whichever caller's start() happens to win the race, the other
/// finding it already satisfied is success, not failure.
/// Start the background mesh listener.
pub fn start(&mut self) -> Result<()> {
if self.listener_handle.is_some() {
return Ok(());
anyhow::bail!("Mesh listener already running");
}
let (shutdown_tx, shutdown_rx) = watch::channel(false);
@ -984,36 +967,8 @@ impl MeshService {
if let Some(tx) = self.shutdown_tx.take() {
let _ = tx.send(true);
}
if let Some(mut handle) = self.listener_handle.take() {
// Bounded wait for graceful shutdown, with a hard abort as
// fallback — confirmed live 2026-07-23: a caller-side timeout
// wrapping stop() (mesh::flash's STOP_LISTENER_TIMEOUT) cancelled
// this await when the listener was slow to notice its shutdown
// signal (mid multi-candidate probe), but `.take()` above had
// already cleared `listener_handle` to None — so MeshService
// believed it was stopped while the task kept running, orphaned
// (dropping a JoinHandle does not abort the task it points to).
// A later start() then spawned a second, fully independent
// listener session racing the orphaned one on the same serial
// port — neither could ever get a clean response, so every
// mesh.configure/probe against that device failed indefinitely
// even though the device itself was fine.
//
// Awaiting `&mut handle` (not `handle` by value) is what makes
// the fallback possible: the Future is polled through the
// reference, so if the timeout fires, this task's own `handle`
// binding is still ours to call `.abort()` on afterward —
// unlike moving `handle` into the timeout future outright, which
// would drop (and thus orphan) it on timeout with nothing left
// to abort.
if tokio::time::timeout(LISTENER_SHUTDOWN_TIMEOUT, &mut handle)
.await
.is_err()
{
warn!("Mesh listener did not shut down gracefully in time — aborting it");
handle.abort();
let _ = handle.await;
}
if let Some(handle) = self.listener_handle.take() {
let _ = handle.await;
}
if let Some(handle) = self.deadman_handle.take() {
handle.abort();

View File

@ -58,20 +58,11 @@ impl MeshcoreDevice {
path
))?;
// See probe_rnode() in reticulum.rs for why: ESP32-S3 native-USB
// boards (and CP2102/CH340-bridged boards wired for Arduino-style
// auto-reset) reset on a DTR/RTS transition, so deassert both and
// settle before the handshake below. 300ms is nowhere near a real
// firmware boot time (LoRa radio init alone can take longer) —
// confirmed live 2026-07-23: with every one of Reticulum/Meshcore/
// Meshtastic's open() doing this same reset, a single auto-detect
// cycle trying multiple protocols in sequence kept re-resetting the
// board before it ever finished booting from the PREVIOUS attempt's
// reset, on both a Heltec V3 and V4, regardless of firmware family —
// a self-sustaining "never finishes booting" loop with a boot-time
// root cause hiding behind what looked like a per-protocol failure.
// boards reset on a DTR/RTS transition, so deassert both and settle
// before the handshake below.
let _ = port.set_dtr(false);
let _ = port.set_rts(false);
tokio::time::sleep(Duration::from_millis(2000)).await;
tokio::time::sleep(Duration::from_millis(300)).await;
info!(path = %path, baud = BAUD_RATE, "Opened serial port");

View File

@ -110,61 +110,6 @@ lands].
2. ❑ Mobile Home: wallet card directly under My Apps (G5 from the voice epic).
3. ❑ Mesh RF settings panel (Mesh → Device) still loads and saves.
## H. LoRa radio firmware flashing (Heltec V3/V4, new — extends Section E)
Full v1 scope is 3 firmware families × 2 boards (6 cells); mark each cell
tested on real hardware vs. code-reviewed only as this is run.
1. ❑ From the hot-swap modal's step 1 (device already probed), press
**Flash Firmware…** → new step shows firmware-family + board pickers and
the erase-confirmation checkbox; "Erase & Flash Now" stays disabled until
family, board, AND the checkbox are all set.
2. ❑ Confirm what's currently on the test stick via the existing probe
BEFORE flashing it — don't flash the only known-good device without a
fallback board on hand.
3. ❑ Prefer a spare Heltec V3/V4 for the first destructive erase+flash run;
only exercise a primary/in-use stick once the flow is proven safe.
4. ❑ MeshCore → Heltec V3: erase + write completes, progress bar and log
tail update live, ends at "Flash complete".
5. ❑ Meshtastic → Heltec V3: same, using the extracted `*.factory.bin` from
the esp32s3 release zip.
6. ❑ Reticulum/RNode → Heltec V3: `archy-rnodeconf --autoinstall` path
completes (no raw esptool erase/write step for this family — see
`mesh/flash.rs` doc comment).
7. ❑ Repeat 4-6 against a Heltec V4. Confirmed 2026-07-23 on real hardware:
V4 uses the ESP32-S3's native-USB JTAG/serial peripheral (vid:pid
303a:1001, generic to every native-USB ESP32-S3 board, not V4-specific)
— so unlike V3's CP2102 bridge chip, V4 is permanently NOT auto-matchable
by vid:pid. Board auto-detect should fail closed for it every time
(manual board selection required, "couldn't confirm automatically"
warning shown) — this is expected steady-state behavior, not a gap to
close later.
8. ❑ After a successful flash, the modal automatically re-probes and shows
the NEW firmware's badge/details — same as unplugging and replugging
(Section E item 3), but without physically touching the cable.
9. ❑ Deliberately test a failure path once (disconnect the board mid-write,
or point at a bad cached asset) — confirm the error surfaces in the
progress log AND that `docs/troubleshooting.md`'s "LoRa radio firmware
flash failed" recovery steps (BOOT+RST bootloader entry, manual esptool/
rnodeconf command) actually get the board back to a flashable state.
10. ❑ Cancel button only appears (and only works) while still in the
"Downloading firmware…" stage — once erasing/writing starts, no cancel
affordance is offered.
11. ❑ **Boot-loop regression (2026-07-23 incident)**: after a *failed* flash
(e.g. kill network access mid-download to force a failure), confirm the
mesh listener does NOT auto-resume — `journalctl -u archipelago` should
show a single `Leaving mesh listener stopped after failed flash` line
and then go quiet for that device, not a repeating `mesh::serial:
Opened serial port... Starting Meshcore handshake` cycle every few
seconds. Reconnect manually via the hot-swap modal afterward and confirm
it connects normally (the board itself should be untouched — the
download fails before esptool/rnodeconf ever runs).
12. ❑ Separately, force a device to flap connected/disconnected a few times
in under 20s each (e.g. a marginal USB connection) and confirm
`reconnect_delay` in the logs actually escalates (5s → 10s → 20s → ...)
rather than resetting to 5s on every attempt — see
`STABLE_SESSION_THRESHOLD` in `mesh/listener/mod.rs`.
---
After this passes: fold the batch + other agent's work into the next release

View File

@ -443,86 +443,6 @@ free -h
- Check Nginx WebSocket proxy config: `/etc/nginx/sites-available/archipelago` must include `proxy_set_header Upgrade $http_upgrade`
- If on WiFi, try wired Ethernet for more stable connectivity
### 21. LoRa radio firmware flash failed / board unresponsive
**Symptoms**: The "Erase & Flash Now" flow in the mesh hot-swap modal reports
an error, or the radio no longer enumerates as a serial device after a flash
attempt.
**Diagnosis**:
```bash
# Poll the flash job's last-known stage/error directly
curl -s http://localhost:5678/rpc/v1 \
-H 'Content-Type: application/json' \
-d '{"method":"mesh.flash-status","params":{}}'
# Confirm the board is still enumerating at all
ls -la /dev/ttyUSB* /dev/ttyACM* /dev/mesh-radio 2>&1
# esptool/rnodeconf binaries present?
which esptool; ls -la /usr/local/bin/archy-rnodeconf
```
**Solutions**:
- A failure during `erasing`/`writing` (MeshCore/Meshtastic) or
`autoinstalling` (Reticulum) can leave the chip erased or half-written —
this is expected risk of the "always erase first" default, not a bug.
- Heltec V3/V4 boards can be forced back into bootloader mode manually: hold
**BOOT**, tap **RST**, then release **BOOT** — this puts the chip in a
state esptool can always talk to, regardless of what firmware (if any) is
currently on it.
- With the board in bootloader mode, a manual recovery flash can be run
directly over SSH without the UI:
```bash
esptool --chip esp32s3 --port /dev/ttyACM0 erase_flash
esptool --chip esp32s3 --port /dev/ttyACM0 write_flash 0x0 <known-good-image.bin>
```
- For Reticulum/RNode boards, the equivalent manual recovery is
`archy-rnodeconf /dev/ttyACM0 --autoinstall` (or `/usr/local/bin/archy-rnodeconf`
if it's not on `PATH`) — it re-runs the same fetch+erase+flash+bootstrap
sequence the UI triggers.
- If `esptool`/`archy-rnodeconf` are missing entirely, they should have been
installed by the last `self-update.sh` run — check
`sudo journalctl -u archipelago-update` for install failures, or install
`esptool` via `sudo apt-get install esptool` directly.
- Once a fresh image is confirmed written, unplug/replug the radio (or wait
for the next detection poll) — the hot-swap modal re-probes automatically
and shows whatever firmware is actually on the board now.
**Known incident (2026-07-23) — reconnect storm / device boot-loop after a
failed flash**: a real Heltec V3 got stuck cycling "connect → partial
handshake → drop" every 5-15s for 5+ minutes after a `mesh.flash-device`
attempt failed with `Reading firmware download stream`. Root cause was two
compounding issues, both now fixed:
1. `spawn_mesh_listener`'s reconnect backoff (`core/archipelago/src/mesh/listener/mod.rs`)
reset to its 5s minimum any time the prior session had been `device_connected`
at all, even for under a second — so a device that connects-then-drops
repeatedly never actually backed off. Every retry's `open()` toggles
DTR/RTS, which resets many ESP32 boards' MCU (native-USB *and*
CP2102/CH340 auto-reset-circuit boards), so the aggressive retries were
themselves *causing* the boot loop, not just observing one. Fixed by only
resetting backoff when a session ran for at least `STABLE_SESSION_THRESHOLD`
(20s) — see that constant's doc comment.
2. `mesh::flash::start_flash_job`'s post-completion handler auto-resumed the
listener unconditionally, even after a *failed* flash, immediately
re-entering the reconnect loop above with no cooldown. Fixed: on failure
the listener is now deliberately left stopped (reconnect manually via the
UI once the board is confirmed alive); on success there's a 5s settle
delay before resuming, so the board finishes booting from the flash
tool's own reset before Archipelago starts probing it again.
3. Separately, the download itself was failing because `mesh::flash`'s HTTP
client had a blanket 30s request timeout that covered the *entire*
download (including streaming a 170MB Meshtastic zip), not just
connection setup — fixed with a per-chunk stall timeout instead of a
fixed total-transfer cap.
If this symptom recurs (rapid repeating `mesh::serial: Opened serial
port... Starting Meshcore handshake` lines in `journalctl -u archipelago`
without a `LoRa firmware flash` job in progress), it's a NEW instance of the
same class of bug, not the one above — check whether backoff is actually
escalating (`Mesh session error: ... (retry in Xs)` — X should grow past 5s
within a few cycles) before assuming it's flashing-related.
---
## General Maintenance

View File

@ -365,11 +365,6 @@ RUN apt-get update && apt-get -y full-upgrade && apt-get install -y --no-install
ca-certificates \
openssl \
chrony \
iputils-ping \
esptool \
python3-venv \
binutils \
libpython3.13 \
locales \
console-setup \
keyboard-configuration \

View File

@ -1,7 +1,7 @@
<template>
<BaseModal
:show="show"
:title="step === 1 ? 'Mesh Radio Detected' : step === 2 ? 'Apply Archipelago Settings' : 'Flash Firmware'"
:title="step === 1 ? 'Mesh Radio Detected' : 'Apply Archipelago Settings'"
max-width="max-w-lg"
content-class="max-h-[90vh] overflow-y-auto"
@close="dismiss"
@ -96,111 +96,9 @@
"Keep As Is" uses the radio exactly as it is nothing on it is changed,
and you can hot-swap radios any time.
</p>
<button
class="w-full text-center text-white/40 hover:text-white/70 text-[11px] mt-3 underline underline-offset-2"
:disabled="!!connecting"
@click="openFlashStep"
>
Flash Firmware
</button>
<p v-if="error" class="text-xs text-red-400 mt-2">{{ error }}</p>
</div>
<!-- Step 3: erase + reflash destructive, opt-in only -->
<div v-else-if="step === 'flash'">
<!-- Once a job exists (started via startFlash), ALWAYS show the
progress/result view below including on failure. The old
condition (`!active && stage !== 'done'`) was also true for a
FAILED job (active:false, stage:'failed'), which silently sent
the user back to this picker instead of showing the error. -->
<template v-if="!flashJob">
<p class="text-white/60 text-xs mb-3">
Downloads the latest firmware from upstream and writes it to
<span class="font-mono text-orange-300">{{ devicePath }}</span>.
</p>
<div class="space-y-4">
<div>
<label class="block text-sm text-white/80 mb-1">Firmware family</label>
<select v-model="flashFamily" class="w-full rounded-lg bg-white/[0.06] border border-white/10 text-white px-3 py-2 text-sm focus:outline-none focus:border-orange-400/60">
<option value="">Choose</option>
<option value="meshcore">MeshCore</option>
<option value="meshtastic">Meshtastic</option>
<option value="reticulum">Reticulum RNode</option>
</select>
</div>
<div>
<label class="block text-sm text-white/80 mb-1">Board</label>
<select v-model="flashBoard" class="w-full rounded-lg bg-white/[0.06] border border-white/10 text-white px-3 py-2 text-sm focus:outline-none focus:border-orange-400/60">
<option value="">Choose</option>
<option value="heltec-v3">Heltec LoRa 32 V3</option>
<option value="heltec-v4">Heltec LoRa 32 V4</option>
</select>
<p v-if="!boardAutoDetected" class="text-[11px] text-amber-400/80 mt-1">
Couldn't confirm the board automatically double check before flashing.
Flashing the wrong board's image can brick it.
</p>
</div>
</div>
<div class="rounded-xl bg-red-500/10 border border-red-500/30 p-3 mt-4">
<label class="flex items-start gap-2 text-xs text-red-300">
<input type="checkbox" v-model="flashConfirmed" class="mt-0.5" />
<span>
This <strong>erases the entire chip</strong>, including any existing
keys, identity, and contacts. This cannot be undone.
</span>
</label>
</div>
<p v-if="error" class="text-xs text-red-400 mt-3">{{ error }}</p>
<div class="flex gap-2 mt-6">
<button class="glass-button px-4 py-2 rounded-lg text-sm" @click="step = 1">Back</button>
<button
class="flex-1 glass-button px-4 py-2 rounded-lg text-sm font-medium bg-red-500/80 hover:bg-red-500 text-white disabled:opacity-50"
:disabled="!flashFamily || !flashBoard || !flashConfirmed || starting"
@click="startFlash"
>
{{ starting ? 'Starting…' : 'Erase & Flash Now' }}
</button>
</div>
</template>
<!-- Progress -->
<template v-else>
<div class="text-center py-2">
<p class="text-white text-sm font-medium">{{ flashStageLabel }}</p>
<div class="mt-3 h-2 rounded-full bg-white/10 overflow-hidden">
<div
class="h-full bg-orange-400 transition-all"
:style="{ width: (flashJob?.percent ?? (flashJob?.stage === 'done' ? 100 : 8)) + '%' }"
></div>
</div>
<p v-if="flashJob?.error" class="text-xs text-red-400 mt-3">{{ flashJob.error }}</p>
</div>
<div class="mt-3 rounded-xl bg-black/30 border border-white/10 p-2 h-32 overflow-y-auto font-mono text-[10px] text-white/50 leading-relaxed">
<div v-for="(line, i) in (flashJob?.log_tail ?? []).slice(-40)" :key="i">{{ line }}</div>
</div>
<div class="flex gap-2 mt-4">
<button
v-if="flashJob?.stage === 'downloading' && flashJob?.active"
class="glass-button px-4 py-2 rounded-lg text-sm"
@click="cancelFlash"
>
Cancel
</button>
<button
v-if="!flashJob?.active"
class="flex-1 glass-button glass-button-warning px-4 py-2 rounded-lg text-sm font-medium"
@click="closeFlashStep"
>
Done
</button>
</div>
</template>
</div>
<!-- Step 2: our latest parameters, shown before anything is written -->
<div v-else>
<p class="text-white/60 text-xs mb-3">
@ -287,7 +185,7 @@
import { ref, computed, watch } from 'vue'
import { useRouter } from 'vue-router'
import BaseModal from '@/components/BaseModal.vue'
import { useMeshStore, type MeshDeviceProbe, type MeshConfigureParams, type FlashFirmwareFamily, type FlashBoard, type FlashJobStatus } from '@/stores/mesh'
import { useMeshStore, type MeshDeviceProbe, type MeshConfigureParams } from '@/stores/mesh'
import { useAppStore } from '@/stores/app'
import { LORA_REGIONS, regionByCode, suggestRegionFromLatLon, meshcorePlanFor } from '@/utils/loraRegions'
import { resolveMeshDeviceImage } from '@/utils/meshDeviceImages'
@ -296,7 +194,7 @@ const mesh = useMeshStore()
const appStore = useAppStore()
const router = useRouter()
const step = ref<1 | 2 | 'flash'>(1)
const step = ref<1 | 2>(1)
const connecting = ref<false | 'keep' | 'setup'>(false)
const error = ref('')
const probing = ref(false)
@ -361,10 +259,7 @@ const rfPreset = computed(() => {
// (Re)probe + (re)apply presets each time a new device surfaces the modal
watch([show, devicePath], async ([visible]) => {
if (!visible) {
stopFlashPoll()
return
}
if (!visible) return
step.value = 1
error.value = ''
imageFailed.value = false
@ -449,118 +344,6 @@ async function applySetup() {
connecting.value = false
}
}
// Step 3: erase + reflash
const flashFamily = ref<FlashFirmwareFamily | ''>('')
const flashBoard = ref<FlashBoard | ''>('')
const flashConfirmed = ref(false)
const starting = ref(false)
const flashJob = ref<FlashJobStatus | null>(null)
let flashPollTimer: ReturnType<typeof setInterval> | null = null
const detectedInfo = computed(() =>
mesh.status?.detected_device_info?.find(d => d.path === devicePath.value)
)
// Mirrors mesh::flash::resolve_flash_board (core/archipelago/src/mesh/flash.rs)
// exactly matching on the display label was wrong: a Heltec V3's CP2102
// bridge chip reports "CP2102 USB to UART Bridge Controller" in its USB
// strings, not "Heltec", so meshDeviceImages.ts falls back to a generic
// "LoRa radio (CP2102 serial)" label that never matched /v3/i, showing the
// "couldn't confirm automatically" warning even though the backend CAN
// safely auto-detect V3 via vid:pid. Heltec V4 deliberately has no entry
// here, same reasoning as the backend: its vid:pid (303a:1001) is the
// ESP32-S3's generic native-USB descriptor, not V4-specific, so it can't be
// safely auto-matched and always requires manual selection.
const resolvedFlashBoard = computed<FlashBoard | ''>(() => {
const info = detectedInfo.value
if (info?.vid?.toLowerCase() === '10c4' && info?.pid?.toLowerCase() === 'ea60') return 'heltec-v3'
return ''
})
const boardAutoDetected = computed(() => !!resolvedFlashBoard.value)
const flashStageLabel = computed(() => {
switch (flashJob.value?.stage) {
case 'downloading': return 'Downloading firmware…'
case 'erasing': return 'Erasing chip…'
case 'writing': return 'Writing firmware…'
case 'autoinstalling': return 'Installing (rnodeconf)…'
case 'done': return 'Flash complete'
case 'failed': return 'Flash failed'
default: return ''
}
})
function openFlashStep() {
flashFamily.value = (probe.value?.kind as FlashFirmwareFamily) ?? ''
flashBoard.value = resolvedFlashBoard.value
flashConfirmed.value = false
flashJob.value = null
error.value = ''
step.value = 'flash'
}
function stopFlashPoll() {
if (flashPollTimer) {
clearInterval(flashPollTimer)
flashPollTimer = null
}
}
async function pollFlashStatus() {
try {
const status = await mesh.flashStatus()
flashJob.value = status
if (!status.active) {
stopFlashPoll()
if (status.done && !status.error) {
// Mirrors the unplug/replug hot-swap flow: re-probe so the details
// card reflects whatever firmware is actually on the board now.
const path = devicePath.value
probing.value = true
try {
probe.value = await mesh.probeDevice(path)
} catch {
probe.value = null
} finally {
probing.value = false
}
}
}
} catch {
stopFlashPoll()
}
}
async function startFlash() {
if (!flashFamily.value || !flashBoard.value || !flashConfirmed.value) return
starting.value = true
error.value = ''
try {
await mesh.flashDevice(devicePath.value, flashFamily.value, flashBoard.value)
flashJob.value = { active: true, stage: 'downloading', log_tail: [] }
stopFlashPoll()
flashPollTimer = setInterval(pollFlashStatus, 1500)
} catch (e) {
error.value = e instanceof Error ? e.message : 'Failed to start flashing'
} finally {
starting.value = false
}
}
async function cancelFlash() {
try {
await mesh.flashCancel()
} catch (e) {
error.value = e instanceof Error ? e.message : 'Failed to cancel'
}
}
function closeFlashStep() {
stopFlashPoll()
step.value = 1
}
</script>
<style scoped>

View File

@ -55,23 +55,6 @@ export interface MeshDeviceProbe {
max_contacts: number | null
}
export type FlashFirmwareFamily = 'meshcore' | 'meshtastic' | 'reticulum'
export type FlashBoard = 'heltec-v3' | 'heltec-v4'
export type FlashStage = 'downloading' | 'erasing' | 'writing' | 'autoinstalling' | 'done' | 'failed'
/** Live progress for the one flash job that can run at a time. */
export interface FlashJobStatus {
active: boolean
board?: FlashBoard
family?: FlashFirmwareFamily
path?: string
stage?: FlashStage
percent?: number | null
log_tail?: string[]
done?: boolean
error?: string | null
}
/** Params accepted by mesh.configure (superset of the status fields). */
export interface MeshConfigureParams {
enabled?: boolean
@ -360,32 +343,6 @@ export const useMeshStore = defineStore('mesh', () => {
timeout: 45000, // serial probes are slow (multi-firmware handshakes)
})
}
/** Available firmware version(s) for a family v1 only ever returns
* ["latest"], since firmware is always fetched from upstream at flash
* time rather than pinned/bundled. */
async function flashListFirmware(family: FlashFirmwareFamily): Promise<string[]> {
const res = await rpcClient.call<{ versions: string[] }>({
method: 'mesh.flash-list-firmware',
params: { family },
})
return res.versions
}
/** Erase + reflash a detected radio. `board` is optional omit it to let
* the backend auto-resolve from the port's USB vid:pid; if that fails
* (e.g. Heltec V4 not yet in the vid:pid table), it errors and the UI
* must ask the user to pick the board explicitly. Always erases first. */
async function flashDevice(path: string, family: FlashFirmwareFamily, board?: FlashBoard): Promise<void> {
await rpcClient.call({
method: 'mesh.flash-device',
params: board ? { path, family, board } : { path, family },
})
}
async function flashStatus(): Promise<FlashJobStatus> {
return rpcClient.call<FlashJobStatus>({ method: 'mesh.flash-status' })
}
async function flashCancel(): Promise<void> {
await rpcClient.call({ method: 'mesh.flash-cancel' })
}
let globalDetectTimer: ReturnType<typeof setInterval> | null = null
/** App-wide light poll so the detected-device modal works on every page
* (the Mesh view's own 5s poll takes over while it is mounted). */
@ -1022,10 +979,6 @@ export const useMeshStore = defineStore('mesh', () => {
undismissedDetectedDevices,
dismissDetectedDevice,
probeDevice,
flashListFirmware,
flashDevice,
flashStatus,
flashCancel,
startGlobalDetection,
fetchPeers,
fetchMessages,

View File

@ -47,16 +47,11 @@ if [ -n "$RNODECONF_SRC" ] && [ -f "$RNODECONF_SRC" ]; then
# exit()/quit() builtins, which only exist in interactive Python (site.py
# injects them) — a frozen app hits NameError right as it tries to quit
# cleanly, after all the real work already succeeded. See
# pyi_rthook_exit_builtins.py. A second hook fixes rnodeconf's board-flash
# step, which shells out to a bundled esptool.py via `sys.executable` —
# under a frozen binary that's the binary itself, not a real interpreter,
# so the flash subprocess call breaks. See
# pyi_rthook_fix_flasher_executable.py.
# pyi_rthook_exit_builtins.py.
.venv/bin/pyinstaller --onefile --name archy-rnodeconf --clean --noconfirm \
--collect-submodules RNS \
--collect-data RNS \
--runtime-hook pyi_rthook_exit_builtins.py \
--runtime-hook pyi_rthook_fix_flasher_executable.py \
-d noarchive \
"$RNODECONF_SRC"
echo "Built dist/archy-rnodeconf ($(du -h dist/archy-rnodeconf | cut -f1))"

View File

@ -1,34 +0,0 @@
# PyInstaller runtime hook — see build.sh.
#
# rnodeconf's own board-flashing code shells out to a bundled esptool.py as
# `[sys.executable, flasher_path, "--chip", ..., "write_flash", ...]` (RNS's
# rnodeconf.py, ~line 2794 as of RNS 1.3.5). That's correct for a normal
# `python rnodeconf.py` invocation, but under a frozen PyInstaller binary
# `sys.executable` is the frozen binary itself, not a real interpreter — so
# the "subprocess" just re-invokes archy-rnodeconf's OWN argparse CLI with
# esptool-shaped flags, which it doesn't recognize, and the flash step fails
# immediately with "unrecognized arguments: --chip ...". Confirmed live
# against a real Heltec V4 (2026-07-23): device selection, band selection,
# and firmware download all worked; only the final `write_flash` subprocess
# call broke this way.
#
# Fix: point sys.executable at a real Python interpreter that has rnodeconf's
# own runtime deps available (esptool.py only needs pyserial, which RNS
# already depends on) before any of rnodeconf's code runs. Prefer the build
# venv this exact binary was frozen from — see build.sh — falling back to a
# bare `python3` on PATH if that venv isn't present on this node.
import os
import sys
if getattr(sys, "frozen", False):
_candidates = [
os.environ.get("ARCHY_RNODECONF_PYTHON", ""),
os.path.join(os.path.dirname(os.path.abspath(sys.argv[0])), "..", "reticulum-daemon", ".venv", "bin", "python3"),
os.path.expanduser("~/archy/reticulum-daemon/.venv/bin/python3"),
]
for _candidate in _candidates:
if _candidate and os.path.isfile(_candidate):
sys.executable = _candidate
break
else:
sys.executable = "python3"

View File

@ -84,64 +84,6 @@ if ! command -v nano >/dev/null 2>&1; then
fi
fi
if ! command -v ping >/dev/null 2>&1; then
log "Installing iputils-ping..."
if sudo apt-get update -qq 2>>"$LOG_FILE" && sudo apt-get install -y -qq iputils-ping 2>>"$LOG_FILE"; then
ok "ping installed"
else
warn "Unable to install ping automatically; continuing update"
fi
fi
if ! command -v esptool >/dev/null 2>&1; then
log "Installing esptool for LoRa radio firmware flashing..."
if sudo apt-get update -qq 2>>"$LOG_FILE" && sudo apt-get install -y -qq esptool 2>>"$LOG_FILE"; then
ok "esptool installed"
else
warn "Unable to install esptool automatically; radio firmware flashing will be unavailable"
fi
fi
# Debian's esptool package (4.7.0+dfsg-0.1) ships without the precompiled
# esp32s3 "stub flasher" blob (stripped for DFSG compliance — no
# buildable-from-source path Debian could verify). Without it, esptool's
# normal stub-loader mode fails outright (FileNotFoundError), and the ROM
# bootloader fallback (--no-stub) doesn't implement a full-chip erase at
# all — confirmed live 2026-07-23 flashing a real Heltec V4, both ways.
# Fetching the exact same file from the matching upstream esptool release
# tag restores full (and correct) flashing behavior — it's the same
# open-source codebase, just the one blob Debian's packaging couldn't
# include.
if command -v esptool >/dev/null 2>&1; then
STUB_DIR="/usr/lib/python3/dist-packages/esptool/targets/stub_flasher"
STUB_FILE="$STUB_DIR/stub_flasher_32s3.json"
if [ ! -f "$STUB_FILE" ]; then
log "Fetching esptool's esp32s3 stub flasher (missing from the Debian package)..."
ESPTOOL_VERSION=$(esptool version 2>/dev/null | tail -1 | tr -d ' \t')
if [ -n "$ESPTOOL_VERSION" ] && sudo curl -fsSL -o "$STUB_FILE" \
"https://raw.githubusercontent.com/espressif/esptool/v${ESPTOOL_VERSION}/esptool/targets/stub_flasher/stub_flasher_32s3.json" \
2>>"$LOG_FILE"; then
sudo chmod 644 "$STUB_FILE"
ok "esp32s3 stub flasher installed"
else
sudo rm -f "$STUB_FILE" 2>/dev/null
warn "Unable to fetch esp32s3 stub flasher; LoRa firmware flashing will be unavailable"
fi
fi
fi
# Build-time prerequisites for reticulum-daemon/build.sh's PyInstaller step
# below (discovered the hard way: ensurepip needs python3-venv, and
# PyInstaller itself needs objdump + libpython3.13.so at build time — none
# of these are pulled in by a bare `python3` package on Debian trixie).
for pkg in python3-venv binutils libpython3.13; do
if ! dpkg -s "$pkg" >/dev/null 2>&1; then
log "Installing $pkg (reticulum-daemon build prerequisite)..."
sudo apt-get update -qq 2>>"$LOG_FILE" && sudo apt-get install -y -qq "$pkg" 2>>"$LOG_FILE" \
|| warn "Unable to install $pkg automatically; reticulum-daemon tools build may fail"
fi
done
# Fetch latest
log "Fetching from origin..."
git fetch origin main --quiet 2>>"$LOG_FILE"
@ -213,30 +155,6 @@ sudo cp "$BUILT_BIN" "$INSTALL_BIN"
sudo chmod +x "$INSTALL_BIN"
ok "Backend installed"
# Build + install reticulum-daemon tools (archy-reticulum-daemon, archy-rnodeconf).
# Non-fatal: archipelago falls back to its dev venv path if the packaged
# binaries aren't present, so a missing/failed build here degrades mesh
# Reticulum support rather than breaking the update. This mirrors
# deploy-to-target.sh's existing manual-deploy step, which until now was the
# only path that ever installed these — a node that only ever received OTA
# self-updates had neither binary.
if [ -f "$REPO_DIR/reticulum-daemon/build.sh" ]; then
log "Building reticulum-daemon tools (archy-reticulum-daemon, archy-rnodeconf)..."
if (cd "$REPO_DIR/reticulum-daemon" && ./build.sh) 2>>"$LOG_FILE"; then
for tool in archy-reticulum-daemon archy-rnodeconf; do
if [ -f "$REPO_DIR/reticulum-daemon/dist/$tool" ]; then
sudo cp "$REPO_DIR/reticulum-daemon/dist/$tool" /usr/local/bin/
sudo chmod +x "/usr/local/bin/$tool"
ok "$tool installed"
else
warn "$tool not built — leaving existing /usr/local/bin/$tool (if any) in place"
fi
done
else
warn "reticulum-daemon tools build failed — continuing without updating them"
fi
fi
# Build frontend
log "Building Vue frontend (production)..."
cd "$FRONTEND_DIR"