Compare commits
12 Commits
8213b0aee3
...
5799c37111
| Author | SHA1 | Date | |
|---|---|---|---|
| 5799c37111 | |||
| b14af20d1a | |||
| 7547d03166 | |||
| d6019e47a5 | |||
| 0ca8f25b1b | |||
| 76d14c3bf9 | |||
| a1cb83dfb2 | |||
| ff532465cf | |||
| 39e88529b3 | |||
| 992bf636e0 | |||
| beff5dd577 | |||
| 7d31ca5d65 |
50
core/Cargo.lock
generated
50
core/Cargo.lock
generated
@ -84,6 +84,15 @@ version = "1.0.100"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a23eb6b1614318a8071c9b2521f36b424b2c83db5eb3a0fead4a6c0809af6e61"
|
||||
|
||||
[[package]]
|
||||
name = "arbitrary"
|
||||
version = "1.4.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c3d036a3c4ab069c7b410a2ce876bd74808d2d0888a82667669f8e783a898bf1"
|
||||
dependencies = [
|
||||
"derive_arbitrary",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "arc-swap"
|
||||
version = "1.9.1"
|
||||
@ -160,6 +169,7 @@ dependencies = [
|
||||
"uuid",
|
||||
"zbase32",
|
||||
"zeroize",
|
||||
"zip",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@ -1174,6 +1184,17 @@ version = "0.5.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c"
|
||||
|
||||
[[package]]
|
||||
name = "derive_arbitrary"
|
||||
version = "1.4.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1e567bd82dcff979e4b03460c307b3cdc9e96fde3d73bed1496d2bc75d9dd62a"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.114",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "derive_builder"
|
||||
version = "0.20.2"
|
||||
@ -6894,8 +6915,37 @@ dependencies = [
|
||||
"syn 2.0.114",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zip"
|
||||
version = "2.4.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fabe6324e908f85a1c52063ce7aa26b68dcb7eb6dbc83a2d148403c9bc3eba50"
|
||||
dependencies = [
|
||||
"arbitrary",
|
||||
"crc32fast",
|
||||
"crossbeam-utils",
|
||||
"displaydoc",
|
||||
"flate2",
|
||||
"indexmap",
|
||||
"memchr",
|
||||
"thiserror 2.0.18",
|
||||
"zopfli",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zmij"
|
||||
version = "1.0.16"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dfcd145825aace48cff44a8844de64bf75feec3080e0aa5cdbde72961ae51a65"
|
||||
|
||||
[[package]]
|
||||
name = "zopfli"
|
||||
version = "0.8.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f05cd8797d63865425ff89b5c4a48804f35ba0ce8d125800027ad6017d2b5249"
|
||||
dependencies = [
|
||||
"bumpalo",
|
||||
"crc32fast",
|
||||
"log",
|
||||
"simd-adler32",
|
||||
]
|
||||
|
||||
@ -105,6 +105,10 @@ bytes = "1"
|
||||
# Mesh networking (Meshcore serial protocol over USB LoRa radios)
|
||||
serial2-tokio = "0.1"
|
||||
|
||||
# LoRa radio firmware flashing: Meshtastic ships per-board images inside a
|
||||
# per-platform release zip (see mesh/flash.rs).
|
||||
zip = { version = "2", default-features = false, features = ["deflate"] }
|
||||
|
||||
# Double Ratchet key derivation (Phase 3: encrypted mesh messaging)
|
||||
hkdf = "0.12.4"
|
||||
|
||||
|
||||
@ -387,6 +387,10 @@ impl RpcHandler {
|
||||
// Mesh networking (Meshcore LoRa)
|
||||
"mesh.status" => self.handle_mesh_status().await,
|
||||
"mesh.probe-device" => self.handle_mesh_probe_device(params).await,
|
||||
"mesh.flash-list-firmware" => self.handle_mesh_flash_list_firmware(params).await,
|
||||
"mesh.flash-device" => self.handle_mesh_flash_device(params).await,
|
||||
"mesh.flash-status" => self.handle_mesh_flash_status().await,
|
||||
"mesh.flash-cancel" => self.handle_mesh_flash_cancel().await,
|
||||
"mesh.peers" => self.handle_mesh_peers().await,
|
||||
"mesh.messages" => self.handle_mesh_messages(params).await,
|
||||
"mesh.debug-dump" => self.handle_mesh_debug_dump().await,
|
||||
|
||||
132
core/archipelago/src/api/rpc/mesh/flash.rs
Normal file
132
core/archipelago/src/api/rpc/mesh/flash.rs
Normal file
@ -0,0 +1,132 @@
|
||||
use super::super::RpcHandler;
|
||||
use crate::mesh;
|
||||
use crate::mesh::flash::{self, FlashBoard, FlashJobStatus};
|
||||
use crate::mesh::types::DeviceType;
|
||||
use anyhow::Result;
|
||||
|
||||
fn parse_family(s: &str) -> Result<DeviceType> {
|
||||
match s.trim().to_lowercase().as_str() {
|
||||
"meshcore" => Ok(DeviceType::Meshcore),
|
||||
"meshtastic" => Ok(DeviceType::Meshtastic),
|
||||
"reticulum" | "rnode" => Ok(DeviceType::Reticulum),
|
||||
other => anyhow::bail!("Unknown firmware family: {other} (expected meshcore|meshtastic|reticulum)"),
|
||||
}
|
||||
}
|
||||
|
||||
fn parse_board(s: &str) -> Result<FlashBoard> {
|
||||
match s.trim().to_lowercase().as_str() {
|
||||
"heltec-v3" | "heltec_v3" | "heltecv3" => Ok(FlashBoard::HeltecV3),
|
||||
"heltec-v4" | "heltec_v4" | "heltecv4" => Ok(FlashBoard::HeltecV4),
|
||||
other => anyhow::bail!("Unknown board: {other} (expected heltec-v3|heltec-v4)"),
|
||||
}
|
||||
}
|
||||
|
||||
impl RpcHandler {
|
||||
/// mesh.flash-list-firmware — resolve the available firmware version(s)
|
||||
/// for a given family. v1 only ever surfaces "latest".
|
||||
pub(in crate::api::rpc) async fn handle_mesh_flash_list_firmware(
|
||||
&self,
|
||||
params: Option<serde_json::Value>,
|
||||
) -> Result<serde_json::Value> {
|
||||
let family = params
|
||||
.as_ref()
|
||||
.and_then(|p| p.get("family"))
|
||||
.and_then(|v| v.as_str())
|
||||
.ok_or_else(|| anyhow::anyhow!("Missing family"))?;
|
||||
let family = parse_family(family)?;
|
||||
let versions = flash::list_firmware(family).await?;
|
||||
Ok(serde_json::json!({ "versions": versions }))
|
||||
}
|
||||
|
||||
/// mesh.flash-device — erase and reflash a detected LoRa radio with the
|
||||
/// latest firmware for the given family, defaulting to a full chip
|
||||
/// erase before write. `board` is optional: if the port's USB vid:pid
|
||||
/// unambiguously resolves to a known board, that's used; otherwise the
|
||||
/// caller must supply it explicitly (see `flash::resolve_flash_board`'s
|
||||
/// doc comment on why we refuse to guess).
|
||||
pub(in crate::api::rpc) async fn handle_mesh_flash_device(
|
||||
&self,
|
||||
params: Option<serde_json::Value>,
|
||||
) -> Result<serde_json::Value> {
|
||||
let path = params
|
||||
.as_ref()
|
||||
.and_then(|p| p.get("path"))
|
||||
.and_then(|v| v.as_str())
|
||||
.ok_or_else(|| anyhow::anyhow!("Missing path"))?
|
||||
.to_string();
|
||||
let family = params
|
||||
.as_ref()
|
||||
.and_then(|p| p.get("family"))
|
||||
.and_then(|v| v.as_str())
|
||||
.ok_or_else(|| anyhow::anyhow!("Missing family"))?;
|
||||
let family = parse_family(family)?;
|
||||
|
||||
let detected = mesh::detect_devices().await;
|
||||
anyhow::ensure!(
|
||||
detected.iter().any(|d| d == &path),
|
||||
"{path} is not a detected mesh-radio candidate port"
|
||||
);
|
||||
|
||||
let board = match params
|
||||
.as_ref()
|
||||
.and_then(|p| p.get("board"))
|
||||
.and_then(|v| v.as_str())
|
||||
{
|
||||
Some(explicit) => parse_board(explicit)?,
|
||||
None => {
|
||||
let info = mesh::detect_devices_info()
|
||||
.await
|
||||
.into_iter()
|
||||
.find(|d| d.path == path);
|
||||
info.as_ref()
|
||||
.and_then(flash::resolve_flash_board)
|
||||
.ok_or_else(|| {
|
||||
anyhow::anyhow!(
|
||||
"Could not auto-detect the board on {path} — specify board explicitly"
|
||||
)
|
||||
})?
|
||||
}
|
||||
};
|
||||
|
||||
flash::start_flash_job(
|
||||
&self.flash_job,
|
||||
&self.mesh_service_arc(),
|
||||
self.config.data_dir.clone(),
|
||||
path,
|
||||
board,
|
||||
family,
|
||||
)
|
||||
.await?;
|
||||
|
||||
Ok(serde_json::json!({ "started": true }))
|
||||
}
|
||||
|
||||
/// mesh.flash-status — poll the current (or most recent) flash job.
|
||||
pub(in crate::api::rpc) async fn handle_mesh_flash_status(&self) -> Result<serde_json::Value> {
|
||||
let job = self.flash_job.read().await;
|
||||
match job.as_ref() {
|
||||
Some(j) => {
|
||||
let status: FlashJobStatus = j.snapshot().await;
|
||||
let mut value = serde_json::to_value(&status)?;
|
||||
if let Some(obj) = value.as_object_mut() {
|
||||
obj.insert("active".into(), (!status.done).into());
|
||||
}
|
||||
Ok(value)
|
||||
}
|
||||
None => Ok(serde_json::json!({ "active": false })),
|
||||
}
|
||||
}
|
||||
|
||||
/// mesh.flash-cancel — best-effort; only honored before erase/write has
|
||||
/// started (see `FlashJob::cancel`'s doc comment).
|
||||
pub(in crate::api::rpc) async fn handle_mesh_flash_cancel(&self) -> Result<serde_json::Value> {
|
||||
let job = self.flash_job.read().await;
|
||||
match job.as_ref() {
|
||||
Some(j) => {
|
||||
j.cancel().await?;
|
||||
Ok(serde_json::json!({ "cancelled": true }))
|
||||
}
|
||||
None => anyhow::bail!("No flash job in progress"),
|
||||
}
|
||||
}
|
||||
}
|
||||
@ -1,5 +1,6 @@
|
||||
mod assistant;
|
||||
mod bitcoin_ops;
|
||||
mod flash;
|
||||
mod messaging;
|
||||
mod safety;
|
||||
mod status;
|
||||
|
||||
@ -88,6 +88,9 @@ pub struct RpcHandler {
|
||||
endpoint_rate_limiter: EndpointRateLimiter,
|
||||
response_cache: ResponseCache,
|
||||
mesh_service: Arc<tokio::sync::RwLock<Option<crate::mesh::MeshService>>>,
|
||||
/// LoRa radio firmware-flash job state, sibling to `mesh_service` — one
|
||||
/// job at a time, since flashing needs exclusive access to the port.
|
||||
flash_job: crate::mesh::flash::FlashJobHandle,
|
||||
transport_router: Arc<tokio::sync::RwLock<Option<Arc<crate::transport::TransportRouter>>>>,
|
||||
/// Shared content-addressed blob store. Set by ApiHandler after construction
|
||||
/// so mesh.send-content / mesh.fetch-content RPCs can reach it without a
|
||||
@ -159,6 +162,7 @@ impl RpcHandler {
|
||||
endpoint_rate_limiter,
|
||||
response_cache: ResponseCache::new(5),
|
||||
mesh_service: Arc::new(tokio::sync::RwLock::new(None)),
|
||||
flash_job: crate::mesh::flash::new_job_handle(),
|
||||
transport_router: Arc::new(tokio::sync::RwLock::new(None)),
|
||||
blob_store: Arc::new(tokio::sync::RwLock::new(None)),
|
||||
self_pubkey_hex: Arc::new(tokio::sync::RwLock::new(None)),
|
||||
|
||||
930
core/archipelago/src/mesh/flash.rs
Normal file
930
core/archipelago/src/mesh/flash.rs
Normal file
@ -0,0 +1,930 @@
|
||||
// WIP mesh/transport protocol — suppress dead code warnings
|
||||
#![allow(dead_code)]
|
||||
//! Firmware flashing for LoRa mesh radios — Heltec V3/V4 in v1, across all
|
||||
//! three firmware families the mesh module already knows how to detect (see
|
||||
//! `mesh::types::DeviceType`). Firmware is always fetched from upstream at
|
||||
//! flash time (never bundled/pinned in the repo), and every flash defaults
|
||||
//! to a full chip erase before write.
|
||||
//!
|
||||
//! MeshCore and Meshtastic are flashed the same way: download a released
|
||||
//! image, `esptool erase_flash`, then `esptool write_flash 0x0 <image>`.
|
||||
//! Reticulum/RNode is different: `archy-rnodeconf --autoinstall` owns the
|
||||
//! whole fetch+erase+flash+EEPROM-bootstrap sequence itself (confirmed live
|
||||
//! via `archy-rnodeconf --help` — there is no raw esptool path exposed for
|
||||
//! this family, so we deliberately don't resolve a firmware URL ourselves
|
||||
//! for Reticulum; rnodeconf already knows how).
|
||||
|
||||
use super::serial::DetectedDeviceInfo;
|
||||
use super::types::DeviceType;
|
||||
use super::MeshService;
|
||||
use anyhow::{Context, Result};
|
||||
use regex::Regex;
|
||||
use serde::Serialize;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::Stdio;
|
||||
use std::sync::{Arc, OnceLock};
|
||||
use tokio::io::{AsyncBufReadExt, AsyncWriteExt, BufReader};
|
||||
use tokio::process::Command;
|
||||
use tokio::sync::RwLock;
|
||||
use tracing::{info, warn};
|
||||
|
||||
/// Boards supported for v1. Both are ESP32-S3 (a single `--chip esp32s3`
|
||||
/// esptool target covers both), but ship different USB identities and
|
||||
/// different per-board firmware assets upstream.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum FlashBoard {
|
||||
HeltecV3,
|
||||
HeltecV4,
|
||||
}
|
||||
|
||||
impl FlashBoard {
|
||||
/// Meshtastic's board id (matches the release manifest's `board` field
|
||||
/// and its per-board asset naming, e.g. `firmware-heltec-v3-<ver>.factory.bin`).
|
||||
fn meshtastic_id(self) -> &'static str {
|
||||
match self {
|
||||
Self::HeltecV3 => "heltec-v3",
|
||||
Self::HeltecV4 => "heltec-v4",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Map a detected USB vid:pid to a known flashable board, using the same
|
||||
/// table as `image-recipe/configs/99-mesh-radio.rules`. CP2102 (10c4:ea60)
|
||||
/// is confirmed there as Heltec V3's USB-UART bridge chip, and is safe to
|
||||
/// auto-match since that vid:pid is bridge-chip-specific.
|
||||
///
|
||||
/// Heltec V4 is NOT auto-matchable and deliberately has no entry here: it
|
||||
/// was confirmed live (real hardware, 2026-07-23) to use the ESP32-S3's
|
||||
/// built-in native-USB JTAG/serial peripheral, reporting vid:pid 303a:1001
|
||||
/// with product string "USB JTAG/serial debug unit" — that descriptor is
|
||||
/// baked into the chip's ROM and is IDENTICAL across every ESP32-S3 board
|
||||
/// with native USB enabled, not just Heltec V4. Adding `303a:1001 =>
|
||||
/// HeltecV4` here would silently misidentify any other native-USB ESP32-S3
|
||||
/// board (a T3-S3, a bare devkit, etc.) as a V4 and risk writing the wrong
|
||||
/// board's image. Callers (the RPC layer / frontend) must let the user pick
|
||||
/// the board manually whenever this returns `None`.
|
||||
pub fn resolve_flash_board(info: &DetectedDeviceInfo) -> Option<FlashBoard> {
|
||||
match (info.vid.as_deref(), info.pid.as_deref()) {
|
||||
(Some("10c4"), Some("ea60")) => Some(FlashBoard::HeltecV3),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum FlashStage {
|
||||
Downloading,
|
||||
Erasing,
|
||||
Writing,
|
||||
Autoinstalling,
|
||||
Done,
|
||||
Failed,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct FlashJobStatus {
|
||||
pub board: FlashBoard,
|
||||
pub family: DeviceType,
|
||||
pub path: String,
|
||||
pub stage: FlashStage,
|
||||
pub percent: Option<u8>,
|
||||
pub log_tail: Vec<String>,
|
||||
pub done: bool,
|
||||
pub error: Option<String>,
|
||||
}
|
||||
|
||||
const LOG_TAIL_MAX: usize = 200;
|
||||
|
||||
/// How long to wait after a successful flash before resuming the mesh
|
||||
/// listener, so the board finishes its own post-flash boot/reset before we
|
||||
/// start opening the port (which itself toggles DTR/RTS) again.
|
||||
const POST_FLASH_SETTLE_DELAY: std::time::Duration = std::time::Duration::from_secs(5);
|
||||
|
||||
/// Absolute ceiling on a whole flash job (download + erase + write, or
|
||||
/// autoinstall), regardless of what it's doing internally. Last-resort
|
||||
/// safety net so a hang anywhere can't wedge the single-flash-job guard
|
||||
/// forever — generous enough to never trigger on a legitimately slow
|
||||
/// multi-hundred-MB transfer.
|
||||
const MAX_JOB_DURATION: std::time::Duration = std::time::Duration::from_secs(15 * 60);
|
||||
|
||||
/// How long to wait for MeshService::stop() to release the serial port
|
||||
/// before giving up. Confirmed live 2026-07-23: the listener's own
|
||||
/// reconnect/multi-candidate-probe loop doesn't check its shutdown signal
|
||||
/// between candidates, so stop() can take a while (or, if the loop is
|
||||
/// wedged, never return) — 20s comfortably covers a normal handshake-probe
|
||||
/// cycle without leaving a flash request hanging indefinitely if the
|
||||
/// listener genuinely won't let go.
|
||||
const STOP_LISTENER_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
|
||||
|
||||
/// Live state for the one flash job that can run at a time. A single global
|
||||
/// slot is sufficient because flashing needs exclusive serial access to the
|
||||
/// one port being flashed — there is no meaningful concept of two concurrent
|
||||
/// flash jobs on this node.
|
||||
pub struct FlashJob {
|
||||
status: RwLock<FlashJobStatus>,
|
||||
/// Set once the background task is spawned. Only used while `stage` is
|
||||
/// still `Downloading` — an interrupted erase/write can leave the chip
|
||||
/// in a worse state than either finished or unstarted, so cancellation
|
||||
/// is refused once erase begins (see `cancel()`).
|
||||
abort_handle: RwLock<Option<tokio::task::AbortHandle>>,
|
||||
}
|
||||
|
||||
impl FlashJob {
|
||||
fn new(board: FlashBoard, family: DeviceType, path: String) -> Arc<Self> {
|
||||
Arc::new(Self {
|
||||
abort_handle: RwLock::new(None),
|
||||
status: RwLock::new(FlashJobStatus {
|
||||
board,
|
||||
family,
|
||||
path,
|
||||
stage: FlashStage::Downloading,
|
||||
percent: None,
|
||||
log_tail: Vec::new(),
|
||||
done: false,
|
||||
error: None,
|
||||
}),
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn snapshot(&self) -> FlashJobStatus {
|
||||
self.status.read().await.clone()
|
||||
}
|
||||
|
||||
async fn set_stage(&self, stage: FlashStage) {
|
||||
let mut s = self.status.write().await;
|
||||
s.stage = stage;
|
||||
s.percent = None;
|
||||
}
|
||||
|
||||
async fn set_percent(&self, percent: u8) {
|
||||
self.status.write().await.percent = Some(percent.min(100));
|
||||
}
|
||||
|
||||
async fn push_log(&self, line: impl Into<String>) {
|
||||
let mut s = self.status.write().await;
|
||||
s.log_tail.push(line.into());
|
||||
let overflow = s.log_tail.len().saturating_sub(LOG_TAIL_MAX);
|
||||
if overflow > 0 {
|
||||
s.log_tail.drain(0..overflow);
|
||||
}
|
||||
}
|
||||
|
||||
async fn fail(&self, err: &anyhow::Error) {
|
||||
let mut s = self.status.write().await;
|
||||
s.stage = FlashStage::Failed;
|
||||
s.error = Some(format!("{err:#}"));
|
||||
s.done = true;
|
||||
}
|
||||
|
||||
async fn finish(&self) {
|
||||
let mut s = self.status.write().await;
|
||||
s.stage = FlashStage::Done;
|
||||
s.done = true;
|
||||
}
|
||||
|
||||
/// Best-effort cancel: only honored before erase/write/autoinstall has
|
||||
/// started (i.e. still in `Downloading`). Once a stage that touches the
|
||||
/// chip begins, this refuses — interrupting an erase or write can leave
|
||||
/// the flash in a state worse than either finished or unstarted.
|
||||
pub async fn cancel(&self) -> Result<()> {
|
||||
let mut s = self.status.write().await;
|
||||
if s.done {
|
||||
anyhow::bail!("Flash job already finished");
|
||||
}
|
||||
if s.stage != FlashStage::Downloading {
|
||||
anyhow::bail!(
|
||||
"Cannot cancel once {:?} has started — let it finish or fail on its own",
|
||||
s.stage
|
||||
);
|
||||
}
|
||||
if let Some(handle) = self.abort_handle.write().await.take() {
|
||||
handle.abort();
|
||||
}
|
||||
s.stage = FlashStage::Failed;
|
||||
s.error = Some("Cancelled by user".to_string());
|
||||
s.done = true;
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Shared handle held by `RpcHandler`, sibling to `mesh_service`.
|
||||
pub type FlashJobHandle = Arc<RwLock<Option<Arc<FlashJob>>>>;
|
||||
|
||||
pub fn new_job_handle() -> FlashJobHandle {
|
||||
Arc::new(RwLock::new(None))
|
||||
}
|
||||
|
||||
fn firmware_cache_dir(data_dir: &Path) -> PathBuf {
|
||||
data_dir.join("mesh").join("firmware-cache")
|
||||
}
|
||||
|
||||
/// No blanket `.timeout()` here on purpose: reqwest's request timeout covers
|
||||
/// the *entire* request including streaming the response body, which would
|
||||
/// kill a legitimate large download partway through (Meshtastic's esp32s3
|
||||
/// zip is ~170MB) — not just a hung connection. `download_to_file` instead
|
||||
/// applies a per-chunk stall timeout, and metadata calls (small JSON
|
||||
/// responses) get their own short timeout at the call site.
|
||||
fn github_client() -> Result<reqwest::Client> {
|
||||
reqwest::Client::builder()
|
||||
.user_agent("archipelago-mesh-flash")
|
||||
.connect_timeout(std::time::Duration::from_secs(10))
|
||||
.build()
|
||||
.context("Failed to build HTTP client")
|
||||
}
|
||||
|
||||
/// Applied per-chunk while streaming a firmware download — if the transfer
|
||||
/// stalls (no bytes for this long) it's treated as a failure, but a slow
|
||||
/// download that's still making progress is never killed just for taking a
|
||||
/// while.
|
||||
const DOWNLOAD_STALL_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(30);
|
||||
|
||||
/// Applied to metadata calls (GitHub release JSON) — these are small
|
||||
/// responses with no reason to ever take this long.
|
||||
const METADATA_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
|
||||
|
||||
/// Resolve what firmware is available for a board+family. v1 only ever
|
||||
/// offers "latest" — MeshCore/Meshtastic latest GitHub release, or, for
|
||||
/// Reticulum, "latest" meaning "whatever archy-rnodeconf --autoinstall
|
||||
/// resolves on its own" (it does its own version checking upstream).
|
||||
pub async fn list_firmware(family: DeviceType) -> Result<Vec<String>> {
|
||||
match family {
|
||||
DeviceType::Reticulum => Ok(vec!["latest".to_string()]),
|
||||
DeviceType::Meshtastic => {
|
||||
let client = github_client()?;
|
||||
let release: GithubRelease = client
|
||||
.get("https://api.github.com/repos/meshtastic/firmware/releases/latest")
|
||||
.send()
|
||||
.await
|
||||
.context("Fetching Meshtastic release list")?
|
||||
.error_for_status()
|
||||
.context("Meshtastic releases API error")?
|
||||
.json()
|
||||
.await
|
||||
.context("Parsing Meshtastic release JSON")?;
|
||||
Ok(vec![release.tag_name])
|
||||
}
|
||||
DeviceType::Meshcore => {
|
||||
let client = github_client()?;
|
||||
let release: GithubRelease = client
|
||||
.get("https://api.github.com/repos/meshcore-dev/MeshCore/releases/latest")
|
||||
.send()
|
||||
.await
|
||||
.context("Fetching MeshCore release list")?
|
||||
.error_for_status()
|
||||
.context("MeshCore releases API error")?
|
||||
.json()
|
||||
.await
|
||||
.context("Parsing MeshCore release JSON")?;
|
||||
Ok(vec![release.tag_name])
|
||||
}
|
||||
DeviceType::Unknown => anyhow::bail!("Pick a firmware family before listing versions"),
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
struct GithubAsset {
|
||||
name: String,
|
||||
browser_download_url: String,
|
||||
}
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
struct GithubRelease {
|
||||
tag_name: String,
|
||||
assets: Vec<GithubAsset>,
|
||||
}
|
||||
|
||||
/// Start a flash job in the background. Returns as soon as the job has been
|
||||
/// registered and the listener released — callers poll `FlashJobHandle` via
|
||||
/// `mesh.flash-status` for progress. Only one job may be in flight at a time.
|
||||
pub async fn start_flash_job(
|
||||
handle: &FlashJobHandle,
|
||||
mesh_service: &Arc<RwLock<Option<MeshService>>>,
|
||||
data_dir: PathBuf,
|
||||
path: String,
|
||||
board: FlashBoard,
|
||||
family: DeviceType,
|
||||
) -> Result<()> {
|
||||
{
|
||||
let existing = handle.read().await;
|
||||
if let Some(job) = existing.as_ref() {
|
||||
if !job.snapshot().await.done {
|
||||
anyhow::bail!("A firmware flash is already in progress on this node");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let job = FlashJob::new(board, family, path.clone());
|
||||
*handle.write().await = Some(Arc::clone(&job));
|
||||
|
||||
let bg_job = Arc::clone(&job);
|
||||
let bg_service = Arc::clone(mesh_service);
|
||||
let task = tokio::spawn(async move {
|
||||
// esptool/archy-rnodeconf need exclusive serial access — release
|
||||
// the listener's hold on the port before touching it. This USED
|
||||
// TO run synchronously in start_flash_job before the job was even
|
||||
// spawned, blocking the RPC call itself on s.stop().await — a real
|
||||
// 2026-07-23 incident: the mesh listener was mid a multi-candidate
|
||||
// reconnect/probe sequence that doesn't check its shutdown signal
|
||||
// between candidates, so stop() never returned. The HTTP request
|
||||
// timed out client-side ("Operation failed"), while the job
|
||||
// (already inserted into `handle`) was permanently wedged — nothing
|
||||
// had been spawned yet to ever mark it done, so every later flash
|
||||
// attempt failed with "already in progress" until a full restart.
|
||||
// Now this runs inside the spawned task with its own bounded
|
||||
// timeout, so the RPC call always returns immediately regardless,
|
||||
// and a slow-to-stop listener fails the job cleanly instead of
|
||||
// hanging everything downstream of it forever.
|
||||
let stop_result = tokio::time::timeout(STOP_LISTENER_TIMEOUT, async {
|
||||
let mut svc = bg_service.write().await;
|
||||
if let Some(s) = svc.as_mut() {
|
||||
s.stop().await;
|
||||
}
|
||||
})
|
||||
.await;
|
||||
if stop_result.is_err() {
|
||||
let err = anyhow::anyhow!(
|
||||
"Mesh listener did not release the serial port within {}s — it may still be mid a reconnect attempt. Try again once mesh.status shows the device idle, or restart the archipelago service if this persists.",
|
||||
STOP_LISTENER_TIMEOUT.as_secs()
|
||||
);
|
||||
bg_job.push_log(format!("ERROR: {err:#}")).await;
|
||||
bg_job.fail(&err).await;
|
||||
return;
|
||||
}
|
||||
|
||||
// Outer ceiling on top of run_flash's own internal timeouts —
|
||||
// belt-and-suspenders so that no future hang (network, subprocess,
|
||||
// anything) can ever wedge the single-flash-job guard permanently
|
||||
// again the way a stuck download did on 2026-07-23 (every
|
||||
// subsequent mesh.flash-device call failed with "already in
|
||||
// progress" until the service was restarted). Generous enough that
|
||||
// a legitimately slow multi-hundred-MB transfer still completes.
|
||||
let result = match tokio::time::timeout(
|
||||
MAX_JOB_DURATION,
|
||||
run_flash(board, family, &data_dir, &path, &bg_job),
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(inner) => inner,
|
||||
Err(_) => Err(anyhow::anyhow!(
|
||||
"Flash job exceeded the {}-minute ceiling — aborted",
|
||||
MAX_JOB_DURATION.as_secs() / 60
|
||||
)),
|
||||
};
|
||||
let succeeded = result.is_ok();
|
||||
|
||||
match &result {
|
||||
Ok(()) => {
|
||||
bg_job.push_log("Flash completed successfully".to_string()).await;
|
||||
bg_job.finish().await;
|
||||
info!(path = %path, board = ?board, family = %family, "LoRa firmware flash succeeded");
|
||||
}
|
||||
Err(e) => {
|
||||
// {:#} (alternate Display) walks the full anyhow context
|
||||
// chain — plain {} / %e only prints the outermost .context()
|
||||
// message, which made a real 2026-07-23 esptool failure
|
||||
// undiagnosable from journalctl alone (just "esptool
|
||||
// erase_flash failed", no actual esptool stderr).
|
||||
warn!(path = %path, error = %format!("{e:#}"), "LoRa firmware flash failed");
|
||||
bg_job.push_log(format!("ERROR: {e:#}")).await;
|
||||
bg_job.fail(e).await;
|
||||
}
|
||||
}
|
||||
|
||||
// The board's firmware may now differ from whatever was pinned
|
||||
// before — clear the pin either way so a later reconnect's strict
|
||||
// auto-detect order picks up reality instead of getting wedged
|
||||
// trying the old protocol first.
|
||||
if let Ok(mut config) = super::load_config(&data_dir).await {
|
||||
config.device_kind = None;
|
||||
if let Err(e) = super::save_config(&data_dir, &config).await {
|
||||
warn!(error = %e, "Failed to clear device_kind pin after flash");
|
||||
}
|
||||
}
|
||||
|
||||
if !succeeded {
|
||||
// Deliberately do NOT auto-restart the listener here. A failed
|
||||
// flash means we can't vouch for the board's state — reopening
|
||||
// the port immediately (esptool/rnodeconf's own reset sequence
|
||||
// plus our open() toggling DTR/RTS again right after) risks
|
||||
// hammering a marginal device with reconnect attempts. Confirmed
|
||||
// live 2026-07-23: exactly this sequence left a real Heltec V3
|
||||
// boot-looping for 5+ minutes after a failed flash. Leave mesh
|
||||
// stopped; the user reconnects explicitly via the hot-swap
|
||||
// modal/Mesh page once they've confirmed the board is alive.
|
||||
warn!(
|
||||
path = %path,
|
||||
"Leaving mesh listener stopped after failed flash — reconnect manually once the board is confirmed responsive"
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
// On success, give the board a moment to finish booting after the
|
||||
// flash tool's own reset sequence before we start hammering it with
|
||||
// connection attempts — same reasoning as above, just the
|
||||
// lower-risk (successful-flash) side of it.
|
||||
tokio::time::sleep(POST_FLASH_SETTLE_DELAY).await;
|
||||
|
||||
let mut svc = bg_service.write().await;
|
||||
if let Some(s) = svc.as_mut() {
|
||||
match super::load_config(&data_dir).await {
|
||||
Ok(config) => {
|
||||
// Only resume if mesh is actually still enabled per the
|
||||
// CURRENT persisted config — confirmed live 2026-07-23:
|
||||
// unconditionally forcing a restart here, regardless of
|
||||
// `enabled`, overrode a user's own concurrent "disable
|
||||
// mesh" toggle and left the listener running while
|
||||
// config said disabled. That inconsistent state is what
|
||||
// made a later legitimate "Keep As Is" click (which
|
||||
// correctly tries to start on a false→true transition)
|
||||
// fail with "already running" — the listener had already
|
||||
// been force-started behind the config's back.
|
||||
let should_run = config.enabled;
|
||||
if let Err(e) = s.configure(config).await {
|
||||
warn!(error = %e, "Failed to resume mesh listener after flash");
|
||||
}
|
||||
if should_run {
|
||||
if let Err(e) = s.start() {
|
||||
warn!(error = %e, "Failed to restart mesh listener after flash");
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => warn!(error = %e, "Failed to load mesh config after flash"),
|
||||
}
|
||||
}
|
||||
});
|
||||
*job.abort_handle.write().await = Some(task.abort_handle());
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn run_flash(
|
||||
board: FlashBoard,
|
||||
family: DeviceType,
|
||||
data_dir: &Path,
|
||||
path: &str,
|
||||
job: &Arc<FlashJob>,
|
||||
) -> Result<()> {
|
||||
match family {
|
||||
DeviceType::Meshtastic | DeviceType::Meshcore => {
|
||||
let image = fetch_esptool_image(board, family, data_dir, job).await?;
|
||||
esptool_erase_and_write(path, &image, job).await
|
||||
}
|
||||
DeviceType::Reticulum => {
|
||||
let lora_region = super::load_config(data_dir)
|
||||
.await
|
||||
.ok()
|
||||
.and_then(|c| c.lora_region);
|
||||
rnodeconf_autoinstall(path, board, lora_region.as_deref(), job).await
|
||||
}
|
||||
DeviceType::Unknown => anyhow::bail!("Pick a firmware family before flashing"),
|
||||
}
|
||||
}
|
||||
|
||||
// ─── MeshCore / Meshtastic: esptool ─────────────────────────────────────
|
||||
|
||||
async fn fetch_esptool_image(
|
||||
board: FlashBoard,
|
||||
family: DeviceType,
|
||||
data_dir: &Path,
|
||||
job: &Arc<FlashJob>,
|
||||
) -> Result<PathBuf> {
|
||||
let cache = firmware_cache_dir(data_dir);
|
||||
tokio::fs::create_dir_all(&cache)
|
||||
.await
|
||||
.context("Creating firmware cache dir")?;
|
||||
let client = github_client()?;
|
||||
|
||||
match family {
|
||||
DeviceType::Meshtastic => fetch_meshtastic_image(&client, board, &cache, job).await,
|
||||
DeviceType::Meshcore => fetch_meshcore_image(&client, board, &cache, job).await,
|
||||
_ => anyhow::bail!("{family} is not flashed via esptool"),
|
||||
}
|
||||
}
|
||||
|
||||
async fn fetch_meshtastic_image(
|
||||
client: &reqwest::Client,
|
||||
board: FlashBoard,
|
||||
cache: &Path,
|
||||
job: &Arc<FlashJob>,
|
||||
) -> Result<PathBuf> {
|
||||
let release: GithubRelease = client
|
||||
.get("https://api.github.com/repos/meshtastic/firmware/releases/latest")
|
||||
.timeout(METADATA_TIMEOUT)
|
||||
.send()
|
||||
.await
|
||||
.context("Fetching Meshtastic release list")?
|
||||
.error_for_status()
|
||||
.context("Meshtastic releases API error")?
|
||||
.json()
|
||||
.await
|
||||
.context("Parsing Meshtastic release JSON")?;
|
||||
|
||||
// Meshtastic bundles all esp32s3 boards' images inside one per-platform
|
||||
// zip rather than shipping per-board top-level assets — both Heltec V3
|
||||
// and V4 are esp32s3, so this is the right zip for both (confirmed live
|
||||
// against v2.7.26.54e0d8d).
|
||||
let zip_asset = release
|
||||
.assets
|
||||
.iter()
|
||||
.find(|a| a.name.starts_with("firmware-esp32s3-") && a.name.ends_with(".zip"))
|
||||
.ok_or_else(|| anyhow::anyhow!("No esp32s3 firmware zip in latest Meshtastic release"))?;
|
||||
|
||||
let version = zip_asset
|
||||
.name
|
||||
.strip_prefix("firmware-esp32s3-")
|
||||
.and_then(|s| s.strip_suffix(".zip"))
|
||||
.ok_or_else(|| anyhow::anyhow!("Unexpected Meshtastic asset name: {}", zip_asset.name))?
|
||||
.to_string();
|
||||
|
||||
let zip_path = cache.join(&zip_asset.name);
|
||||
if tokio::fs::metadata(&zip_path).await.is_err() {
|
||||
download_to_file(client, &zip_asset.browser_download_url, &zip_path, job).await?;
|
||||
} else {
|
||||
job.push_log(format!("Using cached {}", zip_asset.name)).await;
|
||||
}
|
||||
|
||||
// "*.factory.bin" is Meshtastic's full merged image (bootloader +
|
||||
// partition table + app) meant to be written at offset 0x0 on a freshly
|
||||
// erased chip — confirmed by inspecting the real zip's contents, as
|
||||
// opposed to the plain "*.bin" OTA-update image which assumes an
|
||||
// existing bootloader/partition table already on the chip.
|
||||
let entry_name = format!(
|
||||
"firmware-{}-{}.factory.bin",
|
||||
board.meshtastic_id(),
|
||||
version
|
||||
);
|
||||
let out_path = cache.join(&entry_name);
|
||||
if tokio::fs::metadata(&out_path).await.is_ok() {
|
||||
return Ok(out_path);
|
||||
}
|
||||
|
||||
job.push_log(format!(
|
||||
"Extracting {entry_name} from {}",
|
||||
zip_asset.name
|
||||
))
|
||||
.await;
|
||||
let zip_path_owned = zip_path.clone();
|
||||
let entry_name_owned = entry_name.clone();
|
||||
let out_path_owned = out_path.clone();
|
||||
tokio::task::spawn_blocking(move || -> Result<()> {
|
||||
let file = std::fs::File::open(&zip_path_owned).context("Opening downloaded firmware zip")?;
|
||||
let mut archive = zip::ZipArchive::new(file).context("Reading firmware zip")?;
|
||||
let mut entry = archive
|
||||
.by_name(&entry_name_owned)
|
||||
.with_context(|| format!("{entry_name_owned} not found in firmware zip"))?;
|
||||
let mut out =
|
||||
std::fs::File::create(&out_path_owned).context("Creating extracted firmware file")?;
|
||||
std::io::copy(&mut entry, &mut out).context("Extracting firmware image")?;
|
||||
Ok(())
|
||||
})
|
||||
.await
|
||||
.context("Firmware extraction task panicked")??;
|
||||
|
||||
Ok(out_path)
|
||||
}
|
||||
|
||||
async fn fetch_meshcore_image(
|
||||
client: &reqwest::Client,
|
||||
board: FlashBoard,
|
||||
cache: &Path,
|
||||
job: &Arc<FlashJob>,
|
||||
) -> Result<PathBuf> {
|
||||
let release: GithubRelease = client
|
||||
.get("https://api.github.com/repos/meshcore-dev/MeshCore/releases/latest")
|
||||
.timeout(METADATA_TIMEOUT)
|
||||
.send()
|
||||
.await
|
||||
.context("Fetching MeshCore release list")?
|
||||
.error_for_status()
|
||||
.context("MeshCore releases API error")?
|
||||
.json()
|
||||
.await
|
||||
.context("Parsing MeshCore release JSON")?;
|
||||
|
||||
// Upstream's casing differs between boards (Heltec_v3_... vs
|
||||
// heltec_v4_...) — match case-insensitively on the exact per-board
|
||||
// substring so V4 isn't accidentally matched by "heltec_v4_tft_..."
|
||||
// variants (there's a "_tft_" in between, so a straight substring match
|
||||
// on "heltec_v4_companion_radio_usb" is already safe).
|
||||
let needle = match board {
|
||||
FlashBoard::HeltecV3 => "heltec_v3_companion_radio_usb",
|
||||
FlashBoard::HeltecV4 => "heltec_v4_companion_radio_usb",
|
||||
};
|
||||
let asset = release
|
||||
.assets
|
||||
.iter()
|
||||
.find(|a| {
|
||||
let lower = a.name.to_lowercase();
|
||||
lower.contains(needle) && lower.ends_with("-merged.bin")
|
||||
})
|
||||
.ok_or_else(|| {
|
||||
anyhow::anyhow!("No matching MeshCore image in release {}", release.tag_name)
|
||||
})?;
|
||||
|
||||
let out_path = cache.join(&asset.name);
|
||||
if tokio::fs::metadata(&out_path).await.is_ok() {
|
||||
job.push_log(format!("Using cached {}", asset.name)).await;
|
||||
return Ok(out_path);
|
||||
}
|
||||
download_to_file(client, &asset.browser_download_url, &out_path, job).await?;
|
||||
Ok(out_path)
|
||||
}
|
||||
|
||||
async fn download_to_file(
|
||||
client: &reqwest::Client,
|
||||
url: &str,
|
||||
dest: &Path,
|
||||
job: &Arc<FlashJob>,
|
||||
) -> Result<()> {
|
||||
job.set_stage(FlashStage::Downloading).await;
|
||||
// Bound only the wait for the response to *start* (headers) — NOT a
|
||||
// request-level `.timeout()`, which would cap the whole body transfer
|
||||
// again (the bug this replaced: a blanket 30s client timeout killed
|
||||
// large downloads mid-stream). If the server never responds at all,
|
||||
// this is what stops the job from hanging forever; the per-chunk stall
|
||||
// timeout below is what guards the body once streaming starts. Without
|
||||
// this, a server that accepts the TCP connection but never sends
|
||||
// headers back hangs this call indefinitely — confirmed live
|
||||
// 2026-07-23: a stuck `.send()` here wedged the single-flash-job guard
|
||||
// for good, permanently blocking every subsequent flash attempt with
|
||||
// "already in progress" until the service was restarted.
|
||||
let resp = tokio::time::timeout(METADATA_TIMEOUT, client.get(url).send())
|
||||
.await
|
||||
.context("Firmware download server did not respond")?
|
||||
.context("Starting firmware download")?
|
||||
.error_for_status()
|
||||
.context("Firmware download returned an error status")?;
|
||||
let total = resp.content_length();
|
||||
let tmp = dest.with_extension("part");
|
||||
let mut file = tokio::fs::File::create(&tmp)
|
||||
.await
|
||||
.context("Creating firmware download file")?;
|
||||
let mut stream = resp.bytes_stream();
|
||||
let mut downloaded: u64 = 0;
|
||||
use futures_util::StreamExt;
|
||||
loop {
|
||||
let next = tokio::time::timeout(DOWNLOAD_STALL_TIMEOUT, stream.next())
|
||||
.await
|
||||
.context("Firmware download stalled")?;
|
||||
let Some(chunk) = next else { break };
|
||||
let chunk = chunk.context("Reading firmware download stream")?;
|
||||
file.write_all(&chunk)
|
||||
.await
|
||||
.context("Writing firmware download")?;
|
||||
downloaded += chunk.len() as u64;
|
||||
if let Some(total) = total {
|
||||
if total > 0 {
|
||||
job.set_percent(((downloaded.saturating_mul(100)) / total) as u8)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
file.flush().await.ok();
|
||||
tokio::fs::rename(&tmp, dest)
|
||||
.await
|
||||
.context("Finalizing firmware download")?;
|
||||
job.push_log(format!(
|
||||
"Downloaded {} ({downloaded} bytes)",
|
||||
dest.display()
|
||||
))
|
||||
.await;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Both Heltec V3 and V4 are ESP32-S3 boards.
|
||||
const ESPTOOL_CHIP: &str = "esp32s3";
|
||||
|
||||
/// esptool's auto-reset-into-bootloader handshake (toggling DTR/RTS in a
|
||||
/// specific timed pattern) is well-known to be flaky on some CP2102/CH340
|
||||
/// board+adapter combinations — esptool's own docs recommend retrying at a
|
||||
/// lower baud rate when this happens. Rather than fail the whole job on the
|
||||
/// first hiccup, retry once at a conservative baud before giving up.
|
||||
const ESPTOOL_FALLBACK_BAUD: &str = "115200";
|
||||
|
||||
/// `write_flash --erase-all` erases the whole chip before writing, in one
|
||||
/// esptool invocation. This needs the esp32s3 stub flasher loaded (see
|
||||
/// esptool_global_args' doc comment) — without it, --erase-all hits the
|
||||
/// exact same ROM limitation a standalone `erase_flash` does ("ESP32-S3 ROM
|
||||
/// does not support function erase_flash", confirmed live 2026-07-23), since
|
||||
/// esptool's --erase-all is implemented as the same full-chip-erase command,
|
||||
/// not a per-sector loop.
|
||||
async fn esptool_erase_and_write(path: &str, image: &Path, job: &Arc<FlashJob>) -> Result<()> {
|
||||
job.set_stage(FlashStage::Writing).await;
|
||||
let image_str = image.to_string_lossy().to_string();
|
||||
esptool_with_retry(
|
||||
path,
|
||||
&["write_flash", "--erase-all", "0x0", &image_str],
|
||||
job,
|
||||
)
|
||||
.await
|
||||
.context("esptool write_flash failed")?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// esptool's global flags (--chip/--port/--baud) MUST precede the subcommand
|
||||
/// token (erase_flash/write_flash/...) — confirmed live 2026-07-23:
|
||||
/// appending `--baud 115200` after the subcommand on the retry path
|
||||
/// produced "esptool: error: unrecognized arguments: --baud 115200" every
|
||||
/// time, so the fallback-baud retry never actually got a chance to run.
|
||||
/// Building global args separately from subcommand args keeps this correct
|
||||
/// by construction instead of relying on call-site ordering.
|
||||
///
|
||||
/// Normal stub-loader mode (no --no-stub) needs the esp32s3 stub flasher
|
||||
/// blob at /usr/lib/python3/dist-packages/esptool/targets/stub_flasher/
|
||||
/// stub_flasher_32s3.json — Debian's `esptool` package (4.7.0+dfsg-0.1)
|
||||
/// ships without it (stripped for DFSG compliance: the prebuilt blob has no
|
||||
/// buildable-from-source path Debian could verify), so scripts/self-update.sh
|
||||
/// fetches the exact same file from the matching upstream esptool release
|
||||
/// tag and installs it alongside the apt package (see the esptool install
|
||||
/// step there). --no-stub (talk directly to the ROM bootloader, skip the
|
||||
/// stub) was tried first and works for connecting, but the ROM bootloader
|
||||
/// doesn't implement a full-chip-erase opcode at all — only the stub does —
|
||||
/// so --no-stub broke our "always erase before write" default outright
|
||||
/// rather than just being slower. Restoring the real stub file is the
|
||||
/// correct fix, not routing around its absence.
|
||||
fn esptool_global_args<'a>(path: &'a str, baud: Option<&'a str>) -> Vec<&'a str> {
|
||||
let mut args = vec!["--chip", ESPTOOL_CHIP, "--port", path];
|
||||
if let Some(b) = baud {
|
||||
args.push("--baud");
|
||||
args.push(b);
|
||||
}
|
||||
args
|
||||
}
|
||||
|
||||
async fn esptool_with_retry(path: &str, subcommand: &[&str], job: &Arc<FlashJob>) -> Result<()> {
|
||||
let mut cmd = Command::new("esptool");
|
||||
cmd.args(esptool_global_args(path, None));
|
||||
cmd.args(subcommand);
|
||||
match run_streamed(cmd, None, job).await {
|
||||
Ok(()) => Ok(()),
|
||||
Err(first_err) => {
|
||||
job.push_log(format!(
|
||||
"First attempt failed ({first_err:#}); retrying once at {ESPTOOL_FALLBACK_BAUD} baud"
|
||||
))
|
||||
.await;
|
||||
let mut retry = Command::new("esptool");
|
||||
retry.args(esptool_global_args(path, Some(ESPTOOL_FALLBACK_BAUD)));
|
||||
retry.args(subcommand);
|
||||
run_streamed(retry, None, job)
|
||||
.await
|
||||
.context(format!("retry also failed (first attempt: {first_err:#})"))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Reticulum/RNode: archy-rnodeconf ───────────────────────────────────
|
||||
|
||||
fn rnodeconf_bin() -> String {
|
||||
std::env::var("ARCHY_RNODECONF_BIN")
|
||||
.unwrap_or_else(|_| "/usr/local/bin/archy-rnodeconf".to_string())
|
||||
}
|
||||
|
||||
/// `--autoinstall`'s "which board is this" step is interactive by design —
|
||||
/// confirmed live against a real Heltec V4 (2026-07-23): even with a board
|
||||
/// given on the command line, rnodeconf can't always tell V3 from V4 apart
|
||||
/// (their bootstrap-time USB identity is often generic, same root cause as
|
||||
/// `resolve_flash_board`'s doc comment), so it always asks. The full prompt
|
||||
/// sequence observed for a Heltec board that already has *some* RNode
|
||||
/// firmware installed (the common case — a truly blank chip likely skips
|
||||
/// straight to the same "Device Selection" menu):
|
||||
/// 1. numbered device-type menu → answer with the menu number
|
||||
/// 2. "Hit enter to continue" → answer with a blank line
|
||||
/// 3. numbered band menu → answer with the menu number
|
||||
/// 4. "Is the above correct? [y/N]" → answer "y"
|
||||
/// Feeding all four answers up front (rather than watching stdout for each
|
||||
/// prompt text) works because the menu is always asked in this fixed order
|
||||
/// for every board that needs (re)provisioning — verified by driving it
|
||||
/// through an unprovisioned real V4 end-to-end (erase → flash → EEPROM
|
||||
/// bootstrap → "Device signature validated" on the next probe).
|
||||
fn rnodeconf_device_menu_number(board: FlashBoard) -> &'static str {
|
||||
match board {
|
||||
FlashBoard::HeltecV3 => "8",
|
||||
FlashBoard::HeltecV4 => "9",
|
||||
}
|
||||
}
|
||||
|
||||
/// rnodeconf's band choice is a coarse RF-frontend bootstrap parameter
|
||||
/// (868/915/923 MHz), not the final operating frequency — that's still
|
||||
/// configured later via the daemon's interface config, same as today. This
|
||||
/// is a best-effort mapping from the node's persisted Meshtastic-style
|
||||
/// region code (see `mesh::meshtastic::region_name_to_code`) down to
|
||||
/// rnodeconf's 3-way menu; regions with no exact 868/923 match fall back to
|
||||
/// 915 MHz as the broadest-compatibility default.
|
||||
fn rnodeconf_band_menu_number(lora_region: Option<&str>) -> &'static str {
|
||||
match lora_region.map(|s| s.trim().to_uppercase()) {
|
||||
Some(r) if r.contains("868") => "1",
|
||||
Some(r) if r.contains("923") => "3",
|
||||
_ => "2",
|
||||
}
|
||||
}
|
||||
|
||||
/// `--autoinstall` fetches, erases, flashes, and bootstraps the EEPROM for
|
||||
/// a detected board as one atomic step (confirmed via `archy-rnodeconf
|
||||
/// --help` AND a real end-to-end flash on real hardware) — this is the
|
||||
/// RNode-side equivalent of our "always erase before write" default, since
|
||||
/// autoinstall doesn't try to preserve any existing on-device state.
|
||||
async fn rnodeconf_autoinstall(
|
||||
path: &str,
|
||||
board: FlashBoard,
|
||||
lora_region: Option<&str>,
|
||||
job: &Arc<FlashJob>,
|
||||
) -> Result<()> {
|
||||
job.set_stage(FlashStage::Autoinstalling).await;
|
||||
let bin = rnodeconf_bin();
|
||||
let mut cmd = if Path::new(&bin).exists() {
|
||||
Command::new(bin)
|
||||
} else {
|
||||
// Dev fallback if only a plain venv/system rnodeconf is on PATH.
|
||||
Command::new("rnodeconf")
|
||||
};
|
||||
cmd.args(["--autoinstall", path]);
|
||||
let stdin = format!(
|
||||
"{}\n\n{}\ny\n",
|
||||
rnodeconf_device_menu_number(board),
|
||||
rnodeconf_band_menu_number(lora_region)
|
||||
);
|
||||
run_streamed(cmd, Some(stdin.into_bytes()), job)
|
||||
.await
|
||||
.context("archy-rnodeconf --autoinstall failed")
|
||||
}
|
||||
|
||||
// ─── Subprocess streaming ────────────────────────────────────────────────
|
||||
|
||||
fn percent_regex() -> &'static Regex {
|
||||
static RE: OnceLock<Regex> = OnceLock::new();
|
||||
RE.get_or_init(|| Regex::new(r"\((\d{1,3})\s*%\)").expect("valid regex"))
|
||||
}
|
||||
|
||||
async fn run_streamed(mut cmd: Command, stdin: Option<Vec<u8>>, job: &Arc<FlashJob>) -> Result<()> {
|
||||
cmd.stdout(Stdio::piped());
|
||||
cmd.stderr(Stdio::piped());
|
||||
if stdin.is_some() {
|
||||
cmd.stdin(Stdio::piped());
|
||||
}
|
||||
// Deliberately NOT kill_on_drop: an interrupted erase/write can leave
|
||||
// the chip in a worse state than either finished or unstarted (see the
|
||||
// cancellation-safety note in mesh flashing docs). The job is expected
|
||||
// to run to completion or fail on its own.
|
||||
let mut child = cmd.spawn().context("Failed to start subprocess")?;
|
||||
|
||||
if let Some(bytes) = stdin {
|
||||
if let Some(mut child_stdin) = child.stdin.take() {
|
||||
child_stdin
|
||||
.write_all(&bytes)
|
||||
.await
|
||||
.context("Writing to subprocess stdin")?;
|
||||
}
|
||||
}
|
||||
|
||||
let mut tasks = Vec::new();
|
||||
if let Some(stdout) = child.stdout.take() {
|
||||
let job = Arc::clone(job);
|
||||
tasks.push(tokio::spawn(async move {
|
||||
let mut lines = BufReader::new(stdout).lines();
|
||||
while let Ok(Some(line)) = lines.next_line().await {
|
||||
if let Some(cap) = percent_regex().captures(&line) {
|
||||
if let Ok(pct) = cap[1].parse::<u8>() {
|
||||
job.set_percent(pct).await;
|
||||
}
|
||||
}
|
||||
job.push_log(line).await;
|
||||
}
|
||||
}));
|
||||
}
|
||||
if let Some(stderr) = child.stderr.take() {
|
||||
let job = Arc::clone(job);
|
||||
tasks.push(tokio::spawn(async move {
|
||||
let mut lines = BufReader::new(stderr).lines();
|
||||
while let Ok(Some(line)) = lines.next_line().await {
|
||||
job.push_log(line).await;
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
let status = child.wait().await.context("Waiting for subprocess")?;
|
||||
for t in tasks {
|
||||
let _ = t.await;
|
||||
}
|
||||
if !status.success() {
|
||||
// Exit status alone isn't diagnosable — the actual esptool/rnodeconf
|
||||
// stderr (already captured into job.log_tail by the reader tasks
|
||||
// above) is what actually explains a failure. Confirmed live
|
||||
// 2026-07-23: a bare "Command exited with exit status: 1" told us
|
||||
// nothing when esptool's real error was sitting in the log tail the
|
||||
// whole time, only visible via the UI's live poll, not journald.
|
||||
let tail: Vec<String> = job
|
||||
.snapshot()
|
||||
.await
|
||||
.log_tail
|
||||
.iter()
|
||||
.rev()
|
||||
.take(10)
|
||||
.rev()
|
||||
.cloned()
|
||||
.collect();
|
||||
anyhow::bail!("Command exited with {status}\n{}", tail.join("\n"));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@ -87,6 +87,18 @@ const RECONNECT_DELAY_INIT: Duration = Duration::from_secs(5);
|
||||
/// Maximum reconnect delay (cap for exponential backoff).
|
||||
const RECONNECT_DELAY_MAX: Duration = Duration::from_secs(60);
|
||||
|
||||
/// Minimum time a session must run before we trust it enough to reset
|
||||
/// backoff to the minimum. Without this gate, a device that connects then
|
||||
/// fails again within a couple of seconds (e.g. mid-boot-loop) never backs
|
||||
/// off — every retry immediately re-opens the port, which toggles DTR/RTS
|
||||
/// (resets many ESP32 boards' MCU on native-USB and CP2102/CH340
|
||||
/// auto-reset-circuit boards alike), turning a device that's merely
|
||||
/// unstable into a self-sustaining boot loop that outlasts whatever
|
||||
/// triggered the original instability. Confirmed live 2026-07-23: a Heltec
|
||||
/// V3 stuck retrying every ~5-15s for 5+ minutes after a failed firmware
|
||||
/// flash left it in a marginal state.
|
||||
const STABLE_SESSION_THRESHOLD: Duration = Duration::from_secs(20);
|
||||
|
||||
/// Number of consecutive write failures before we consider the device dead
|
||||
/// and trigger a reconnection cycle.
|
||||
const MAX_CONSECUTIVE_WRITE_FAILURES: u32 = 3;
|
||||
@ -550,6 +562,7 @@ pub fn spawn_mesh_listener(
|
||||
return;
|
||||
}
|
||||
|
||||
let session_start = std::time::Instant::now();
|
||||
match session::run_mesh_session(
|
||||
&state,
|
||||
&data_dir,
|
||||
@ -572,13 +585,14 @@ pub fn spawn_mesh_listener(
|
||||
{
|
||||
Ok(()) => {
|
||||
info!("Mesh session ended cleanly");
|
||||
// Session was established before ending — reset backoff
|
||||
// Only trust a session that actually ran for a while —
|
||||
// see STABLE_SESSION_THRESHOLD's doc comment.
|
||||
if session_start.elapsed() >= STABLE_SESSION_THRESHOLD {
|
||||
reconnect_delay = RECONNECT_DELAY_INIT;
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
// Check if session was ever connected (vs failed to open)
|
||||
let was_connected = state.status.read().await.device_connected;
|
||||
if was_connected {
|
||||
if session_start.elapsed() >= STABLE_SESSION_THRESHOLD {
|
||||
reconnect_delay = RECONNECT_DELAY_INIT;
|
||||
}
|
||||
error!("Mesh session error: {} (retry in {:?})", e, reconnect_delay);
|
||||
|
||||
@ -214,11 +214,20 @@ impl MeshtasticDevice {
|
||||
path
|
||||
))?;
|
||||
// See probe_rnode() in reticulum.rs for why: ESP32-S3 native-USB
|
||||
// boards reset on a DTR/RTS transition, so deassert both and settle
|
||||
// before the handshake below.
|
||||
// boards (and CP2102/CH340-bridged boards wired for Arduino-style
|
||||
// auto-reset) reset on a DTR/RTS transition, so deassert both and
|
||||
// settle before the handshake below. 300ms is nowhere near a real
|
||||
// firmware boot time (LoRa radio init alone can take longer) —
|
||||
// confirmed live 2026-07-23: with every one of Reticulum/Meshcore/
|
||||
// Meshtastic's open() doing this same reset, a single auto-detect
|
||||
// cycle trying multiple protocols in sequence kept re-resetting the
|
||||
// board before it ever finished booting from the PREVIOUS attempt's
|
||||
// reset, on both a Heltec V3 and V4, regardless of firmware family —
|
||||
// a self-sustaining "never finishes booting" loop with a boot-time
|
||||
// root cause hiding behind what looked like a per-protocol failure.
|
||||
let _ = port.set_dtr(false);
|
||||
let _ = port.set_rts(false);
|
||||
tokio::time::sleep(Duration::from_millis(300)).await;
|
||||
tokio::time::sleep(Duration::from_millis(2000)).await;
|
||||
info!(path = %path, baud = BAUD_RATE, "Opened Meshtastic serial port");
|
||||
|
||||
Ok(Self {
|
||||
|
||||
@ -8,6 +8,7 @@
|
||||
pub mod alerts;
|
||||
pub mod bitcoin_relay;
|
||||
pub mod crypto;
|
||||
pub mod flash;
|
||||
pub mod listener;
|
||||
pub mod meshtastic;
|
||||
pub mod message_types;
|
||||
@ -38,6 +39,14 @@ use tokio::sync::watch;
|
||||
use tracing::{error, info, warn};
|
||||
|
||||
const MESH_CONFIG_FILE: &str = "mesh-config.json";
|
||||
|
||||
/// How long `MeshService::stop()` waits for the listener task to notice its
|
||||
/// shutdown signal and exit gracefully before force-aborting it. See
|
||||
/// `stop()`'s doc comment for the real incident this guards against: without
|
||||
/// a hard abort fallback, a slow-to-notice listener could be left running
|
||||
/// forever, orphaned, racing a later independently-started listener on the
|
||||
/// same serial port.
|
||||
const LISTENER_SHUTDOWN_TIMEOUT: Duration = Duration::from_secs(15);
|
||||
const MESH_IGNORED_RADIO_FILE: &str = "mesh-ignored-radio-contacts.json";
|
||||
const MESH_CONTACTS_FILE: &str = "mesh-contacts.json";
|
||||
|
||||
@ -734,10 +743,18 @@ impl MeshService {
|
||||
self.server_name = name;
|
||||
}
|
||||
|
||||
/// Start the background mesh listener.
|
||||
/// Start the background mesh listener. Idempotent: if the listener is
|
||||
/// already running, this is a harmless no-op rather than an error —
|
||||
/// confirmed live 2026-07-23, a real race between the flash job's own
|
||||
/// post-flash restart and a concurrent user "Keep As Is" click (both
|
||||
/// legitimately trying to ensure the listener is running) surfaced this
|
||||
/// as a user-facing "Mesh listener already running" RPC error. Ensuring
|
||||
/// the listener is running is the intent every caller actually has;
|
||||
/// whichever caller's start() happens to win the race, the other
|
||||
/// finding it already satisfied is success, not failure.
|
||||
pub fn start(&mut self) -> Result<()> {
|
||||
if self.listener_handle.is_some() {
|
||||
anyhow::bail!("Mesh listener already running");
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let (shutdown_tx, shutdown_rx) = watch::channel(false);
|
||||
@ -967,9 +984,37 @@ impl MeshService {
|
||||
if let Some(tx) = self.shutdown_tx.take() {
|
||||
let _ = tx.send(true);
|
||||
}
|
||||
if let Some(handle) = self.listener_handle.take() {
|
||||
if let Some(mut handle) = self.listener_handle.take() {
|
||||
// Bounded wait for graceful shutdown, with a hard abort as
|
||||
// fallback — confirmed live 2026-07-23: a caller-side timeout
|
||||
// wrapping stop() (mesh::flash's STOP_LISTENER_TIMEOUT) cancelled
|
||||
// this await when the listener was slow to notice its shutdown
|
||||
// signal (mid multi-candidate probe), but `.take()` above had
|
||||
// already cleared `listener_handle` to None — so MeshService
|
||||
// believed it was stopped while the task kept running, orphaned
|
||||
// (dropping a JoinHandle does not abort the task it points to).
|
||||
// A later start() then spawned a second, fully independent
|
||||
// listener session racing the orphaned one on the same serial
|
||||
// port — neither could ever get a clean response, so every
|
||||
// mesh.configure/probe against that device failed indefinitely
|
||||
// even though the device itself was fine.
|
||||
//
|
||||
// Awaiting `&mut handle` (not `handle` by value) is what makes
|
||||
// the fallback possible: the Future is polled through the
|
||||
// reference, so if the timeout fires, this task's own `handle`
|
||||
// binding is still ours to call `.abort()` on afterward —
|
||||
// unlike moving `handle` into the timeout future outright, which
|
||||
// would drop (and thus orphan) it on timeout with nothing left
|
||||
// to abort.
|
||||
if tokio::time::timeout(LISTENER_SHUTDOWN_TIMEOUT, &mut handle)
|
||||
.await
|
||||
.is_err()
|
||||
{
|
||||
warn!("Mesh listener did not shut down gracefully in time — aborting it");
|
||||
handle.abort();
|
||||
let _ = handle.await;
|
||||
}
|
||||
}
|
||||
if let Some(handle) = self.deadman_handle.take() {
|
||||
handle.abort();
|
||||
let _ = handle.await;
|
||||
|
||||
@ -58,11 +58,20 @@ impl MeshcoreDevice {
|
||||
path
|
||||
))?;
|
||||
// See probe_rnode() in reticulum.rs for why: ESP32-S3 native-USB
|
||||
// boards reset on a DTR/RTS transition, so deassert both and settle
|
||||
// before the handshake below.
|
||||
// boards (and CP2102/CH340-bridged boards wired for Arduino-style
|
||||
// auto-reset) reset on a DTR/RTS transition, so deassert both and
|
||||
// settle before the handshake below. 300ms is nowhere near a real
|
||||
// firmware boot time (LoRa radio init alone can take longer) —
|
||||
// confirmed live 2026-07-23: with every one of Reticulum/Meshcore/
|
||||
// Meshtastic's open() doing this same reset, a single auto-detect
|
||||
// cycle trying multiple protocols in sequence kept re-resetting the
|
||||
// board before it ever finished booting from the PREVIOUS attempt's
|
||||
// reset, on both a Heltec V3 and V4, regardless of firmware family —
|
||||
// a self-sustaining "never finishes booting" loop with a boot-time
|
||||
// root cause hiding behind what looked like a per-protocol failure.
|
||||
let _ = port.set_dtr(false);
|
||||
let _ = port.set_rts(false);
|
||||
tokio::time::sleep(Duration::from_millis(300)).await;
|
||||
tokio::time::sleep(Duration::from_millis(2000)).await;
|
||||
|
||||
info!(path = %path, baud = BAUD_RATE, "Opened serial port");
|
||||
|
||||
|
||||
@ -110,6 +110,61 @@ lands].
|
||||
2. ❑ Mobile Home: wallet card directly under My Apps (G5 from the voice epic).
|
||||
3. ❑ Mesh RF settings panel (Mesh → Device) still loads and saves.
|
||||
|
||||
## H. LoRa radio firmware flashing (Heltec V3/V4, new — extends Section E)
|
||||
|
||||
Full v1 scope is 3 firmware families × 2 boards (6 cells); mark each cell
|
||||
tested on real hardware vs. code-reviewed only as this is run.
|
||||
|
||||
1. ❑ From the hot-swap modal's step 1 (device already probed), press
|
||||
**Flash Firmware…** → new step shows firmware-family + board pickers and
|
||||
the erase-confirmation checkbox; "Erase & Flash Now" stays disabled until
|
||||
family, board, AND the checkbox are all set.
|
||||
2. ❑ Confirm what's currently on the test stick via the existing probe
|
||||
BEFORE flashing it — don't flash the only known-good device without a
|
||||
fallback board on hand.
|
||||
3. ❑ Prefer a spare Heltec V3/V4 for the first destructive erase+flash run;
|
||||
only exercise a primary/in-use stick once the flow is proven safe.
|
||||
4. ❑ MeshCore → Heltec V3: erase + write completes, progress bar and log
|
||||
tail update live, ends at "Flash complete".
|
||||
5. ❑ Meshtastic → Heltec V3: same, using the extracted `*.factory.bin` from
|
||||
the esp32s3 release zip.
|
||||
6. ❑ Reticulum/RNode → Heltec V3: `archy-rnodeconf --autoinstall` path
|
||||
completes (no raw esptool erase/write step for this family — see
|
||||
`mesh/flash.rs` doc comment).
|
||||
7. ❑ Repeat 4-6 against a Heltec V4. Confirmed 2026-07-23 on real hardware:
|
||||
V4 uses the ESP32-S3's native-USB JTAG/serial peripheral (vid:pid
|
||||
303a:1001, generic to every native-USB ESP32-S3 board, not V4-specific)
|
||||
— so unlike V3's CP2102 bridge chip, V4 is permanently NOT auto-matchable
|
||||
by vid:pid. Board auto-detect should fail closed for it every time
|
||||
(manual board selection required, "couldn't confirm automatically"
|
||||
warning shown) — this is expected steady-state behavior, not a gap to
|
||||
close later.
|
||||
8. ❑ After a successful flash, the modal automatically re-probes and shows
|
||||
the NEW firmware's badge/details — same as unplugging and replugging
|
||||
(Section E item 3), but without physically touching the cable.
|
||||
9. ❑ Deliberately test a failure path once (disconnect the board mid-write,
|
||||
or point at a bad cached asset) — confirm the error surfaces in the
|
||||
progress log AND that `docs/troubleshooting.md`'s "LoRa radio firmware
|
||||
flash failed" recovery steps (BOOT+RST bootloader entry, manual esptool/
|
||||
rnodeconf command) actually get the board back to a flashable state.
|
||||
10. ❑ Cancel button only appears (and only works) while still in the
|
||||
"Downloading firmware…" stage — once erasing/writing starts, no cancel
|
||||
affordance is offered.
|
||||
11. ❑ **Boot-loop regression (2026-07-23 incident)**: after a *failed* flash
|
||||
(e.g. kill network access mid-download to force a failure), confirm the
|
||||
mesh listener does NOT auto-resume — `journalctl -u archipelago` should
|
||||
show a single `Leaving mesh listener stopped after failed flash` line
|
||||
and then go quiet for that device, not a repeating `mesh::serial:
|
||||
Opened serial port... Starting Meshcore handshake` cycle every few
|
||||
seconds. Reconnect manually via the hot-swap modal afterward and confirm
|
||||
it connects normally (the board itself should be untouched — the
|
||||
download fails before esptool/rnodeconf ever runs).
|
||||
12. ❑ Separately, force a device to flap connected/disconnected a few times
|
||||
in under 20s each (e.g. a marginal USB connection) and confirm
|
||||
`reconnect_delay` in the logs actually escalates (5s → 10s → 20s → ...)
|
||||
rather than resetting to 5s on every attempt — see
|
||||
`STABLE_SESSION_THRESHOLD` in `mesh/listener/mod.rs`.
|
||||
|
||||
---
|
||||
|
||||
After this passes: fold the batch + other agent's work into the next release
|
||||
|
||||
@ -443,6 +443,86 @@ free -h
|
||||
- Check Nginx WebSocket proxy config: `/etc/nginx/sites-available/archipelago` must include `proxy_set_header Upgrade $http_upgrade`
|
||||
- If on WiFi, try wired Ethernet for more stable connectivity
|
||||
|
||||
### 21. LoRa radio firmware flash failed / board unresponsive
|
||||
|
||||
**Symptoms**: The "Erase & Flash Now" flow in the mesh hot-swap modal reports
|
||||
an error, or the radio no longer enumerates as a serial device after a flash
|
||||
attempt.
|
||||
|
||||
**Diagnosis**:
|
||||
```bash
|
||||
# Poll the flash job's last-known stage/error directly
|
||||
curl -s http://localhost:5678/rpc/v1 \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"method":"mesh.flash-status","params":{}}'
|
||||
|
||||
# Confirm the board is still enumerating at all
|
||||
ls -la /dev/ttyUSB* /dev/ttyACM* /dev/mesh-radio 2>&1
|
||||
|
||||
# esptool/rnodeconf binaries present?
|
||||
which esptool; ls -la /usr/local/bin/archy-rnodeconf
|
||||
```
|
||||
|
||||
**Solutions**:
|
||||
- A failure during `erasing`/`writing` (MeshCore/Meshtastic) or
|
||||
`autoinstalling` (Reticulum) can leave the chip erased or half-written —
|
||||
this is expected risk of the "always erase first" default, not a bug.
|
||||
- Heltec V3/V4 boards can be forced back into bootloader mode manually: hold
|
||||
**BOOT**, tap **RST**, then release **BOOT** — this puts the chip in a
|
||||
state esptool can always talk to, regardless of what firmware (if any) is
|
||||
currently on it.
|
||||
- With the board in bootloader mode, a manual recovery flash can be run
|
||||
directly over SSH without the UI:
|
||||
```bash
|
||||
esptool --chip esp32s3 --port /dev/ttyACM0 erase_flash
|
||||
esptool --chip esp32s3 --port /dev/ttyACM0 write_flash 0x0 <known-good-image.bin>
|
||||
```
|
||||
- For Reticulum/RNode boards, the equivalent manual recovery is
|
||||
`archy-rnodeconf /dev/ttyACM0 --autoinstall` (or `/usr/local/bin/archy-rnodeconf`
|
||||
if it's not on `PATH`) — it re-runs the same fetch+erase+flash+bootstrap
|
||||
sequence the UI triggers.
|
||||
- If `esptool`/`archy-rnodeconf` are missing entirely, they should have been
|
||||
installed by the last `self-update.sh` run — check
|
||||
`sudo journalctl -u archipelago-update` for install failures, or install
|
||||
`esptool` via `sudo apt-get install esptool` directly.
|
||||
- Once a fresh image is confirmed written, unplug/replug the radio (or wait
|
||||
for the next detection poll) — the hot-swap modal re-probes automatically
|
||||
and shows whatever firmware is actually on the board now.
|
||||
|
||||
**Known incident (2026-07-23) — reconnect storm / device boot-loop after a
|
||||
failed flash**: a real Heltec V3 got stuck cycling "connect → partial
|
||||
handshake → drop" every 5-15s for 5+ minutes after a `mesh.flash-device`
|
||||
attempt failed with `Reading firmware download stream`. Root cause was two
|
||||
compounding issues, both now fixed:
|
||||
1. `spawn_mesh_listener`'s reconnect backoff (`core/archipelago/src/mesh/listener/mod.rs`)
|
||||
reset to its 5s minimum any time the prior session had been `device_connected`
|
||||
at all, even for under a second — so a device that connects-then-drops
|
||||
repeatedly never actually backed off. Every retry's `open()` toggles
|
||||
DTR/RTS, which resets many ESP32 boards' MCU (native-USB *and*
|
||||
CP2102/CH340 auto-reset-circuit boards), so the aggressive retries were
|
||||
themselves *causing* the boot loop, not just observing one. Fixed by only
|
||||
resetting backoff when a session ran for at least `STABLE_SESSION_THRESHOLD`
|
||||
(20s) — see that constant's doc comment.
|
||||
2. `mesh::flash::start_flash_job`'s post-completion handler auto-resumed the
|
||||
listener unconditionally, even after a *failed* flash, immediately
|
||||
re-entering the reconnect loop above with no cooldown. Fixed: on failure
|
||||
the listener is now deliberately left stopped (reconnect manually via the
|
||||
UI once the board is confirmed alive); on success there's a 5s settle
|
||||
delay before resuming, so the board finishes booting from the flash
|
||||
tool's own reset before Archipelago starts probing it again.
|
||||
3. Separately, the download itself was failing because `mesh::flash`'s HTTP
|
||||
client had a blanket 30s request timeout that covered the *entire*
|
||||
download (including streaming a 170MB Meshtastic zip), not just
|
||||
connection setup — fixed with a per-chunk stall timeout instead of a
|
||||
fixed total-transfer cap.
|
||||
|
||||
If this symptom recurs (rapid repeating `mesh::serial: Opened serial
|
||||
port... Starting Meshcore handshake` lines in `journalctl -u archipelago`
|
||||
without a `LoRa firmware flash` job in progress), it's a NEW instance of the
|
||||
same class of bug, not the one above — check whether backoff is actually
|
||||
escalating (`Mesh session error: ... (retry in Xs)` — X should grow past 5s
|
||||
within a few cycles) before assuming it's flashing-related.
|
||||
|
||||
---
|
||||
|
||||
## General Maintenance
|
||||
|
||||
@ -365,6 +365,11 @@ RUN apt-get update && apt-get -y full-upgrade && apt-get install -y --no-install
|
||||
ca-certificates \
|
||||
openssl \
|
||||
chrony \
|
||||
iputils-ping \
|
||||
esptool \
|
||||
python3-venv \
|
||||
binutils \
|
||||
libpython3.13 \
|
||||
locales \
|
||||
console-setup \
|
||||
keyboard-configuration \
|
||||
|
||||
@ -1,7 +1,7 @@
|
||||
<template>
|
||||
<BaseModal
|
||||
:show="show"
|
||||
:title="step === 1 ? 'Mesh Radio Detected' : 'Apply Archipelago Settings'"
|
||||
:title="step === 1 ? 'Mesh Radio Detected' : step === 2 ? 'Apply Archipelago Settings' : 'Flash Firmware'"
|
||||
max-width="max-w-lg"
|
||||
content-class="max-h-[90vh] overflow-y-auto"
|
||||
@close="dismiss"
|
||||
@ -96,9 +96,111 @@
|
||||
"Keep As Is" uses the radio exactly as it is — nothing on it is changed,
|
||||
and you can hot-swap radios any time.
|
||||
</p>
|
||||
<button
|
||||
class="w-full text-center text-white/40 hover:text-white/70 text-[11px] mt-3 underline underline-offset-2"
|
||||
:disabled="!!connecting"
|
||||
@click="openFlashStep"
|
||||
>
|
||||
Flash Firmware…
|
||||
</button>
|
||||
<p v-if="error" class="text-xs text-red-400 mt-2">{{ error }}</p>
|
||||
</div>
|
||||
|
||||
<!-- Step 3: erase + reflash — destructive, opt-in only -->
|
||||
<div v-else-if="step === 'flash'">
|
||||
<!-- Once a job exists (started via startFlash), ALWAYS show the
|
||||
progress/result view below — including on failure. The old
|
||||
condition (`!active && stage !== 'done'`) was also true for a
|
||||
FAILED job (active:false, stage:'failed'), which silently sent
|
||||
the user back to this picker instead of showing the error. -->
|
||||
<template v-if="!flashJob">
|
||||
<p class="text-white/60 text-xs mb-3">
|
||||
Downloads the latest firmware from upstream and writes it to
|
||||
<span class="font-mono text-orange-300">{{ devicePath }}</span>.
|
||||
</p>
|
||||
|
||||
<div class="space-y-4">
|
||||
<div>
|
||||
<label class="block text-sm text-white/80 mb-1">Firmware family</label>
|
||||
<select v-model="flashFamily" class="w-full rounded-lg bg-white/[0.06] border border-white/10 text-white px-3 py-2 text-sm focus:outline-none focus:border-orange-400/60">
|
||||
<option value="">Choose…</option>
|
||||
<option value="meshcore">MeshCore</option>
|
||||
<option value="meshtastic">Meshtastic</option>
|
||||
<option value="reticulum">Reticulum RNode</option>
|
||||
</select>
|
||||
</div>
|
||||
<div>
|
||||
<label class="block text-sm text-white/80 mb-1">Board</label>
|
||||
<select v-model="flashBoard" class="w-full rounded-lg bg-white/[0.06] border border-white/10 text-white px-3 py-2 text-sm focus:outline-none focus:border-orange-400/60">
|
||||
<option value="">Choose…</option>
|
||||
<option value="heltec-v3">Heltec LoRa 32 V3</option>
|
||||
<option value="heltec-v4">Heltec LoRa 32 V4</option>
|
||||
</select>
|
||||
<p v-if="!boardAutoDetected" class="text-[11px] text-amber-400/80 mt-1">
|
||||
Couldn't confirm the board automatically — double check before flashing.
|
||||
Flashing the wrong board's image can brick it.
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="rounded-xl bg-red-500/10 border border-red-500/30 p-3 mt-4">
|
||||
<label class="flex items-start gap-2 text-xs text-red-300">
|
||||
<input type="checkbox" v-model="flashConfirmed" class="mt-0.5" />
|
||||
<span>
|
||||
This <strong>erases the entire chip</strong>, including any existing
|
||||
keys, identity, and contacts. This cannot be undone.
|
||||
</span>
|
||||
</label>
|
||||
</div>
|
||||
|
||||
<p v-if="error" class="text-xs text-red-400 mt-3">{{ error }}</p>
|
||||
|
||||
<div class="flex gap-2 mt-6">
|
||||
<button class="glass-button px-4 py-2 rounded-lg text-sm" @click="step = 1">Back</button>
|
||||
<button
|
||||
class="flex-1 glass-button px-4 py-2 rounded-lg text-sm font-medium bg-red-500/80 hover:bg-red-500 text-white disabled:opacity-50"
|
||||
:disabled="!flashFamily || !flashBoard || !flashConfirmed || starting"
|
||||
@click="startFlash"
|
||||
>
|
||||
{{ starting ? 'Starting…' : 'Erase & Flash Now' }}
|
||||
</button>
|
||||
</div>
|
||||
</template>
|
||||
|
||||
<!-- Progress -->
|
||||
<template v-else>
|
||||
<div class="text-center py-2">
|
||||
<p class="text-white text-sm font-medium">{{ flashStageLabel }}</p>
|
||||
<div class="mt-3 h-2 rounded-full bg-white/10 overflow-hidden">
|
||||
<div
|
||||
class="h-full bg-orange-400 transition-all"
|
||||
:style="{ width: (flashJob?.percent ?? (flashJob?.stage === 'done' ? 100 : 8)) + '%' }"
|
||||
></div>
|
||||
</div>
|
||||
<p v-if="flashJob?.error" class="text-xs text-red-400 mt-3">{{ flashJob.error }}</p>
|
||||
</div>
|
||||
<div class="mt-3 rounded-xl bg-black/30 border border-white/10 p-2 h-32 overflow-y-auto font-mono text-[10px] text-white/50 leading-relaxed">
|
||||
<div v-for="(line, i) in (flashJob?.log_tail ?? []).slice(-40)" :key="i">{{ line }}</div>
|
||||
</div>
|
||||
<div class="flex gap-2 mt-4">
|
||||
<button
|
||||
v-if="flashJob?.stage === 'downloading' && flashJob?.active"
|
||||
class="glass-button px-4 py-2 rounded-lg text-sm"
|
||||
@click="cancelFlash"
|
||||
>
|
||||
Cancel
|
||||
</button>
|
||||
<button
|
||||
v-if="!flashJob?.active"
|
||||
class="flex-1 glass-button glass-button-warning px-4 py-2 rounded-lg text-sm font-medium"
|
||||
@click="closeFlashStep"
|
||||
>
|
||||
Done
|
||||
</button>
|
||||
</div>
|
||||
</template>
|
||||
</div>
|
||||
|
||||
<!-- Step 2: our latest parameters, shown before anything is written -->
|
||||
<div v-else>
|
||||
<p class="text-white/60 text-xs mb-3">
|
||||
@ -185,7 +287,7 @@
|
||||
import { ref, computed, watch } from 'vue'
|
||||
import { useRouter } from 'vue-router'
|
||||
import BaseModal from '@/components/BaseModal.vue'
|
||||
import { useMeshStore, type MeshDeviceProbe, type MeshConfigureParams } from '@/stores/mesh'
|
||||
import { useMeshStore, type MeshDeviceProbe, type MeshConfigureParams, type FlashFirmwareFamily, type FlashBoard, type FlashJobStatus } from '@/stores/mesh'
|
||||
import { useAppStore } from '@/stores/app'
|
||||
import { LORA_REGIONS, regionByCode, suggestRegionFromLatLon, meshcorePlanFor } from '@/utils/loraRegions'
|
||||
import { resolveMeshDeviceImage } from '@/utils/meshDeviceImages'
|
||||
@ -194,7 +296,7 @@ const mesh = useMeshStore()
|
||||
const appStore = useAppStore()
|
||||
const router = useRouter()
|
||||
|
||||
const step = ref<1 | 2>(1)
|
||||
const step = ref<1 | 2 | 'flash'>(1)
|
||||
const connecting = ref<false | 'keep' | 'setup'>(false)
|
||||
const error = ref('')
|
||||
const probing = ref(false)
|
||||
@ -259,7 +361,10 @@ const rfPreset = computed(() => {
|
||||
|
||||
// (Re)probe + (re)apply presets each time a new device surfaces the modal
|
||||
watch([show, devicePath], async ([visible]) => {
|
||||
if (!visible) return
|
||||
if (!visible) {
|
||||
stopFlashPoll()
|
||||
return
|
||||
}
|
||||
step.value = 1
|
||||
error.value = ''
|
||||
imageFailed.value = false
|
||||
@ -344,6 +449,118 @@ async function applySetup() {
|
||||
connecting.value = false
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Step 3: erase + reflash ─────────────────────────────────────────────
|
||||
const flashFamily = ref<FlashFirmwareFamily | ''>('')
|
||||
const flashBoard = ref<FlashBoard | ''>('')
|
||||
const flashConfirmed = ref(false)
|
||||
const starting = ref(false)
|
||||
const flashJob = ref<FlashJobStatus | null>(null)
|
||||
let flashPollTimer: ReturnType<typeof setInterval> | null = null
|
||||
|
||||
const detectedInfo = computed(() =>
|
||||
mesh.status?.detected_device_info?.find(d => d.path === devicePath.value)
|
||||
)
|
||||
|
||||
// Mirrors mesh::flash::resolve_flash_board (core/archipelago/src/mesh/flash.rs)
|
||||
// exactly — matching on the display label was wrong: a Heltec V3's CP2102
|
||||
// bridge chip reports "CP2102 USB to UART Bridge Controller" in its USB
|
||||
// strings, not "Heltec", so meshDeviceImages.ts falls back to a generic
|
||||
// "LoRa radio (CP2102 serial)" label that never matched /v3/i, showing the
|
||||
// "couldn't confirm automatically" warning even though the backend CAN
|
||||
// safely auto-detect V3 via vid:pid. Heltec V4 deliberately has no entry
|
||||
// here, same reasoning as the backend: its vid:pid (303a:1001) is the
|
||||
// ESP32-S3's generic native-USB descriptor, not V4-specific, so it can't be
|
||||
// safely auto-matched and always requires manual selection.
|
||||
const resolvedFlashBoard = computed<FlashBoard | ''>(() => {
|
||||
const info = detectedInfo.value
|
||||
if (info?.vid?.toLowerCase() === '10c4' && info?.pid?.toLowerCase() === 'ea60') return 'heltec-v3'
|
||||
return ''
|
||||
})
|
||||
|
||||
const boardAutoDetected = computed(() => !!resolvedFlashBoard.value)
|
||||
|
||||
const flashStageLabel = computed(() => {
|
||||
switch (flashJob.value?.stage) {
|
||||
case 'downloading': return 'Downloading firmware…'
|
||||
case 'erasing': return 'Erasing chip…'
|
||||
case 'writing': return 'Writing firmware…'
|
||||
case 'autoinstalling': return 'Installing (rnodeconf)…'
|
||||
case 'done': return 'Flash complete'
|
||||
case 'failed': return 'Flash failed'
|
||||
default: return ''
|
||||
}
|
||||
})
|
||||
|
||||
function openFlashStep() {
|
||||
flashFamily.value = (probe.value?.kind as FlashFirmwareFamily) ?? ''
|
||||
flashBoard.value = resolvedFlashBoard.value
|
||||
flashConfirmed.value = false
|
||||
flashJob.value = null
|
||||
error.value = ''
|
||||
step.value = 'flash'
|
||||
}
|
||||
|
||||
function stopFlashPoll() {
|
||||
if (flashPollTimer) {
|
||||
clearInterval(flashPollTimer)
|
||||
flashPollTimer = null
|
||||
}
|
||||
}
|
||||
|
||||
async function pollFlashStatus() {
|
||||
try {
|
||||
const status = await mesh.flashStatus()
|
||||
flashJob.value = status
|
||||
if (!status.active) {
|
||||
stopFlashPoll()
|
||||
if (status.done && !status.error) {
|
||||
// Mirrors the unplug/replug hot-swap flow: re-probe so the details
|
||||
// card reflects whatever firmware is actually on the board now.
|
||||
const path = devicePath.value
|
||||
probing.value = true
|
||||
try {
|
||||
probe.value = await mesh.probeDevice(path)
|
||||
} catch {
|
||||
probe.value = null
|
||||
} finally {
|
||||
probing.value = false
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
stopFlashPoll()
|
||||
}
|
||||
}
|
||||
|
||||
async function startFlash() {
|
||||
if (!flashFamily.value || !flashBoard.value || !flashConfirmed.value) return
|
||||
starting.value = true
|
||||
error.value = ''
|
||||
try {
|
||||
await mesh.flashDevice(devicePath.value, flashFamily.value, flashBoard.value)
|
||||
flashJob.value = { active: true, stage: 'downloading', log_tail: [] }
|
||||
stopFlashPoll()
|
||||
flashPollTimer = setInterval(pollFlashStatus, 1500)
|
||||
} catch (e) {
|
||||
error.value = e instanceof Error ? e.message : 'Failed to start flashing'
|
||||
} finally {
|
||||
starting.value = false
|
||||
}
|
||||
}
|
||||
|
||||
async function cancelFlash() {
|
||||
try {
|
||||
await mesh.flashCancel()
|
||||
} catch (e) {
|
||||
error.value = e instanceof Error ? e.message : 'Failed to cancel'
|
||||
}
|
||||
}
|
||||
|
||||
function closeFlashStep() {
|
||||
stopFlashPoll()
|
||||
step.value = 1
|
||||
}
|
||||
</script>
|
||||
|
||||
<style scoped>
|
||||
|
||||
@ -55,6 +55,23 @@ export interface MeshDeviceProbe {
|
||||
max_contacts: number | null
|
||||
}
|
||||
|
||||
export type FlashFirmwareFamily = 'meshcore' | 'meshtastic' | 'reticulum'
|
||||
export type FlashBoard = 'heltec-v3' | 'heltec-v4'
|
||||
export type FlashStage = 'downloading' | 'erasing' | 'writing' | 'autoinstalling' | 'done' | 'failed'
|
||||
|
||||
/** Live progress for the one flash job that can run at a time. */
|
||||
export interface FlashJobStatus {
|
||||
active: boolean
|
||||
board?: FlashBoard
|
||||
family?: FlashFirmwareFamily
|
||||
path?: string
|
||||
stage?: FlashStage
|
||||
percent?: number | null
|
||||
log_tail?: string[]
|
||||
done?: boolean
|
||||
error?: string | null
|
||||
}
|
||||
|
||||
/** Params accepted by mesh.configure (superset of the status fields). */
|
||||
export interface MeshConfigureParams {
|
||||
enabled?: boolean
|
||||
@ -343,6 +360,32 @@ export const useMeshStore = defineStore('mesh', () => {
|
||||
timeout: 45000, // serial probes are slow (multi-firmware handshakes)
|
||||
})
|
||||
}
|
||||
/** Available firmware version(s) for a family — v1 only ever returns
|
||||
* ["latest"], since firmware is always fetched from upstream at flash
|
||||
* time rather than pinned/bundled. */
|
||||
async function flashListFirmware(family: FlashFirmwareFamily): Promise<string[]> {
|
||||
const res = await rpcClient.call<{ versions: string[] }>({
|
||||
method: 'mesh.flash-list-firmware',
|
||||
params: { family },
|
||||
})
|
||||
return res.versions
|
||||
}
|
||||
/** Erase + reflash a detected radio. `board` is optional — omit it to let
|
||||
* the backend auto-resolve from the port's USB vid:pid; if that fails
|
||||
* (e.g. Heltec V4 not yet in the vid:pid table), it errors and the UI
|
||||
* must ask the user to pick the board explicitly. Always erases first. */
|
||||
async function flashDevice(path: string, family: FlashFirmwareFamily, board?: FlashBoard): Promise<void> {
|
||||
await rpcClient.call({
|
||||
method: 'mesh.flash-device',
|
||||
params: board ? { path, family, board } : { path, family },
|
||||
})
|
||||
}
|
||||
async function flashStatus(): Promise<FlashJobStatus> {
|
||||
return rpcClient.call<FlashJobStatus>({ method: 'mesh.flash-status' })
|
||||
}
|
||||
async function flashCancel(): Promise<void> {
|
||||
await rpcClient.call({ method: 'mesh.flash-cancel' })
|
||||
}
|
||||
let globalDetectTimer: ReturnType<typeof setInterval> | null = null
|
||||
/** App-wide light poll so the detected-device modal works on every page
|
||||
* (the Mesh view's own 5s poll takes over while it is mounted). */
|
||||
@ -979,6 +1022,10 @@ export const useMeshStore = defineStore('mesh', () => {
|
||||
undismissedDetectedDevices,
|
||||
dismissDetectedDevice,
|
||||
probeDevice,
|
||||
flashListFirmware,
|
||||
flashDevice,
|
||||
flashStatus,
|
||||
flashCancel,
|
||||
startGlobalDetection,
|
||||
fetchPeers,
|
||||
fetchMessages,
|
||||
|
||||
@ -47,11 +47,16 @@ if [ -n "$RNODECONF_SRC" ] && [ -f "$RNODECONF_SRC" ]; then
|
||||
# exit()/quit() builtins, which only exist in interactive Python (site.py
|
||||
# injects them) — a frozen app hits NameError right as it tries to quit
|
||||
# cleanly, after all the real work already succeeded. See
|
||||
# pyi_rthook_exit_builtins.py.
|
||||
# pyi_rthook_exit_builtins.py. A second hook fixes rnodeconf's board-flash
|
||||
# step, which shells out to a bundled esptool.py via `sys.executable` —
|
||||
# under a frozen binary that's the binary itself, not a real interpreter,
|
||||
# so the flash subprocess call breaks. See
|
||||
# pyi_rthook_fix_flasher_executable.py.
|
||||
.venv/bin/pyinstaller --onefile --name archy-rnodeconf --clean --noconfirm \
|
||||
--collect-submodules RNS \
|
||||
--collect-data RNS \
|
||||
--runtime-hook pyi_rthook_exit_builtins.py \
|
||||
--runtime-hook pyi_rthook_fix_flasher_executable.py \
|
||||
-d noarchive \
|
||||
"$RNODECONF_SRC"
|
||||
echo "Built dist/archy-rnodeconf ($(du -h dist/archy-rnodeconf | cut -f1))"
|
||||
|
||||
34
reticulum-daemon/pyi_rthook_fix_flasher_executable.py
Normal file
34
reticulum-daemon/pyi_rthook_fix_flasher_executable.py
Normal file
@ -0,0 +1,34 @@
|
||||
# PyInstaller runtime hook — see build.sh.
|
||||
#
|
||||
# rnodeconf's own board-flashing code shells out to a bundled esptool.py as
|
||||
# `[sys.executable, flasher_path, "--chip", ..., "write_flash", ...]` (RNS's
|
||||
# rnodeconf.py, ~line 2794 as of RNS 1.3.5). That's correct for a normal
|
||||
# `python rnodeconf.py` invocation, but under a frozen PyInstaller binary
|
||||
# `sys.executable` is the frozen binary itself, not a real interpreter — so
|
||||
# the "subprocess" just re-invokes archy-rnodeconf's OWN argparse CLI with
|
||||
# esptool-shaped flags, which it doesn't recognize, and the flash step fails
|
||||
# immediately with "unrecognized arguments: --chip ...". Confirmed live
|
||||
# against a real Heltec V4 (2026-07-23): device selection, band selection,
|
||||
# and firmware download all worked; only the final `write_flash` subprocess
|
||||
# call broke this way.
|
||||
#
|
||||
# Fix: point sys.executable at a real Python interpreter that has rnodeconf's
|
||||
# own runtime deps available (esptool.py only needs pyserial, which RNS
|
||||
# already depends on) before any of rnodeconf's code runs. Prefer the build
|
||||
# venv this exact binary was frozen from — see build.sh — falling back to a
|
||||
# bare `python3` on PATH if that venv isn't present on this node.
|
||||
import os
|
||||
import sys
|
||||
|
||||
if getattr(sys, "frozen", False):
|
||||
_candidates = [
|
||||
os.environ.get("ARCHY_RNODECONF_PYTHON", ""),
|
||||
os.path.join(os.path.dirname(os.path.abspath(sys.argv[0])), "..", "reticulum-daemon", ".venv", "bin", "python3"),
|
||||
os.path.expanduser("~/archy/reticulum-daemon/.venv/bin/python3"),
|
||||
]
|
||||
for _candidate in _candidates:
|
||||
if _candidate and os.path.isfile(_candidate):
|
||||
sys.executable = _candidate
|
||||
break
|
||||
else:
|
||||
sys.executable = "python3"
|
||||
@ -84,6 +84,64 @@ if ! command -v nano >/dev/null 2>&1; then
|
||||
fi
|
||||
fi
|
||||
|
||||
if ! command -v ping >/dev/null 2>&1; then
|
||||
log "Installing iputils-ping..."
|
||||
if sudo apt-get update -qq 2>>"$LOG_FILE" && sudo apt-get install -y -qq iputils-ping 2>>"$LOG_FILE"; then
|
||||
ok "ping installed"
|
||||
else
|
||||
warn "Unable to install ping automatically; continuing update"
|
||||
fi
|
||||
fi
|
||||
|
||||
if ! command -v esptool >/dev/null 2>&1; then
|
||||
log "Installing esptool for LoRa radio firmware flashing..."
|
||||
if sudo apt-get update -qq 2>>"$LOG_FILE" && sudo apt-get install -y -qq esptool 2>>"$LOG_FILE"; then
|
||||
ok "esptool installed"
|
||||
else
|
||||
warn "Unable to install esptool automatically; radio firmware flashing will be unavailable"
|
||||
fi
|
||||
fi
|
||||
|
||||
# Debian's esptool package (4.7.0+dfsg-0.1) ships without the precompiled
|
||||
# esp32s3 "stub flasher" blob (stripped for DFSG compliance — no
|
||||
# buildable-from-source path Debian could verify). Without it, esptool's
|
||||
# normal stub-loader mode fails outright (FileNotFoundError), and the ROM
|
||||
# bootloader fallback (--no-stub) doesn't implement a full-chip erase at
|
||||
# all — confirmed live 2026-07-23 flashing a real Heltec V4, both ways.
|
||||
# Fetching the exact same file from the matching upstream esptool release
|
||||
# tag restores full (and correct) flashing behavior — it's the same
|
||||
# open-source codebase, just the one blob Debian's packaging couldn't
|
||||
# include.
|
||||
if command -v esptool >/dev/null 2>&1; then
|
||||
STUB_DIR="/usr/lib/python3/dist-packages/esptool/targets/stub_flasher"
|
||||
STUB_FILE="$STUB_DIR/stub_flasher_32s3.json"
|
||||
if [ ! -f "$STUB_FILE" ]; then
|
||||
log "Fetching esptool's esp32s3 stub flasher (missing from the Debian package)..."
|
||||
ESPTOOL_VERSION=$(esptool version 2>/dev/null | tail -1 | tr -d ' \t')
|
||||
if [ -n "$ESPTOOL_VERSION" ] && sudo curl -fsSL -o "$STUB_FILE" \
|
||||
"https://raw.githubusercontent.com/espressif/esptool/v${ESPTOOL_VERSION}/esptool/targets/stub_flasher/stub_flasher_32s3.json" \
|
||||
2>>"$LOG_FILE"; then
|
||||
sudo chmod 644 "$STUB_FILE"
|
||||
ok "esp32s3 stub flasher installed"
|
||||
else
|
||||
sudo rm -f "$STUB_FILE" 2>/dev/null
|
||||
warn "Unable to fetch esp32s3 stub flasher; LoRa firmware flashing will be unavailable"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# Build-time prerequisites for reticulum-daemon/build.sh's PyInstaller step
|
||||
# below (discovered the hard way: ensurepip needs python3-venv, and
|
||||
# PyInstaller itself needs objdump + libpython3.13.so at build time — none
|
||||
# of these are pulled in by a bare `python3` package on Debian trixie).
|
||||
for pkg in python3-venv binutils libpython3.13; do
|
||||
if ! dpkg -s "$pkg" >/dev/null 2>&1; then
|
||||
log "Installing $pkg (reticulum-daemon build prerequisite)..."
|
||||
sudo apt-get update -qq 2>>"$LOG_FILE" && sudo apt-get install -y -qq "$pkg" 2>>"$LOG_FILE" \
|
||||
|| warn "Unable to install $pkg automatically; reticulum-daemon tools build may fail"
|
||||
fi
|
||||
done
|
||||
|
||||
# Fetch latest
|
||||
log "Fetching from origin..."
|
||||
git fetch origin main --quiet 2>>"$LOG_FILE"
|
||||
@ -155,6 +213,30 @@ sudo cp "$BUILT_BIN" "$INSTALL_BIN"
|
||||
sudo chmod +x "$INSTALL_BIN"
|
||||
ok "Backend installed"
|
||||
|
||||
# Build + install reticulum-daemon tools (archy-reticulum-daemon, archy-rnodeconf).
|
||||
# Non-fatal: archipelago falls back to its dev venv path if the packaged
|
||||
# binaries aren't present, so a missing/failed build here degrades mesh
|
||||
# Reticulum support rather than breaking the update. This mirrors
|
||||
# deploy-to-target.sh's existing manual-deploy step, which until now was the
|
||||
# only path that ever installed these — a node that only ever received OTA
|
||||
# self-updates had neither binary.
|
||||
if [ -f "$REPO_DIR/reticulum-daemon/build.sh" ]; then
|
||||
log "Building reticulum-daemon tools (archy-reticulum-daemon, archy-rnodeconf)..."
|
||||
if (cd "$REPO_DIR/reticulum-daemon" && ./build.sh) 2>>"$LOG_FILE"; then
|
||||
for tool in archy-reticulum-daemon archy-rnodeconf; do
|
||||
if [ -f "$REPO_DIR/reticulum-daemon/dist/$tool" ]; then
|
||||
sudo cp "$REPO_DIR/reticulum-daemon/dist/$tool" /usr/local/bin/
|
||||
sudo chmod +x "/usr/local/bin/$tool"
|
||||
ok "$tool installed"
|
||||
else
|
||||
warn "$tool not built — leaving existing /usr/local/bin/$tool (if any) in place"
|
||||
fi
|
||||
done
|
||||
else
|
||||
warn "reticulum-daemon tools build failed — continuing without updating them"
|
||||
fi
|
||||
fi
|
||||
|
||||
# Build frontend
|
||||
log "Building Vue frontend (production)..."
|
||||
cd "$FRONTEND_DIR"
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user