Compare commits

..
Author SHA1 Message Date
Archipelago dc80b552a1 Archipelago — open-source initial import 2026-08-12 10:55:49 +00:00
42 changed files with 2892 additions and 259 deletions
+2 -2
View File
@@ -11,8 +11,8 @@ android {
applicationId = "com.archipelago.app"
minSdk = 26
targetSdk = 35
versionCode = 20
versionName = "0.5.0"
versionCode = 26
versionName = "0.5.6"
vectorDrawables {
useSupportLibrary = true
@@ -21,6 +21,10 @@ data class ServerEntry(
val name: String = "",
/** Node's FIPS mesh ULA (IPv6) — reachable from anywhere once meshed. */
val meshIp: String = "",
/** Node's FIPS npub — the durable identity. When present it, not the
* address, is what identifies the entry: FIPS peers on npubs, IPs are
* only dial hints (docs/companion-pairing-qr.md, npub-first contract). */
val npub: String = "",
) {
/** Label to show in lists — the user-given name, or the address if unnamed. */
fun displayName(): String = name.ifBlank { address }
@@ -45,9 +49,15 @@ data class ServerEntry(
fun toMeshUrl(): String? =
meshIp.takeIf { it.isNotBlank() }?.let { "http://${urlHost(it)}" }
// name/meshIp are trailing fields so entries saved before they existed
// (4/5 fields) still deserialize, defaulting to "".
fun serialize(): String = "$address|$useHttps|$port|$password|$name|$meshIp"
// name/meshIp/npub are trailing fields so entries saved before they
// existed (4/5/6 fields) still deserialize, defaulting to "".
fun serialize(): String = "$address|$useHttps|$port|$password|$name|$meshIp|$npub"
/** Same node as [other]? npub identity wins; address/port/scheme is the
* fallback for LAN-only entries that never advertised FIPS. */
fun sameNode(other: ServerEntry): Boolean =
(npub.isNotBlank() && npub == other.npub) ||
(address == other.address && port == other.port && useHttps == other.useHttps)
companion object {
fun deserialize(raw: String): ServerEntry? {
@@ -60,6 +70,7 @@ data class ServerEntry(
password = parts.getOrElse(3) { "" },
name = parts.getOrElse(4) { "" },
meshIp = parts.getOrElse(5) { "" },
npub = parts.getOrElse(6) { "" },
)
}
}
@@ -73,6 +84,7 @@ class ServerPreferences(private val context: Context) {
private val activePasswordKey = stringPreferencesKey("active_password")
private val activeNameKey = stringPreferencesKey("active_name")
private val activeMeshIpKey = stringPreferencesKey("active_mesh_ip")
private val activeNpubKey = stringPreferencesKey("active_npub")
private val savedServersKey = stringSetPreferencesKey("saved_servers")
private val introSeenKey = booleanPreferencesKey("intro_seen")
private val gestureHintSeenKey = booleanPreferencesKey("gesture_hint_seen")
@@ -86,6 +98,7 @@ class ServerPreferences(private val context: Context) {
password = prefs[activePasswordKey] ?: "",
name = prefs[activeNameKey] ?: "",
meshIp = prefs[activeMeshIpKey] ?: "",
npub = prefs[activeNpubKey] ?: "",
)
}
@@ -111,6 +124,7 @@ class ServerPreferences(private val context: Context) {
prefs[activePasswordKey] = server.password
prefs[activeNameKey] = server.name
prefs[activeMeshIpKey] = server.meshIp
prefs[activeNpubKey] = server.npub
}
addSavedServer(server)
}
@@ -123,6 +137,7 @@ class ServerPreferences(private val context: Context) {
prefs.remove(activePasswordKey)
prefs.remove(activeNameKey)
prefs.remove(activeMeshIpKey)
prefs.remove(activeNpubKey)
}
}
@@ -134,65 +149,66 @@ class ServerPreferences(private val context: Context) {
}
/**
* Replace a saved server in place. Matches the existing entry by connection
* identity (address/port/scheme) so edits that change the name or password —
* or that touch a legacy 4-field entry — still update the right record. If the
* edited server is also the active one, the active record is kept in sync.
* Replace a saved server in place. Matches the existing entry by node
* identity — npub first, address/port/scheme as the LAN-only fallback
* (ServerEntry.sameNode) — so edits that change the name, password or even
* every address still update the right record. An edit form that doesn't
* carry the npub keeps the stored one. If the edited server is also the
* active one, the active record is kept in sync.
*/
suspend fun updateSavedServer(original: ServerEntry, updated: ServerEntry) {
val toStore = updated.copy(npub = updated.npub.ifBlank { original.npub })
context.dataStore.edit { prefs ->
val current = prefs[savedServersKey] ?: emptySet()
val filtered = current.filterNot { raw ->
val e = ServerEntry.deserialize(raw)
e != null &&
e.address == original.address &&
e.port == original.port &&
e.useHttps == original.useHttps
ServerEntry.deserialize(raw)?.sameNode(original) == true
}.toSet()
prefs[savedServersKey] = filtered + updated.serialize()
prefs[savedServersKey] = filtered + toStore.serialize()
val isActive = prefs[activeAddressKey] == original.address &&
(prefs[activePortKey] ?: "") == original.port &&
(prefs[activeHttpsKey] ?: false) == original.useHttps
val activeNpub = prefs[activeNpubKey] ?: ""
val isActive = (activeNpub.isNotBlank() && activeNpub == original.npub) ||
(
prefs[activeAddressKey] == original.address &&
(prefs[activePortKey] ?: "") == original.port &&
(prefs[activeHttpsKey] ?: false) == original.useHttps
)
if (isActive) {
prefs[activeAddressKey] = updated.address
prefs[activeHttpsKey] = updated.useHttps
prefs[activePortKey] = updated.port
prefs[activePasswordKey] = updated.password
prefs[activeNameKey] = updated.name
prefs[activeMeshIpKey] = updated.meshIp
prefs[activeAddressKey] = toStore.address
prefs[activeHttpsKey] = toStore.useHttps
prefs[activePortKey] = toStore.port
prefs[activePasswordKey] = toStore.password
prefs[activeNameKey] = toStore.name
prefs[activeMeshIpKey] = toStore.meshIp
prefs[activeNpubKey] = toStore.npub
}
}
}
/**
* Add a server, or update the entry with the same connection identity
* (address/port/scheme) — used by QR pairing so re-scanning a node never
* duplicates it. A blank incoming password/name keeps the stored value
* (a real node's QR never carries the password). Returns the merged entry.
* Add a server, or update the entry for the same node — npub first,
* address/port/scheme as the LAN-only fallback (ServerEntry.sameNode) —
* used by QR pairing so re-scanning a node never duplicates it, even after
* the LAN renumbered and every address changed (npub-first contract in
* docs/companion-pairing-qr.md). A blank incoming password/name keeps the
* stored value (a real node's QR never carries the password). Returns the
* merged entry.
*/
suspend fun upsertServer(server: ServerEntry): ServerEntry {
var merged = server
context.dataStore.edit { prefs ->
val current = prefs[savedServersKey] ?: emptySet()
val existing = current.mapNotNull { ServerEntry.deserialize(it) }.firstOrNull {
it.address == server.address &&
it.port == server.port &&
it.useHttps == server.useHttps
}
val existing = current.mapNotNull { ServerEntry.deserialize(it) }
.firstOrNull { it.sameNode(server) }
if (existing != null) {
merged = server.copy(
password = server.password.ifBlank { existing.password },
name = server.name.ifBlank { existing.name },
meshIp = server.meshIp.ifBlank { existing.meshIp },
npub = server.npub.ifBlank { existing.npub },
)
}
val filtered = current.filterNot { raw ->
val e = ServerEntry.deserialize(raw)
e != null &&
e.address == server.address &&
e.port == server.port &&
e.useHttps == server.useHttps
ServerEntry.deserialize(raw)?.sameNode(merged) == true
}.toSet()
prefs[savedServersKey] = filtered + merged.serialize()
}
@@ -202,15 +218,11 @@ class ServerPreferences(private val context: Context) {
suspend fun removeSavedServer(server: ServerEntry) {
context.dataStore.edit { prefs ->
val current = prefs[savedServersKey] ?: emptySet()
// Match by connection identity (address/port/scheme) rather than the
// exact serialized string, so a rename — or the legacy 4-field format
// saved before names existed — still removes the right entry.
// Match by node identity (npub, else address/port/scheme) rather
// than the exact serialized string, so a rename — or a legacy
// short-format entry — still removes the right record.
prefs[savedServersKey] = current.filterNot { raw ->
val e = ServerEntry.deserialize(raw)
e != null &&
e.address == server.address &&
e.port == server.port &&
e.useHttps == server.useHttps
ServerEntry.deserialize(raw)?.sameNode(server) == true
}.toSet()
}
}
@@ -78,6 +78,9 @@ object ServerQrParser {
password = credential,
name = uri.getQueryParameter("name") ?: "",
meshIp = fips?.ula ?: "",
// npub is the durable identity — saved-server upserts match on
// it, so re-scanning after a LAN renumber updates in place.
npub = fips?.npub ?: "",
),
fips = fips,
)
@@ -10,10 +10,15 @@ import android.os.Build
import android.util.Log
import com.archipelago.app.MainActivity
import com.archipelago.app.R
import com.archipelago.app.data.ServerPreferences
import kotlinx.coroutines.CoroutineScope
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.Job
import kotlinx.coroutines.SupervisorJob
import kotlinx.coroutines.cancel
import kotlinx.coroutines.delay
import kotlinx.coroutines.flow.first
import kotlinx.coroutines.isActive
import kotlinx.coroutines.launch
/**
@@ -27,6 +32,7 @@ import kotlinx.coroutines.launch
class ArchyVpnService : VpnService() {
private val scope = CoroutineScope(SupervisorJob() + Dispatchers.IO)
private var warmerJob: Job? = null
override fun onStartCommand(intent: Intent?, flags: Int, startId: Int): Int {
if (intent?.action == ACTION_STOP) {
@@ -61,6 +67,21 @@ class ArchyVpnService : VpnService() {
.setMtu(1280)
.addAddress(identity.address, 128)
.addRoute("fd00::", 8)
// The TUN is IPv6-only. Android blocks every address family
// the VPN has no address for — without this, bringing the
// mesh up cut ALL of the phone's IPv4 internet.
.allowFamily(android.system.OsConstants.AF_INET)
// And let apps that bind their own network skip the TUN
// entirely — this is mesh reachability, not a privacy VPN.
.allowBypass()
.apply {
// Android 10+ treats VPN networks as METERED by default,
// which flips the whole phone into data-saver behaviour
// (background sync off, "metered" warnings) while the
// mesh is up. It inherits the underlying network's real
// metered state instead.
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.Q) setMetered(false)
}
.establish()
} catch (e: Exception) {
Log.e(TAG, "VPN establish failed", e)
@@ -74,10 +95,57 @@ class ArchyVpnService : VpnService() {
val fd = pfd.detachFd()
val result = FipsNative.start(identity.secret, peersJson, fd)
Log.i(TAG, "mesh start: $result")
if (result.contains("\"error\"")) shutdown()
if (result.contains("\"error\"")) {
shutdown()
} else {
startSessionWarmer()
}
}
/**
* Pre-warm + keep-warm mesh sessions to every known node ULA.
*
* Discovery + first session through the public tree can take 15s+
* (HANDOFF-2026-07-23 node diagnosis) — paying that cost here, the
* moment the tunnel is up, means the connect probe and WebView hit an
* established session instead of timing out on a cold one. The periodic
* touch afterwards keeps the session from idling out. Failed connects
* are expected and cheap; the attempt itself is what drives discovery.
*/
private fun startSessionWarmer() {
warmerJob?.cancel()
warmerJob = scope.launch {
val prefs = ServerPreferences(this@ArchyVpnService)
var round = 0
while (isActive && FipsNative.isRunning()) {
val ulas = try {
prefs.savedServers.first().mapNotNull { it.meshIp.ifBlank { null } }.distinct()
} catch (_: Exception) {
emptyList()
}
for (ula in ulas) {
try {
java.net.Socket().use { s ->
s.connect(
java.net.InetSocketAddress(java.net.InetAddress.getByName(ula), 80),
20_000,
)
}
} catch (_: Exception) {
// Cold path / node away — the connect attempt still
// drove session establishment; try again next round.
}
}
round++
// Aggressive for the first ~minute (session bring-up), then a
// slow keep-warm tick that costs nearly nothing.
delay(if (round < 12) 5_000 else 60_000)
}
}
}
private fun shutdown() {
warmerJob?.cancel()
FipsNative.stop()
stopForeground(STOP_FOREGROUND_REMOVE)
stopSelf()
@@ -12,6 +12,16 @@ import androidx.datastore.preferences.preferencesDataStore
private val Context.fipsDataStore: DataStore<Preferences> by preferencesDataStore(name = "fips_prefs")
// Archipelago-operated public anchor (vps2). Baked in so EVERY pairing yields
// both paths — direct LAN p2p to the node AND a public rendezvous for
// away-from-home — even when the scanned node is old enough that its QR
// carries no fanchors. Keep in lockstep with
// core/archipelago/src/fips/anchors.rs (ARCHY_ANCHOR_*).
internal const val ARCHY_ANCHOR_NPUB =
"npub1dptaktwxv0mm245g2lqjykwm5ll0jpc6m3r4242ydfa9z7qe6urs3jvrak"
internal const val ARCHY_ANCHOR_ADDR = "146.59.87.168:8444"
internal const val ARCHY_ANCHOR_TRANSPORT = "tcp"
/**
* Mesh identity + known node peers. Follows the same plaintext-DataStore
* storage model as ServerPreferences (the server password lives there the
@@ -91,6 +101,21 @@ class FipsPreferences(private val context: Context) {
}))
}
}
// Guarantee the public anchor: without it, a QR from an older node
// leaves the phone LAN-only and pairing/connecting dies off-LAN.
if (info.npub != ARCHY_ANCHOR_NPUB &&
incoming.none { it.optString("npub") == ARCHY_ANCHOR_NPUB }
) {
incoming += JSONObject().apply {
put("npub", ARCHY_ANCHOR_NPUB)
put("alias", "Archipelago anchor")
put("addresses", JSONArray().put(JSONObject().apply {
put("transport", ARCHY_ANCHOR_TRANSPORT)
put("addr", ARCHY_ANCHOR_ADDR)
put("priority", 40)
}))
}
}
val incomingNpubs = incoming.map { it.optString("npub") }.toSet()
context.fipsDataStore.edit { prefs ->
val current = JSONArray(prefs[peersKey] ?: "[]")
@@ -85,6 +85,7 @@ import com.archipelago.app.ui.theme.TextMuted
import com.archipelago.app.ui.theme.TextPrimary
import com.archipelago.app.ui.theme.TextSecondary
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.delay
import kotlinx.coroutines.launch
import kotlinx.coroutines.withContext
import java.net.HttpURLConnection
@@ -171,10 +172,35 @@ fun ServerConnectScreen(
errorMessage = null
scope.launch {
val result = testConnection(server)
var reachable = testConnection(server)
// LAN address didn't answer — phone off-LAN (5G) or DHCP moved the
// node. The scanned IP was only ever a dial hint; the node's real
// identity is its npub and its ULA is reachable from anywhere over
// the mesh. Bring the tunnel up and probe the ULA before failing.
if (!reachable && server.meshIp.isNotBlank()) {
FipsManager.autoStartIfReady(context)
val meshServer = server.copy(
address = server.meshIp,
useHttps = false,
port = "",
)
// Mesh discovery + first session can take 15s+ through the
// public tree (HANDOFF-2026-07-23 node diagnosis), and on a
// first-ever pairing the VPN consent dialog is on screen at
// the same time — so probe patiently inside a 60s budget with
// per-attempt timeouts wide enough to ride out TCP
// retransmit backoff. The VPN service pre-warms the session
// in parallel (ArchyVpnService.startSessionWarmer).
val deadline = System.currentTimeMillis() + 60_000
while (!reachable && System.currentTimeMillis() < deadline) {
reachable = testConnection(meshServer, timeoutMs = 15_000)
if (!reachable) delay(3000)
}
}
isConnecting = false
if (result) {
if (reachable) {
prefs.setActiveServer(server)
onConnected(server.toUrl())
} else {
@@ -651,8 +677,10 @@ private fun sanitizeAddress(input: String): String {
.trimEnd('/')
}
/** Test RPC connectivity. Accepts self-signed certs for local LAN servers. */
private suspend fun testConnection(server: ServerEntry): Boolean {
/** Test RPC connectivity. Accepts self-signed certs for local LAN servers.
* [timeoutMs] is per-phase (connect / read) — mesh probes need far more
* patience than LAN ones (first session through the tree can take 15s+). */
private suspend fun testConnection(server: ServerEntry, timeoutMs: Int = 5000): Boolean {
return withContext(Dispatchers.IO) {
try {
val url = URL("${server.toUrl()}/rpc/v1")
@@ -672,8 +700,8 @@ private suspend fun testConnection(server: ServerEntry): Boolean {
}
connection.requestMethod = "POST"
connection.connectTimeout = 5000
connection.readTimeout = 5000
connection.connectTimeout = timeoutMs
connection.readTimeout = timeoutMs
connection.setRequestProperty("Content-Type", "application/json")
connection.doOutput = true
val body = """{"method":"server.echo","params":{"message":"ping"}}"""
@@ -87,6 +87,28 @@ import kotlinx.coroutines.launch
import kotlinx.coroutines.withContext
import org.json.JSONObject
/** True when a TCP listener answers at [base]'s host:port within [timeoutMs]. */
private fun tcpAnswers(base: String, timeoutMs: Int): Boolean = try {
val u = android.net.Uri.parse(base)
val port = if (u.port != -1) u.port else if (u.scheme == "https") 443 else 80
java.net.Socket().use {
it.connect(java.net.InetSocketAddress(u.host, port), timeoutMs)
true
}
} catch (_: Exception) {
false
}
/** Fastest answering origin: LAN inside a short window, else the mesh ULA
* (patient — a cold session may still be establishing), else LAN anyway so
* the existing error/fallback path handles it. */
private suspend fun pickStartUrl(lanUrl: String, meshUrl: String?): String =
withContext(Dispatchers.IO) {
if (tcpAnswers(lanUrl, 2500)) return@withContext lanUrl
if (meshUrl != null && tcpAnswers(meshUrl, 12_000)) return@withContext meshUrl
lanUrl
}
/** Open a URL in the phone's default browser (genuinely external links). */
private fun openExternalUrl(context: android.content.Context, url: String) {
try {
@@ -168,6 +190,20 @@ fun WebViewScreen(
var hasError by remember { mutableStateOf(false) }
var webView by remember { mutableStateOf<WebView?>(null) }
// Race LAN vs mesh BEFORE the WebView exists: Chromium pointed at an
// unreachable LAN IP burns a minute+ in connect retries before
// onReceivedError fires the mesh fallback — the "stuck connecting"
// stall. A raw TCP probe answers in milliseconds at home and fails in
// ~2.5s off-LAN, so startup lands on the right origin in seconds.
var startUrl by remember(serverUrl) { mutableStateOf<String?>(null) }
var raceNonce by remember { mutableIntStateOf(0) }
LaunchedEffect(serverUrl, meshFallbackUrl, raceNonce) {
val picked = pickStartUrl(serverUrl, meshFallbackUrl)
// Starting on the mesh: don't bounce back to it on error (it IS it).
if (picked != serverUrl) triedMeshFallback = true
startUrl = picked
}
// Web-page camera access (wallet QR scanner). The WebView's default
// WebChromeClient silently denies getUserMedia, so grant video capture —
// asking for the app-level CAMERA permission first when needed.
@@ -187,6 +223,14 @@ fun WebViewScreen(
// while this is shown, so closing it returns instantly with no reload.
var inAppUrl by remember { mutableStateOf<String?>(null) }
// Same node = EITHER of its addresses. Over the mesh the kiosk's host is
// the ULA while app links may carry the LAN IP (and vice versa) —
// comparing against one host bounced same-node apps (Pine, Home
// Assistant) out to the phone's external browser.
fun isSameNode(url: String): Boolean =
isSameHost(url, serverUrl) ||
(meshFallbackUrl != null && isSameHost(url, meshFallbackUrl))
// Native wallet QR scanner, opened by the web UI via the ArchipelagoQr
// bridge; status lines stream back from the page while it's up.
var walletScannerVisible by remember { mutableStateOf(false) }
@@ -268,9 +312,13 @@ fun WebViewScreen(
GlassButton(
text = stringResource(R.string.retry),
onClick = {
// Re-race LAN vs mesh — the network we're on may have
// changed since the last pick.
hasError = false
isLoading = true
webView?.loadUrl(serverUrl)
triedMeshFallback = false
startUrl = null
raceNonce++
},
modifier = Modifier.fillMaxWidth().height(56.dp),
)
@@ -283,9 +331,16 @@ fun WebViewScreen(
modifier = Modifier.fillMaxWidth().height(48.dp),
)
}
} else if (startUrl == null) {
// Racing LAN vs mesh (≤2.5s at home, a few seconds off-LAN) —
// far cheaper than letting Chromium retry a dead LAN IP.
Box(Modifier.fillMaxSize(), contentAlignment = Alignment.Center) {
CircularProgressIndicator(color = BitcoinOrange)
}
} else {
// Edge-to-edge WebView — background bleeds behind status bar.
// Safe area values injected as CSS env() polyfill on each page load.
val initialUrl = startUrl ?: serverUrl
AndroidView(
modifier = Modifier.fillMaxSize(),
factory = { context ->
@@ -319,7 +374,7 @@ fun WebViewScreen(
// kiosk couldn't iframe — keep the user inside the app)
// - different host → the phone's real browser
fun routeOutbound(url: String) {
if (isSameHost(url, serverUrl)) {
if (isSameNode(url)) {
inAppUrl = url
} else {
openExternalUrl(context, url)
@@ -458,7 +513,7 @@ fun WebViewScreen(
error: android.net.http.SslError?,
) {
val u = error?.url
if (u != null && isSameHost(u, serverUrl)) {
if (u != null && isSameNode(u)) {
handler?.proceed()
} else {
handler?.cancel()
@@ -595,7 +650,7 @@ fun WebViewScreen(
}
webView = this
loadUrl(serverUrl)
loadUrl(initialUrl)
}
},
)
@@ -620,6 +675,7 @@ fun WebViewScreen(
InAppBrowser(
url = target,
serverUrl = serverUrl,
meshUrl = meshFallbackUrl,
onClose = { inAppUrl = null },
)
}
@@ -693,9 +749,14 @@ private fun fetchFavicon(pageUrl: String): Bitmap? {
private fun InAppBrowser(
url: String,
serverUrl: String,
meshUrl: String? = null,
onClose: () -> Unit,
) {
val context = LocalContext.current
// Same-node check across BOTH node addresses (LAN + mesh ULA) — see the
// kiosk's isSameNode; a mismatch here bounced app links to the browser.
fun isSameNode(u: String): Boolean =
isSameHost(u, serverUrl) || (meshUrl != null && isSameHost(u, meshUrl))
var browser by remember { mutableStateOf<WebView?>(null) }
var title by remember { mutableStateOf(android.net.Uri.parse(url).host ?: url) }
var favicon by remember { mutableStateOf<Bitmap?>(null) }
@@ -809,7 +870,7 @@ private fun InAppBrowser(
error: android.net.http.SslError?,
) {
val u = error?.url
if (u != null && isSameHost(u, serverUrl)) {
if (u != null && isSameNode(u)) {
handler?.proceed()
} else {
handler?.cancel()
@@ -823,7 +884,7 @@ private fun InAppBrowser(
val u = request?.url?.toString() ?: return false
// Stay in the overlay for same-node navigation;
// hand genuinely external links to the real browser.
if (isSameHost(u, serverUrl)) return false
if (isSameNode(u)) return false
openExternalUrl(ctx, u)
return true
}
+1
View File
@@ -86,6 +86,7 @@ dependencies = [
"getrandom 0.2.17",
"hex",
"jni",
"libc",
"paranoid-android",
"serde_json",
"tokio",
+2
View File
@@ -31,6 +31,8 @@ hex = "0.4"
getrandom = "0.2"
tokio = { version = "1", features = ["rt-multi-thread", "sync", "time", "macros"] }
tracing = "0.1"
# fcntl: force the VpnService TUN fd into blocking mode (see mesh::start).
libc = "0.2"
# The JNI surface only exists on Android; host builds skip it and drive the
# mesh module directly (tests).
+13
View File
@@ -97,6 +97,19 @@ pub fn parse_peers(peers_json: &str) -> Result<Vec<PeerConfig>> {
pub fn start(secret: &str, peers_json: &str, tun_fd: i32) -> Result<(String, String)> {
stop();
// Android hands the VpnService TUN fd over in non-blocking mode on some
// OS builds. The fips TUN reader is a dedicated blocking-read thread that
// treats EAGAIN as fatal — the loop died at startup ("TUN read error …
// Try again (os error 11)" on-device), so sessions came up but no packet
// ever entered the mesh. Force the fd into the blocking mode the reader
// is designed for.
unsafe {
let flags = libc::fcntl(tun_fd, libc::F_GETFL);
if flags >= 0 && (flags & libc::O_NONBLOCK) != 0 {
libc::fcntl(tun_fd, libc::F_SETFL, flags & !libc::O_NONBLOCK);
}
}
let peers = parse_peers(peers_json)?;
let config = build_config(secret, peers);
let mut node = Node::new(config).map_err(|e| anyhow!("node init: {e}"))?;
+51
View File
@@ -84,6 +84,15 @@ version = "1.0.100"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a23eb6b1614318a8071c9b2521f36b424b2c83db5eb3a0fead4a6c0809af6e61"
[[package]]
name = "arbitrary"
version = "1.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c3d036a3c4ab069c7b410a2ce876bd74808d2d0888a82667669f8e783a898bf1"
dependencies = [
"derive_arbitrary",
]
[[package]]
name = "arc-swap"
version = "1.9.1"
@@ -145,6 +154,7 @@ dependencies = [
"serde_yaml",
"serial2-tokio",
"sha2 0.10.9",
"socket2 0.5.10",
"tar",
"tempfile",
"thiserror 1.0.69",
@@ -160,6 +170,7 @@ dependencies = [
"uuid",
"zbase32",
"zeroize",
"zip",
]
[[package]]
@@ -1174,6 +1185,17 @@ version = "0.5.8"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c"
[[package]]
name = "derive_arbitrary"
version = "1.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1e567bd82dcff979e4b03460c307b3cdc9e96fde3d73bed1496d2bc75d9dd62a"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.114",
]
[[package]]
name = "derive_builder"
version = "0.20.2"
@@ -6894,8 +6916,37 @@ dependencies = [
"syn 2.0.114",
]
[[package]]
name = "zip"
version = "2.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fabe6324e908f85a1c52063ce7aa26b68dcb7eb6dbc83a2d148403c9bc3eba50"
dependencies = [
"arbitrary",
"crc32fast",
"crossbeam-utils",
"displaydoc",
"flate2",
"indexmap",
"memchr",
"thiserror 2.0.18",
"zopfli",
]
[[package]]
name = "zmij"
version = "1.0.16"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dfcd145825aace48cff44a8844de64bf75feec3080e0aa5cdbde72961ae51a65"
[[package]]
name = "zopfli"
version = "0.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f05cd8797d63865425ff89b5c4a48804f35ba0ce8d125800027ad6017d2b5249"
dependencies = [
"bumpalo",
"crc32fast",
"log",
"simd-adler32",
]
+7
View File
@@ -22,6 +22,9 @@ iroh-swarm = ["dep:iroh", "dep:iroh-blobs"]
[dependencies]
# Core dependencies
tokio = { version = "1", features = ["full"] }
# Mesh port mirror: needs IPV6_V6ONLY on [::] listeners so they coexist with
# the containers' own 0.0.0.0 binds (std/tokio don't expose the sockopt).
socket2 = "0.5"
libc = "0.2" # process-group signalling for the supervised reticulum daemon
serde = { version = "1.0", features = ["derive"] }
serde_json = "1.0"
@@ -105,6 +108,10 @@ bytes = "1"
# Mesh networking (Meshcore serial protocol over USB LoRa radios)
serial2-tokio = "0.1"
# LoRa radio firmware flashing: Meshtastic ships per-board images inside a
# per-platform release zip (see mesh/flash.rs).
zip = { version = "2", default-features = false, features = ["deflate"] }
# Double Ratchet key derivation (Phase 3: encrypted mesh messaging)
hkdf = "0.12.4"
@@ -388,6 +388,10 @@ impl RpcHandler {
// Mesh networking (Meshcore LoRa)
"mesh.status" => self.handle_mesh_status().await,
"mesh.probe-device" => self.handle_mesh_probe_device(params).await,
"mesh.flash-list-firmware" => self.handle_mesh_flash_list_firmware(params).await,
"mesh.flash-device" => self.handle_mesh_flash_device(params).await,
"mesh.flash-status" => self.handle_mesh_flash_status().await,
"mesh.flash-cancel" => self.handle_mesh_flash_cancel().await,
"mesh.peers" => self.handle_mesh_peers().await,
"mesh.messages" => self.handle_mesh_messages(params).await,
"mesh.debug-dump" => self.handle_mesh_debug_dump().await,
+132
View File
@@ -0,0 +1,132 @@
use super::super::RpcHandler;
use crate::mesh;
use crate::mesh::flash::{self, FlashBoard, FlashJobStatus};
use crate::mesh::types::DeviceType;
use anyhow::Result;
fn parse_family(s: &str) -> Result<DeviceType> {
match s.trim().to_lowercase().as_str() {
"meshcore" => Ok(DeviceType::Meshcore),
"meshtastic" => Ok(DeviceType::Meshtastic),
"reticulum" | "rnode" => Ok(DeviceType::Reticulum),
other => anyhow::bail!("Unknown firmware family: {other} (expected meshcore|meshtastic|reticulum)"),
}
}
fn parse_board(s: &str) -> Result<FlashBoard> {
match s.trim().to_lowercase().as_str() {
"heltec-v3" | "heltec_v3" | "heltecv3" => Ok(FlashBoard::HeltecV3),
"heltec-v4" | "heltec_v4" | "heltecv4" => Ok(FlashBoard::HeltecV4),
other => anyhow::bail!("Unknown board: {other} (expected heltec-v3|heltec-v4)"),
}
}
impl RpcHandler {
/// mesh.flash-list-firmware — resolve the available firmware version(s)
/// for a given family. v1 only ever surfaces "latest".
pub(in crate::api::rpc) async fn handle_mesh_flash_list_firmware(
&self,
params: Option<serde_json::Value>,
) -> Result<serde_json::Value> {
let family = params
.as_ref()
.and_then(|p| p.get("family"))
.and_then(|v| v.as_str())
.ok_or_else(|| anyhow::anyhow!("Missing family"))?;
let family = parse_family(family)?;
let versions = flash::list_firmware(family).await?;
Ok(serde_json::json!({ "versions": versions }))
}
/// mesh.flash-device — erase and reflash a detected LoRa radio with the
/// latest firmware for the given family, defaulting to a full chip
/// erase before write. `board` is optional: if the port's USB vid:pid
/// unambiguously resolves to a known board, that's used; otherwise the
/// caller must supply it explicitly (see `flash::resolve_flash_board`'s
/// doc comment on why we refuse to guess).
pub(in crate::api::rpc) async fn handle_mesh_flash_device(
&self,
params: Option<serde_json::Value>,
) -> Result<serde_json::Value> {
let path = params
.as_ref()
.and_then(|p| p.get("path"))
.and_then(|v| v.as_str())
.ok_or_else(|| anyhow::anyhow!("Missing path"))?
.to_string();
let family = params
.as_ref()
.and_then(|p| p.get("family"))
.and_then(|v| v.as_str())
.ok_or_else(|| anyhow::anyhow!("Missing family"))?;
let family = parse_family(family)?;
let detected = mesh::detect_devices().await;
anyhow::ensure!(
detected.iter().any(|d| d == &path),
"{path} is not a detected mesh-radio candidate port"
);
let board = match params
.as_ref()
.and_then(|p| p.get("board"))
.and_then(|v| v.as_str())
{
Some(explicit) => parse_board(explicit)?,
None => {
let info = mesh::detect_devices_info()
.await
.into_iter()
.find(|d| d.path == path);
info.as_ref()
.and_then(flash::resolve_flash_board)
.ok_or_else(|| {
anyhow::anyhow!(
"Could not auto-detect the board on {path} — specify board explicitly"
)
})?
}
};
flash::start_flash_job(
&self.flash_job,
&self.mesh_service_arc(),
self.config.data_dir.clone(),
path,
board,
family,
)
.await?;
Ok(serde_json::json!({ "started": true }))
}
/// mesh.flash-status — poll the current (or most recent) flash job.
pub(in crate::api::rpc) async fn handle_mesh_flash_status(&self) -> Result<serde_json::Value> {
let job = self.flash_job.read().await;
match job.as_ref() {
Some(j) => {
let status: FlashJobStatus = j.snapshot().await;
let mut value = serde_json::to_value(&status)?;
if let Some(obj) = value.as_object_mut() {
obj.insert("active".into(), (!status.done).into());
}
Ok(value)
}
None => Ok(serde_json::json!({ "active": false })),
}
}
/// mesh.flash-cancel — best-effort; only honored before erase/write has
/// started (see `FlashJob::cancel`'s doc comment).
pub(in crate::api::rpc) async fn handle_mesh_flash_cancel(&self) -> Result<serde_json::Value> {
let job = self.flash_job.read().await;
match job.as_ref() {
Some(j) => {
j.cancel().await?;
Ok(serde_json::json!({ "cancelled": true }))
}
None => anyhow::bail!("No flash job in progress"),
}
}
}
+1
View File
@@ -1,5 +1,6 @@
mod assistant;
mod bitcoin_ops;
mod flash;
mod messaging;
mod safety;
mod status;
+30 -6
View File
@@ -101,12 +101,36 @@ impl RpcHandler {
detected.iter().any(|d| d == &path),
"{path} is not a detected mesh-radio candidate port"
);
let service = self.mesh_service.read().await;
let probe = match service.as_ref() {
Some(svc) => svc.probe_device(&path).await?,
// No mesh service yet (radio never enabled) — probe directly.
None => mesh::listener::probe_device(&path).await?,
};
// Refuse to probe while a firmware flash is in flight. Confirmed
// live 2026-07-23: esptool ("multiple access on port?") and
// rnodeconf (OSError Errno 71 Protocol error on an RTS ioctl) both
// failed with symptoms consistent with a second process holding the
// same serial fd — the flash subprocess runs for minutes outside
// our own async runtime, so nothing previously stopped a concurrent
// `mesh.probe-device` call (e.g. the hot-swap modal's own re-probe)
// from opening the identical port at the same time and corrupting
// both operations' handshakes.
if let Some(job) = self.flash_job.read().await.as_ref() {
anyhow::ensure!(
job.snapshot().await.done,
"A firmware flash is in progress — refusing to probe the serial port until it finishes"
);
}
// Only hold the mesh_service lock long enough for the quick
// active-path guard check — NEVER across the actual probe, which
// can take 15-60s across its internal collision retries. Confirmed
// live 2026-07-23: holding this read lock for the full probe starved
// a concurrent firmware-flash job's MeshService::stop() (which needs
// the write lock) well past its own bounded timeout, surfacing as
// "Mesh listener did not release the serial port" even though
// stop() itself was fast.
{
let service = self.mesh_service.read().await;
if let Some(svc) = service.as_ref() {
svc.ensure_probe_allowed(&path).await?;
}
}
let probe = mesh::listener::probe_device(&path).await?;
Ok(serde_json::to_value(probe)?)
}
+4
View File
@@ -89,6 +89,9 @@ pub struct RpcHandler {
endpoint_rate_limiter: EndpointRateLimiter,
response_cache: ResponseCache,
mesh_service: Arc<tokio::sync::RwLock<Option<crate::mesh::MeshService>>>,
/// LoRa radio firmware-flash job state, sibling to `mesh_service` — one
/// job at a time, since flashing needs exclusive access to the port.
flash_job: crate::mesh::flash::FlashJobHandle,
transport_router: Arc<tokio::sync::RwLock<Option<Arc<crate::transport::TransportRouter>>>>,
/// Shared content-addressed blob store. Set by ApiHandler after construction
/// so mesh.send-content / mesh.fetch-content RPCs can reach it without a
@@ -160,6 +163,7 @@ impl RpcHandler {
endpoint_rate_limiter,
response_cache: ResponseCache::new(5),
mesh_service: Arc::new(tokio::sync::RwLock::new(None)),
flash_job: crate::mesh::flash::new_job_handle(),
transport_router: Arc::new(tokio::sync::RwLock::new(None)),
blob_store: Arc::new(tokio::sync::RwLock::new(None)),
self_pubkey_hex: Arc::new(tokio::sync::RwLock::new(None)),
+144 -82
View File
@@ -38,7 +38,19 @@ impl RpcHandler {
.unwrap_or("")
.to_string();
let routers = detect::scan_subnet(subnet, prefix, &ssh_user, &ssh_password).await;
// `scan_subnet` is `async fn` but loops over up to a full /24 of
// blocking TCP probes + SSH handshakes with no real await points —
// same blocking-on-a-worker-thread hazard as the other openwrt
// handlers (see handle_openwrt_get_status), just larger in scope.
let routers = tokio::task::spawn_blocking(move || {
tokio::runtime::Handle::current().block_on(detect::scan_subnet(
subnet,
prefix,
&ssh_user,
&ssh_password,
))
})
.await?;
let ips: Vec<String> = routers.iter().map(|ip| ip.to_string()).collect();
Ok(serde_json::json!({ "routers": ips }))
@@ -87,8 +99,84 @@ impl RpcHandler {
.or_else(|| saved.password.clone())
.unwrap_or_default();
let router = Router::connect_password(&host, 22, &ssh_user, &ssh_password)?;
router.verify_openwrt()?;
// Router/Session (ssh2) is a fully synchronous, blocking API with no
// timeout on the initial TCP connect — run it on the blocking pool,
// not directly on a tokio worker thread. Inlined here, a single
// unreachable router (e.g. after physically relocating the node, so
// the configured router is on a different/unreachable network) hangs
// for the OS's default TCP connect timeout (routinely 2+ minutes),
// and every concurrent poll of this endpoint eats another worker
// thread — with only a handful of worker threads total, that starves
// every other in-flight request in the whole process. This was a
// real full-node outage (2026-07-24), diagnosed via a live gdb
// backtrace showing 3 of 4 worker threads blocked in this exact
// `TcpStream::connect` → `Router::connect_password` call chain.
let host_for_task = host.clone();
let ssh_user_for_task = ssh_user.clone();
let ssh_password_for_task = ssh_password.clone();
let status = tokio::task::spawn_blocking(move || -> Result<serde_json::Value> {
let router =
Router::connect_password(&host_for_task, 22, &ssh_user_for_task, &ssh_password_for_task)?;
router.verify_openwrt()?;
// System info
let release = router
.run_ok("cat /etc/openwrt_release")
.unwrap_or_default();
let hostname = router
.uci_get("system.@system[0].hostname")
.unwrap_or_else(|_| "unknown".into());
let uptime_secs: u64 = router
.run_ok("cat /proc/uptime")
.unwrap_or_default()
.split_whitespace()
.next()
.and_then(|s| s.split('.').next())
.and_then(|s| s.parse().ok())
.unwrap_or(0);
// TollGate — check via opkg (≤24.x) or binary presence (25.x apk-native).
// The service binary is /usr/bin/tollgate-wrt (per its init.d script),
// not /usr/bin/tollgate-module-basic-go — that's only the opkg/apk
// *package* name, never an on-disk filename.
let tollgate_installed = router
.run("/usr/bin/opkg list-installed 2>/dev/null | grep -q '^tollgate-module-basic-go ' || \
test -f /usr/bin/tollgate-wrt 2>/dev/null")
.map(|(_, code)| code == 0)
.unwrap_or(false);
let tollgate = if tollgate_installed {
serde_json::json!({
"installed": true,
"enabled": router.uci_get("tollgate.main.enabled").map(|v| v == "1").unwrap_or(false),
"metric": router.uci_get("tollgate.main.metric").unwrap_or_default(),
"step_size_ms": router.uci_get("tollgate.main.step_size").ok().and_then(|v| v.parse::<u64>().ok()).unwrap_or(0),
"price_per_step":router.uci_get("tollgate.main.price_per_step").ok().and_then(|v| v.parse::<u64>().ok()).unwrap_or(0),
"min_steps": router.uci_get("tollgate.main.min_steps").ok().and_then(|v| v.parse::<u32>().ok()).unwrap_or(1),
"currency": router.uci_get("tollgate.main.currency").unwrap_or_default(),
"mint_url": router.uci_get("tollgate.main.mint_url").unwrap_or_default(),
})
} else {
serde_json::json!({ "installed": false })
};
// WiFi interfaces
let wifi_raw = router.run_ok("uci show wireless").unwrap_or_default();
let wifi_interfaces = parse_wifi_interfaces(&wifi_raw);
let wan_status = wan::get_wan_status(&router);
Ok(serde_json::json!({
"host": host_for_task,
"hostname": hostname,
"uptime_secs": uptime_secs,
"release": parse_release(&release),
"tollgate": tollgate,
"wifi_interfaces": wifi_interfaces,
"wan": wan_status,
}))
})
.await??;
// Persist the connection so other views (e.g. the Home dashboard's
// Network tile) can poll `openwrt.get-status` with no params instead
@@ -107,62 +195,7 @@ impl RpcHandler {
.await;
}
// System info
let release = router
.run_ok("cat /etc/openwrt_release")
.unwrap_or_default();
let hostname = router
.uci_get("system.@system[0].hostname")
.unwrap_or_else(|_| "unknown".into());
let uptime_secs: u64 = router
.run_ok("cat /proc/uptime")
.unwrap_or_default()
.split_whitespace()
.next()
.and_then(|s| s.split('.').next())
.and_then(|s| s.parse().ok())
.unwrap_or(0);
// TollGate — check via opkg (≤24.x) or binary presence (25.x apk-native).
// The service binary is /usr/bin/tollgate-wrt (per its init.d script),
// not /usr/bin/tollgate-module-basic-go — that's only the opkg/apk
// *package* name, never an on-disk filename.
let tollgate_installed = router
.run("/usr/bin/opkg list-installed 2>/dev/null | grep -q '^tollgate-module-basic-go ' || \
test -f /usr/bin/tollgate-wrt 2>/dev/null")
.map(|(_, code)| code == 0)
.unwrap_or(false);
let tollgate = if tollgate_installed {
serde_json::json!({
"installed": true,
"enabled": router.uci_get("tollgate.main.enabled").map(|v| v == "1").unwrap_or(false),
"metric": router.uci_get("tollgate.main.metric").unwrap_or_default(),
"step_size_ms": router.uci_get("tollgate.main.step_size").ok().and_then(|v| v.parse::<u64>().ok()).unwrap_or(0),
"price_per_step":router.uci_get("tollgate.main.price_per_step").ok().and_then(|v| v.parse::<u64>().ok()).unwrap_or(0),
"min_steps": router.uci_get("tollgate.main.min_steps").ok().and_then(|v| v.parse::<u32>().ok()).unwrap_or(1),
"currency": router.uci_get("tollgate.main.currency").unwrap_or_default(),
"mint_url": router.uci_get("tollgate.main.mint_url").unwrap_or_default(),
})
} else {
serde_json::json!({ "installed": false })
};
// WiFi interfaces
let wifi_raw = router.run_ok("uci show wireless").unwrap_or_default();
let wifi_interfaces = parse_wifi_interfaces(&wifi_raw);
let wan_status = wan::get_wan_status(&router);
Ok(serde_json::json!({
"host": host,
"hostname": hostname,
"uptime_secs": uptime_secs,
"release": parse_release(&release),
"tollgate": tollgate,
"wifi_interfaces": wifi_interfaces,
"wan": wan_status,
}))
Ok(status)
}
/// Provision TollGate on an OpenWrt router and create the "archipelago" SSID.
@@ -228,15 +261,32 @@ impl RpcHandler {
enabled: p.get("enabled").and_then(|v| v.as_bool()).unwrap_or(true),
};
let router = Router::connect_password(&host, 22, &ssh_user, &ssh_password)?;
router.verify_openwrt()?;
tollgate::provision(&router, &config).await?;
let response_ssid = config.ssid.clone();
let response_mint_url = config.mint_url.clone();
// Blocking ssh2 I/O — see handle_openwrt_get_status for why this
// must run on the blocking pool rather than a tokio worker thread.
// `tollgate::provision` is `async fn` but has no real await points
// (every op inside it is a synchronous SSH round trip) — block_on
// here just runs it to completion on this blocking-pool thread
// instead of pretending it yields on a tokio worker.
let host_for_task = host.clone();
let ssh_user_for_task = ssh_user.clone();
let ssh_password_for_task = ssh_password.clone();
tokio::task::spawn_blocking(move || -> Result<()> {
let router =
Router::connect_password(&host_for_task, 22, &ssh_user_for_task, &ssh_password_for_task)?;
router.verify_openwrt()?;
tokio::runtime::Handle::current().block_on(tollgate::provision(&router, &config))?;
Ok(())
})
.await??;
Ok(serde_json::json!({
"ok": true,
"host": host,
"ssid": config.ssid,
"mint_url": config.mint_url,
"ssid": response_ssid,
"mint_url": response_mint_url,
}))
}
@@ -279,22 +329,27 @@ impl RpcHandler {
.or_else(|| saved.password.clone())
.unwrap_or_default();
let router = Router::connect_password(&host, 22, &ssh_user, &ssh_password)?;
router.verify_openwrt()?;
// Blocking ssh2 I/O — see handle_openwrt_get_status for why this
// must run on the blocking pool rather than a tokio worker thread.
let result = tokio::task::spawn_blocking(move || -> Result<Vec<serde_json::Value>> {
let router = Router::connect_password(&host, 22, &ssh_user, &ssh_password)?;
router.verify_openwrt()?;
let networks = wifi_scan::scan_networks(&router)?;
let result: Vec<serde_json::Value> = networks
.iter()
.map(|n| {
serde_json::json!({
"ssid": n.ssid,
"bssid": n.bssid,
"signal": n.signal,
"channel": n.channel,
"encryption": n.encryption,
let networks = wifi_scan::scan_networks(&router)?;
Ok(networks
.iter()
.map(|n| {
serde_json::json!({
"ssid": n.ssid,
"bssid": n.bssid,
"signal": n.signal,
"channel": n.channel,
"encryption": n.encryption,
})
})
})
.collect();
.collect())
})
.await??;
Ok(serde_json::json!({ "networks": result }))
}
@@ -357,9 +412,6 @@ impl RpcHandler {
let dhcp_limit = p.get("dhcp_limit").and_then(|v| v.as_u64()).unwrap_or(150) as u32;
let masq = p.get("masq").and_then(|v| v.as_bool()).unwrap_or(true);
let router = Router::connect_password(&host, 22, &ssh_user, &ssh_password)?;
router.verify_openwrt()?;
let config = wan::WispConfig {
ssid: ssid.clone(),
password,
@@ -368,7 +420,17 @@ impl RpcHandler {
dhcp_limit,
masq,
};
wan::configure_wisp(&router, &config)?;
// Blocking ssh2 I/O — see handle_openwrt_get_status for why this
// must run on the blocking pool rather than a tokio worker thread.
let host_for_task = host.clone();
tokio::task::spawn_blocking(move || -> Result<()> {
let router = Router::connect_password(&host_for_task, 22, &ssh_user, &ssh_password)?;
router.verify_openwrt()?;
wan::configure_wisp(&router, &config)?;
Ok(())
})
.await??;
Ok(serde_json::json!({ "ok": true, "host": host, "ssid": ssid }))
}
+22
View File
@@ -991,6 +991,13 @@ async fn patch_nginx_conf(path: &str) -> Result<bool> {
// B13: fedimint block present but lacking the asset-rewrite sub_filters.
let needs_fedimint_css = content.contains("location /app/fedimint/")
&& !content.contains("'href=\"/' 'href=\"/app/fedimint/'");
// Companion mesh access: phones reach this node over FIPS at its fips0
// ULA (http://[fdxx:…]). Configs shipped before 2026-07-23 listened on
// IPv4 only, so the ULA could never connect — nothing answered [::]:80.
let missing_v6_http = content.contains("listen 80 default_server;")
&& !content.contains("listen [::]:80");
let missing_v6_https = content.contains("listen 443 ssl default_server;")
&& !content.contains("listen [::]:443");
if !missing_app_catalog
&& !missing_bitcoin_status
&& !missing_lnd_proxy
@@ -998,12 +1005,27 @@ async fn patch_nginx_conf(path: &str) -> Result<bool> {
&& !missing_pine_status
&& !has_lnd_dup_cors
&& !needs_fedimint_css
&& !missing_v6_http
&& !missing_v6_https
{
return Ok(false);
}
let mut patched = content.clone();
if missing_v6_http {
patched = patched.replace(
"listen 80 default_server;",
"listen 80 default_server;\n listen [::]:80 default_server;",
);
}
if missing_v6_https {
patched = patched.replace(
"listen 443 ssl default_server;",
"listen 443 ssl default_server;\n listen [::]:443 ssl default_server;",
);
}
if has_lnd_dup_cors {
// Drop the redundant nginx-side CORS headers so the backend's single
// validated Access-Control-Allow-Origin is the only one returned.
+5
View File
@@ -34,6 +34,7 @@ mod bitcoin_rpc;
mod bitcoin_status;
mod blobs;
mod bootstrap;
mod mesh_ports;
mod ceremony;
mod config;
mod constants;
@@ -395,6 +396,10 @@ async fn main() -> Result<()> {
// iframe on kiosk nodes (docs/tv-input-iframe-apps.md).
tokio::spawn(bootstrap::ensure_gamepad_keys());
// Mesh access: mirror IPv4-published app ports onto [::] so direct-port
// app URLs (http://[<fips0 ULA>]:<port>) work from the companion.
tokio::spawn(mesh_ports::run_mesh_port_mirror());
// Pine voice: re-point IP-pinned Wyoming satellite entries (speakers) when
// DHCP renumbering strands them — HA never re-resolves on its own.
tokio::spawn(api::rpc::wyoming_satellite_keeper());
+975
View File
@@ -0,0 +1,975 @@
// WIP mesh/transport protocol — suppress dead code warnings
#![allow(dead_code)]
//! Firmware flashing for LoRa mesh radios — Heltec V3/V4 in v1, across all
//! three firmware families the mesh module already knows how to detect (see
//! `mesh::types::DeviceType`). Firmware is always fetched from upstream at
//! flash time (never bundled/pinned in the repo), and every flash defaults
//! to a full chip erase before write.
//!
//! MeshCore and Meshtastic are flashed the same way: download a released
//! image, `esptool erase_flash`, then `esptool write_flash 0x0 <image>`.
//! Reticulum/RNode is different: `archy-rnodeconf --autoinstall` owns the
//! whole fetch+erase+flash+EEPROM-bootstrap sequence itself (confirmed live
//! via `archy-rnodeconf --help` — there is no raw esptool path exposed for
//! this family, so we deliberately don't resolve a firmware URL ourselves
//! for Reticulum; rnodeconf already knows how).
use super::serial::DetectedDeviceInfo;
use super::types::DeviceType;
use super::MeshService;
use anyhow::{Context, Result};
use regex::Regex;
use serde::Serialize;
use std::path::{Path, PathBuf};
use std::process::Stdio;
use std::sync::{Arc, OnceLock};
use tokio::io::{AsyncBufReadExt, AsyncWriteExt, BufReader};
use tokio::process::Command;
use tokio::sync::RwLock;
use tracing::{info, warn};
/// Boards supported for v1. Both are ESP32-S3 (a single `--chip esp32s3`
/// esptool target covers both), but ship different USB identities and
/// different per-board firmware assets upstream.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
#[serde(rename_all = "lowercase")]
pub enum FlashBoard {
HeltecV3,
HeltecV4,
}
impl FlashBoard {
/// Meshtastic's board id (matches the release manifest's `board` field
/// and its per-board asset naming, e.g. `firmware-heltec-v3-<ver>.factory.bin`).
fn meshtastic_id(self) -> &'static str {
match self {
Self::HeltecV3 => "heltec-v3",
Self::HeltecV4 => "heltec-v4",
}
}
}
/// Map a detected USB vid:pid to a known flashable board, using the same
/// table as `image-recipe/configs/99-mesh-radio.rules`. CP2102 (10c4:ea60)
/// is confirmed there as Heltec V3's USB-UART bridge chip, and is safe to
/// auto-match since that vid:pid is bridge-chip-specific.
///
/// Heltec V4 is NOT auto-matchable and deliberately has no entry here: it
/// was confirmed live (real hardware, 2026-07-23) to use the ESP32-S3's
/// built-in native-USB JTAG/serial peripheral, reporting vid:pid 303a:1001
/// with product string "USB JTAG/serial debug unit" — that descriptor is
/// baked into the chip's ROM and is IDENTICAL across every ESP32-S3 board
/// with native USB enabled, not just Heltec V4. Adding `303a:1001 =>
/// HeltecV4` here would silently misidentify any other native-USB ESP32-S3
/// board (a T3-S3, a bare devkit, etc.) as a V4 and risk writing the wrong
/// board's image. Callers (the RPC layer / frontend) must let the user pick
/// the board manually whenever this returns `None`.
pub fn resolve_flash_board(info: &DetectedDeviceInfo) -> Option<FlashBoard> {
match (info.vid.as_deref(), info.pid.as_deref()) {
(Some("10c4"), Some("ea60")) => Some(FlashBoard::HeltecV3),
_ => None,
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
#[serde(rename_all = "lowercase")]
pub enum FlashStage {
Downloading,
Erasing,
Writing,
Autoinstalling,
Done,
Failed,
}
#[derive(Debug, Clone, Serialize)]
pub struct FlashJobStatus {
pub board: FlashBoard,
pub family: DeviceType,
pub path: String,
pub stage: FlashStage,
pub percent: Option<u8>,
pub log_tail: Vec<String>,
pub done: bool,
pub error: Option<String>,
}
const LOG_TAIL_MAX: usize = 200;
/// How long to wait after a successful flash before resuming the mesh
/// listener, so the board finishes its own post-flash boot/reset before we
/// start opening the port (which itself toggles DTR/RTS) again.
const POST_FLASH_SETTLE_DELAY: std::time::Duration = std::time::Duration::from_secs(5);
/// Absolute ceiling on a whole flash job (download + erase + write, or
/// autoinstall), regardless of what it's doing internally. Last-resort
/// safety net so a hang anywhere can't wedge the single-flash-job guard
/// forever — generous enough to never trigger on a legitimately slow
/// multi-hundred-MB transfer.
const MAX_JOB_DURATION: std::time::Duration = std::time::Duration::from_secs(15 * 60);
/// How long to wait for MeshService::stop() to release the serial port
/// before giving up. Confirmed live 2026-07-23: the listener's own
/// reconnect/multi-candidate-probe loop doesn't check its shutdown signal
/// between candidates, so stop() can take a while (or, if the loop is
/// wedged, never return) — 20s comfortably covers a normal handshake-probe
/// cycle without leaving a flash request hanging indefinitely if the
/// listener genuinely won't let go.
const STOP_LISTENER_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
/// How long to keep retrying the port-free check before giving up.
const PORT_FREE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(10);
/// Confirm nothing else has `path` open by actually opening (and immediately
/// closing) it ourselves. Retries across the timeout since a just-stopped
/// listener's fd can take a moment to actually release even after `stop()`
/// returns (task abort is a request, not an instant guarantee the OS-level
/// resource is gone yet).
async fn wait_for_port_free(path: &str) -> Result<()> {
let deadline = tokio::time::Instant::now() + PORT_FREE_TIMEOUT;
let mut last_err = None;
loop {
match serial2_tokio::SerialPort::open(path, 115200) {
Ok(_) => return Ok(()),
Err(e) => last_err = Some(e),
}
if tokio::time::Instant::now() >= deadline {
break;
}
tokio::time::sleep(std::time::Duration::from_millis(500)).await;
}
Err(anyhow::anyhow!(
"{path} is still held open by something else after {}s (last error: {}) — refusing to start the flasher against a contended port",
PORT_FREE_TIMEOUT.as_secs(),
last_err.map(|e| e.to_string()).unwrap_or_default()
))
}
/// Live state for the one flash job that can run at a time. A single global
/// slot is sufficient because flashing needs exclusive serial access to the
/// one port being flashed — there is no meaningful concept of two concurrent
/// flash jobs on this node.
pub struct FlashJob {
status: RwLock<FlashJobStatus>,
/// Set once the background task is spawned. Only used while `stage` is
/// still `Downloading` — an interrupted erase/write can leave the chip
/// in a worse state than either finished or unstarted, so cancellation
/// is refused once erase begins (see `cancel()`).
abort_handle: RwLock<Option<tokio::task::AbortHandle>>,
}
impl FlashJob {
fn new(board: FlashBoard, family: DeviceType, path: String) -> Arc<Self> {
Arc::new(Self {
abort_handle: RwLock::new(None),
status: RwLock::new(FlashJobStatus {
board,
family,
path,
stage: FlashStage::Downloading,
percent: None,
log_tail: Vec::new(),
done: false,
error: None,
}),
})
}
pub async fn snapshot(&self) -> FlashJobStatus {
self.status.read().await.clone()
}
async fn set_stage(&self, stage: FlashStage) {
let mut s = self.status.write().await;
s.stage = stage;
s.percent = None;
}
async fn set_percent(&self, percent: u8) {
self.status.write().await.percent = Some(percent.min(100));
}
async fn push_log(&self, line: impl Into<String>) {
let mut s = self.status.write().await;
s.log_tail.push(line.into());
let overflow = s.log_tail.len().saturating_sub(LOG_TAIL_MAX);
if overflow > 0 {
s.log_tail.drain(0..overflow);
}
}
async fn fail(&self, err: &anyhow::Error) {
let mut s = self.status.write().await;
s.stage = FlashStage::Failed;
s.error = Some(format!("{err:#}"));
s.done = true;
}
async fn finish(&self) {
let mut s = self.status.write().await;
s.stage = FlashStage::Done;
s.done = true;
}
/// Best-effort cancel: only honored before erase/write/autoinstall has
/// started (i.e. still in `Downloading`). Once a stage that touches the
/// chip begins, this refuses — interrupting an erase or write can leave
/// the flash in a state worse than either finished or unstarted.
pub async fn cancel(&self) -> Result<()> {
let mut s = self.status.write().await;
if s.done {
anyhow::bail!("Flash job already finished");
}
if s.stage != FlashStage::Downloading {
anyhow::bail!(
"Cannot cancel once {:?} has started — let it finish or fail on its own",
s.stage
);
}
if let Some(handle) = self.abort_handle.write().await.take() {
handle.abort();
}
s.stage = FlashStage::Failed;
s.error = Some("Cancelled by user".to_string());
s.done = true;
Ok(())
}
}
/// Shared handle held by `RpcHandler`, sibling to `mesh_service`.
pub type FlashJobHandle = Arc<RwLock<Option<Arc<FlashJob>>>>;
pub fn new_job_handle() -> FlashJobHandle {
Arc::new(RwLock::new(None))
}
fn firmware_cache_dir(data_dir: &Path) -> PathBuf {
data_dir.join("mesh").join("firmware-cache")
}
/// No blanket `.timeout()` here on purpose: reqwest's request timeout covers
/// the *entire* request including streaming the response body, which would
/// kill a legitimate large download partway through (Meshtastic's esp32s3
/// zip is ~170MB) — not just a hung connection. `download_to_file` instead
/// applies a per-chunk stall timeout, and metadata calls (small JSON
/// responses) get their own short timeout at the call site.
fn github_client() -> Result<reqwest::Client> {
reqwest::Client::builder()
.user_agent("archipelago-mesh-flash")
.connect_timeout(std::time::Duration::from_secs(10))
.build()
.context("Failed to build HTTP client")
}
/// Applied per-chunk while streaming a firmware download — if the transfer
/// stalls (no bytes for this long) it's treated as a failure, but a slow
/// download that's still making progress is never killed just for taking a
/// while.
const DOWNLOAD_STALL_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(30);
/// Applied to metadata calls (GitHub release JSON) — these are small
/// responses with no reason to ever take this long.
const METADATA_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
/// Resolve what firmware is available for a board+family. v1 only ever
/// offers "latest" — MeshCore/Meshtastic latest GitHub release, or, for
/// Reticulum, "latest" meaning "whatever archy-rnodeconf --autoinstall
/// resolves on its own" (it does its own version checking upstream).
pub async fn list_firmware(family: DeviceType) -> Result<Vec<String>> {
match family {
DeviceType::Reticulum => Ok(vec!["latest".to_string()]),
DeviceType::Meshtastic => {
let client = github_client()?;
let release: GithubRelease = client
.get("https://api.github.com/repos/meshtastic/firmware/releases/latest")
.send()
.await
.context("Fetching Meshtastic release list")?
.error_for_status()
.context("Meshtastic releases API error")?
.json()
.await
.context("Parsing Meshtastic release JSON")?;
Ok(vec![release.tag_name])
}
DeviceType::Meshcore => {
let client = github_client()?;
let release: GithubRelease = client
.get("https://api.github.com/repos/meshcore-dev/MeshCore/releases/latest")
.send()
.await
.context("Fetching MeshCore release list")?
.error_for_status()
.context("MeshCore releases API error")?
.json()
.await
.context("Parsing MeshCore release JSON")?;
Ok(vec![release.tag_name])
}
DeviceType::Unknown => anyhow::bail!("Pick a firmware family before listing versions"),
}
}
#[derive(serde::Deserialize)]
struct GithubAsset {
name: String,
browser_download_url: String,
}
#[derive(serde::Deserialize)]
struct GithubRelease {
tag_name: String,
assets: Vec<GithubAsset>,
}
/// Start a flash job in the background. Returns as soon as the job has been
/// registered and the listener released — callers poll `FlashJobHandle` via
/// `mesh.flash-status` for progress. Only one job may be in flight at a time.
pub async fn start_flash_job(
handle: &FlashJobHandle,
mesh_service: &Arc<RwLock<Option<MeshService>>>,
data_dir: PathBuf,
path: String,
board: FlashBoard,
family: DeviceType,
) -> Result<()> {
{
let existing = handle.read().await;
if let Some(job) = existing.as_ref() {
if !job.snapshot().await.done {
anyhow::bail!("A firmware flash is already in progress on this node");
}
}
}
let job = FlashJob::new(board, family, path.clone());
*handle.write().await = Some(Arc::clone(&job));
let bg_job = Arc::clone(&job);
let bg_service = Arc::clone(mesh_service);
let task = tokio::spawn(async move {
// esptool/archy-rnodeconf need exclusive serial access — release
// the listener's hold on the port before touching it. This USED
// TO run synchronously in start_flash_job before the job was even
// spawned, blocking the RPC call itself on s.stop().await — a real
// 2026-07-23 incident: the mesh listener was mid a multi-candidate
// reconnect/probe sequence that doesn't check its shutdown signal
// between candidates, so stop() never returned. The HTTP request
// timed out client-side ("Operation failed"), while the job
// (already inserted into `handle`) was permanently wedged — nothing
// had been spawned yet to ever mark it done, so every later flash
// attempt failed with "already in progress" until a full restart.
// Now this runs inside the spawned task with its own bounded
// timeout, so the RPC call always returns immediately regardless,
// and a slow-to-stop listener fails the job cleanly instead of
// hanging everything downstream of it forever.
let stop_result = tokio::time::timeout(STOP_LISTENER_TIMEOUT, async {
let mut svc = bg_service.write().await;
if let Some(s) = svc.as_mut() {
s.stop().await;
}
})
.await;
if stop_result.is_err() {
let err = anyhow::anyhow!(
"Mesh listener did not release the serial port within {}s — it may still be mid a reconnect attempt. Try again once mesh.status shows the device idle, or restart the archipelago service if this persists.",
STOP_LISTENER_TIMEOUT.as_secs()
);
bg_job.push_log(format!("ERROR: {err:#}")).await;
bg_job.fail(&err).await;
return;
}
// Belt-and-suspenders port-free check. `stop()` above should have
// fully released the port, but esptool/rnodeconf run as external
// subprocesses for minutes outside our own async runtime — if
// ANYTHING else still has it open (a racing probe, a not-yet-dropped
// fd from an aborted task, anything we haven't anticipated), handing
// the port to the flasher anyway risks exactly the corruption
// confirmed live 2026-07-23: esptool's "device disconnected or
// multiple access on port?" and rnodeconf's raw `OSError: [Errno 71]
// Protocol error` on an RTS ioctl are both textbook two-openers-on-
// one-fd symptoms. Verify by actually opening it ourselves — cheap,
// and definitive — before ever starting the flasher.
if let Err(e) = wait_for_port_free(&path).await {
bg_job.push_log(format!("ERROR: {e:#}")).await;
bg_job.fail(&e).await;
return;
}
// Outer ceiling on top of run_flash's own internal timeouts —
// belt-and-suspenders so that no future hang (network, subprocess,
// anything) can ever wedge the single-flash-job guard permanently
// again the way a stuck download did on 2026-07-23 (every
// subsequent mesh.flash-device call failed with "already in
// progress" until the service was restarted). Generous enough that
// a legitimately slow multi-hundred-MB transfer still completes.
let result = match tokio::time::timeout(
MAX_JOB_DURATION,
run_flash(board, family, &data_dir, &path, &bg_job),
)
.await
{
Ok(inner) => inner,
Err(_) => Err(anyhow::anyhow!(
"Flash job exceeded the {}-minute ceiling — aborted",
MAX_JOB_DURATION.as_secs() / 60
)),
};
let succeeded = result.is_ok();
match &result {
Ok(()) => {
bg_job.push_log("Flash completed successfully".to_string()).await;
bg_job.finish().await;
info!(path = %path, board = ?board, family = %family, "LoRa firmware flash succeeded");
}
Err(e) => {
// {:#} (alternate Display) walks the full anyhow context
// chain — plain {} / %e only prints the outermost .context()
// message, which made a real 2026-07-23 esptool failure
// undiagnosable from journalctl alone (just "esptool
// erase_flash failed", no actual esptool stderr).
warn!(path = %path, error = %format!("{e:#}"), "LoRa firmware flash failed");
bg_job.push_log(format!("ERROR: {e:#}")).await;
bg_job.fail(e).await;
}
}
// The board's firmware may now differ from whatever was pinned
// before — clear the pin either way so a later reconnect's strict
// auto-detect order picks up reality instead of getting wedged
// trying the old protocol first.
if let Ok(mut config) = super::load_config(&data_dir).await {
config.device_kind = None;
if let Err(e) = super::save_config(&data_dir, &config).await {
warn!(error = %e, "Failed to clear device_kind pin after flash");
}
}
if !succeeded {
// Deliberately do NOT auto-restart the listener here. A failed
// flash means we can't vouch for the board's state — reopening
// the port immediately (esptool/rnodeconf's own reset sequence
// plus our open() toggling DTR/RTS again right after) risks
// hammering a marginal device with reconnect attempts. Confirmed
// live 2026-07-23: exactly this sequence left a real Heltec V3
// boot-looping for 5+ minutes after a failed flash. Leave mesh
// stopped; the user reconnects explicitly via the hot-swap
// modal/Mesh page once they've confirmed the board is alive.
warn!(
path = %path,
"Leaving mesh listener stopped after failed flash — reconnect manually once the board is confirmed responsive"
);
return;
}
// On success, give the board a moment to finish booting after the
// flash tool's own reset sequence before we start hammering it with
// connection attempts — same reasoning as above, just the
// lower-risk (successful-flash) side of it.
tokio::time::sleep(POST_FLASH_SETTLE_DELAY).await;
let mut svc = bg_service.write().await;
if let Some(s) = svc.as_mut() {
match super::load_config(&data_dir).await {
Ok(config) => {
// Only resume if mesh is actually still enabled per the
// CURRENT persisted config — confirmed live 2026-07-23:
// unconditionally forcing a restart here, regardless of
// `enabled`, overrode a user's own concurrent "disable
// mesh" toggle and left the listener running while
// config said disabled. That inconsistent state is what
// made a later legitimate "Keep As Is" click (which
// correctly tries to start on a false→true transition)
// fail with "already running" — the listener had already
// been force-started behind the config's back.
let should_run = config.enabled;
if let Err(e) = s.configure(config).await {
warn!(error = %e, "Failed to resume mesh listener after flash");
}
if should_run {
if let Err(e) = s.start() {
warn!(error = %e, "Failed to restart mesh listener after flash");
}
}
}
Err(e) => warn!(error = %e, "Failed to load mesh config after flash"),
}
}
});
*job.abort_handle.write().await = Some(task.abort_handle());
Ok(())
}
async fn run_flash(
board: FlashBoard,
family: DeviceType,
data_dir: &Path,
path: &str,
job: &Arc<FlashJob>,
) -> Result<()> {
match family {
DeviceType::Meshtastic | DeviceType::Meshcore => {
let image = fetch_esptool_image(board, family, data_dir, job).await?;
esptool_erase_and_write(path, &image, job).await
}
DeviceType::Reticulum => {
let lora_region = super::load_config(data_dir)
.await
.ok()
.and_then(|c| c.lora_region);
rnodeconf_autoinstall(path, board, lora_region.as_deref(), job).await
}
DeviceType::Unknown => anyhow::bail!("Pick a firmware family before flashing"),
}
}
// ─── MeshCore / Meshtastic: esptool ─────────────────────────────────────
async fn fetch_esptool_image(
board: FlashBoard,
family: DeviceType,
data_dir: &Path,
job: &Arc<FlashJob>,
) -> Result<PathBuf> {
let cache = firmware_cache_dir(data_dir);
tokio::fs::create_dir_all(&cache)
.await
.context("Creating firmware cache dir")?;
let client = github_client()?;
match family {
DeviceType::Meshtastic => fetch_meshtastic_image(&client, board, &cache, job).await,
DeviceType::Meshcore => fetch_meshcore_image(&client, board, &cache, job).await,
_ => anyhow::bail!("{family} is not flashed via esptool"),
}
}
async fn fetch_meshtastic_image(
client: &reqwest::Client,
board: FlashBoard,
cache: &Path,
job: &Arc<FlashJob>,
) -> Result<PathBuf> {
let release: GithubRelease = client
.get("https://api.github.com/repos/meshtastic/firmware/releases/latest")
.timeout(METADATA_TIMEOUT)
.send()
.await
.context("Fetching Meshtastic release list")?
.error_for_status()
.context("Meshtastic releases API error")?
.json()
.await
.context("Parsing Meshtastic release JSON")?;
// Meshtastic bundles all esp32s3 boards' images inside one per-platform
// zip rather than shipping per-board top-level assets — both Heltec V3
// and V4 are esp32s3, so this is the right zip for both (confirmed live
// against v2.7.26.54e0d8d).
let zip_asset = release
.assets
.iter()
.find(|a| a.name.starts_with("firmware-esp32s3-") && a.name.ends_with(".zip"))
.ok_or_else(|| anyhow::anyhow!("No esp32s3 firmware zip in latest Meshtastic release"))?;
let version = zip_asset
.name
.strip_prefix("firmware-esp32s3-")
.and_then(|s| s.strip_suffix(".zip"))
.ok_or_else(|| anyhow::anyhow!("Unexpected Meshtastic asset name: {}", zip_asset.name))?
.to_string();
let zip_path = cache.join(&zip_asset.name);
if tokio::fs::metadata(&zip_path).await.is_err() {
download_to_file(client, &zip_asset.browser_download_url, &zip_path, job).await?;
} else {
job.push_log(format!("Using cached {}", zip_asset.name)).await;
}
// "*.factory.bin" is Meshtastic's full merged image (bootloader +
// partition table + app) meant to be written at offset 0x0 on a freshly
// erased chip — confirmed by inspecting the real zip's contents, as
// opposed to the plain "*.bin" OTA-update image which assumes an
// existing bootloader/partition table already on the chip.
let entry_name = format!(
"firmware-{}-{}.factory.bin",
board.meshtastic_id(),
version
);
let out_path = cache.join(&entry_name);
if tokio::fs::metadata(&out_path).await.is_ok() {
return Ok(out_path);
}
job.push_log(format!(
"Extracting {entry_name} from {}",
zip_asset.name
))
.await;
let zip_path_owned = zip_path.clone();
let entry_name_owned = entry_name.clone();
let out_path_owned = out_path.clone();
tokio::task::spawn_blocking(move || -> Result<()> {
let file = std::fs::File::open(&zip_path_owned).context("Opening downloaded firmware zip")?;
let mut archive = zip::ZipArchive::new(file).context("Reading firmware zip")?;
let mut entry = archive
.by_name(&entry_name_owned)
.with_context(|| format!("{entry_name_owned} not found in firmware zip"))?;
let mut out =
std::fs::File::create(&out_path_owned).context("Creating extracted firmware file")?;
std::io::copy(&mut entry, &mut out).context("Extracting firmware image")?;
Ok(())
})
.await
.context("Firmware extraction task panicked")??;
Ok(out_path)
}
async fn fetch_meshcore_image(
client: &reqwest::Client,
board: FlashBoard,
cache: &Path,
job: &Arc<FlashJob>,
) -> Result<PathBuf> {
let release: GithubRelease = client
.get("https://api.github.com/repos/meshcore-dev/MeshCore/releases/latest")
.timeout(METADATA_TIMEOUT)
.send()
.await
.context("Fetching MeshCore release list")?
.error_for_status()
.context("MeshCore releases API error")?
.json()
.await
.context("Parsing MeshCore release JSON")?;
// Upstream's casing differs between boards (Heltec_v3_... vs
// heltec_v4_...) — match case-insensitively on the exact per-board
// substring so V4 isn't accidentally matched by "heltec_v4_tft_..."
// variants (there's a "_tft_" in between, so a straight substring match
// on "heltec_v4_companion_radio_usb" is already safe).
let needle = match board {
FlashBoard::HeltecV3 => "heltec_v3_companion_radio_usb",
FlashBoard::HeltecV4 => "heltec_v4_companion_radio_usb",
};
let asset = release
.assets
.iter()
.find(|a| {
let lower = a.name.to_lowercase();
lower.contains(needle) && lower.ends_with("-merged.bin")
})
.ok_or_else(|| {
anyhow::anyhow!("No matching MeshCore image in release {}", release.tag_name)
})?;
let out_path = cache.join(&asset.name);
if tokio::fs::metadata(&out_path).await.is_ok() {
job.push_log(format!("Using cached {}", asset.name)).await;
return Ok(out_path);
}
download_to_file(client, &asset.browser_download_url, &out_path, job).await?;
Ok(out_path)
}
async fn download_to_file(
client: &reqwest::Client,
url: &str,
dest: &Path,
job: &Arc<FlashJob>,
) -> Result<()> {
job.set_stage(FlashStage::Downloading).await;
// Bound only the wait for the response to *start* (headers) — NOT a
// request-level `.timeout()`, which would cap the whole body transfer
// again (the bug this replaced: a blanket 30s client timeout killed
// large downloads mid-stream). If the server never responds at all,
// this is what stops the job from hanging forever; the per-chunk stall
// timeout below is what guards the body once streaming starts. Without
// this, a server that accepts the TCP connection but never sends
// headers back hangs this call indefinitely — confirmed live
// 2026-07-23: a stuck `.send()` here wedged the single-flash-job guard
// for good, permanently blocking every subsequent flash attempt with
// "already in progress" until the service was restarted.
let resp = tokio::time::timeout(METADATA_TIMEOUT, client.get(url).send())
.await
.context("Firmware download server did not respond")?
.context("Starting firmware download")?
.error_for_status()
.context("Firmware download returned an error status")?;
let total = resp.content_length();
let tmp = dest.with_extension("part");
let mut file = tokio::fs::File::create(&tmp)
.await
.context("Creating firmware download file")?;
let mut stream = resp.bytes_stream();
let mut downloaded: u64 = 0;
use futures_util::StreamExt;
loop {
let next = tokio::time::timeout(DOWNLOAD_STALL_TIMEOUT, stream.next())
.await
.context("Firmware download stalled")?;
let Some(chunk) = next else { break };
let chunk = chunk.context("Reading firmware download stream")?;
file.write_all(&chunk)
.await
.context("Writing firmware download")?;
downloaded += chunk.len() as u64;
if let Some(total) = total {
if total > 0 {
job.set_percent(((downloaded.saturating_mul(100)) / total) as u8)
.await;
}
}
}
file.flush().await.ok();
tokio::fs::rename(&tmp, dest)
.await
.context("Finalizing firmware download")?;
job.push_log(format!(
"Downloaded {} ({downloaded} bytes)",
dest.display()
))
.await;
Ok(())
}
/// Both Heltec V3 and V4 are ESP32-S3 boards.
const ESPTOOL_CHIP: &str = "esp32s3";
/// esptool's auto-reset-into-bootloader handshake (toggling DTR/RTS in a
/// specific timed pattern) is well-known to be flaky on some CP2102/CH340
/// board+adapter combinations — esptool's own docs recommend retrying at a
/// lower baud rate when this happens. Rather than fail the whole job on the
/// first hiccup, retry once at a conservative baud before giving up.
const ESPTOOL_FALLBACK_BAUD: &str = "115200";
/// `write_flash --erase-all` erases the whole chip before writing, in one
/// esptool invocation. This needs the esp32s3 stub flasher loaded (see
/// esptool_global_args' doc comment) — without it, --erase-all hits the
/// exact same ROM limitation a standalone `erase_flash` does ("ESP32-S3 ROM
/// does not support function erase_flash", confirmed live 2026-07-23), since
/// esptool's --erase-all is implemented as the same full-chip-erase command,
/// not a per-sector loop.
async fn esptool_erase_and_write(path: &str, image: &Path, job: &Arc<FlashJob>) -> Result<()> {
job.set_stage(FlashStage::Writing).await;
let image_str = image.to_string_lossy().to_string();
esptool_with_retry(
path,
&["write_flash", "--erase-all", "0x0", &image_str],
job,
)
.await
.context("esptool write_flash failed")?;
Ok(())
}
/// esptool's global flags (--chip/--port/--baud) MUST precede the subcommand
/// token (erase_flash/write_flash/...) — confirmed live 2026-07-23:
/// appending `--baud 115200` after the subcommand on the retry path
/// produced "esptool: error: unrecognized arguments: --baud 115200" every
/// time, so the fallback-baud retry never actually got a chance to run.
/// Building global args separately from subcommand args keeps this correct
/// by construction instead of relying on call-site ordering.
///
/// Normal stub-loader mode (no --no-stub) needs the esp32s3 stub flasher
/// blob at /usr/lib/python3/dist-packages/esptool/targets/stub_flasher/
/// stub_flasher_32s3.json — Debian's `esptool` package (4.7.0+dfsg-0.1)
/// ships without it (stripped for DFSG compliance: the prebuilt blob has no
/// buildable-from-source path Debian could verify), so scripts/self-update.sh
/// fetches the exact same file from the matching upstream esptool release
/// tag and installs it alongside the apt package (see the esptool install
/// step there). --no-stub (talk directly to the ROM bootloader, skip the
/// stub) was tried first and works for connecting, but the ROM bootloader
/// doesn't implement a full-chip-erase opcode at all — only the stub does —
/// so --no-stub broke our "always erase before write" default outright
/// rather than just being slower. Restoring the real stub file is the
/// correct fix, not routing around its absence.
fn esptool_global_args<'a>(path: &'a str, baud: Option<&'a str>) -> Vec<&'a str> {
let mut args = vec!["--chip", ESPTOOL_CHIP, "--port", path];
if let Some(b) = baud {
args.push("--baud");
args.push(b);
}
args
}
async fn esptool_with_retry(path: &str, subcommand: &[&str], job: &Arc<FlashJob>) -> Result<()> {
let mut cmd = Command::new("esptool");
cmd.args(esptool_global_args(path, None));
cmd.args(subcommand);
match run_streamed(cmd, None, job).await {
Ok(()) => Ok(()),
Err(first_err) => {
job.push_log(format!(
"First attempt failed ({first_err:#}); retrying once at {ESPTOOL_FALLBACK_BAUD} baud"
))
.await;
let mut retry = Command::new("esptool");
retry.args(esptool_global_args(path, Some(ESPTOOL_FALLBACK_BAUD)));
retry.args(subcommand);
run_streamed(retry, None, job)
.await
.context(format!("retry also failed (first attempt: {first_err:#})"))
}
}
}
// ─── Reticulum/RNode: archy-rnodeconf ───────────────────────────────────
fn rnodeconf_bin() -> String {
std::env::var("ARCHY_RNODECONF_BIN")
.unwrap_or_else(|_| "/usr/local/bin/archy-rnodeconf".to_string())
}
/// `--autoinstall`'s "which board is this" step is interactive by design —
/// confirmed live against a real Heltec V4 (2026-07-23): even with a board
/// given on the command line, rnodeconf can't always tell V3 from V4 apart
/// (their bootstrap-time USB identity is often generic, same root cause as
/// `resolve_flash_board`'s doc comment), so it always asks. The full prompt
/// sequence observed for a Heltec board that already has *some* RNode
/// firmware installed (the common case — a truly blank chip likely skips
/// straight to the same "Device Selection" menu):
/// 1. numbered device-type menu → answer with the menu number
/// 2. "Hit enter to continue" → answer with a blank line
/// 3. numbered band menu → answer with the menu number
/// 4. "Is the above correct? [y/N]" → answer "y"
/// Feeding all four answers up front (rather than watching stdout for each
/// prompt text) works because the menu is always asked in this fixed order
/// for every board that needs (re)provisioning — verified by driving it
/// through an unprovisioned real V4 end-to-end (erase → flash → EEPROM
/// bootstrap → "Device signature validated" on the next probe).
fn rnodeconf_device_menu_number(board: FlashBoard) -> &'static str {
match board {
FlashBoard::HeltecV3 => "8",
FlashBoard::HeltecV4 => "9",
}
}
/// rnodeconf's band choice is a coarse RF-frontend bootstrap parameter
/// (868/915/923 MHz), not the final operating frequency — that's still
/// configured later via the daemon's interface config, same as today. This
/// is a best-effort mapping from the node's persisted Meshtastic-style
/// region code (see `mesh::meshtastic::region_name_to_code`) down to
/// rnodeconf's 3-way menu; regions with no exact 868/923 match fall back to
/// 915 MHz as the broadest-compatibility default.
fn rnodeconf_band_menu_number(lora_region: Option<&str>) -> &'static str {
match lora_region.map(|s| s.trim().to_uppercase()) {
Some(r) if r.contains("868") => "1",
Some(r) if r.contains("923") => "3",
_ => "2",
}
}
/// `--autoinstall` fetches, erases, flashes, and bootstraps the EEPROM for
/// a detected board as one atomic step (confirmed via `archy-rnodeconf
/// --help` AND a real end-to-end flash on real hardware) — this is the
/// RNode-side equivalent of our "always erase before write" default, since
/// autoinstall doesn't try to preserve any existing on-device state.
async fn rnodeconf_autoinstall(
path: &str,
board: FlashBoard,
lora_region: Option<&str>,
job: &Arc<FlashJob>,
) -> Result<()> {
job.set_stage(FlashStage::Autoinstalling).await;
let bin = rnodeconf_bin();
let mut cmd = if Path::new(&bin).exists() {
Command::new(bin)
} else {
// Dev fallback if only a plain venv/system rnodeconf is on PATH.
Command::new("rnodeconf")
};
cmd.args(["--autoinstall", path]);
let stdin = format!(
"{}\n\n{}\ny\n",
rnodeconf_device_menu_number(board),
rnodeconf_band_menu_number(lora_region)
);
run_streamed(cmd, Some(stdin.into_bytes()), job)
.await
.context("archy-rnodeconf --autoinstall failed")
}
// ─── Subprocess streaming ────────────────────────────────────────────────
fn percent_regex() -> &'static Regex {
static RE: OnceLock<Regex> = OnceLock::new();
RE.get_or_init(|| Regex::new(r"\((\d{1,3})\s*%\)").expect("valid regex"))
}
async fn run_streamed(mut cmd: Command, stdin: Option<Vec<u8>>, job: &Arc<FlashJob>) -> Result<()> {
cmd.stdout(Stdio::piped());
cmd.stderr(Stdio::piped());
if stdin.is_some() {
cmd.stdin(Stdio::piped());
}
// Deliberately NOT kill_on_drop: an interrupted erase/write can leave
// the chip in a worse state than either finished or unstarted (see the
// cancellation-safety note in mesh flashing docs). The job is expected
// to run to completion or fail on its own.
let mut child = cmd.spawn().context("Failed to start subprocess")?;
if let Some(bytes) = stdin {
if let Some(mut child_stdin) = child.stdin.take() {
child_stdin
.write_all(&bytes)
.await
.context("Writing to subprocess stdin")?;
}
}
let mut tasks = Vec::new();
if let Some(stdout) = child.stdout.take() {
let job = Arc::clone(job);
tasks.push(tokio::spawn(async move {
let mut lines = BufReader::new(stdout).lines();
while let Ok(Some(line)) = lines.next_line().await {
if let Some(cap) = percent_regex().captures(&line) {
if let Ok(pct) = cap[1].parse::<u8>() {
job.set_percent(pct).await;
}
}
job.push_log(line).await;
}
}));
}
if let Some(stderr) = child.stderr.take() {
let job = Arc::clone(job);
tasks.push(tokio::spawn(async move {
let mut lines = BufReader::new(stderr).lines();
while let Ok(Some(line)) = lines.next_line().await {
job.push_log(line).await;
}
}));
}
let status = child.wait().await.context("Waiting for subprocess")?;
for t in tasks {
let _ = t.await;
}
if !status.success() {
// Exit status alone isn't diagnosable — the actual esptool/rnodeconf
// stderr (already captured into job.log_tail by the reader tasks
// above) is what actually explains a failure. Confirmed live
// 2026-07-23: a bare "Command exited with exit status: 1" told us
// nothing when esptool's real error was sitting in the log tail the
// whole time, only visible via the UI's live poll, not journald.
let tail: Vec<String> = job
.snapshot()
.await
.log_tail
.iter()
.rev()
.take(10)
.rev()
.cloned()
.collect();
anyhow::bail!("Command exited with {status}\n{}", tail.join("\n"));
}
Ok(())
}
+62 -5
View File
@@ -87,6 +87,18 @@ const RECONNECT_DELAY_INIT: Duration = Duration::from_secs(5);
/// Maximum reconnect delay (cap for exponential backoff).
const RECONNECT_DELAY_MAX: Duration = Duration::from_secs(60);
/// Minimum time a session must run before we trust it enough to reset
/// backoff to the minimum. Without this gate, a device that connects then
/// fails again within a couple of seconds (e.g. mid-boot-loop) never backs
/// off — every retry immediately re-opens the port, which toggles DTR/RTS
/// (resets many ESP32 boards' MCU on native-USB and CP2102/CH340
/// auto-reset-circuit boards alike), turning a device that's merely
/// unstable into a self-sustaining boot loop that outlasts whatever
/// triggered the original instability. Confirmed live 2026-07-23: a Heltec
/// V3 stuck retrying every ~5-15s for 5+ minutes after a failed firmware
/// flash left it in a marginal state.
const STABLE_SESSION_THRESHOLD: Duration = Duration::from_secs(20);
/// Number of consecutive write failures before we consider the device dead
/// and trigger a reconnection cycle.
const MAX_CONSECUTIVE_WRITE_FAILURES: u32 = 3;
@@ -535,6 +547,10 @@ pub fn spawn_mesh_listener(
let mut shutdown = shutdown;
let mut cmd_rx = cmd_rx;
let mut reconnect_delay = RECONNECT_DELAY_INIT;
// Mutable so a successful auto-detect can pin the firmware kind for
// the rest of this listener's lifetime — see the pin-on-first-success
// block below for why.
let mut device_kind = device_kind;
// Backlog #12 hot-swap re-binding: each run_mesh_session call already
// builds a fresh device struct (contacts/current_region/etc. all
// start empty), so per-device session state is naturally isolated
@@ -550,6 +566,7 @@ pub fn spawn_mesh_listener(
return;
}
let session_start = std::time::Instant::now();
match session::run_mesh_session(
&state,
&data_dir,
@@ -572,13 +589,14 @@ pub fn spawn_mesh_listener(
{
Ok(()) => {
info!("Mesh session ended cleanly");
// Session was established before ending — reset backoff
reconnect_delay = RECONNECT_DELAY_INIT;
// Only trust a session that actually ran for a while —
// see STABLE_SESSION_THRESHOLD's doc comment.
if session_start.elapsed() >= STABLE_SESSION_THRESHOLD {
reconnect_delay = RECONNECT_DELAY_INIT;
}
}
Err(e) => {
// Check if session was ever connected (vs failed to open)
let was_connected = state.status.read().await.device_connected;
if was_connected {
if session_start.elapsed() >= STABLE_SESSION_THRESHOLD {
reconnect_delay = RECONNECT_DELAY_INIT;
}
error!("Mesh session error: {} (retry in {:?})", e, reconnect_delay);
@@ -604,6 +622,45 @@ pub fn spawn_mesh_listener(
}
}
// Pin the firmware kind after the first successful auto-detect.
// Confirmed live 2026-07-23: with device_kind left unpinned (e.g.
// after clearing a stale pin), EVERY reconnect re-runs the full
// Reticulum→Meshcore→Meshtastic auto-detect cascade — each
// candidate past the first does its own open() with the DTR/RTS
// reset both boards need, so a device correctly identified as
// Meshtastic still gets reset once for the failed Meshcore
// attempt before Meshtastic's own open() resets it again. That
// doubled the reset count on every single reconnect indefinitely,
// not just during initial detection. Once auto-detect has
// identified the device this listener is actually talking to,
// there's no reason to keep guessing on subsequent reconnects —
// pin it, both in this task's own loop (takes effect
// immediately) and on disk (survives a service restart). A
// genuine hot-swap to different firmware is still handled: the
// setup modal's `mesh.probe-device` always re-probes unpinned,
// and the flash flow already clears this pin on its own.
if device_kind.is_none() {
let detected = state.status.read().await.device_type;
if detected != super::types::DeviceType::Unknown {
device_kind = Some(detected);
match super::load_config(&data_dir).await {
Ok(mut cfg) if cfg.device_kind.is_none() => {
cfg.device_kind = Some(detected);
if let Err(e) = super::save_config(&data_dir, &cfg).await {
warn!("Failed to persist auto-detected device_kind: {}", e);
} else {
info!(
kind = %detected,
"Pinned auto-detected firmware kind to avoid repeated multi-protocol resets on reconnect"
);
}
}
Ok(_) => {}
Err(e) => warn!("Failed to load mesh config to persist device_kind: {}", e),
}
}
}
// Update status to disconnected. device_type/firmware_version are
// reset too — they were previously left holding the LAST radio's
// identity, so after a hot-swap the UI showed the old firmware
+14 -34
View File
@@ -274,6 +274,7 @@ async fn auto_detect_and_open(
if paths.is_empty() {
anyhow::bail!("No serial devices found in /dev");
}
info!(candidates = ?paths, "Auto-detect candidate ports for this attempt");
for path in &paths {
debug!(path = %path, "Probing for mesh radio device");
// Tried FIRST: `ReticulumLink::open()` gates its expensive daemon
@@ -487,40 +488,19 @@ async fn open_preferred_path(
};
}
// Reticulum first — see the matching comment on auto_detect_and_open:
// its cheap probe_rnode gate fails in ~1s for non-RNode firmware, while
// trying Meshcore/Meshtastic first was observed leaving a real RNode
// board unresponsive by the time Reticulum's turn came.
match ReticulumLink::open(
path,
data_dir,
Some(our_ed_pubkey_hex),
Some(our_x25519_pubkey_hex),
)
.await
{
Ok(mut dev) => match dev.initialize().await {
Ok(info) => return Ok((MeshRadioDevice::Reticulum(dev), info)),
Err(e) => {
debug!(path = %path, error = %e, "Preferred path is not a working Reticulum RNode")
}
},
Err(e) => debug!(path = %path, error = %e, "Could not open preferred path as Reticulum"),
}
match MeshcoreDevice::open(path).await {
Ok(mut dev) => match dev.initialize().await {
Ok(info) => return Ok((MeshRadioDevice::Meshcore(dev), info)),
Err(e) => debug!(path = %path, error = %e, "Preferred path is not Meshcore"),
},
Err(e) => debug!(path = %path, error = %e, "Could not open preferred path as Meshcore"),
}
match MeshtasticDevice::open(path).await {
Ok(mut dev) => match dev.initialize().await {
Ok(info) => Ok((MeshRadioDevice::Meshtastic(dev), info)),
Err(e) => Err(e).context("Preferred path is not a working Meshtastic device"),
},
Err(e) => Err(e).context("Could not open preferred path as Meshtastic"),
}
// Unpinned: don't probe this path ourselves at all. Confirmed live
// 2026-07-23 — this function used to run its own Reticulum→Meshcore→
// Meshtastic sequence here, and the caller (run_mesh_session) falls
// back to `auto_detect_and_open` on any error, which scans every
// candidate path (this one included) with the exact same three-protocol
// sequence. With a single physical radio — the overwhelmingly common
// case — `path` here IS the one candidate `auto_detect_and_open` is
// about to try, so every unpinned reconnect was resetting the board via
// Reticulum/Meshcore/Meshtastic's DTR/RTS toggle TWICE: once here, once
// again moments later in auto-detect. Bailing immediately (no port
// access at all) means auto-detect's single pass is the only one that
// ever touches the port when nothing is pinned yet.
anyhow::bail!("No device_kind pin — deferring to auto-detect for {path}")
}
/// Bring up a Reticulum daemon over plain TCP — no physical RNode, no
+12 -3
View File
@@ -214,11 +214,20 @@ impl MeshtasticDevice {
path
))?;
// See probe_rnode() in reticulum.rs for why: ESP32-S3 native-USB
// boards reset on a DTR/RTS transition, so deassert both and settle
// before the handshake below.
// boards (and CP2102/CH340-bridged boards wired for Arduino-style
// auto-reset) reset on a DTR/RTS transition, so deassert both and
// settle before the handshake below. 300ms is nowhere near a real
// firmware boot time (LoRa radio init alone can take longer) —
// confirmed live 2026-07-23: with every one of Reticulum/Meshcore/
// Meshtastic's open() doing this same reset, a single auto-detect
// cycle trying multiple protocols in sequence kept re-resetting the
// board before it ever finished booting from the PREVIOUS attempt's
// reset, on both a Heltec V3 and V4, regardless of firmware family —
// a self-sustaining "never finishes booting" loop with a boot-time
// root cause hiding behind what looked like a per-protocol failure.
let _ = port.set_dtr(false);
let _ = port.set_rts(false);
tokio::time::sleep(Duration::from_millis(300)).await;
tokio::time::sleep(Duration::from_millis(2000)).await;
info!(path = %path, baud = BAUD_RATE, "Opened Meshtastic serial port");
Ok(Self {
+55 -12
View File
@@ -8,6 +8,7 @@
pub mod alerts;
pub mod bitcoin_relay;
pub mod crypto;
pub mod flash;
pub mod listener;
pub mod meshtastic;
pub mod message_types;
@@ -38,6 +39,14 @@ use tokio::sync::watch;
use tracing::{error, info, warn};
const MESH_CONFIG_FILE: &str = "mesh-config.json";
/// How long `MeshService::stop()` waits for the listener task to notice its
/// shutdown signal and exit gracefully before force-aborting it. See
/// `stop()`'s doc comment for the real incident this guards against: without
/// a hard abort fallback, a slow-to-notice listener could be left running
/// forever, orphaned, racing a later independently-started listener on the
/// same serial port.
const LISTENER_SHUTDOWN_TIMEOUT: Duration = Duration::from_secs(15);
const MESH_IGNORED_RADIO_FILE: &str = "mesh-ignored-radio-contacts.json";
const MESH_CONTACTS_FILE: &str = "mesh-contacts.json";
@@ -734,10 +743,18 @@ impl MeshService {
self.server_name = name;
}
/// Start the background mesh listener.
/// Start the background mesh listener. Idempotent: if the listener is
/// already running, this is a harmless no-op rather than an error —
/// confirmed live 2026-07-23, a real race between the flash job's own
/// post-flash restart and a concurrent user "Keep As Is" click (both
/// legitimately trying to ensure the listener is running) surfaced this
/// as a user-facing "Mesh listener already running" RPC error. Ensuring
/// the listener is running is the intent every caller actually has;
/// whichever caller's start() happens to win the race, the other
/// finding it already satisfied is success, not failure.
pub fn start(&mut self) -> Result<()> {
if self.listener_handle.is_some() {
anyhow::bail!("Mesh listener already running");
return Ok(());
}
let (shutdown_tx, shutdown_rx) = watch::channel(false);
@@ -967,8 +984,36 @@ impl MeshService {
if let Some(tx) = self.shutdown_tx.take() {
let _ = tx.send(true);
}
if let Some(handle) = self.listener_handle.take() {
let _ = handle.await;
if let Some(mut handle) = self.listener_handle.take() {
// Bounded wait for graceful shutdown, with a hard abort as
// fallback — confirmed live 2026-07-23: a caller-side timeout
// wrapping stop() (mesh::flash's STOP_LISTENER_TIMEOUT) cancelled
// this await when the listener was slow to notice its shutdown
// signal (mid multi-candidate probe), but `.take()` above had
// already cleared `listener_handle` to None — so MeshService
// believed it was stopped while the task kept running, orphaned
// (dropping a JoinHandle does not abort the task it points to).
// A later start() then spawned a second, fully independent
// listener session racing the orphaned one on the same serial
// port — neither could ever get a clean response, so every
// mesh.configure/probe against that device failed indefinitely
// even though the device itself was fine.
//
// Awaiting `&mut handle` (not `handle` by value) is what makes
// the fallback possible: the Future is polled through the
// reference, so if the timeout fires, this task's own `handle`
// binding is still ours to call `.abort()` on afterward —
// unlike moving `handle` into the timeout future outright, which
// would drop (and thus orphan) it on timeout with nothing left
// to abort.
if tokio::time::timeout(LISTENER_SHUTDOWN_TIMEOUT, &mut handle)
.await
.is_err()
{
warn!("Mesh listener did not shut down gracefully in time — aborting it");
handle.abort();
let _ = handle.await;
}
}
if let Some(handle) = self.deadman_handle.take() {
handle.abort();
@@ -1019,19 +1064,17 @@ impl MeshService {
self.state.peers.read().await.values().cloned().collect()
}
/// Probe a serial port for a mesh radio without provisioning or keeping
/// it — powers the hot-swap "device detected" modal's current-details
/// view. Refuses to probe the port the live session currently occupies
/// (the probe would steal the serial port from under the session); a
/// Refuse to probe the port the live session currently occupies (the
/// probe would steal the serial port from under the session); a
/// detected-but-not-connected port is fair game, accepting a benign race
/// with the reconnect loop (whichever loses just retries).
pub async fn probe_device(&self, path: &str) -> Result<listener::DeviceProbe> {
/// with the reconnect loop (whichever loses just retries). Split out from
/// the actual probe on purpose — see `probe_device`'s doc comment.
pub async fn ensure_probe_allowed(&self, path: &str) -> Result<()> {
let status = self.state.status.read().await;
if status.device_connected && status.device_path.as_deref() == Some(path) {
anyhow::bail!("{path} is the active mesh radio — already connected");
}
drop(status);
listener::probe_device(path).await
Ok(())
}
/// Get message history.
+36 -3
View File
@@ -58,11 +58,20 @@ impl MeshcoreDevice {
path
))?;
// See probe_rnode() in reticulum.rs for why: ESP32-S3 native-USB
// boards reset on a DTR/RTS transition, so deassert both and settle
// before the handshake below.
// boards (and CP2102/CH340-bridged boards wired for Arduino-style
// auto-reset) reset on a DTR/RTS transition, so deassert both and
// settle before the handshake below. 300ms is nowhere near a real
// firmware boot time (LoRa radio init alone can take longer) —
// confirmed live 2026-07-23: with every one of Reticulum/Meshcore/
// Meshtastic's open() doing this same reset, a single auto-detect
// cycle trying multiple protocols in sequence kept re-resetting the
// board before it ever finished booting from the PREVIOUS attempt's
// reset, on both a Heltec V3 and V4, regardless of firmware family —
// a self-sustaining "never finishes booting" loop with a boot-time
// root cause hiding behind what looked like a per-protocol failure.
let _ = port.set_dtr(false);
let _ = port.set_rts(false);
tokio::time::sleep(Duration::from_millis(300)).await;
tokio::time::sleep(Duration::from_millis(2000)).await;
info!(path = %path, baud = BAUD_RATE, "Opened serial port");
@@ -547,14 +556,38 @@ fn likely_non_mesh_serial_device(path: &str) -> bool {
/// Scan for serial devices that could be Meshcore radios.
/// Returns paths to existing serial device files.
///
/// Dedupes by canonical (symlink-resolved) target: `/dev/mesh-radio` is a
/// stable udev symlink to whatever `/dev/ttyUSB*`/`/dev/ttyACM*` node the
/// primary radio currently enumerates as, so both names always pointed at
/// the same candidate list entry and both passed this scan — confirmed live
/// 2026-07-23, this made an already-connected, working radio (connected via
/// its `/dev/mesh-radio` alias) simultaneously appear as a second, separate
/// "detected but unclaimed" device under its raw `/dev/ttyUSBn` name. The
/// hot-swap UI's active-session guard compares path strings, so it didn't
/// recognize the two aliases as the same port, showed the "device detected"
/// modal for a radio that was already set up, and probing it there opened
/// (and DTR/RTS-reset) the exact port the live session was mid-conversation
/// with — a continuous, UI-driven reset loop that only ran while that view
/// was open (matches the reported "stops when I leave, resumes when I come
/// back"). SERIAL_CANDIDATES lists `/dev/mesh-radio` first, so it wins the
/// dedup and is what's reported when both alias and target are present.
pub async fn detect_serial_devices() -> Vec<String> {
let mut devices = Vec::new();
let mut seen_real_paths = std::collections::HashSet::new();
for path in SERIAL_CANDIDATES {
if tokio::fs::metadata(path).await.is_ok() {
if likely_non_mesh_serial_device(path) {
debug!(path = %path, "Skipping known non-mesh serial device");
continue;
}
let real_path = tokio::fs::canonicalize(path)
.await
.unwrap_or_else(|_| std::path::PathBuf::from(path));
if !seen_real_paths.insert(real_path.clone()) {
debug!(path = %path, real_path = %real_path.display(), "Skipping duplicate alias for an already-listed device");
continue;
}
devices.push(path.to_string());
}
}
+145
View File
@@ -0,0 +1,145 @@
//! Mirror IPv4-published app ports onto IPv6 for mesh access.
//!
//! The companion (and anything else on the FIPS mesh) reaches this node at
//! its fips0 ULA. The web UI builds app links from the current host + each
//! app's DIRECT port (Direct Port Rule) — but rootless-podman published
//! ports bind 0.0.0.0 only, so `http://[<ULA>]:8334` was refused for every
//! catalog app even after nginx :80 learned IPv6 (2026-07-23, third
//! "the node wasn't listening on v6" bug of the night).
//!
//! Instead of touching the container layer (port-publish changes would
//! recreate every container), a reconcile loop mirrors the host's public
//! IPv4 listeners: any port ≥1024 listening on 0.0.0.0 without an IPv6
//! any-address listener gets a v6-only `[::]:<port>` forwarder to
//! `127.0.0.1:<port>`. Ports appear/disappear with app install/remove, so
//! the mirror follows `/proc/net/tcp*` rather than the catalog — companion
//! containers with hardcoded ports (bitcoin-ui :8334) are covered the same
//! as manifest-declared ones. Nothing is exposed that IPv4 didn't already
//! expose to the LAN; the ULA is the established remote-access surface.
use std::collections::{HashMap, HashSet};
use std::net::{Ipv6Addr, SocketAddrV6};
use std::time::Duration;
use anyhow::{Context, Result};
use tokio::task::JoinHandle;
use tracing::{debug, info, warn};
const RECONCILE_INTERVAL: Duration = Duration::from_secs(15);
/// Never mirror system ports; nginx (80/443) handles its own v6 listeners.
const MIN_PORT: u16 = 1024;
/// Entry point — spawned from main; runs for the process lifetime.
pub async fn run_mesh_port_mirror() {
let mut active: HashMap<u16, JoinHandle<()>> = HashMap::new();
loop {
match reconcile(&mut active).await {
Ok(()) => {}
Err(e) => debug!("mesh-ports: reconcile skipped: {e:#}"),
}
tokio::time::sleep(RECONCILE_INTERVAL).await;
}
}
async fn reconcile(active: &mut HashMap<u16, JoinHandle<()>>) -> Result<()> {
let v4 = listening_ports("/proc/net/tcp", 8).await?;
let v6 = listening_ports("/proc/net/tcp6", 32).await?;
// Drop mirrors whose backing IPv4 listener vanished (app removed/stopped).
active.retain(|port, handle| {
if v4.contains(port) && !handle.is_finished() {
true
} else {
handle.abort();
info!("mesh-ports: released [::]:{port} (backing 0.0.0.0 listener gone)");
false
}
});
for port in v4 {
if port < MIN_PORT || active.contains_key(&port) {
continue;
}
// A foreign IPv6 any-listener already serves this port (our own
// mirrors are excluded above via `active`).
if v6.contains(&port) {
continue;
}
match spawn_forwarder(port) {
Ok(handle) => {
info!("mesh-ports: mirroring [::]:{port} -> 127.0.0.1:{port}");
active.insert(port, handle);
}
Err(e) => debug!("mesh-ports: cannot mirror port {port}: {e:#}"),
}
}
Ok(())
}
/// Ports in LISTEN state bound to the any-address, parsed from /proc/net/tcp
/// (`addr_hex_len` 8) or /proc/net/tcp6 (32).
async fn listening_ports(path: &str, addr_hex_len: usize) -> Result<HashSet<u16>> {
let raw = tokio::fs::read_to_string(path)
.await
.with_context(|| format!("read {path}"))?;
let any_addr = "0".repeat(addr_hex_len);
let mut ports = HashSet::new();
for line in raw.lines().skip(1) {
let mut cols = line.split_whitespace();
let (Some(_sl), Some(local), Some(_rem), Some(state)) =
(cols.next(), cols.next(), cols.next(), cols.next())
else {
continue;
};
if state != "0A" {
continue; // not LISTEN
}
let Some((addr, port_hex)) = local.split_once(':') else {
continue;
};
if addr != any_addr {
continue;
}
if let Ok(port) = u16::from_str_radix(port_hex, 16) {
ports.insert(port);
}
}
Ok(ports)
}
/// A v6-only listener on [::]:port forwarding each connection to 127.0.0.1:port.
fn spawn_forwarder(port: u16) -> Result<JoinHandle<()>> {
use socket2::{Domain, Protocol, Socket, Type};
let socket = Socket::new(Domain::IPV6, Type::STREAM, Some(Protocol::TCP))
.context("create v6 socket")?;
// v6only so we coexist with the app's own 0.0.0.0:<port> bind.
socket.set_only_v6(true).context("set v6only")?;
socket.set_reuse_address(true).ok();
socket.set_nonblocking(true).context("set nonblocking")?;
let addr = SocketAddrV6::new(Ipv6Addr::UNSPECIFIED, port, 0, 0);
socket.bind(&addr.into()).context("bind [::]")?;
socket.listen(128).context("listen")?;
let listener = tokio::net::TcpListener::from_std(socket.into())
.context("register with tokio")?;
Ok(tokio::spawn(async move {
loop {
let (mut inbound, _peer) = match listener.accept().await {
Ok(conn) => conn,
Err(e) => {
warn!("mesh-ports: accept failed on [::]:{port}: {e}");
tokio::time::sleep(Duration::from_millis(500)).await;
continue;
}
};
tokio::spawn(async move {
match tokio::net::TcpStream::connect(("127.0.0.1", port)).await {
Ok(mut upstream) => {
let _ = tokio::io::copy_bidirectional(&mut inbound, &mut upstream).await;
}
Err(e) => debug!("mesh-ports: upstream 127.0.0.1:{port} refused: {e}"),
}
});
}
}))
}
+113 -39
View File
@@ -97,42 +97,57 @@ async fn get_wan_ip() -> Option<String> {
}
/// Check if UPnP is available by attempting SSDP discovery.
///
/// The socket I/O here is plain blocking `std::net` (its 3s read timeout is
/// enforced by the OS, not by yielding to the async runtime), so it must run
/// on the blocking-pool via `spawn_blocking` — inlined into this "async fn"
/// directly, it used to occupy a tokio worker thread for the full 3s on
/// every call. With only as many worker threads as CPU cores, a handful of
/// concurrent `network.diagnostics` calls (e.g. several Server-settings page
/// loads) could starve the whole runtime and freeze every other in-flight
/// request — root cause of a full-node outage (2026-07-24) that had nothing
/// to do with connection limits and everything to do with blocking sockets
/// on async worker threads.
async fn check_upnp_available() -> bool {
use std::net::UdpSocket;
tokio::task::spawn_blocking(|| {
use std::net::UdpSocket;
let ssdp_request = "M-SEARCH * HTTP/1.1\r\n\
HOST: 239.255.255.250:1900\r\n\
MAN: \"ssdp:discover\"\r\n\
MX: 2\r\n\
ST: urn:schemas-upnp-org:device:InternetGatewayDevice:1\r\n\r\n";
let ssdp_request = "M-SEARCH * HTTP/1.1\r\n\
HOST: 239.255.255.250:1900\r\n\
MAN: \"ssdp:discover\"\r\n\
MX: 2\r\n\
ST: urn:schemas-upnp-org:device:InternetGatewayDevice:1\r\n\r\n";
let socket = match UdpSocket::bind("0.0.0.0:0") {
Ok(s) => s,
Err(_) => return false,
};
let socket = match UdpSocket::bind("0.0.0.0:0") {
Ok(s) => s,
Err(_) => return false,
};
if socket
.set_read_timeout(Some(std::time::Duration::from_secs(3)))
.is_err()
{
return false;
}
if socket
.send_to(ssdp_request.as_bytes(), "239.255.255.250:1900")
.is_err()
{
return false;
}
let mut buf = [0u8; 2048];
match socket.recv_from(&mut buf) {
Ok((len, _)) => {
let response = String::from_utf8_lossy(&buf[..len]);
response.contains("InternetGatewayDevice") || response.contains("200 OK")
if socket
.set_read_timeout(Some(std::time::Duration::from_secs(3)))
.is_err()
{
return false;
}
Err(_) => false,
}
if socket
.send_to(ssdp_request.as_bytes(), "239.255.255.250:1900")
.is_err()
{
return false;
}
let mut buf = [0u8; 2048];
match socket.recv_from(&mut buf) {
Ok((len, _)) => {
let response = String::from_utf8_lossy(&buf[..len]);
response.contains("InternetGatewayDevice") || response.contains("200 OK")
}
Err(_) => false,
}
})
.await
.unwrap_or(false)
}
/// Add a port forward (stored locally; actual UPnP mapping done on request).
@@ -281,19 +296,41 @@ pub async fn run_diagnostics() -> Result<NetworkDiagnostics> {
}
/// Check if Tor SOCKS proxy is reachable.
///
/// `TcpStream::connect_timeout` blocks the calling OS thread for up to its
/// timeout — same blocking-on-a-worker-thread hazard as `check_upnp_available`
/// above, so this also runs on the blocking pool.
async fn check_tor_connectivity() -> bool {
use std::net::TcpStream;
TcpStream::connect_timeout(
&"127.0.0.1:9050".parse().unwrap(),
std::time::Duration::from_secs(2),
)
.is_ok()
tokio::task::spawn_blocking(|| {
use std::net::TcpStream;
TcpStream::connect_timeout(
&"127.0.0.1:9050".parse().unwrap(),
std::time::Duration::from_secs(2),
)
.is_ok()
})
.await
.unwrap_or(false)
}
/// Check DNS resolution works.
///
/// `to_socket_addrs()` is a blocking libc resolver call with no timeout of
/// its own — on a network with a slow/unreachable DNS server (e.g. right
/// after relocating to a new network) it can hang far longer than the other
/// checks here. Runs on the blocking pool (same reason as the checks above)
/// AND under an explicit timeout, since unlike UPnP/Tor there's no built-in
/// bound to rely on.
async fn check_dns() -> bool {
use std::net::ToSocketAddrs;
"cloudflare.com:443".to_socket_addrs().is_ok()
let lookup = tokio::task::spawn_blocking(|| {
use std::net::ToSocketAddrs;
"cloudflare.com:443".to_socket_addrs().is_ok()
});
tokio::time::timeout(std::time::Duration::from_secs(5), lookup)
.await
.ok()
.and_then(|r| r.ok())
.unwrap_or(false)
}
// --- Router Compatibility Abstraction ---
@@ -469,3 +506,40 @@ pub async fn get_router_info(data_dir: &Path) -> Result<serde_json::Value> {
},
}))
}
#[cfg(test)]
mod blocking_io_tests {
use super::*;
// Regression test for the 2026-07-24 outage: check_upnp_available,
// check_tor_connectivity, and check_dns each did blocking std::net I/O
// directly on their calling task instead of via spawn_blocking. On a
// small worker pool (4 threads in production), a handful of concurrent
// network.diagnostics calls tied up every worker thread for seconds,
// freezing every other in-flight RPC request. Proves a cheap task
// spawned alongside these checks still gets scheduled promptly, which
// only holds if the checks aren't monopolizing worker threads.
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn network_checks_do_not_starve_other_tasks() {
let cheap = tokio::spawn(async {
tokio::time::sleep(std::time::Duration::from_millis(5)).await;
std::time::Instant::now()
});
let _ = tokio::join!(
check_upnp_available(),
check_tor_connectivity(),
check_dns()
);
// The assertion is that `cheap` — spawned before the blocking checks
// and sleeping only 5ms — finishes within 1s of being spawned. If the
// checks were still blocking worker threads directly, a 2-worker
// runtime running 3 blocking checks concurrently would starve this
// task well past 1s.
tokio::time::timeout(std::time::Duration::from_secs(1), cheap)
.await
.expect("a concurrently-spawned cheap task must not be starved by blocking network checks")
.expect("cheap task panicked");
}
}
+47 -5
View File
@@ -976,13 +976,21 @@ impl Server {
main_addr: SocketAddr,
shutdown: impl std::future::Future<Output = ()>,
) -> Result<()> {
let active_connections = Arc::new(tokio::sync::Semaphore::new(1024));
// Separate pools per listener. Federation/peer connections (fips0,
// often over Tor, from nodes we don't control the behavior of) used
// to share one pool with the local web UI listener — when peer
// connections piled up in CLOSE-WAIT without ever completing, they
// starved the shared pool and took the web UI down with them
// (production outage, 2026-07-24). Peer congestion must never be
// able to block a local login.
let main_connections = Arc::new(tokio::sync::Semaphore::new(1024));
let peer_connections = Arc::new(tokio::sync::Semaphore::new(256));
let (tx, rx_main) = tokio::sync::watch::channel(false);
let main_task = tokio::spawn(accept_loop(
self.api_handler.clone(),
TcpListener::bind(main_addr).await?,
active_connections.clone(),
main_connections.clone(),
false, // main listener: no path filter
rx_main,
main_addr,
@@ -992,7 +1000,7 @@ impl Server {
// restart when fips0 comes up after onboarding.
let peer_task = tokio::spawn(peer_late_bind_loop(
self.api_handler.clone(),
active_connections.clone(),
peer_connections.clone(),
tx.subscribe(),
));
@@ -1003,7 +1011,9 @@ impl Server {
// Wait up to 5s for in-flight requests.
let drain_start = std::time::Instant::now();
let drain_timeout = std::time::Duration::from_secs(5);
while active_connections.available_permits() < 1024 {
while main_connections.available_permits() < 1024
|| peer_connections.available_permits() < 256
{
if drain_start.elapsed() > drain_timeout {
warn!("Drain timeout reached, forcing shutdown");
break;
@@ -1096,6 +1106,11 @@ pub fn is_peer_allowed_path(path: &str) -> bool {
|| path.starts_with("/content/")
}
/// How long a freshly-accepted connection will wait for a connection-pool
/// permit before it's dropped. Bounds worst-case fd/task growth if the pool
/// is ever genuinely saturated; under normal load this never triggers.
const PERMIT_ACQUIRE_TIMEOUT: Duration = Duration::from_secs(30);
async fn accept_loop(
handler: Arc<ApiHandler>,
listener: TcpListener,
@@ -1115,8 +1130,35 @@ async fn accept_loop(
}
};
let handler = handler.clone();
let permit = active_connections.clone().acquire_owned().await;
let active_connections = active_connections.clone();
// Acquire the permit *inside* the spawned task, not here.
// This loop must never block on anything but accept()/shutdown:
// the main (5678) and FIPS peer (5679) listeners share one
// semaphore, and a single slow/hung connection holding the
// last permit used to freeze this whole loop — including for
// the OTHER listener — since accept() couldn't be called
// again until a permit freed up. That took down the entire
// web UI in production (2026-07-24) when federation peer
// connections piled up. Now a saturated pool just delays
// (and, past PERMIT_ACQUIRE_TIMEOUT, drops) individual
// connections instead of wedging the acceptor itself.
tokio::spawn(async move {
let permit = match tokio::time::timeout(
PERMIT_ACQUIRE_TIMEOUT,
active_connections.acquire_owned(),
)
.await
{
Ok(Ok(permit)) => permit,
Ok(Err(_)) => return, // semaphore closed during shutdown
Err(_) => {
warn!(
"{} connection from {} dropped — connection pool saturated for {}s",
local_addr, peer_addr, PERMIT_ACQUIRE_TIMEOUT.as_secs()
);
return;
}
};
let _permit = permit;
let service = service_fn(move |mut req: hyper::Request<hyper::Body>| {
let handler = handler.clone();
@@ -0,0 +1,205 @@
# HANDOFF — deploy companion APK 0.5.1 (vc21) to nodes
**For: the agent on archi-dev-box.** User-reported failure this evening:
pairing flow on Framework PT — downloaded the companion from the node's
QR, then the pairing scan didn't work. The APK the node serves predates
today's scanner fixes; the pipeline below gets the fixed build into the
user's hands.
## What changed on main today (all merged)
- **Pairing-scanner fix** (`QrScannerOverlay.kt`): ZXing decode attempts are
frame-gated (~7/s, was every frame — the CPU contention made the preview
stutter badly enough to never decode) and PreviewView uses TextureView (no
more black flash on open). This is the likely fix for "doesn't scan".
- Three-finger menu gesture (was two-finger, collided with scroll) + one-time
teaching overlay ~2 min after login.
- Native wallet QR scanner behind `window.ArchipelagoQr` + WebView file-chooser
support; web scan modal hands live scanning to it.
- npub-keyed saved servers (pairing contract item 1, PR #106).
- Served APK refreshed: `neode-ui/public/packages/archipelago-companion.apk`
is now **0.5.3 / versionCode 23**. On top of the 0.5.1 scanner fixes it
guarantees dual-path peering — the node's LAN endpoint (direct p2p, npub-
keyed dial hints) AND the Archipelago public anchor (vps2, baked into the
app so even an old node's QR can't leave the phone LAN-only) — and fixes
the two field failures from the user's 5G test (screenshots, 21:54):
- **Mesh VPN no longer kills the phone's internet** — the IPv6-only TUN
never called `allowFamily(AF_INET)`, so Android blocked all IPv4 while
the mesh was up. Now allowed (+ `allowBypass`).
- **Off-LAN connect works**`connect()` no longer hard-fails when the
scanned LAN IP doesn't answer; it brings the mesh up and probes the
node's ULA (`meshIp`) with retries before reporting failure.
## What to do
1. Redeploy the web-ui bundle from current main to the active nodes —
web root `/opt/archipelago/web-ui/` (NOT a neode-ui/ subfolder), all
nodes the user pairs against, at minimum the one Framework PT scans.
2. Verify the served artifact really updated:
`curl -sI http://<node>/packages/archipelago-companion.apk` — size should
change (~27 MB build of 2026-07-23), or pull it and check
`aapt dump badging` shows `versionCode='21' versionName='0.5.1'`.
3. The demo stack gets its images from CI (run 100 pushed today with the new
web bundle) — confirm the Portainer stack re-pulled, or trigger its
redeploy, so the demo QR also serves vc21.
4. **Node side is half the 5G story**: away-from-home reachability needs the
NODE connected to the public anchor too. On Framework PT (and any test
node): deploy current main (node-side npub-first `fips.pair-info`), then
verify `sudo -n fipsctl show status` reports the anchor connected —
`fips.reconnect` RPC if not. A phone can dial the anchor perfectly and
still fail if the node never enrolled with it.
5. Re-test the user's exact flows with **vc24** (updates any older install in
place): (a) pair ON the LAN, then switch the phone to 5G — the UI must
come up via the mesh ULA; (b) pair while ALREADY on 5G (never on the
node's LAN) — scan, VPN consent, and the connect must succeed through
the anchor.
## Live diagnosis update (22:3022:50, phone on adb — Mac agent)
vc23 on-device testing found and fixed the phone-side blocker, and narrowed
what remains to the node side. State as of vc24:
- **Fixed: TUN reader died at startup.** Android hands the VpnService fd over
non-blocking; the fips fork's blocking reader thread treats EAGAIN as fatal
("TUN read error … Try again (os error 11)") — so mesh sessions came up but
NO packet ever entered the tunnel. archy-fips-core now forces the fd
blocking before `start_with_tun_fd`. Verified on-device: reader survives,
and the 30s anchor-link flap disappeared with it (stable 8+ min on 5G).
- **Fixed: VPN marked not-metered** (`setMetered(false)`) — Android 10+
defaults VPNs to metered, putting the phone into data-restricted behaviour
while the mesh is up. Note the user's phone also has system **always-on
VPN** enabled for the app (`always_on_vpn_app`), a Settings-side toggle.
- **Verified good on-device**: peer store has node (LAN udp/tcp hints) + vps2
anchor; saved server is npub-keyed with ULA; anchor session establishes
from 5G in ~6s; VPN is bypassable, VALIDATED, only fd00::/8 routed.
- **REMAINING BLOCKER (node side)**: from the phone (app uid), ping6 and
HTTP to the node's ULA `fd79:1aa:b9e9:4c9f:1f80:5376:9385:1824` get no
reply — packets enter the mesh, nothing returns. Phone↔anchor works, so
suspect phone-fork ↔ node-daemon session/routing mismatch (FIPS wire
format is not stable across revs; phone pins fips-native fork 46494a74).
From Framework PT please capture:
- `fipsctl show status` (daemon version + anchor state)
- `fipsctl show sessions` and `show bloom` while the phone pings
- `ping6 <phone ULA fd68:496d:fe34:a06d:cf1:6e4:b6a4:3586>` from the
node (tests the reverse path)
- `ip addr show fips0` + confirm the web server listens on `[::]:80`
Report whether the node ever sees a session attempt from
`npub132c5whrsa6ccs0eylcpzaejq9uxul5ldvczz0axq78dh7fxkqj9st4uvzu`.
## Node-side diagnosis complete (23:0023:30, archi-dev-box agent)
Chain of findings, each verified live:
1. **FIXED: nginx had no IPv6 listener anywhere** — every shipped config
listened on 0.0.0.0 only, so `http://[<ULA>]` could NEVER connect on any
node, ever. Live-fixed on framework-pt + .116, canonical conf + bootstrap
self-heal shipped (`1e89362e`), heal binary deployed. ULA HTTP verified
answering on both nodes (local + over-mesh).
2. Firewall clean, fd00::/8 routes correct both ends, wire compat proven
(node + vps2 anchor both run fips 0.4.1 rev 15db6471db — the latest
upstream stable; nothing newer exists).
3. **THE REMAINING PROBLEM IS MESH SESSION PATH QUALITY.** From the vps2
anchor — a DIRECT connected peer — `GET /health` on the node's ULA takes
**1517 s per request and intermittently fails outright** (nginx logs
show 499 client-gave-up then 200; TCP SYN-retransmit backoff signature).
Session MMP is wildly asymmetric: node→vps2 srtt 204 ms, **vps2→node
srtt 4270 ms** — on a direct link whose raw RTT is 4 ms. Session traffic
is not riding the direct link; it appears to route through the ~1271-node
public tree (node's tree root is the public 00001a8c, depth 8; the node's
log also shows chronic "Discovery lookup timed out" for other targets).
4. The phone's npub never appears in the node's sessions — consistent with
discovery/handshake dying on the same degraded tree path, and the app's
~8 s probe window being far smaller than the observed 15 s+ first-request
latency even on the GOOD path.
### Recommendations
- **App side (Mac agent):** widen the ULA probe/connect window to ≥30 s
with retransmit-friendly pacing, and PRE-WARM the mesh session (start
pinging the node ULA as soon as the VPN is up, decoupled from the UI
probe) so the WebView hits a warm session.
- **Infra decision (user):** consider detaching the fleet from the public
v0l mesh — private tree rooted at the vps2 anchor (drop the legacy
185.18.221.160 seed anchor fleet-wide AND vps2's public peering). A
2-hop private tree would make session paths ride the direct links and
should collapse latency to ms. Trade-off: no reachability to/from the
broader public mesh.
- **Upstream:** report the direct-peer session-path asymmetry to
jmcorgan/fips (0.4.1).
## App-side recommendations implemented (23:3023:50, Mac agent — 0.5.5/vc25)
- Connect probe: mesh ULA now probed inside a **60s budget** with 15s
per-phase timeouts (rides out TCP retransmit backoff), replacing the old
~8s window.
- **Session pre-warm**: the VPN service starts probing every saved node ULA
the moment the tunnel is up (5s cadence for the first minute, then a 60s
keep-warm tick) — discovery/handshake cost is paid in the background, and
the session never idles out while the mesh is connected.
- (A phone-side ping test in this window still showed zero replies — that
measurement predated the vps2 daemon restart below and is superseded.)
## RESOLVED — root cause was vps2's degraded daemon, NOT the public tree (23:45)
The privatize-the-mesh recommendation above is WITHDRAWN. Final diagnosis:
vps2's fips daemon (3 days uptime, 0.2% CPU, idle box) had internally
degraded — EVERY link it carried showed ~4.5 s RTT (even to peers 30 ms
away), and since the anchor sits on the phone↔node path, everything through
it inherited that. `systemctl restart fips` on vps2 restored link RTTs to
40340 ms, and anchor→node mesh HTTP went from 1417 s (intermittent hard
fails) to a steady **165275 ms**. Node↔node direct sessions were always
fine (.116→framework-pt ULA HTTP: 894 ms cold, sub-second warm) — the
user's read was correct.
Actions taken: dead legacy anchor (185.18.221.160) removed from
framework-pt + .116 seed files (fleet keeps vps2 + public-mesh membership
via vps2 — we stay in the open mesh); fresh daemons on both nodes;
**vps2 fips now has RuntimeMaxSec=1d + Restart=always** so a wedged anchor
daemon can never rot for days again. Report the slow-degradation behaviour
upstream (jmcorgan/fips, 0.4.1): long-running daemon in a ~1400-node mesh
accumulates multi-second link latency at idle CPU, cleared by restart.
Phone side: vc25's 60 s probe + pre-warm now has a millisecond-latency mesh
to work with. Ready for the user's 5G test.
## NEXT (00:05, Mac agent → dev-box agent): app direct ports are IPv4-only over the mesh
The kiosk loads over the ULA now — but opening any APP dies with
`ERR_CONNECTION_REFUSED` at `http://[<ULA>]:<port>/`. User-hit first on
**Bitcoin Knots (:8334)**, and it will be every catalog app: the web UI
builds app URLs from the current host + the app's DIRECT port (Direct Port
Rule), and container-published ports only bind 0.0.0.0. Verified:
`192.168.63.249:8334` → HTTP 200 (nginx), ULA:8334 → refused. Same disease
as your :80 nginx fix, one layer down.
Fix must cover EVERY catalog app port and survive app install/remove. Two
shapes; pick what fits the container layer best:
1. **IPv6 publish at the container layer** — publish on `[::]` too
(pasta/rootless podman support address-specific `-p`), wired into the
container manager so new apps inherit it; or
2. **Host-side v6→v4 forwarders** — generated nginx `stream {}` (or
systemd-socket) units: `listen [::]:<port>``127.0.0.1:<port>`, one per
catalog app port, regenerated on app install/remove, boot-time
self-healed like the :80 fix. Keeps the Direct Port Rule URL contract
without touching containers.
Either way: extend the bootstrap self-heal, and verify from the MESH side
(curl the ULA on 23 app ports incl. :8334 from vps2 or .116) — not just
from the LAN.
## DONE (00:30, dev-box agent): app direct ports live over the mesh
Shape 2-variant implemented INSIDE the backend (`mesh_ports.rs`, `2ad57c63`):
a reconcile loop mirrors every public IPv4 listener (>=1024, bound 0.0.0.0,
no existing IPv6 any-listener) as a v6-ONLY `[::]:<port>` forwarder to
`127.0.0.1:<port>`, following `/proc/net/tcp*` every 15s — so app
install/remove and hardcoded companion ports (bitcoin-ui :8334) are covered
with zero container changes and no generated units; self-healing because it
lives in the binary. Strictly ADDITIVE: IPv4/LAN/Tor paths untouched, v6only
cannot intercept v4, foreign IPv6 listeners win.
Verified FROM THE MESH (vps2 → node ULA): :8334 HTTP 200 (466ms),
:18083 200 (306ms), :50002 200 (239ms); LAN :8334 still 200. Deployed to
framework-pt + .116 (binary sha 52ac0d8a…). Direct-port apps should now
open in the companion over 5G.
+55
View File
@@ -110,6 +110,61 @@ lands].
2. ❑ Mobile Home: wallet card directly under My Apps (G5 from the voice epic).
3. ❑ Mesh RF settings panel (Mesh → Device) still loads and saves.
## H. LoRa radio firmware flashing (Heltec V3/V4, new — extends Section E)
Full v1 scope is 3 firmware families × 2 boards (6 cells); mark each cell
tested on real hardware vs. code-reviewed only as this is run.
1. ❑ From the hot-swap modal's step 1 (device already probed), press
**Flash Firmware…** → new step shows firmware-family + board pickers and
the erase-confirmation checkbox; "Erase & Flash Now" stays disabled until
family, board, AND the checkbox are all set.
2. ❑ Confirm what's currently on the test stick via the existing probe
BEFORE flashing it — don't flash the only known-good device without a
fallback board on hand.
3. ❑ Prefer a spare Heltec V3/V4 for the first destructive erase+flash run;
only exercise a primary/in-use stick once the flow is proven safe.
4. ❑ MeshCore → Heltec V3: erase + write completes, progress bar and log
tail update live, ends at "Flash complete".
5. ❑ Meshtastic → Heltec V3: same, using the extracted `*.factory.bin` from
the esp32s3 release zip.
6. ❑ Reticulum/RNode → Heltec V3: `archy-rnodeconf --autoinstall` path
completes (no raw esptool erase/write step for this family — see
`mesh/flash.rs` doc comment).
7. ❑ Repeat 4-6 against a Heltec V4. Confirmed 2026-07-23 on real hardware:
V4 uses the ESP32-S3's native-USB JTAG/serial peripheral (vid:pid
303a:1001, generic to every native-USB ESP32-S3 board, not V4-specific)
— so unlike V3's CP2102 bridge chip, V4 is permanently NOT auto-matchable
by vid:pid. Board auto-detect should fail closed for it every time
(manual board selection required, "couldn't confirm automatically"
warning shown) — this is expected steady-state behavior, not a gap to
close later.
8. ❑ After a successful flash, the modal automatically re-probes and shows
the NEW firmware's badge/details — same as unplugging and replugging
(Section E item 3), but without physically touching the cable.
9. ❑ Deliberately test a failure path once (disconnect the board mid-write,
or point at a bad cached asset) — confirm the error surfaces in the
progress log AND that `docs/troubleshooting.md`'s "LoRa radio firmware
flash failed" recovery steps (BOOT+RST bootloader entry, manual esptool/
rnodeconf command) actually get the board back to a flashable state.
10. ❑ Cancel button only appears (and only works) while still in the
"Downloading firmware…" stage — once erasing/writing starts, no cancel
affordance is offered.
11. ❑ **Boot-loop regression (2026-07-23 incident)**: after a *failed* flash
(e.g. kill network access mid-download to force a failure), confirm the
mesh listener does NOT auto-resume — `journalctl -u archipelago` should
show a single `Leaving mesh listener stopped after failed flash` line
and then go quiet for that device, not a repeating `mesh::serial:
Opened serial port... Starting Meshcore handshake` cycle every few
seconds. Reconnect manually via the hot-swap modal afterward and confirm
it connects normally (the board itself should be untouched — the
download fails before esptool/rnodeconf ever runs).
12. ❑ Separately, force a device to flap connected/disconnected a few times
in under 20s each (e.g. a marginal USB connection) and confirm
`reconnect_delay` in the logs actually escalates (5s → 10s → 20s → ...)
rather than resetting to 5s on every attempt — see
`STABLE_SESSION_THRESHOLD` in `mesh/listener/mod.rs`.
---
After this passes: fold the batch + other agent's work into the next release
+7
View File
@@ -112,6 +112,13 @@ broke as soon as the LAN renumbered or the phone left home. FIPS peers on
the vps2 public anchor, and the web UI prepends the node's self-anchor to
`fanchors`.)
App-side status: items 2/3 were already covered by `FipsPreferences.
upsertNodePeer` (npub-matched peers, self-anchor dedup) and item 4 by the
WebView's `meshFallbackUrl` retry. Item 1 shipped 2026-07-23: `ServerEntry`
carries `npub` (trailing serialization field, legacy entries still parse) and
`ServerPreferences` matches saved/active entries via `sameNode` — npub first,
address/port/scheme only as the LAN-only fallback.
## Testing checklist (app side)
- [ ] Scan demo QR from https://demo.archipelago-foundation.org → auto-connected demo session.
+80
View File
@@ -443,6 +443,86 @@ free -h
- Check Nginx WebSocket proxy config: `/etc/nginx/sites-available/archipelago` must include `proxy_set_header Upgrade $http_upgrade`
- If on WiFi, try wired Ethernet for more stable connectivity
### 21. LoRa radio firmware flash failed / board unresponsive
**Symptoms**: The "Erase & Flash Now" flow in the mesh hot-swap modal reports
an error, or the radio no longer enumerates as a serial device after a flash
attempt.
**Diagnosis**:
```bash
# Poll the flash job's last-known stage/error directly
curl -s http://localhost:5678/rpc/v1 \
-H 'Content-Type: application/json' \
-d '{"method":"mesh.flash-status","params":{}}'
# Confirm the board is still enumerating at all
ls -la /dev/ttyUSB* /dev/ttyACM* /dev/mesh-radio 2>&1
# esptool/rnodeconf binaries present?
which esptool; ls -la /usr/local/bin/archy-rnodeconf
```
**Solutions**:
- A failure during `erasing`/`writing` (MeshCore/Meshtastic) or
`autoinstalling` (Reticulum) can leave the chip erased or half-written —
this is expected risk of the "always erase first" default, not a bug.
- Heltec V3/V4 boards can be forced back into bootloader mode manually: hold
**BOOT**, tap **RST**, then release **BOOT** — this puts the chip in a
state esptool can always talk to, regardless of what firmware (if any) is
currently on it.
- With the board in bootloader mode, a manual recovery flash can be run
directly over SSH without the UI:
```bash
esptool --chip esp32s3 --port /dev/ttyACM0 erase_flash
esptool --chip esp32s3 --port /dev/ttyACM0 write_flash 0x0 <known-good-image.bin>
```
- For Reticulum/RNode boards, the equivalent manual recovery is
`archy-rnodeconf /dev/ttyACM0 --autoinstall` (or `/usr/local/bin/archy-rnodeconf`
if it's not on `PATH`) — it re-runs the same fetch+erase+flash+bootstrap
sequence the UI triggers.
- If `esptool`/`archy-rnodeconf` are missing entirely, they should have been
installed by the last `self-update.sh` run — check
`sudo journalctl -u archipelago-update` for install failures, or install
`esptool` via `sudo apt-get install esptool` directly.
- Once a fresh image is confirmed written, unplug/replug the radio (or wait
for the next detection poll) — the hot-swap modal re-probes automatically
and shows whatever firmware is actually on the board now.
**Known incident (2026-07-23) — reconnect storm / device boot-loop after a
failed flash**: a real Heltec V3 got stuck cycling "connect → partial
handshake → drop" every 5-15s for 5+ minutes after a `mesh.flash-device`
attempt failed with `Reading firmware download stream`. Root cause was two
compounding issues, both now fixed:
1. `spawn_mesh_listener`'s reconnect backoff (`core/archipelago/src/mesh/listener/mod.rs`)
reset to its 5s minimum any time the prior session had been `device_connected`
at all, even for under a second — so a device that connects-then-drops
repeatedly never actually backed off. Every retry's `open()` toggles
DTR/RTS, which resets many ESP32 boards' MCU (native-USB *and*
CP2102/CH340 auto-reset-circuit boards), so the aggressive retries were
themselves *causing* the boot loop, not just observing one. Fixed by only
resetting backoff when a session ran for at least `STABLE_SESSION_THRESHOLD`
(20s) — see that constant's doc comment.
2. `mesh::flash::start_flash_job`'s post-completion handler auto-resumed the
listener unconditionally, even after a *failed* flash, immediately
re-entering the reconnect loop above with no cooldown. Fixed: on failure
the listener is now deliberately left stopped (reconnect manually via the
UI once the board is confirmed alive); on success there's a 5s settle
delay before resuming, so the board finishes booting from the flash
tool's own reset before Archipelago starts probing it again.
3. Separately, the download itself was failing because `mesh::flash`'s HTTP
client had a blanket 30s request timeout that covered the *entire*
download (including streaming a 170MB Meshtastic zip), not just
connection setup — fixed with a per-chunk stall timeout instead of a
fixed total-transfer cap.
If this symptom recurs (rapid repeating `mesh::serial: Opened serial
port... Starting Meshcore handshake` lines in `journalctl -u archipelago`
without a `LoRa firmware flash` job in progress), it's a NEW instance of the
same class of bug, not the one above — check whether backoff is actually
escalating (`Mesh session error: ... (retry in Xs)` — X should grow past 5s
within a few cycles) before assuming it's flashing-related.
---
## General Maintenance
@@ -365,6 +365,11 @@ RUN apt-get update && apt-get -y full-upgrade && apt-get install -y --no-install
ca-certificates \
openssl \
chrony \
iputils-ping \
esptool \
python3-venv \
binutils \
libpython3.13 \
locales \
console-setup \
keyboard-configuration \
@@ -9,6 +9,10 @@ resolver_timeout 5s;
server {
listen 80 default_server;
# IPv6 listener is REQUIRED: companion phones reach this node over the
# FIPS mesh at its fips0 ULA (http://[fdxx:…]) — without [::]:80 that
# address can never connect (found live 2026-07-23, framework-pt).
listen [::]:80 default_server;
server_name _;
root /opt/archipelago/web-ui;
@@ -901,6 +905,7 @@ server {
# HTTPS - required for PWA install (Add to Home Screen) from dev servers
server {
listen 443 ssl default_server;
listen [::]:443 ssl default_server;
server_name _;
ssl_certificate /etc/archipelago/ssl/archipelago.crt;
Binary file not shown.
@@ -357,10 +357,15 @@ async function buildPairingUrl(): Promise<string> {
const selfAnchor = info.tcp_port
? [{ npub: info.npub, addr: `${params.get('fhost')}:${info.tcp_port}`, transport: 'tcp' }]
: []
const anchors = [
...selfAnchor,
...(info.anchors || []).filter((a) => a.npub !== info.npub),
].slice(0, 4)
// HARD CAP at 2 anchors: every extra npub adds ~100 URL-encoded chars
// and pushed the QR past what phone cameras decode off a screen
// (3 anchors 600 chars QR v23 observed unscannable 2026-07-23).
// Self + one public rendezvous is all the phone needs; prefer the
// Archipelago-operated vps2 anchor as the public one.
const others = (info.anchors || []).filter((a) => a.npub !== info.npub)
const publicAnchor =
others.find((a) => a.addr.startsWith('146.59.87.168')) ?? others[0]
const anchors = [...selfAnchor, ...(publicAnchor ? [publicAnchor] : [])].slice(0, 2)
if (anchors.length) {
params.set(
'fanchors',
@@ -385,10 +390,13 @@ async function showPairScreen() {
pairingUrl.value = await buildPairingUrl()
// Large source + a real quiet zone; this QR is scanned by the companion
// app's camera, so give it every advantage (see download QR note above).
// EC level L: the payload is long (token + npub + anchors) and screen
// scans don't suffer the damage EC-M protects against L drops the
// module count a full version tier, which is what makes it scannable.
pairQrDataUrl.value = await QRCode.toDataURL(pairingUrl.value, {
width: 512,
width: 768,
margin: 3,
errorCorrectionLevel: 'M',
errorCorrectionLevel: 'L',
color: {
dark: '#111111',
light: '#ffffff',
@@ -1,7 +1,7 @@
<template>
<BaseModal
:show="show"
:title="step === 1 ? 'Mesh Radio Detected' : 'Apply Archipelago Settings'"
:title="step === 1 ? 'Mesh Radio Detected' : step === 2 ? 'Apply Archipelago Settings' : 'Flash Firmware'"
max-width="max-w-lg"
content-class="max-h-[90vh] overflow-y-auto"
@close="dismiss"
@@ -101,9 +101,111 @@
"Keep As Is" uses the radio exactly as it is nothing on it is changed,
and you can hot-swap radios any time.
</p>
<button
class="w-full text-center text-white/40 hover:text-white/70 text-[11px] mt-3 underline underline-offset-2"
:disabled="!!connecting"
@click="openFlashStep"
>
Flash Firmware
</button>
<p v-if="error" class="text-xs text-red-400 mt-2">{{ error }}</p>
</div>
<!-- Step 3: erase + reflash destructive, opt-in only -->
<div v-else-if="step === 'flash'">
<!-- Once a job exists (started via startFlash), ALWAYS show the
progress/result view below including on failure. The old
condition (`!active && stage !== 'done'`) was also true for a
FAILED job (active:false, stage:'failed'), which silently sent
the user back to this picker instead of showing the error. -->
<template v-if="!flashJob">
<p class="text-white/60 text-xs mb-3">
Downloads the latest firmware from upstream and writes it to
<span class="font-mono text-orange-300">{{ devicePath }}</span>.
</p>
<div class="space-y-4">
<div>
<label class="block text-sm text-white/80 mb-1">Firmware family</label>
<select v-model="flashFamily" class="w-full rounded-lg bg-white/[0.06] border border-white/10 text-white px-3 py-2 text-sm focus:outline-none focus:border-orange-400/60">
<option value="">Choose</option>
<option value="meshcore">MeshCore</option>
<option value="meshtastic">Meshtastic</option>
<option value="reticulum">Reticulum RNode</option>
</select>
</div>
<div>
<label class="block text-sm text-white/80 mb-1">Board</label>
<select v-model="flashBoard" class="w-full rounded-lg bg-white/[0.06] border border-white/10 text-white px-3 py-2 text-sm focus:outline-none focus:border-orange-400/60">
<option value="">Choose</option>
<option value="heltec-v3">Heltec LoRa 32 V3</option>
<option value="heltec-v4">Heltec LoRa 32 V4</option>
</select>
<p v-if="!boardAutoDetected" class="text-[11px] text-amber-400/80 mt-1">
Couldn't confirm the board automatically double check before flashing.
Flashing the wrong board's image can brick it.
</p>
</div>
</div>
<div class="rounded-xl bg-red-500/10 border border-red-500/30 p-3 mt-4">
<label class="flex items-start gap-2 text-xs text-red-300">
<input type="checkbox" v-model="flashConfirmed" class="mt-0.5" />
<span>
This <strong>erases the entire chip</strong>, including any existing
keys, identity, and contacts. This cannot be undone.
</span>
</label>
</div>
<p v-if="error" class="text-xs text-red-400 mt-3">{{ error }}</p>
<div class="flex gap-2 mt-6">
<button class="glass-button px-4 py-2 rounded-lg text-sm" @click="step = 1">Back</button>
<button
class="flex-1 glass-button px-4 py-2 rounded-lg text-sm font-medium bg-red-500/80 hover:bg-red-500 text-white disabled:opacity-50"
:disabled="!flashFamily || !flashBoard || !flashConfirmed || starting"
@click="startFlash"
>
{{ starting ? 'Starting…' : 'Erase & Flash Now' }}
</button>
</div>
</template>
<!-- Progress -->
<template v-else>
<div class="text-center py-2">
<p class="text-white text-sm font-medium">{{ flashStageLabel }}</p>
<div class="mt-3 h-2 rounded-full bg-white/10 overflow-hidden">
<div
class="h-full bg-orange-400 transition-all"
:style="{ width: (flashJob?.percent ?? (flashJob?.stage === 'done' ? 100 : 8)) + '%' }"
></div>
</div>
<p v-if="flashJob?.error" class="text-xs text-red-400 mt-3">{{ flashJob.error }}</p>
</div>
<div class="mt-3 rounded-xl bg-black/30 border border-white/10 p-2 h-32 overflow-y-auto font-mono text-[10px] text-white/50 leading-relaxed">
<div v-for="(line, i) in (flashJob?.log_tail ?? []).slice(-40)" :key="i">{{ line }}</div>
</div>
<div class="flex gap-2 mt-4">
<button
v-if="flashJob?.stage === 'downloading' && flashJob?.active"
class="glass-button px-4 py-2 rounded-lg text-sm"
@click="cancelFlash"
>
Cancel
</button>
<button
v-if="!flashJob?.active"
class="flex-1 glass-button glass-button-warning px-4 py-2 rounded-lg text-sm font-medium"
@click="closeFlashStep"
>
Done
</button>
</div>
</template>
</div>
<!-- Step 2: our latest parameters, shown before anything is written -->
<div v-else>
<p class="text-white/60 text-xs mb-3">
@@ -190,7 +292,7 @@
import { ref, computed, watch } from 'vue'
import { useRouter } from 'vue-router'
import BaseModal from '@/components/BaseModal.vue'
import { useMeshStore, type MeshDeviceProbe, type MeshConfigureParams } from '@/stores/mesh'
import { useMeshStore, type MeshDeviceProbe, type MeshConfigureParams, type FlashFirmwareFamily, type FlashBoard, type FlashJobStatus } from '@/stores/mesh'
import { useAppStore } from '@/stores/app'
import { LORA_REGIONS, regionByCode, suggestRegionFromLatLon, meshcorePlanFor } from '@/utils/loraRegions'
import { resolveMeshDeviceImage } from '@/utils/meshDeviceImages'
@@ -199,7 +301,7 @@ const mesh = useMeshStore()
const appStore = useAppStore()
const router = useRouter()
const step = ref<1 | 2>(1)
const step = ref<1 | 2 | 'flash'>(1)
const connecting = ref<false | 'keep' | 'setup'>(false)
const error = ref('')
const probing = ref(false)
@@ -264,7 +366,10 @@ const rfPreset = computed(() => {
// (Re)probe + (re)apply presets each time a new device surfaces the modal
watch([show, devicePath], async ([visible]) => {
if (!visible) return
if (!visible) {
stopFlashPoll()
return
}
step.value = 1
error.value = ''
imageFailed.value = false
@@ -349,6 +454,118 @@ async function applySetup() {
connecting.value = false
}
}
// Step 3: erase + reflash
const flashFamily = ref<FlashFirmwareFamily | ''>('')
const flashBoard = ref<FlashBoard | ''>('')
const flashConfirmed = ref(false)
const starting = ref(false)
const flashJob = ref<FlashJobStatus | null>(null)
let flashPollTimer: ReturnType<typeof setInterval> | null = null
const detectedInfo = computed(() =>
mesh.status?.detected_device_info?.find(d => d.path === devicePath.value)
)
// Mirrors mesh::flash::resolve_flash_board (core/archipelago/src/mesh/flash.rs)
// exactly matching on the display label was wrong: a Heltec V3's CP2102
// bridge chip reports "CP2102 USB to UART Bridge Controller" in its USB
// strings, not "Heltec", so meshDeviceImages.ts falls back to a generic
// "LoRa radio (CP2102 serial)" label that never matched /v3/i, showing the
// "couldn't confirm automatically" warning even though the backend CAN
// safely auto-detect V3 via vid:pid. Heltec V4 deliberately has no entry
// here, same reasoning as the backend: its vid:pid (303a:1001) is the
// ESP32-S3's generic native-USB descriptor, not V4-specific, so it can't be
// safely auto-matched and always requires manual selection.
const resolvedFlashBoard = computed<FlashBoard | ''>(() => {
const info = detectedInfo.value
if (info?.vid?.toLowerCase() === '10c4' && info?.pid?.toLowerCase() === 'ea60') return 'heltec-v3'
return ''
})
const boardAutoDetected = computed(() => !!resolvedFlashBoard.value)
const flashStageLabel = computed(() => {
switch (flashJob.value?.stage) {
case 'downloading': return 'Downloading firmware…'
case 'erasing': return 'Erasing chip…'
case 'writing': return 'Writing firmware…'
case 'autoinstalling': return 'Installing (rnodeconf)…'
case 'done': return 'Flash complete'
case 'failed': return 'Flash failed'
default: return ''
}
})
function openFlashStep() {
flashFamily.value = (probe.value?.kind as FlashFirmwareFamily) ?? ''
flashBoard.value = resolvedFlashBoard.value
flashConfirmed.value = false
flashJob.value = null
error.value = ''
step.value = 'flash'
}
function stopFlashPoll() {
if (flashPollTimer) {
clearInterval(flashPollTimer)
flashPollTimer = null
}
}
async function pollFlashStatus() {
try {
const status = await mesh.flashStatus()
flashJob.value = status
if (!status.active) {
stopFlashPoll()
if (status.done && !status.error) {
// Mirrors the unplug/replug hot-swap flow: re-probe so the details
// card reflects whatever firmware is actually on the board now.
const path = devicePath.value
probing.value = true
try {
probe.value = await mesh.probeDevice(path)
} catch {
probe.value = null
} finally {
probing.value = false
}
}
}
} catch {
stopFlashPoll()
}
}
async function startFlash() {
if (!flashFamily.value || !flashBoard.value || !flashConfirmed.value) return
starting.value = true
error.value = ''
try {
await mesh.flashDevice(devicePath.value, flashFamily.value, flashBoard.value)
flashJob.value = { active: true, stage: 'downloading', log_tail: [] }
stopFlashPoll()
flashPollTimer = setInterval(pollFlashStatus, 1500)
} catch (e) {
error.value = e instanceof Error ? e.message : 'Failed to start flashing'
} finally {
starting.value = false
}
}
async function cancelFlash() {
try {
await mesh.flashCancel()
} catch (e) {
error.value = e instanceof Error ? e.message : 'Failed to cancel'
}
}
function closeFlashStep() {
stopFlashPoll()
step.value = 1
}
</script>
<style scoped>
+47
View File
@@ -57,6 +57,23 @@ export interface MeshDeviceProbe {
max_contacts: number | null
}
export type FlashFirmwareFamily = 'meshcore' | 'meshtastic' | 'reticulum'
export type FlashBoard = 'heltec-v3' | 'heltec-v4'
export type FlashStage = 'downloading' | 'erasing' | 'writing' | 'autoinstalling' | 'done' | 'failed'
/** Live progress for the one flash job that can run at a time. */
export interface FlashJobStatus {
active: boolean
board?: FlashBoard
family?: FlashFirmwareFamily
path?: string
stage?: FlashStage
percent?: number | null
log_tail?: string[]
done?: boolean
error?: string | null
}
/** Params accepted by mesh.configure (superset of the status fields). */
export interface MeshConfigureParams {
enabled?: boolean
@@ -358,6 +375,32 @@ export const useMeshStore = defineStore('mesh', () => {
timeout: 45000, // serial probes are slow (multi-firmware handshakes)
})
}
/** Available firmware version(s) for a family v1 only ever returns
* ["latest"], since firmware is always fetched from upstream at flash
* time rather than pinned/bundled. */
async function flashListFirmware(family: FlashFirmwareFamily): Promise<string[]> {
const res = await rpcClient.call<{ versions: string[] }>({
method: 'mesh.flash-list-firmware',
params: { family },
})
return res.versions
}
/** Erase + reflash a detected radio. `board` is optional omit it to let
* the backend auto-resolve from the port's USB vid:pid; if that fails
* (e.g. Heltec V4 not yet in the vid:pid table), it errors and the UI
* must ask the user to pick the board explicitly. Always erases first. */
async function flashDevice(path: string, family: FlashFirmwareFamily, board?: FlashBoard): Promise<void> {
await rpcClient.call({
method: 'mesh.flash-device',
params: board ? { path, family, board } : { path, family },
})
}
async function flashStatus(): Promise<FlashJobStatus> {
return rpcClient.call<FlashJobStatus>({ method: 'mesh.flash-status' })
}
async function flashCancel(): Promise<void> {
await rpcClient.call({ method: 'mesh.flash-cancel' })
}
let globalDetectTimer: ReturnType<typeof setInterval> | null = null
/** App-wide light poll so the detected-device modal works on every page
* (the Mesh view's own 5s poll takes over while it is mounted). */
@@ -994,6 +1037,10 @@ export const useMeshStore = defineStore('mesh', () => {
undismissedDetectedDevices,
dismissDetectedDevice,
probeDevice,
flashListFirmware,
flashDevice,
flashStatus,
flashCancel,
startGlobalDetection,
fetchPeers,
fetchMessages,
+6 -1
View File
@@ -47,11 +47,16 @@ if [ -n "$RNODECONF_SRC" ] && [ -f "$RNODECONF_SRC" ]; then
# exit()/quit() builtins, which only exist in interactive Python (site.py
# injects them) — a frozen app hits NameError right as it tries to quit
# cleanly, after all the real work already succeeded. See
# pyi_rthook_exit_builtins.py.
# pyi_rthook_exit_builtins.py. A second hook fixes rnodeconf's board-flash
# step, which shells out to a bundled esptool.py via `sys.executable` —
# under a frozen binary that's the binary itself, not a real interpreter,
# so the flash subprocess call breaks. See
# pyi_rthook_fix_flasher_executable.py.
.venv/bin/pyinstaller --onefile --name archy-rnodeconf --clean --noconfirm \
--collect-submodules RNS \
--collect-data RNS \
--runtime-hook pyi_rthook_exit_builtins.py \
--runtime-hook pyi_rthook_fix_flasher_executable.py \
-d noarchive \
"$RNODECONF_SRC"
echo "Built dist/archy-rnodeconf ($(du -h dist/archy-rnodeconf | cut -f1))"
@@ -0,0 +1,34 @@
# PyInstaller runtime hook — see build.sh.
#
# rnodeconf's own board-flashing code shells out to a bundled esptool.py as
# `[sys.executable, flasher_path, "--chip", ..., "write_flash", ...]` (RNS's
# rnodeconf.py, ~line 2794 as of RNS 1.3.5). That's correct for a normal
# `python rnodeconf.py` invocation, but under a frozen PyInstaller binary
# `sys.executable` is the frozen binary itself, not a real interpreter — so
# the "subprocess" just re-invokes archy-rnodeconf's OWN argparse CLI with
# esptool-shaped flags, which it doesn't recognize, and the flash step fails
# immediately with "unrecognized arguments: --chip ...". Confirmed live
# against a real Heltec V4 (2026-07-23): device selection, band selection,
# and firmware download all worked; only the final `write_flash` subprocess
# call broke this way.
#
# Fix: point sys.executable at a real Python interpreter that has rnodeconf's
# own runtime deps available (esptool.py only needs pyserial, which RNS
# already depends on) before any of rnodeconf's code runs. Prefer the build
# venv this exact binary was frozen from — see build.sh — falling back to a
# bare `python3` on PATH if that venv isn't present on this node.
import os
import sys
if getattr(sys, "frozen", False):
_candidates = [
os.environ.get("ARCHY_RNODECONF_PYTHON", ""),
os.path.join(os.path.dirname(os.path.abspath(sys.argv[0])), "..", "reticulum-daemon", ".venv", "bin", "python3"),
os.path.expanduser("~/archy/reticulum-daemon/.venv/bin/python3"),
]
for _candidate in _candidates:
if _candidate and os.path.isfile(_candidate):
sys.executable = _candidate
break
else:
sys.executable = "python3"
+82
View File
@@ -84,6 +84,64 @@ if ! command -v nano >/dev/null 2>&1; then
fi
fi
if ! command -v ping >/dev/null 2>&1; then
log "Installing iputils-ping..."
if sudo apt-get update -qq 2>>"$LOG_FILE" && sudo apt-get install -y -qq iputils-ping 2>>"$LOG_FILE"; then
ok "ping installed"
else
warn "Unable to install ping automatically; continuing update"
fi
fi
if ! command -v esptool >/dev/null 2>&1; then
log "Installing esptool for LoRa radio firmware flashing..."
if sudo apt-get update -qq 2>>"$LOG_FILE" && sudo apt-get install -y -qq esptool 2>>"$LOG_FILE"; then
ok "esptool installed"
else
warn "Unable to install esptool automatically; radio firmware flashing will be unavailable"
fi
fi
# Debian's esptool package (4.7.0+dfsg-0.1) ships without the precompiled
# esp32s3 "stub flasher" blob (stripped for DFSG compliance — no
# buildable-from-source path Debian could verify). Without it, esptool's
# normal stub-loader mode fails outright (FileNotFoundError), and the ROM
# bootloader fallback (--no-stub) doesn't implement a full-chip erase at
# all — confirmed live 2026-07-23 flashing a real Heltec V4, both ways.
# Fetching the exact same file from the matching upstream esptool release
# tag restores full (and correct) flashing behavior — it's the same
# open-source codebase, just the one blob Debian's packaging couldn't
# include.
if command -v esptool >/dev/null 2>&1; then
STUB_DIR="/usr/lib/python3/dist-packages/esptool/targets/stub_flasher"
STUB_FILE="$STUB_DIR/stub_flasher_32s3.json"
if [ ! -f "$STUB_FILE" ]; then
log "Fetching esptool's esp32s3 stub flasher (missing from the Debian package)..."
ESPTOOL_VERSION=$(esptool version 2>/dev/null | tail -1 | tr -d ' \t')
if [ -n "$ESPTOOL_VERSION" ] && sudo curl -fsSL -o "$STUB_FILE" \
"https://raw.githubusercontent.com/espressif/esptool/v${ESPTOOL_VERSION}/esptool/targets/stub_flasher/stub_flasher_32s3.json" \
2>>"$LOG_FILE"; then
sudo chmod 644 "$STUB_FILE"
ok "esp32s3 stub flasher installed"
else
sudo rm -f "$STUB_FILE" 2>/dev/null
warn "Unable to fetch esp32s3 stub flasher; LoRa firmware flashing will be unavailable"
fi
fi
fi
# Build-time prerequisites for reticulum-daemon/build.sh's PyInstaller step
# below (discovered the hard way: ensurepip needs python3-venv, and
# PyInstaller itself needs objdump + libpython3.13.so at build time — none
# of these are pulled in by a bare `python3` package on Debian trixie).
for pkg in python3-venv binutils libpython3.13; do
if ! dpkg -s "$pkg" >/dev/null 2>&1; then
log "Installing $pkg (reticulum-daemon build prerequisite)..."
sudo apt-get update -qq 2>>"$LOG_FILE" && sudo apt-get install -y -qq "$pkg" 2>>"$LOG_FILE" \
|| warn "Unable to install $pkg automatically; reticulum-daemon tools build may fail"
fi
done
# Fetch latest
log "Fetching from origin..."
git fetch origin main --quiet 2>>"$LOG_FILE"
@@ -155,6 +213,30 @@ sudo cp "$BUILT_BIN" "$INSTALL_BIN"
sudo chmod +x "$INSTALL_BIN"
ok "Backend installed"
# Build + install reticulum-daemon tools (archy-reticulum-daemon, archy-rnodeconf).
# Non-fatal: archipelago falls back to its dev venv path if the packaged
# binaries aren't present, so a missing/failed build here degrades mesh
# Reticulum support rather than breaking the update. This mirrors
# deploy-to-target.sh's existing manual-deploy step, which until now was the
# only path that ever installed these — a node that only ever received OTA
# self-updates had neither binary.
if [ -f "$REPO_DIR/reticulum-daemon/build.sh" ]; then
log "Building reticulum-daemon tools (archy-reticulum-daemon, archy-rnodeconf)..."
if (cd "$REPO_DIR/reticulum-daemon" && ./build.sh) 2>>"$LOG_FILE"; then
for tool in archy-reticulum-daemon archy-rnodeconf; do
if [ -f "$REPO_DIR/reticulum-daemon/dist/$tool" ]; then
sudo cp "$REPO_DIR/reticulum-daemon/dist/$tool" /usr/local/bin/
sudo chmod +x "/usr/local/bin/$tool"
ok "$tool installed"
else
warn "$tool not built — leaving existing /usr/local/bin/$tool (if any) in place"
fi
done
else
warn "reticulum-daemon tools build failed — continuing without updating them"
fi
fi
# Build frontend
log "Building Vue frontend (production)..."
cd "$FRONTEND_DIR"