Archipelago — open-source initial import
This commit is contained in:
@@ -0,0 +1,497 @@
|
||||
//! Seed-anchor management for FIPS bootstrap.
|
||||
//!
|
||||
//! A freshly-installed node can't reach the global mesh via npub
|
||||
//! routing until it's connected to at least one peer that's already in
|
||||
//! the DHT. Upstream `fips` solves this by dialing a public anchor
|
||||
//! (e.g. `fips.v0l.io`) on first start. That's a single point of
|
||||
//! failure and doesn't help nodes behind restrictive firewalls or
|
||||
//! intermittent networks — archipelago operators reported fresh
|
||||
//! installs failing to reach any public anchor.
|
||||
//!
|
||||
//! This module adds a local, operator-editable seed-anchor list. Each
|
||||
//! entry is a `{npub, address, transport}` triple that archipelago
|
||||
//! pushes into the running daemon via `fipsctl connect` on startup and
|
||||
//! periodically thereafter. If one anchor falls over, the next one
|
||||
//! seeds the DHT instead. A well-configured cluster (e.g. a VPS
|
||||
//! running fips in anchor mode + a couple of home nodes) stops
|
||||
//! depending on the global anchor entirely.
|
||||
//!
|
||||
//! The list is persisted at `<data_dir>/seed-anchors.json`. The
|
||||
//! archipelago service user owns that directory, so no sudo is needed
|
||||
//! to read or write it.
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::path::{Path, PathBuf};
|
||||
use tokio::process::Command;
|
||||
|
||||
/// On-disk filename under `data_dir/`.
|
||||
const SEED_ANCHORS_FILE: &str = "seed-anchors.json";
|
||||
|
||||
/// Public anchor (`fips.v0l.io`) carried as a default seed for every
|
||||
/// node — it bootstraps DHT routing so a fresh node isn't isolated.
|
||||
/// Operators can remove it from the UI once their own cluster has
|
||||
/// independent anchors (removal persists, see `load`/`remove`).
|
||||
///
|
||||
/// IMPORTANT transport details, learned the hard way (see git history /
|
||||
/// the 2026-06-15 debugging on .116):
|
||||
/// - The anchor answers ONLY on **TCP port 8443**. UDP 8668 is dead
|
||||
/// (host pings on both IP families but never completes a UDP FIPS
|
||||
/// handshake). `fips/config.rs` always knew this; the old default
|
||||
/// here (`fips.v0l.io:8668`/udp) silently never connected fleet-wide.
|
||||
/// - We use the **IPv4 literal** rather than the `fips.v0l.io` hostname
|
||||
/// on purpose: the hostname resolves IPv6-first, but the daemon binds
|
||||
/// its transports IPv4-only (`0.0.0.0:8443`), so a v6 target makes the
|
||||
/// daemon fail to send the handshake with `EAFNOSUPPORT (os error 97)`.
|
||||
/// An IPv4 literal sidesteps the resolver entirely.
|
||||
pub const DEFAULT_PUBLIC_ANCHOR_NPUB: &str =
|
||||
"npub1zv58cn7v83mxvttl70w5fwjwuclfmntv9cnmv5wmz2nzz88u5urqvdx96n";
|
||||
pub const DEFAULT_PUBLIC_ANCHOR_ADDR: &str = "185.18.221.160:8443";
|
||||
pub const DEFAULT_PUBLIC_ANCHOR_TRANSPORT: &str = "tcp";
|
||||
|
||||
/// The upstream public anchor as a ready-to-apply `SeedAnchor`.
|
||||
pub fn default_public_anchor() -> SeedAnchor {
|
||||
SeedAnchor {
|
||||
npub: DEFAULT_PUBLIC_ANCHOR_NPUB.to_string(),
|
||||
address: DEFAULT_PUBLIC_ANCHOR_ADDR.to_string(),
|
||||
transport: DEFAULT_PUBLIC_ANCHOR_TRANSPORT.to_string(),
|
||||
label: "Public anchor (fips.v0l.io)".to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
// Archipelago-operated anchor on vps2 (the OTA/registry host, 146.59.87.168).
|
||||
// Every node already reaches this host for updates, so it is reachable from
|
||||
// networks that the upstream anchor is not — which is most of them (the
|
||||
// upstream anchor answers on one IPv4 that many home/office networks can't
|
||||
// reach, and its DNS resolves IPv6-first while the daemon is IPv4-only).
|
||||
// TCP because that traverses NAT/firewalls best; 8444 because 8443 on that
|
||||
// host is already taken by a container.
|
||||
pub const ARCHY_ANCHOR_NPUB: &str =
|
||||
"npub1dptaktwxv0mm245g2lqjykwm5ll0jpc6m3r4242ydfa9z7qe6urs3jvrak";
|
||||
pub const ARCHY_ANCHOR_ADDR: &str = "146.59.87.168:8444";
|
||||
pub const ARCHY_ANCHOR_TRANSPORT: &str = "tcp";
|
||||
|
||||
/// The Archipelago-operated anchor as a ready-to-apply `SeedAnchor`.
|
||||
pub fn archy_anchor() -> SeedAnchor {
|
||||
SeedAnchor {
|
||||
npub: ARCHY_ANCHOR_NPUB.to_string(),
|
||||
address: ARCHY_ANCHOR_ADDR.to_string(),
|
||||
transport: ARCHY_ANCHOR_TRANSPORT.to_string(),
|
||||
label: "Archipelago anchor (vps2)".to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Public FIPS-network anchors from join.fips.network (adverts published as
|
||||
/// Nostr kind-37195 `fips-overlay-v1` events, refreshed hourly). Of the eight
|
||||
/// live test anchors (2026-07-22), these two are the dual-transport ones —
|
||||
/// they answer on TCP as well as UDP, and TCP is what traverses restrictive
|
||||
/// networks. The remaining six are UDP-only; add them per-node via
|
||||
/// `fips.add-seed-anchor` if more rendezvous diversity is wanted.
|
||||
pub fn fips_network_anchors() -> Vec<SeedAnchor> {
|
||||
vec![
|
||||
SeedAnchor {
|
||||
npub: "npub10yffd020a4ag8zcy75f9pruq3rnghvvhd5hphl9s62zgp35s560qrksp9u".to_string(),
|
||||
address: "23.182.128.74:443".to_string(),
|
||||
transport: "tcp".to_string(),
|
||||
label: "FIPS network anchor (join.fips.network)".to_string(),
|
||||
},
|
||||
SeedAnchor {
|
||||
npub: "npub1qmc3cvfz0yu2hx96nq3gp55zdan2qclealn7xshgr448d3nh6lks7zel98".to_string(),
|
||||
address: "217.77.8.91:443".to_string(),
|
||||
transport: "tcp".to_string(),
|
||||
label: "FIPS network anchor (join.fips.network)".to_string(),
|
||||
},
|
||||
]
|
||||
}
|
||||
|
||||
/// The default anchor set carried implicitly by `load()` on nodes that have
|
||||
/// never edited their anchor list, so every node dials them without operator
|
||||
/// action. Multiple anchors so one unreachable rendezvous host can't strand a
|
||||
/// node: `fipsctl connect` is attempted for each, and whichever the node's
|
||||
/// network can reach wins. The Archipelago-operated anchor is listed first
|
||||
/// because it is reachable from the widest set of networks.
|
||||
pub fn default_public_anchors() -> Vec<SeedAnchor> {
|
||||
let mut anchors = vec![archy_anchor(), default_public_anchor()];
|
||||
anchors.extend(fips_network_anchors());
|
||||
anchors
|
||||
}
|
||||
|
||||
/// One seed-anchor entry. `address` must be directly dialable (IP or
|
||||
/// resolvable hostname + UDP port); `transport` is one of "udp", "tcp",
|
||||
/// "tor", "ethernet" (the values upstream `fipsctl connect` accepts).
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
pub struct SeedAnchor {
|
||||
/// Bech32 `npub1...` of the anchor's FIPS identity.
|
||||
pub npub: String,
|
||||
/// Directly-dialable transport address, e.g. `192.0.2.12:8668`.
|
||||
pub address: String,
|
||||
/// Transport to use — almost always `"udp"`.
|
||||
#[serde(default = "default_transport")]
|
||||
pub transport: String,
|
||||
/// Human-readable note shown in the UI (e.g. "Home anchor", "VPS").
|
||||
#[serde(default)]
|
||||
pub label: String,
|
||||
}
|
||||
|
||||
fn default_transport() -> String {
|
||||
"udp".to_string()
|
||||
}
|
||||
|
||||
fn anchors_path(data_dir: &Path) -> PathBuf {
|
||||
data_dir.join(SEED_ANCHORS_FILE)
|
||||
}
|
||||
|
||||
/// Load the seed-anchor list. A node that has never edited its anchor
|
||||
/// list (no file yet) gets the default public anchor so it can bootstrap
|
||||
/// the mesh out of the box. Once the operator edits anchors — including
|
||||
/// removing the default — a file exists and is authoritative, so removal
|
||||
/// persists and we never silently re-add it.
|
||||
pub async fn load(data_dir: &Path) -> Result<Vec<SeedAnchor>> {
|
||||
let path = anchors_path(data_dir);
|
||||
if !path.exists() {
|
||||
return Ok(default_public_anchors());
|
||||
}
|
||||
let bytes = tokio::fs::read(&path)
|
||||
.await
|
||||
.with_context(|| format!("read {}", path.display()))?;
|
||||
let anchors: Vec<SeedAnchor> =
|
||||
serde_json::from_slice(&bytes).with_context(|| format!("parse {}", path.display()))?;
|
||||
Ok(repair_legacy_anchor_set(anchors))
|
||||
}
|
||||
|
||||
/// Add the newer redundant public TCP anchors to legacy files that only carry
|
||||
/// the Archipelago-operated vps2 anchor. A deliberately custom/private anchor
|
||||
/// list remains authoritative; this only repairs the exact stale shape shipped
|
||||
/// before the join.fips.network anchors became defaults.
|
||||
fn repair_legacy_anchor_set(mut anchors: Vec<SeedAnchor>) -> Vec<SeedAnchor> {
|
||||
let has_archy = anchors.iter().any(|a| a.npub == ARCHY_ANCHOR_NPUB);
|
||||
let has_fips_network = fips_network_anchors()
|
||||
.iter()
|
||||
.any(|default| anchors.iter().any(|a| a.npub == default.npub));
|
||||
if has_archy && !has_fips_network {
|
||||
for anchor in fips_network_anchors() {
|
||||
anchors.push(anchor);
|
||||
}
|
||||
}
|
||||
anchors
|
||||
}
|
||||
|
||||
/// Persist the list. Overwrites atomically via write-then-rename so a
|
||||
/// crashed archipelago never leaves a half-written config.
|
||||
pub async fn save(data_dir: &Path, anchors: &[SeedAnchor]) -> Result<()> {
|
||||
tokio::fs::create_dir_all(data_dir)
|
||||
.await
|
||||
.with_context(|| format!("mkdir -p {}", data_dir.display()))?;
|
||||
let path = anchors_path(data_dir);
|
||||
let tmp = path.with_extension("json.tmp");
|
||||
let json = serde_json::to_vec_pretty(anchors).context("serialize seed anchors")?;
|
||||
tokio::fs::write(&tmp, json)
|
||||
.await
|
||||
.with_context(|| format!("write {}", tmp.display()))?;
|
||||
tokio::fs::rename(&tmp, &path)
|
||||
.await
|
||||
.with_context(|| format!("rename {} -> {}", tmp.display(), path.display()))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Add (or update) one anchor, keyed by npub. Returns the resulting list.
|
||||
pub async fn add(data_dir: &Path, anchor: SeedAnchor) -> Result<Vec<SeedAnchor>> {
|
||||
let mut list = load(data_dir).await?;
|
||||
if let Some(existing) = list.iter_mut().find(|a| a.npub == anchor.npub) {
|
||||
*existing = anchor;
|
||||
} else {
|
||||
list.push(anchor);
|
||||
}
|
||||
save(data_dir, &list).await?;
|
||||
Ok(list)
|
||||
}
|
||||
|
||||
/// Remove an anchor by npub. Returns the resulting list.
|
||||
pub async fn remove(data_dir: &Path, npub: &str) -> Result<Vec<SeedAnchor>> {
|
||||
let mut list = load(data_dir).await?;
|
||||
list.retain(|a| a.npub != npub);
|
||||
save(data_dir, &list).await?;
|
||||
Ok(list)
|
||||
}
|
||||
|
||||
/// Apply the seed anchors to the running FIPS daemon. For each entry,
|
||||
/// asks `fipsctl connect` to dial the peer. Errors are logged but don't
|
||||
/// fail the whole operation — a single unreachable anchor shouldn't
|
||||
/// block the others.
|
||||
///
|
||||
/// `fipsctl connect` is idempotent-ish: calling it for an already-
|
||||
/// connected peer is a no-op at the protocol layer, so re-applying on
|
||||
/// a timer is safe. Returns a list of per-anchor results for logging.
|
||||
///
|
||||
/// Invoked through `sudo -n`: the upstream daemon's control socket
|
||||
/// (`/run/fips/control.sock`) is owned `root:fips` 0660, and the
|
||||
/// archipelago service user is not in the `fips` group, so a bare
|
||||
/// `fipsctl connect` fails with EACCES. This matches the privileged
|
||||
/// `sudo -n fipsctl show peers` call in `service::peer_connectivity_summary`.
|
||||
/// Without it, seed anchors persist to disk but never actually dial,
|
||||
/// leaving `anchor_connected=false` and every peer dial falling back to
|
||||
/// a slow Tor timeout.
|
||||
pub async fn apply(anchors: &[SeedAnchor]) -> Vec<ApplyResult> {
|
||||
// Concurrent, each connect hard-capped: the old serial loop waited
|
||||
// unbounded on every `sudo fipsctl connect`, so one hung subprocess
|
||||
// stalled the whole apply — and the periodic anchor tick behind it,
|
||||
// which is exactly when a wedged daemon most needs the re-apply.
|
||||
let futs = anchors.iter().cloned().map(|anchor| async move {
|
||||
let out = tokio::time::timeout(
|
||||
std::time::Duration::from_secs(15),
|
||||
Command::new("sudo")
|
||||
.args([
|
||||
"-n",
|
||||
"fipsctl",
|
||||
"connect",
|
||||
&anchor.npub,
|
||||
&anchor.address,
|
||||
&anchor.transport,
|
||||
])
|
||||
.output(),
|
||||
)
|
||||
.await;
|
||||
let result = match out {
|
||||
Ok(Ok(o)) if o.status.success() => ApplyResult {
|
||||
npub: anchor.npub.clone(),
|
||||
ok: true,
|
||||
message: String::from_utf8_lossy(&o.stdout).trim().to_string(),
|
||||
},
|
||||
Ok(Ok(o)) => ApplyResult {
|
||||
npub: anchor.npub.clone(),
|
||||
ok: false,
|
||||
message: format!(
|
||||
"sudo fipsctl connect exited {}: {}",
|
||||
o.status,
|
||||
String::from_utf8_lossy(&o.stderr).trim()
|
||||
),
|
||||
},
|
||||
Ok(Err(e)) => ApplyResult {
|
||||
npub: anchor.npub.clone(),
|
||||
ok: false,
|
||||
message: format!("sudo fipsctl launch failed: {}", e),
|
||||
},
|
||||
Err(_) => ApplyResult {
|
||||
npub: anchor.npub.clone(),
|
||||
ok: false,
|
||||
message: "sudo fipsctl connect timed out after 15s".to_string(),
|
||||
},
|
||||
};
|
||||
if result.ok {
|
||||
tracing::debug!(npub = %result.npub, "Seed anchor applied");
|
||||
} else {
|
||||
tracing::warn!(
|
||||
npub = %result.npub,
|
||||
message = %result.message,
|
||||
"Seed anchor apply failed (non-fatal)"
|
||||
);
|
||||
}
|
||||
result
|
||||
});
|
||||
futures_util::future::join_all(futs).await
|
||||
}
|
||||
|
||||
/// Outcome of a single `fipsctl connect` call.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ApplyResult {
|
||||
pub npub: String,
|
||||
pub ok: bool,
|
||||
pub message: String,
|
||||
}
|
||||
|
||||
/// FIPS UDP transport port (matches `transports.udp.bind_addr` in the generated
|
||||
/// `fips.yaml`). Direct peer links dial this, NOT the HTTP/LAN messaging port.
|
||||
const FIPS_UDP_PORT: u16 = crate::fips::PUBLISHED_UDP_PORT;
|
||||
|
||||
/// Build transient seed-anchor entries that dial LAN-discovered federation peers
|
||||
/// directly over their FIPS UDP transport. For each peer the registry knows both
|
||||
/// a LAN socket address AND a FIPS npub for, point a `udp` anchor at
|
||||
/// `<lan-ip>:<FIPS_UDP_PORT>`. This lets co-located federation nodes form a DIRECT FIPS link
|
||||
/// instead of depending on the global anchor's spanning tree to route between
|
||||
/// them (the cause of every dial falling back to Tor when the anchor link flaps).
|
||||
///
|
||||
/// This is FIPS's own UDP transport over the LAN — not Tailscale, not the LAN
|
||||
/// HTTP messaging port. NOT persisted to `seed-anchors.json`: recomputed each
|
||||
/// apply tick from live LAN discovery, so a peer's changing IP self-corrects and
|
||||
/// stale entries never accumulate. `fipsctl connect` is idempotent, so
|
||||
/// re-applying just keeps the link warm.
|
||||
pub fn lan_fips_anchors(peers: &[crate::transport::PeerRecord]) -> Vec<SeedAnchor> {
|
||||
let mut out = Vec::new();
|
||||
for p in peers {
|
||||
let (Some(lan), Some(npub)) = (p.lan_address.as_deref(), p.fips_npub.as_deref()) else {
|
||||
continue;
|
||||
};
|
||||
// lan_address is the peer's HTTP/LAN socket ("ip:port"); reuse only its IP
|
||||
// and target the FIPS UDP port. SocketAddr::new(...).to_string() formats
|
||||
// IPv6 with brackets correctly.
|
||||
let Ok(sa) = lan.parse::<std::net::SocketAddr>() else {
|
||||
continue;
|
||||
};
|
||||
out.push(SeedAnchor {
|
||||
npub: npub.to_string(),
|
||||
address: std::net::SocketAddr::new(sa.ip(), FIPS_UDP_PORT).to_string(),
|
||||
transport: "udp".to_string(),
|
||||
label: "LAN federation peer (direct FIPS)".to_string(),
|
||||
});
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn mk(npub: &str) -> SeedAnchor {
|
||||
SeedAnchor {
|
||||
npub: npub.to_string(),
|
||||
address: "example.test:8668".to_string(),
|
||||
transport: "udp".to_string(),
|
||||
label: "test".to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn load_missing_seeds_default_public_anchors() {
|
||||
// A node that has never edited its anchor list should still get the
|
||||
// full default anchor set so it can bootstrap the mesh out of the box.
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let got = load(dir.path()).await.unwrap();
|
||||
assert_eq!(got, default_public_anchors());
|
||||
// The Archipelago-operated anchor must come first (widest reachability)
|
||||
// and the upstream anchor must remain present as a fallback.
|
||||
assert_eq!(got[0], archy_anchor());
|
||||
assert!(got.contains(&default_public_anchor()));
|
||||
// Every default must be a TCP form (traverses NAT/firewalls), never the
|
||||
// dead udp:8668 the upstream anchor never answers on.
|
||||
assert!(got.iter().all(|a| a.transport == "tcp"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn load_repairs_legacy_archy_only_anchor_file() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
save(dir.path(), &[archy_anchor()]).await.unwrap();
|
||||
|
||||
let got = load(dir.path()).await.unwrap();
|
||||
assert!(got.iter().any(|a| a.npub == ARCHY_ANCHOR_NPUB));
|
||||
for anchor in fips_network_anchors() {
|
||||
assert!(got.iter().any(|a| a.npub == anchor.npub));
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn load_keeps_private_anchor_file_authoritative() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let private = mk("npub1private");
|
||||
save(dir.path(), std::slice::from_ref(&private))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let got = load(dir.path()).await.unwrap();
|
||||
assert_eq!(got, vec![private]);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn removing_one_default_persists_and_keeps_the_other() {
|
||||
// Editing the anchor list (here removing one default) makes the file
|
||||
// authoritative: the removed anchor must not be silently re-seeded on
|
||||
// next load, and the remaining default must stay.
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let list = remove(dir.path(), ARCHY_ANCHOR_NPUB).await.unwrap();
|
||||
assert!(!list.iter().any(|a| a.npub == ARCHY_ANCHOR_NPUB));
|
||||
assert!(list.contains(&default_public_anchor()));
|
||||
let got = load(dir.path()).await.unwrap();
|
||||
assert_eq!(got, list, "edited list is authoritative; no re-seed");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn removing_all_defaults_persists_as_empty() {
|
||||
// Removing every default leaves an empty authoritative list that must
|
||||
// not be re-seeded on next load.
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let mut list = Vec::new();
|
||||
for anchor in default_public_anchors() {
|
||||
list = remove(dir.path(), &anchor.npub).await.unwrap();
|
||||
}
|
||||
assert!(list.is_empty());
|
||||
let got = load(dir.path()).await.unwrap();
|
||||
assert!(got.is_empty(), "defaults must stay removed once edited");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn save_and_load_roundtrip() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let a = mk("npub1aaa");
|
||||
let b = mk("npub1bbb");
|
||||
save(dir.path(), &[a.clone(), b.clone()]).await.unwrap();
|
||||
let got = load(dir.path()).await.unwrap();
|
||||
assert_eq!(got, vec![a, b]);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn add_replaces_existing_by_npub() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let mut a = mk("npub1aaa");
|
||||
save(dir.path(), &[a.clone()]).await.unwrap();
|
||||
a.address = "newhost:8668".to_string();
|
||||
let list = add(dir.path(), a.clone()).await.unwrap();
|
||||
assert_eq!(list.len(), 1);
|
||||
assert_eq!(list[0].address, "newhost:8668");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn remove_by_npub() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
save(
|
||||
dir.path(),
|
||||
&[mk("npub1aaa"), mk("npub1bbb"), mk("npub1ccc")],
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let list = remove(dir.path(), "npub1bbb").await.unwrap();
|
||||
assert_eq!(list.len(), 2);
|
||||
assert!(list.iter().all(|a| a.npub != "npub1bbb"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn seed_anchor_uses_udp_by_default() {
|
||||
let json = r#"{"npub":"npub1x","address":"h:8668"}"#;
|
||||
let a: SeedAnchor = serde_json::from_str(json).unwrap();
|
||||
assert_eq!(a.transport, "udp");
|
||||
assert_eq!(a.label, "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn lan_fips_anchor_port_matches_daemon_bind() {
|
||||
// Drift guard: direct LAN anchors must dial the UDP port the
|
||||
// generated fips.yaml actually binds. These were out of sync for
|
||||
// months (anchors dialed 8668, the daemon bound 2121), making the
|
||||
// whole direct-peering feature dial a dead port.
|
||||
let yaml = crate::fips::config::render_config_yaml();
|
||||
assert!(
|
||||
yaml.contains(&format!("0.0.0.0:{FIPS_UDP_PORT}")),
|
||||
"lan_fips_anchors dials :{FIPS_UDP_PORT} but the daemon config binds elsewhere"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn lan_fips_anchors_builds_direct_entry() {
|
||||
let peer = crate::transport::PeerRecord {
|
||||
did: "did:key:zpeer".to_string(),
|
||||
lan_address: Some("192.0.2.198:5678".to_string()),
|
||||
fips_npub: Some("npub1peer".to_string()),
|
||||
..Default::default()
|
||||
};
|
||||
let out = lan_fips_anchors(&[peer]);
|
||||
assert_eq!(out.len(), 1);
|
||||
assert_eq!(out[0].address, format!("192.0.2.198:{FIPS_UDP_PORT}"));
|
||||
assert_eq!(out[0].transport, "udp");
|
||||
|
||||
// Peers missing either the LAN address or the npub produce nothing.
|
||||
let no_npub = crate::transport::PeerRecord {
|
||||
did: "did:key:zother".to_string(),
|
||||
lan_address: Some("192.0.2.199:5678".to_string()),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(lan_fips_anchors(&[no_npub]).is_empty());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
//! Generated by scripts/generate-app-catalog.py. Do not edit manually.
|
||||
//!
|
||||
//! Catalog app launch ports (the web UIs the companion opens by direct
|
||||
//! port). Used to write the fips0 firewall allowance drop-in so app UIs
|
||||
//! are reachable over the mesh; ports of apps that aren't installed have
|
||||
//! no listener, so allowing them is inert.
|
||||
|
||||
pub const APP_LAUNCH_PORTS: &[u16] = &[
|
||||
2283, 2342, 3000, 3001, 3002, 4080, 5180, 7778, 8080, 8081, 8082, 8083, 8084, 8085, 8087, 8088,
|
||||
8089, 8090, 8096, 8123, 8175, 8176, 8240, 8334, 8336, 8888, 8999, 9000, 9100, 10380, 11434,
|
||||
18081, 18083, 23000, 32838, 50002,
|
||||
];
|
||||
@@ -0,0 +1,499 @@
|
||||
//! FIPS daemon config + key materialisation.
|
||||
//!
|
||||
//! Writes `/etc/fips/fips.yaml`, `/etc/fips/fips.key`, and
|
||||
//! `/etc/fips/fips.pub` from the archipelago node's seed-derived FIPS
|
||||
//! keypair, then chmod 0600 the private key.
|
||||
//!
|
||||
//! Privileged filesystem writes go through a `sudo install` invocation
|
||||
//! rather than opening `/etc/fips/*` directly — the archipelago service
|
||||
//! user cannot write `/etc` itself. The sudoers policy in the ISO
|
||||
//! whitelists `install` into `/etc/fips/`.
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use serde::Serialize;
|
||||
use std::path::Path;
|
||||
use tokio::process::Command;
|
||||
|
||||
use super::{
|
||||
DAEMON_CONFIG_PATH, DAEMON_KEY_PATH, DAEMON_PUB_PATH, DEFAULT_TCP_PORT, PUBLISHED_UDP_PORT,
|
||||
};
|
||||
|
||||
/// Header prepended to the generated YAML. serde doesn't emit comments, so
|
||||
/// this is concatenated onto the serialised body.
|
||||
const CONFIG_HEADER: &str = "# Generated by archipelago — do not edit by hand.\n\
|
||||
# Regenerated on every key change and daemon upgrade.\n";
|
||||
|
||||
/// Typed mirror of the subset of upstream `fips.yaml` that archipelago owns.
|
||||
///
|
||||
/// This was previously built by `format!`-ing a string literal. Upstream's
|
||||
/// config structs are `#[serde(deny_unknown_fields)]`, so a key we get wrong
|
||||
/// doesn't degrade gracefully — the daemon refuses to start and the node drops
|
||||
/// off the mesh. Serialising from typed structs lets the compiler and the
|
||||
/// tests below catch drift, instead of a node discovering it at boot after an
|
||||
/// upgrade.
|
||||
///
|
||||
/// Schema verified field-by-field against jmcorgan/fips **v0.4.1** (2026-07-20).
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct FipsConfig {
|
||||
pub node: NodeSection,
|
||||
pub tun: TunSection,
|
||||
pub dns: DnsSection,
|
||||
pub transports: TransportsSection,
|
||||
/// Static peers. Always empty: archipelago feeds peers dynamically via the
|
||||
/// seed-anchors apply loop and federation-invite hooks.
|
||||
pub peers: Vec<PeerEntry>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct NodeSection {
|
||||
pub identity: IdentitySection,
|
||||
pub discovery: DiscoverySection,
|
||||
pub retry: RetrySection,
|
||||
pub rate_limit: RateLimitSection,
|
||||
}
|
||||
|
||||
/// Fast-reconnect profile (`node.retry.*`). Upstream defaults (5s base
|
||||
/// doubling to 300s) are tuned for stable always-on links; a node redialing
|
||||
/// a recycled anchor sat off-mesh for ~90s. Field names verified live on
|
||||
/// 2026-07-24: a daemon restarted with these keys in fips.yaml and peered.
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct RetrySection {
|
||||
pub base_interval_secs: u64,
|
||||
pub max_backoff_secs: u64,
|
||||
pub max_retries: u32,
|
||||
}
|
||||
|
||||
/// Session-handshake resend pacing (`node.rate_limit.*`). Stock 1s x2.0
|
||||
/// gaps out to 8-16s between resends exactly when a route has just
|
||||
/// appeared; 400ms x1.5 keeps continuous coverage through the connect
|
||||
/// window (phone measured session-after-route: 225ms).
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct RateLimitSection {
|
||||
pub handshake_resend_interval_ms: u64,
|
||||
pub handshake_resend_backoff: f64,
|
||||
pub handshake_max_resends: u32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct IdentitySection {
|
||||
/// With `persistent: true` the daemon reuses the key file at
|
||||
/// config-dir/fips.key (= `DAEMON_KEY_PATH`) instead of generating an
|
||||
/// ephemeral identity on every start.
|
||||
pub persistent: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct DiscoverySection {
|
||||
/// Lookup completion timeout. Lookups fired while the tree position is
|
||||
/// still settling are doomed; failing them fast (5s, not 10s) lets the
|
||||
/// 1s-backoff retry find the route the moment it exists.
|
||||
pub timeout_secs: u64,
|
||||
pub backoff_base_secs: u64,
|
||||
pub backoff_max_secs: u64,
|
||||
pub retry_interval_secs: u64,
|
||||
pub max_attempts: u8,
|
||||
pub lan: LanDiscoverySection,
|
||||
}
|
||||
|
||||
/// mDNS / DNS-SD discovery on the local link (`node.discovery.lan.*`), added
|
||||
/// upstream in v0.4.0 and opt-in there (upstream default is `false`).
|
||||
///
|
||||
/// We enable it so co-located nodes peer directly instead of depending on the
|
||||
/// public anchor being reachable — an anchor blackhole on one network segment
|
||||
/// otherwise islands a node completely.
|
||||
///
|
||||
/// Emitted unconditionally rather than version-gated: v0.3.0's `DiscoveryConfig`
|
||||
/// has no `lan` field *and* no `deny_unknown_fields`, so a v0.3.0 daemon ignores
|
||||
/// this key harmlessly (verified against the v0.3.0 source). It therefore starts
|
||||
/// working on its own when a node upgrades, with no second config migration.
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct LanDiscoverySection {
|
||||
pub enabled: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct TunSection {
|
||||
pub enabled: bool,
|
||||
pub name: String,
|
||||
pub mtu: u16,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct DnsSection {
|
||||
pub enabled: bool,
|
||||
pub bind_addr: String,
|
||||
}
|
||||
|
||||
/// Both UDP and TCP are enabled: the public anchor answers on TCP/8443 only,
|
||||
/// and networks that block outbound UDP can still bootstrap over TCP.
|
||||
/// Upstream dropped the `tor:` transport variant — archipelago's own Tor
|
||||
/// fallback handles that layer.
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct TransportsSection {
|
||||
pub udp: TransportBind,
|
||||
pub tcp: TransportBind,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct TransportBind {
|
||||
/// Upstream takes `bind_addr` ("host:port"), not `enabled` + `port`.
|
||||
pub bind_addr: String,
|
||||
}
|
||||
|
||||
/// A static peer entry. Never constructed today (see `FipsConfig::peers`), but
|
||||
/// typed so the shape is checked if static peering is ever needed.
|
||||
#[derive(Debug, Clone, PartialEq, Serialize)]
|
||||
pub struct PeerEntry {
|
||||
pub npub: String,
|
||||
pub address: String,
|
||||
pub transport: String,
|
||||
}
|
||||
|
||||
impl Default for FipsConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
node: NodeSection {
|
||||
identity: IdentitySection { persistent: true },
|
||||
discovery: DiscoverySection {
|
||||
timeout_secs: 5,
|
||||
backoff_base_secs: 1,
|
||||
backoff_max_secs: 30,
|
||||
retry_interval_secs: 2,
|
||||
max_attempts: 3,
|
||||
lan: LanDiscoverySection { enabled: true },
|
||||
},
|
||||
retry: RetrySection {
|
||||
base_interval_secs: 1,
|
||||
max_backoff_secs: 30,
|
||||
max_retries: 30,
|
||||
},
|
||||
rate_limit: RateLimitSection {
|
||||
handshake_resend_interval_ms: 400,
|
||||
handshake_resend_backoff: 1.5,
|
||||
handshake_max_resends: 10,
|
||||
},
|
||||
},
|
||||
tun: TunSection {
|
||||
enabled: true,
|
||||
name: "fips0".to_string(),
|
||||
mtu: 1280,
|
||||
},
|
||||
dns: DnsSection {
|
||||
enabled: true,
|
||||
bind_addr: "127.0.0.1".to_string(),
|
||||
},
|
||||
transports: TransportsSection {
|
||||
udp: TransportBind {
|
||||
bind_addr: format!("0.0.0.0:{PUBLISHED_UDP_PORT}"),
|
||||
},
|
||||
tcp: TransportBind {
|
||||
bind_addr: format!("0.0.0.0:{DEFAULT_TCP_PORT}"),
|
||||
},
|
||||
},
|
||||
peers: Vec::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Render the FIPS daemon config. Overwrites any existing file — callers
|
||||
/// re-run this whenever the key or daemon version changes.
|
||||
///
|
||||
/// Node identity comes from the key file on disk; the static peer list stays
|
||||
/// empty because peers are fed dynamically at runtime.
|
||||
pub fn render_config_yaml() -> String {
|
||||
let body = serde_yaml::to_string(&FipsConfig::default())
|
||||
.expect("FipsConfig is a plain struct tree and cannot fail to serialise");
|
||||
format!("{CONFIG_HEADER}{body}")
|
||||
}
|
||||
|
||||
/// Install the local FIPS key + rendered config into `/etc/fips/`.
|
||||
/// Requires the seed-derived key to already exist at `identity_dir/fips_key`.
|
||||
pub async fn install(identity_dir: &Path) -> Result<()> {
|
||||
let src_key = identity_dir.join("fips_key");
|
||||
let src_pub = identity_dir.join("fips_key.pub");
|
||||
if !src_key.exists() {
|
||||
anyhow::bail!(
|
||||
"FIPS key not materialised at {} — run seed onboarding first",
|
||||
src_key.display()
|
||||
);
|
||||
}
|
||||
|
||||
// Ensure /etc/fips exists with mode 0755.
|
||||
sudo_install_dir("/etc/fips").await?;
|
||||
|
||||
// Render + write the yaml via a staging file the archipelago user owns,
|
||||
// then `sudo install` it into place so we never need to write to
|
||||
// /etc directly.
|
||||
let yaml = render_config_yaml();
|
||||
let stage = std::env::temp_dir().join(format!("fips-{}.yaml", std::process::id()));
|
||||
tokio::fs::write(&stage, yaml)
|
||||
.await
|
||||
.context("Failed to stage fips.yaml")?;
|
||||
let install_result = sudo_install_file(&stage, DAEMON_CONFIG_PATH, "0644").await;
|
||||
let _ = tokio::fs::remove_file(&stage).await;
|
||||
install_result?;
|
||||
|
||||
// The release-hardening firewall (/etc/fips/fips.nft, provisioned
|
||||
// out-of-band) default-denies inbound on fips0 — without an explicit
|
||||
// allowance the node's web UI is unreachable over the mesh (phones got
|
||||
// RST on :80 with a healthy session; root-caused 2026-07-26 on
|
||||
// a test node). Ship the allowance as a fips.d drop-in on every
|
||||
// install/upgrade so no node ever regresses to a UI-less mesh.
|
||||
sudo_install_dir("/etc/fips/fips.d").await?;
|
||||
// PEER_PORT (5679) carries ALL federation sync, cloud browse/download,
|
||||
// mesh envelopes, DWN and invoices. It was missing from this allowlist
|
||||
// while the comment claimed "web UI + peer API" — so every hardened
|
||||
// node silently dropped peers' FIPS dials at the firewall and the whole
|
||||
// fleet fell back to Tor (root-caused live 2026-07-27: 28k drops on
|
||||
// .198's counter; :5679 answered in 0.35s once the rule was inserted).
|
||||
let dropin = format!(
|
||||
"# Written by archipelago on every daemon config install.\n\
|
||||
# Allows the web UI + peer API through the fips0\n\
|
||||
# default-deny inbound baseline (fips.nft).\n\
|
||||
tcp dport 80 accept\n\
|
||||
tcp dport 8443 accept\n\
|
||||
tcp dport {peer_port} accept\n",
|
||||
peer_port = crate::fips::dial::PEER_PORT
|
||||
);
|
||||
let nft_stage = std::env::temp_dir().join(format!("fips-webui-{}.nft", std::process::id()));
|
||||
tokio::fs::write(&nft_stage, dropin)
|
||||
.await
|
||||
.context("Failed to stage web-ui nft drop-in")?;
|
||||
let nft_install = sudo_install_file(&nft_stage, "/etc/fips/fips.d/80-web-ui.nft", "0644").await;
|
||||
let _ = tokio::fs::remove_file(&nft_stage).await;
|
||||
nft_install?;
|
||||
|
||||
// App launch ports: the companion opens catalog apps by direct port
|
||||
// over the mesh. Ports of apps that aren't installed have no listener,
|
||||
// so the allowance is inert until an app exists to answer.
|
||||
let port_list = super::app_ports::APP_LAUNCH_PORTS
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect::<Vec<_>>()
|
||||
.join(", ");
|
||||
let app_dropin = format!(
|
||||
"# Written by archipelago on every daemon config install.\n\
|
||||
# Catalog app launch ports (web UIs) allowed through the fips0\n\
|
||||
# default-deny inbound baseline. Service/RPC ports stay closed.\n\
|
||||
tcp dport {{ {port_list} }} accept\n"
|
||||
);
|
||||
let app_stage = std::env::temp_dir().join(format!("fips-appports-{}.nft", std::process::id()));
|
||||
tokio::fs::write(&app_stage, app_dropin)
|
||||
.await
|
||||
.context("Failed to stage app-ports nft drop-in")?;
|
||||
let app_install =
|
||||
sudo_install_file(&app_stage, "/etc/fips/fips.d/85-app-ports.nft", "0644").await;
|
||||
let _ = tokio::fs::remove_file(&app_stage).await;
|
||||
app_install?;
|
||||
// Make the allowance live immediately; a no-op error when the
|
||||
// hardening baseline isn't installed on this node yet.
|
||||
if tokio::fs::try_exists("/etc/fips/fips.nft")
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
{
|
||||
match Command::new("sudo")
|
||||
.args(["nft", "-f", "/etc/fips/fips.nft"])
|
||||
.output()
|
||||
.await
|
||||
{
|
||||
Ok(out) if !out.status.success() => tracing::warn!(
|
||||
"nft reload after web-ui drop-in failed: {}",
|
||||
String::from_utf8_lossy(&out.stderr).trim()
|
||||
),
|
||||
Err(e) => tracing::warn!("nft reload after web-ui drop-in failed: {e}"),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
sudo_install_file(&src_key, DAEMON_KEY_PATH, "0600").await?;
|
||||
// Heal a legacy fips_key.pub that was written as bech32 npub text
|
||||
// (pre-fix identity::write_fips_key_from_seed did this). Upstream
|
||||
// fips expects 32 raw bytes; a text file silently passes through
|
||||
// and then the daemon can't identify itself to peers. This
|
||||
// rewrites the source file in place with the correct binary form
|
||||
// derived from fips_key before staging it to /etc/fips/fips.pub.
|
||||
normalize_pub_file(&src_key, &src_pub).await?;
|
||||
sudo_install_file(&src_pub, DAEMON_PUB_PATH, "0644").await?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Ensure `fips_key.pub` is 32 raw bytes. If it's a bech32 npub text
|
||||
/// file (from the pre-fix writer), decode it and rewrite in place. If
|
||||
/// the file is missing or its content doesn't match either format,
|
||||
/// re-derive the public key from `fips_key` and write that.
|
||||
pub async fn normalize_pub_file(key_path: &Path, pub_path: &Path) -> Result<()> {
|
||||
// Happy path: already 32 raw bytes.
|
||||
if let Ok(bytes) = tokio::fs::read(pub_path).await {
|
||||
if bytes.len() == 32 {
|
||||
return Ok(());
|
||||
}
|
||||
// bech32 npub text from the pre-fix writer: decode in place.
|
||||
if let Ok(s) = std::str::from_utf8(&bytes) {
|
||||
let trimmed = s.trim();
|
||||
if trimmed.starts_with("npub1") {
|
||||
if let Ok(pk) = nostr_sdk::PublicKey::parse(trimmed) {
|
||||
let raw: [u8; 32] = pk.to_bytes();
|
||||
tokio::fs::write(pub_path, raw)
|
||||
.await
|
||||
.context("rewriting fips_key.pub as 32 raw bytes")?;
|
||||
tracing::info!(
|
||||
"Migrated legacy bech32 fips_key.pub to raw-byte form at {}",
|
||||
pub_path.display()
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback: no pub file, or unreadable format. Re-derive from the
|
||||
// private key file (already validated by load_fips_keys).
|
||||
let secret_bytes = tokio::fs::read(key_path)
|
||||
.await
|
||||
.with_context(|| format!("read {} to derive public", key_path.display()))?;
|
||||
let text = std::str::from_utf8(&secret_bytes)
|
||||
.context("fips_key is not UTF-8 — can't derive public")?;
|
||||
let secret = nostr_sdk::SecretKey::parse(text.trim())
|
||||
.context("fips_key not parseable as bech32 nsec")?;
|
||||
let keys = nostr_sdk::Keys::new(secret);
|
||||
let raw: [u8; 32] = keys.public_key().to_bytes();
|
||||
tokio::fs::write(pub_path, raw)
|
||||
.await
|
||||
.context("writing re-derived fips_key.pub")?;
|
||||
tracing::info!("Re-derived fips_key.pub from fips_key");
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn sudo_install_dir(path: &str) -> Result<()> {
|
||||
let out = Command::new("sudo")
|
||||
.args(["install", "-d", "-m", "0755", path])
|
||||
.output()
|
||||
.await
|
||||
.with_context(|| format!("sudo install -d {}", path))?;
|
||||
if !out.status.success() {
|
||||
anyhow::bail!(
|
||||
"sudo install -d {}: {}",
|
||||
path,
|
||||
String::from_utf8_lossy(&out.stderr).trim()
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn sudo_install_file(src: &Path, dest: &str, mode: &str) -> Result<()> {
|
||||
let out = Command::new("sudo")
|
||||
.args([
|
||||
"install",
|
||||
"-m",
|
||||
mode,
|
||||
src.to_str().context("Non-UTF8 source path")?,
|
||||
dest,
|
||||
])
|
||||
.output()
|
||||
.await
|
||||
.with_context(|| format!("sudo install {} -> {}", src.display(), dest))?;
|
||||
if !out.status.success() {
|
||||
anyhow::bail!(
|
||||
"sudo install {} -> {}: {}",
|
||||
src.display(),
|
||||
dest,
|
||||
String::from_utf8_lossy(&out.stderr).trim()
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_rendered_yaml_matches_upstream_schema() {
|
||||
let yaml = render_config_yaml();
|
||||
assert!(yaml.contains("persistent: true"));
|
||||
assert!(yaml.contains(&format!("0.0.0.0:{}", PUBLISHED_UDP_PORT)));
|
||||
assert!(yaml.contains(&format!("0.0.0.0:{}", DEFAULT_TCP_PORT)));
|
||||
assert!(yaml.contains("udp:"));
|
||||
assert!(yaml.contains("tcp:"));
|
||||
assert!(yaml.contains("tun:"));
|
||||
assert!(yaml.contains("name: fips0"));
|
||||
// Upstream fips dropped the `tor:` transport variant; archipelago
|
||||
// handles Tor fallback itself. Make sure we didn't regress.
|
||||
assert!(!yaml.contains("tor:"));
|
||||
}
|
||||
|
||||
/// Exact-output snapshot. Upstream's config structs are
|
||||
/// `deny_unknown_fields`, so an accidental key rename/addition means the
|
||||
/// daemon won't start. Pinning the full rendering makes any such change
|
||||
/// fail here — where it's cheap — instead of on a node after an upgrade.
|
||||
/// If this fails, re-verify against the upstream schema before updating it.
|
||||
#[test]
|
||||
fn test_rendered_yaml_exact_snapshot() {
|
||||
let expected = "\
|
||||
# Generated by archipelago — do not edit by hand.
|
||||
# Regenerated on every key change and daemon upgrade.
|
||||
node:
|
||||
identity:
|
||||
persistent: true
|
||||
discovery:
|
||||
timeout_secs: 5
|
||||
backoff_base_secs: 1
|
||||
backoff_max_secs: 30
|
||||
retry_interval_secs: 2
|
||||
max_attempts: 3
|
||||
lan:
|
||||
enabled: true
|
||||
retry:
|
||||
base_interval_secs: 1
|
||||
max_backoff_secs: 30
|
||||
max_retries: 30
|
||||
rate_limit:
|
||||
handshake_resend_interval_ms: 400
|
||||
handshake_resend_backoff: 1.5
|
||||
handshake_max_resends: 10
|
||||
tun:
|
||||
enabled: true
|
||||
name: fips0
|
||||
mtu: 1280
|
||||
dns:
|
||||
enabled: true
|
||||
bind_addr: 127.0.0.1
|
||||
transports:
|
||||
udp:
|
||||
bind_addr: 0.0.0.0:2121
|
||||
tcp:
|
||||
bind_addr: 0.0.0.0:8443
|
||||
peers: []
|
||||
";
|
||||
assert_eq!(render_config_yaml(), expected);
|
||||
}
|
||||
|
||||
/// The rendered config must parse as YAML and carry the mDNS opt-in at the
|
||||
/// exact path upstream reads (`node.discovery.lan.enabled`) — a typo there
|
||||
/// would silently leave LAN discovery off rather than erroring.
|
||||
#[test]
|
||||
fn test_lan_discovery_enabled_at_upstream_path() {
|
||||
let yaml = render_config_yaml();
|
||||
let parsed: serde_yaml::Value = serde_yaml::from_str(&yaml).expect("renders valid YAML");
|
||||
assert_eq!(
|
||||
parsed["node"]["discovery"]["lan"]["enabled"],
|
||||
serde_yaml::Value::Bool(true),
|
||||
);
|
||||
}
|
||||
|
||||
/// Rendering is deterministic: the startup drift check in server.rs compares
|
||||
/// the freshly rendered config against what's on disk, so any instability
|
||||
/// here would cause an endless reinstall+restart loop of the daemon.
|
||||
#[test]
|
||||
fn test_render_is_deterministic() {
|
||||
assert_eq!(render_config_yaml(), render_config_yaml());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_install_refuses_when_key_missing() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let err = install(dir.path()).await.unwrap_err();
|
||||
assert!(err.to_string().contains("FIPS key not materialised"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,762 @@
|
||||
//! Dial peers over the FIPS mesh.
|
||||
//!
|
||||
//! The FIPS daemon exposes a local DNS resolver on `127.0.0.1:5354` that
|
||||
//! answers AAAA queries for `<npub>.fips` with the peer's ULA address on
|
||||
//! the `fips0` TUN. Once resolved we speak plain HTTP to the peer on
|
||||
//! [`PEER_PORT`] — the same port `127.0.0.1:5678` where the archipelago
|
||||
//! backend serves the existing signed peer-to-peer endpoints
|
||||
//! (`/rpc/v1`, `/archipelago/node-message`, `/content/{id}`, …). The
|
||||
//! server-side binding to the `fips0` address is handled in `server.rs`.
|
||||
//!
|
||||
//! The module is deliberately dependency-free for DNS — one packet in,
|
||||
//! one packet out, standard RFC 1035 wire format — to avoid pulling
|
||||
//! hickory-resolver's transitive tree for a single AAAA query.
|
||||
//!
|
||||
//! On any failure (daemon down, peer not in the identity cache, TUN
|
||||
//! unreachable) callers fall back to the Tor transport.
|
||||
//!
|
||||
//! # Examples
|
||||
//! ```ignore
|
||||
//! let base = crate::fips::dial::peer_base_url("npub1…").await?;
|
||||
//! // base = "http://[fd9d:…]:5678"
|
||||
//! let client = crate::fips::dial::client();
|
||||
//! let resp = client.get(format!("{}/content/abc", base)).send().await?;
|
||||
//! ```
|
||||
#![allow(dead_code)]
|
||||
|
||||
use super::telemetry::{self, FallbackReason};
|
||||
use anyhow::{Context, Result};
|
||||
use std::net::{IpAddr, Ipv6Addr};
|
||||
use std::time::Duration;
|
||||
use tokio::net::UdpSocket;
|
||||
|
||||
/// Port the archipelago backend listens on for FIPS peer-to-peer traffic.
|
||||
/// Separate from the localhost-only internal port (5678) so the per-listener
|
||||
/// path filter can restrict the exposed surface.
|
||||
pub const PEER_PORT: u16 = 5679;
|
||||
|
||||
/// Whether a FIPS-side HTTP status should trigger a fall-back to Tor in
|
||||
/// `Auto` mode. A `404` over FIPS often means the peer's mesh listener
|
||||
/// doesn't expose that path (e.g. a peer on an older build with a stricter
|
||||
/// `is_peer_allowed_path`), and `5xx` is a server-side error — both are
|
||||
/// worth retrying over Tor, which reaches a different (less-filtered) route.
|
||||
/// Success, redirects, and other 4xx (auth / bad request) are authoritative
|
||||
/// and are returned as-is so we neither mask real errors nor double latency.
|
||||
fn fips_should_fall_back(status: reqwest::StatusCode) -> bool {
|
||||
status == reqwest::StatusCode::NOT_FOUND || status.is_server_error()
|
||||
}
|
||||
|
||||
/// DNS suffix appended to a peer's bech32 npub.
|
||||
pub const FIPS_DNS_SUFFIX: &str = "fips";
|
||||
|
||||
/// FIPS daemon's local DNS resolver.
|
||||
pub const FIPS_DNS_ADDR: &str = "127.0.0.1:5354";
|
||||
|
||||
/// Short DNS query timeout — FIPS DNS is a local process; a slow answer
|
||||
/// almost certainly means the daemon is gone.
|
||||
const DNS_TIMEOUT: Duration = Duration::from_secs(2);
|
||||
|
||||
/// DNS AAAA query type.
|
||||
const QTYPE_AAAA: u16 = 28;
|
||||
|
||||
/// DNS IN class.
|
||||
const QCLASS_IN: u16 = 1;
|
||||
|
||||
/// Resolve a peer's bech32 npub to their `fips0` ULA address via the local
|
||||
/// FIPS DNS resolver.
|
||||
pub async fn resolve(npub: &str) -> Result<Ipv6Addr> {
|
||||
let sock = UdpSocket::bind("127.0.0.1:0")
|
||||
.await
|
||||
.context("bind UDP socket for FIPS DNS")?;
|
||||
sock.connect(FIPS_DNS_ADDR)
|
||||
.await
|
||||
.context("connect to FIPS DNS")?;
|
||||
|
||||
// KEY-05: source named. A 2-byte DNS transaction id, not key material, so it
|
||||
// is drawn unguarded — an "all bytes identical" predicate on two bytes
|
||||
// false-positives once in 256, which would be worse than the defect.
|
||||
let id: u16 = rand::RngCore::next_u32(&mut rand::rngs::OsRng) as u16;
|
||||
let query = encode_query(id, npub)?;
|
||||
tokio::time::timeout(DNS_TIMEOUT, sock.send(&query))
|
||||
.await
|
||||
.context("FIPS DNS query timed out on send")?
|
||||
.context("FIPS DNS send")?;
|
||||
|
||||
let mut buf = [0u8; 512];
|
||||
let n = tokio::time::timeout(DNS_TIMEOUT, sock.recv(&mut buf))
|
||||
.await
|
||||
.context("FIPS DNS query timed out on recv")?
|
||||
.context("FIPS DNS recv")?;
|
||||
|
||||
decode_response(id, &buf[..n], npub)
|
||||
}
|
||||
|
||||
/// Return a peer's base URL on the FIPS overlay, e.g. `http://[fd9d:…]:5678`.
|
||||
pub async fn peer_base_url(npub: &str) -> Result<String> {
|
||||
let ip = resolve(npub).await?;
|
||||
Ok(format!("http://[{}]:{}", ip, PEER_PORT))
|
||||
}
|
||||
|
||||
/// Build an HTTP client tuned for FIPS peer-to-peer dialing. No proxy.
|
||||
/// `connect_timeout` is generous enough to let NAT hole-punching complete on
|
||||
/// the first dial (FIPS is UDP hole-punched; the path often isn't established
|
||||
/// until the first packets flow), so a reachable-but-cold peer isn't abandoned
|
||||
/// to Tor prematurely. Reliability over latency — FIPS is the preferred path.
|
||||
pub fn client() -> reqwest::Client {
|
||||
client_with_timeout(Duration::from_secs(20))
|
||||
}
|
||||
|
||||
/// FIPS client with a caller-chosen overall request timeout. The static 20s
|
||||
/// `client()` budget is fine for catalog browses and short calls, but a large
|
||||
/// content download (#38) needs the per-request timeout the caller asked for —
|
||||
/// otherwise a 178MB transfer is aborted at 20s and the whole download fails
|
||||
/// before the Tor fallback ever gets a chance. The generous `connect_timeout`
|
||||
/// is preserved so a cold hole-punched path still gets time to establish.
|
||||
pub fn client_with_timeout(timeout: Duration) -> reqwest::Client {
|
||||
reqwest::Client::builder()
|
||||
.timeout(timeout)
|
||||
.connect_timeout(Duration::from_secs(8))
|
||||
.user_agent("archipelago-fips/1")
|
||||
.build()
|
||||
.expect("static reqwest client config")
|
||||
}
|
||||
|
||||
/// Send a FIPS request with ONE retry on a connect/timeout error.
|
||||
///
|
||||
/// The first dial to a peer typically triggers NAT hole-punching and can time
|
||||
/// out before the overlay path is established; a quick retry then lands on the
|
||||
/// now-warm path. Without this, a single cold-path failure drops the call to
|
||||
/// Tor even though the peer is FIPS-reachable — the main reason FIPS "isn't
|
||||
/// robust". Only connect/timeout errors are retried (a real HTTP response,
|
||||
/// including 4xx/5xx, is returned as-is for the caller to interpret).
|
||||
async fn send_with_retry(rb: reqwest::RequestBuilder) -> Result<reqwest::Response, reqwest::Error> {
|
||||
let retry = rb.try_clone();
|
||||
match rb.send().await {
|
||||
Ok(resp) => Ok(resp),
|
||||
Err(e) if (e.is_connect() || e.is_timeout()) && retry.is_some() => {
|
||||
// Brief pause so the hole-punch packets from the first attempt can
|
||||
// traverse before we re-dial onto the warmed path.
|
||||
tokio::time::sleep(Duration::from_millis(600)).await;
|
||||
retry.expect("retry builder present").send().await
|
||||
}
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
}
|
||||
|
||||
/// Proactively warm the hole-punched FIPS path to a peer: resolve its overlay
|
||||
/// address and open a short connection to its peer listener. Hole-punched
|
||||
/// paths and NAT mappings go cold after ~30-60s of no traffic, after which the
|
||||
/// next real dial pays the full re-punch cost and often falls back to Tor.
|
||||
/// Keeping the path warm is what makes FIPS the transport that actually gets
|
||||
/// used. Best-effort: any error (peer offline, UDP blocked) is ignored — the
|
||||
/// connection attempt itself is what re-punches and refreshes the path.
|
||||
pub async fn warm_path(npub: &str) {
|
||||
if !is_service_active().await {
|
||||
return;
|
||||
}
|
||||
warm_path_unchecked(npub).await
|
||||
}
|
||||
|
||||
/// [`warm_path`] without the service-active check — for callers (the warm
|
||||
/// tick) that already verified the daemon once for the whole batch.
|
||||
pub async fn warm_path_unchecked(npub: &str) {
|
||||
let Ok(base) = peer_base_url(npub).await else {
|
||||
return;
|
||||
};
|
||||
let c = client();
|
||||
// The response status is irrelevant; establishing the connection warms it.
|
||||
let _ = tokio::time::timeout(Duration::from_secs(8), c.get(&base).send()).await;
|
||||
}
|
||||
|
||||
// ── DNS wire-format helpers ─────────────────────────────────────────────
|
||||
|
||||
fn encode_query(id: u16, npub: &str) -> Result<Vec<u8>> {
|
||||
let mut out = Vec::with_capacity(64 + npub.len());
|
||||
// Header
|
||||
out.extend_from_slice(&id.to_be_bytes());
|
||||
out.extend_from_slice(&0x0100u16.to_be_bytes()); // RD=1, std query
|
||||
out.extend_from_slice(&1u16.to_be_bytes()); // QDCOUNT
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // ANCOUNT
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // NSCOUNT
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // ARCOUNT
|
||||
|
||||
// QNAME — two labels: "<npub>" and "fips".
|
||||
encode_label(&mut out, npub)?;
|
||||
encode_label(&mut out, FIPS_DNS_SUFFIX)?;
|
||||
out.push(0); // root
|
||||
// QTYPE + QCLASS
|
||||
out.extend_from_slice(&QTYPE_AAAA.to_be_bytes());
|
||||
out.extend_from_slice(&QCLASS_IN.to_be_bytes());
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
fn encode_label(out: &mut Vec<u8>, label: &str) -> Result<()> {
|
||||
if label.is_empty() || label.len() > 63 {
|
||||
anyhow::bail!("invalid DNS label length: {}", label.len());
|
||||
}
|
||||
out.push(label.len() as u8);
|
||||
out.extend_from_slice(label.as_bytes());
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn decode_response(expected_id: u16, buf: &[u8], npub: &str) -> Result<Ipv6Addr> {
|
||||
if buf.len() < 12 {
|
||||
anyhow::bail!("DNS response too short");
|
||||
}
|
||||
let id = u16::from_be_bytes([buf[0], buf[1]]);
|
||||
if id != expected_id {
|
||||
anyhow::bail!("DNS response id mismatch");
|
||||
}
|
||||
let rcode = buf[3] & 0x0F;
|
||||
if rcode != 0 {
|
||||
anyhow::bail!("DNS rcode {} resolving {}.fips", rcode, npub);
|
||||
}
|
||||
let qdcount = u16::from_be_bytes([buf[4], buf[5]]) as usize;
|
||||
let ancount = u16::from_be_bytes([buf[6], buf[7]]) as usize;
|
||||
if ancount == 0 {
|
||||
anyhow::bail!("no AAAA record for {}.fips", npub);
|
||||
}
|
||||
|
||||
let mut pos = 12;
|
||||
// Skip question section(s)
|
||||
for _ in 0..qdcount {
|
||||
pos = skip_name(buf, pos)?;
|
||||
pos = pos
|
||||
.checked_add(4)
|
||||
.ok_or_else(|| anyhow::anyhow!("qsection overflow"))?;
|
||||
if pos > buf.len() {
|
||||
anyhow::bail!("qsection past end");
|
||||
}
|
||||
}
|
||||
|
||||
// Walk answers; return the first valid AAAA rdata.
|
||||
for _ in 0..ancount {
|
||||
pos = skip_name(buf, pos)?;
|
||||
if pos + 10 > buf.len() {
|
||||
anyhow::bail!("answer RR past end");
|
||||
}
|
||||
let rtype = u16::from_be_bytes([buf[pos], buf[pos + 1]]);
|
||||
let rclass = u16::from_be_bytes([buf[pos + 2], buf[pos + 3]]);
|
||||
let rdlength = u16::from_be_bytes([buf[pos + 8], buf[pos + 9]]) as usize;
|
||||
pos += 10;
|
||||
if pos + rdlength > buf.len() {
|
||||
anyhow::bail!("rdata past end");
|
||||
}
|
||||
if rtype == QTYPE_AAAA && rclass == QCLASS_IN && rdlength == 16 {
|
||||
let mut octets = [0u8; 16];
|
||||
octets.copy_from_slice(&buf[pos..pos + 16]);
|
||||
return Ok(Ipv6Addr::from(octets));
|
||||
}
|
||||
pos += rdlength;
|
||||
}
|
||||
anyhow::bail!("no AAAA answer for {}.fips", npub)
|
||||
}
|
||||
|
||||
/// Advance past a DNS name (handles compressed pointers). Returns the
|
||||
/// position immediately after the name.
|
||||
fn skip_name(buf: &[u8], mut pos: usize) -> Result<usize> {
|
||||
loop {
|
||||
if pos >= buf.len() {
|
||||
anyhow::bail!("name past end");
|
||||
}
|
||||
let len = buf[pos];
|
||||
if len == 0 {
|
||||
return Ok(pos + 1);
|
||||
}
|
||||
if len & 0xC0 == 0xC0 {
|
||||
// Compressed pointer — 2 bytes total, no further labels.
|
||||
if pos + 2 > buf.len() {
|
||||
anyhow::bail!("pointer past end");
|
||||
}
|
||||
return Ok(pos + 2);
|
||||
}
|
||||
if len & 0xC0 != 0 {
|
||||
anyhow::bail!("reserved label type");
|
||||
}
|
||||
pos = pos
|
||||
.checked_add(1 + len as usize)
|
||||
.ok_or_else(|| anyhow::anyhow!("name overflow"))?;
|
||||
}
|
||||
}
|
||||
|
||||
/// Treat `IpAddr::V6` as the raw address for ergonomic callers.
|
||||
pub fn as_ip_addr(v6: Ipv6Addr) -> IpAddr {
|
||||
IpAddr::V6(v6)
|
||||
}
|
||||
|
||||
// ── High-level peer request helpers ────────────────────────────────────
|
||||
|
||||
/// TTL for the [`is_service_active`] cache. Every FIPS dial attempt and
|
||||
/// every warm-tick peer used to spawn up to two `systemctl` subprocesses;
|
||||
/// service state changes on human timescales, so 10s staleness is free.
|
||||
const SERVICE_ACTIVE_TTL_MS: u64 = 10_000;
|
||||
static SERVICE_ACTIVE: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||
static SERVICE_PROBED_AT_MS: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0);
|
||||
|
||||
/// Quick poll: is the FIPS daemon (archipelago-supervised OR upstream)
|
||||
/// currently `systemctl is-active`? Cached for [`SERVICE_ACTIVE_TTL_MS`];
|
||||
/// concurrent refreshes are harmless (idempotent probe, last write wins).
|
||||
pub async fn is_service_active() -> bool {
|
||||
use std::sync::atomic::Ordering;
|
||||
let now_ms = std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|d| d.as_millis() as u64)
|
||||
.unwrap_or(0);
|
||||
let probed_at = SERVICE_PROBED_AT_MS.load(Ordering::Relaxed);
|
||||
if probed_at != 0 && now_ms.saturating_sub(probed_at) < SERVICE_ACTIVE_TTL_MS {
|
||||
return SERVICE_ACTIVE.load(Ordering::Relaxed);
|
||||
}
|
||||
let mut active = false;
|
||||
for unit in [
|
||||
crate::fips::SERVICE_UNIT,
|
||||
crate::fips::UPSTREAM_SERVICE_UNIT,
|
||||
] {
|
||||
if crate::fips::service::unit_state(unit).await == "active" {
|
||||
active = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
SERVICE_ACTIVE.store(active, Ordering::Relaxed);
|
||||
SERVICE_PROBED_AT_MS.store(now_ms, Ordering::Relaxed);
|
||||
active
|
||||
}
|
||||
|
||||
/// Builder for a peer request that may be sent over FIPS (preferred) or
|
||||
/// Tor (fallback). The call sites migrating off direct-Tor dialing build
|
||||
/// one of these and call [`send_json`] / [`send_get`]; the helper handles
|
||||
/// dial, timeout, fallback, and cross-transport auth headers.
|
||||
///
|
||||
/// The optional `service` field ties the request to a user-configurable
|
||||
/// transport preference (see `crate::settings::transport`). Leaving it
|
||||
/// unset picks Auto (FIPS preferred, Tor fallback) — the same default as
|
||||
/// before the Settings UI landed.
|
||||
pub struct PeerRequest<'a> {
|
||||
pub fips_npub: Option<&'a str>,
|
||||
pub onion_host: &'a str,
|
||||
pub path: &'a str,
|
||||
pub headers: Vec<(&'a str, String)>,
|
||||
pub timeout: std::time::Duration,
|
||||
/// Optional shorter cap on the FIPS *attempt* only. When set, a cold or hung
|
||||
/// FIPS overlay fails fast within this budget so the Tor fallback still gets
|
||||
/// its full `timeout` — without it, a stuck FIPS dial can consume the whole
|
||||
/// caller budget (e.g. a 60s frontend RPC) and the request "times out" even
|
||||
/// though Tor would have answered (#6, the Pay-with-QR invoice request).
|
||||
/// `None` keeps the legacy behavior (FIPS uses the full `timeout`), which a
|
||||
/// large content download needs so its long FIPS transfer isn't truncated.
|
||||
pub fips_timeout: Option<std::time::Duration>,
|
||||
pub service: Option<crate::settings::transport::PeerService>,
|
||||
/// When set, the transport that actually served this request is written
|
||||
/// to federation storage (`record_peer_transport`, matched by onion) so
|
||||
/// the per-peer FIPS/Tor badge reflects reality. Opt-in because not
|
||||
/// every caller has a data dir in scope.
|
||||
pub record_data_dir: Option<std::path::PathBuf>,
|
||||
}
|
||||
|
||||
impl<'a> PeerRequest<'a> {
|
||||
pub fn new(fips_npub: Option<&'a str>, onion_host: &'a str, path: &'a str) -> Self {
|
||||
Self {
|
||||
fips_npub,
|
||||
onion_host,
|
||||
path,
|
||||
headers: Vec::new(),
|
||||
timeout: std::time::Duration::from_secs(30),
|
||||
fips_timeout: None,
|
||||
service: None,
|
||||
record_data_dir: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Record the transport that serves this request into federation storage
|
||||
/// (matched by this request's onion host). Best-effort, off the hot path.
|
||||
pub fn record_transport(mut self, data_dir: impl Into<std::path::PathBuf>) -> Self {
|
||||
self.record_data_dir = Some(data_dir.into());
|
||||
self
|
||||
}
|
||||
|
||||
fn spawn_record(&self, kind: crate::transport::TransportKind) {
|
||||
if let Some(dir) = &self.record_data_dir {
|
||||
let dir = dir.clone();
|
||||
let onion = self.onion_host.to_string();
|
||||
let transport = kind.to_string();
|
||||
tokio::spawn(async move {
|
||||
let _ =
|
||||
crate::federation::record_peer_transport(&dir, None, Some(&onion), &transport)
|
||||
.await;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/// Cap the FIPS attempt to a shorter budget than the overall `timeout`, so a
|
||||
/// cold/hung overlay path fails fast and the Tor fallback keeps its full
|
||||
/// budget. Use on short request/response calls (invoice, status); leave
|
||||
/// unset for large downloads that legitimately need a long FIPS transfer.
|
||||
pub fn fips_timeout(mut self, t: std::time::Duration) -> Self {
|
||||
self.fips_timeout = Some(t);
|
||||
self
|
||||
}
|
||||
|
||||
/// Timeout to apply to the FIPS attempt — the explicit cap if set, else the
|
||||
/// overall request timeout.
|
||||
fn fips_attempt_timeout(&self) -> std::time::Duration {
|
||||
self.fips_timeout.unwrap_or(self.timeout)
|
||||
}
|
||||
|
||||
/// Tie this request to a user-configurable service preference. If
|
||||
/// the user has set that service to `Fips` or `Tor`, the builder
|
||||
/// respects it.
|
||||
pub fn service(mut self, s: crate::settings::transport::PeerService) -> Self {
|
||||
self.service = Some(s);
|
||||
self
|
||||
}
|
||||
|
||||
pub fn header(mut self, name: &'a str, value: impl Into<String>) -> Self {
|
||||
self.headers.push((name, value.into()));
|
||||
self
|
||||
}
|
||||
|
||||
pub fn timeout(mut self, t: std::time::Duration) -> Self {
|
||||
self.timeout = t;
|
||||
self
|
||||
}
|
||||
|
||||
/// Resolved preference: user setting if `service` was set, else Auto.
|
||||
async fn preference(&self) -> crate::settings::transport::TransportPref {
|
||||
match self.service {
|
||||
Some(s) => crate::settings::transport::get(s).await,
|
||||
None => crate::settings::transport::TransportPref::Auto,
|
||||
}
|
||||
}
|
||||
|
||||
/// POST a JSON body. Returns the `reqwest::Response` — caller decides
|
||||
/// how to interpret the status code.
|
||||
pub async fn send_json<B: serde::Serialize>(
|
||||
&self,
|
||||
body: &B,
|
||||
) -> Result<(reqwest::Response, crate::transport::TransportKind)> {
|
||||
use crate::settings::transport::TransportPref;
|
||||
let pref = self.preference().await;
|
||||
// FIPS-only or Auto: try FIPS first.
|
||||
if matches!(pref, TransportPref::Auto | TransportPref::Fips) {
|
||||
match self.try_fips_post_json(body).await? {
|
||||
Some(resp) => {
|
||||
// Use the FIPS reply unless it's one a Tor retry could
|
||||
// fix (404 path-not-served / 5xx) and we're allowed to
|
||||
// fall back. FIPS-only never falls back.
|
||||
if pref == TransportPref::Fips || !fips_should_fall_back(resp.status()) {
|
||||
telemetry::record_fips_ok();
|
||||
self.spawn_record(crate::transport::TransportKind::Fips);
|
||||
return Ok((resp, crate::transport::TransportKind::Fips));
|
||||
}
|
||||
let reason = if resp.status() == reqwest::StatusCode::NOT_FOUND {
|
||||
FallbackReason::Http404
|
||||
} else {
|
||||
FallbackReason::Http5xx
|
||||
};
|
||||
telemetry::record_fallback(reason);
|
||||
tracing::info!(
|
||||
reason = reason.key(),
|
||||
status = %resp.status(),
|
||||
"FIPS POST {} answered but status triggers Tor fallback",
|
||||
self.path
|
||||
);
|
||||
}
|
||||
None => {
|
||||
if pref == TransportPref::Fips {
|
||||
anyhow::bail!(
|
||||
"User set transport preference to FIPS only, but peer is unreachable over FIPS"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
let resp = self.send_tor_post_json(body).await?;
|
||||
self.spawn_record(crate::transport::TransportKind::Tor);
|
||||
Ok((resp, crate::transport::TransportKind::Tor))
|
||||
}
|
||||
|
||||
/// GET with optional header-based auth.
|
||||
pub async fn send_get(&self) -> Result<(reqwest::Response, crate::transport::TransportKind)> {
|
||||
use crate::settings::transport::TransportPref;
|
||||
let pref = self.preference().await;
|
||||
if matches!(pref, TransportPref::Auto | TransportPref::Fips) {
|
||||
match self.try_fips_get().await? {
|
||||
Some(resp) => {
|
||||
if pref == TransportPref::Fips || !fips_should_fall_back(resp.status()) {
|
||||
telemetry::record_fips_ok();
|
||||
self.spawn_record(crate::transport::TransportKind::Fips);
|
||||
return Ok((resp, crate::transport::TransportKind::Fips));
|
||||
}
|
||||
let reason = if resp.status() == reqwest::StatusCode::NOT_FOUND {
|
||||
FallbackReason::Http404
|
||||
} else {
|
||||
FallbackReason::Http5xx
|
||||
};
|
||||
telemetry::record_fallback(reason);
|
||||
tracing::info!(
|
||||
reason = reason.key(),
|
||||
status = %resp.status(),
|
||||
"FIPS GET {} answered but status triggers Tor fallback",
|
||||
self.path
|
||||
);
|
||||
}
|
||||
None => {
|
||||
if pref == TransportPref::Fips {
|
||||
anyhow::bail!(
|
||||
"User set transport preference to FIPS only, but peer is unreachable over FIPS"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
let resp = self.send_tor_get().await?;
|
||||
self.spawn_record(crate::transport::TransportKind::Tor);
|
||||
Ok((resp, crate::transport::TransportKind::Tor))
|
||||
}
|
||||
|
||||
async fn try_fips_post_json<B: serde::Serialize>(
|
||||
&self,
|
||||
body: &B,
|
||||
) -> Result<Option<reqwest::Response>> {
|
||||
let Some(npub) = self.fips_npub else {
|
||||
telemetry::record_fallback(FallbackReason::NoNpub);
|
||||
return Ok(None);
|
||||
};
|
||||
if !is_service_active().await {
|
||||
telemetry::record_fallback(FallbackReason::ServiceInactive);
|
||||
return Ok(None);
|
||||
}
|
||||
let base = match peer_base_url(npub).await {
|
||||
Ok(b) => b,
|
||||
Err(e) => {
|
||||
telemetry::record_fallback(FallbackReason::DnsFail);
|
||||
tracing::info!(
|
||||
reason = FallbackReason::DnsFail.key(),
|
||||
"FIPS resolve for {} failed: {}, falling back to Tor",
|
||||
npub,
|
||||
e
|
||||
);
|
||||
return Ok(None);
|
||||
}
|
||||
};
|
||||
let url = format!("{}{}", base, self.path);
|
||||
let budget = self.fips_attempt_timeout();
|
||||
// With an explicit fast-fail cap, halve the per-attempt client
|
||||
// timeout so the one-retry path in send_with_retry fits inside the
|
||||
// budget instead of silently doubling it ("fips_timeout(6s)" used
|
||||
// to really mean ~12.6s). Without one (long streaming downloads),
|
||||
// keep the full budget per attempt — the client timeout also
|
||||
// governs body streaming and must not truncate a real transfer.
|
||||
let per_attempt = if self.fips_timeout.is_some() {
|
||||
budget / 2
|
||||
} else {
|
||||
budget
|
||||
};
|
||||
let c = client_with_timeout(per_attempt);
|
||||
let mut rb = c.post(&url).json(body);
|
||||
for (k, v) in &self.headers {
|
||||
rb = rb.header(*k, v);
|
||||
}
|
||||
match tokio::time::timeout(budget, send_with_retry(rb)).await {
|
||||
Ok(Ok(r)) => Ok(Some(r)),
|
||||
Ok(Err(e)) => {
|
||||
telemetry::record_fallback(FallbackReason::ConnectFail);
|
||||
tracing::info!(
|
||||
reason = FallbackReason::ConnectFail.key(),
|
||||
"FIPS POST {} failed after retry: {}, falling back to Tor",
|
||||
url,
|
||||
e
|
||||
);
|
||||
Ok(None)
|
||||
}
|
||||
Err(_) => {
|
||||
telemetry::record_fallback(FallbackReason::ConnectFail);
|
||||
tracing::info!(
|
||||
reason = FallbackReason::ConnectFail.key(),
|
||||
"FIPS POST {} exceeded attempt budget {:?}, falling back to Tor",
|
||||
url,
|
||||
budget
|
||||
);
|
||||
Ok(None)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn try_fips_get(&self) -> Result<Option<reqwest::Response>> {
|
||||
let Some(npub) = self.fips_npub else {
|
||||
telemetry::record_fallback(FallbackReason::NoNpub);
|
||||
return Ok(None);
|
||||
};
|
||||
if !is_service_active().await {
|
||||
telemetry::record_fallback(FallbackReason::ServiceInactive);
|
||||
return Ok(None);
|
||||
}
|
||||
let base = match peer_base_url(npub).await {
|
||||
Ok(b) => b,
|
||||
Err(e) => {
|
||||
telemetry::record_fallback(FallbackReason::DnsFail);
|
||||
tracing::info!(
|
||||
reason = FallbackReason::DnsFail.key(),
|
||||
"FIPS resolve for {} failed: {}, falling back to Tor",
|
||||
npub,
|
||||
e
|
||||
);
|
||||
return Ok(None);
|
||||
}
|
||||
};
|
||||
let url = format!("{}{}", base, self.path);
|
||||
let budget = self.fips_attempt_timeout();
|
||||
// Same budget discipline as the POST path: halve per attempt only
|
||||
// under an explicit fast-fail cap; hard-cap the retry sequence.
|
||||
let per_attempt = if self.fips_timeout.is_some() {
|
||||
budget / 2
|
||||
} else {
|
||||
budget
|
||||
};
|
||||
let c = client_with_timeout(per_attempt);
|
||||
let mut rb = c.get(&url);
|
||||
for (k, v) in &self.headers {
|
||||
rb = rb.header(*k, v);
|
||||
}
|
||||
match tokio::time::timeout(budget, send_with_retry(rb)).await {
|
||||
Ok(Ok(r)) => Ok(Some(r)),
|
||||
Ok(Err(e)) => {
|
||||
telemetry::record_fallback(FallbackReason::ConnectFail);
|
||||
tracing::info!(
|
||||
reason = FallbackReason::ConnectFail.key(),
|
||||
"FIPS GET {} failed after retry: {}, falling back to Tor",
|
||||
url,
|
||||
e
|
||||
);
|
||||
Ok(None)
|
||||
}
|
||||
Err(_) => {
|
||||
telemetry::record_fallback(FallbackReason::ConnectFail);
|
||||
tracing::info!(
|
||||
reason = FallbackReason::ConnectFail.key(),
|
||||
"FIPS GET {} exceeded attempt budget {:?}, falling back to Tor",
|
||||
url,
|
||||
budget
|
||||
);
|
||||
Ok(None)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn send_tor_post_json<B: serde::Serialize>(&self, body: &B) -> Result<reqwest::Response> {
|
||||
let url = self.tor_url();
|
||||
let client = self.tor_client()?;
|
||||
let mut rb = client.post(&url).json(body);
|
||||
for (k, v) in &self.headers {
|
||||
rb = rb.header(*k, v);
|
||||
}
|
||||
rb.send().await.with_context(|| format!("Tor POST {}", url))
|
||||
}
|
||||
|
||||
async fn send_tor_get(&self) -> Result<reqwest::Response> {
|
||||
let url = self.tor_url();
|
||||
let client = self.tor_client()?;
|
||||
let mut rb = client.get(&url);
|
||||
for (k, v) in &self.headers {
|
||||
rb = rb.header(*k, v);
|
||||
}
|
||||
rb.send().await.with_context(|| format!("Tor GET {}", url))
|
||||
}
|
||||
|
||||
fn tor_url(&self) -> String {
|
||||
let host = if self.onion_host.ends_with(".onion") {
|
||||
self.onion_host.to_string()
|
||||
} else {
|
||||
format!("{}.onion", self.onion_host)
|
||||
};
|
||||
format!("http://{}{}", host, self.path)
|
||||
}
|
||||
|
||||
fn tor_client(&self) -> Result<reqwest::Client> {
|
||||
let proxy = reqwest::Proxy::all(crate::constants::TOR_SOCKS_PROXY)
|
||||
.context("Invalid Tor SOCKS proxy URL")?;
|
||||
reqwest::Client::builder()
|
||||
.proxy(proxy)
|
||||
.timeout(self.timeout)
|
||||
.build()
|
||||
.context("Build Tor HTTP client")
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn encode_query_round_trip_header_is_correct() {
|
||||
let q = encode_query(0x1234, "npub1abc").unwrap();
|
||||
assert_eq!(&q[0..2], &[0x12, 0x34]);
|
||||
assert_eq!(&q[2..4], &[0x01, 0x00]); // flags RD=1
|
||||
assert_eq!(&q[4..6], &[0x00, 0x01]); // QDCOUNT=1
|
||||
// Tail: QTYPE=28, QCLASS=1
|
||||
assert_eq!(&q[q.len() - 4..], &[0x00, 0x1C, 0x00, 0x01]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encode_query_includes_both_labels() {
|
||||
let q = encode_query(0, "npub1xyz").unwrap();
|
||||
assert!(q.windows(9).any(|w| w == b"\x08npub1xyz"));
|
||||
assert!(q.windows(5).any(|w| w == b"\x04fips"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_response_returns_aaaa_rdata() {
|
||||
// Minimal crafted response: header + qsection + one AAAA answer.
|
||||
let id = 0xBEEFu16;
|
||||
let mut r = Vec::new();
|
||||
r.extend_from_slice(&id.to_be_bytes());
|
||||
r.extend_from_slice(&0x8180u16.to_be_bytes()); // QR=1, RD=1, RA=1, rcode=0
|
||||
r.extend_from_slice(&1u16.to_be_bytes()); // QDCOUNT
|
||||
r.extend_from_slice(&1u16.to_be_bytes()); // ANCOUNT
|
||||
r.extend_from_slice(&0u16.to_be_bytes()); // NSCOUNT
|
||||
r.extend_from_slice(&0u16.to_be_bytes()); // ARCOUNT
|
||||
// Question: 1 label "a" + "fips"
|
||||
r.extend_from_slice(b"\x01a\x04fips\x00");
|
||||
r.extend_from_slice(&QTYPE_AAAA.to_be_bytes());
|
||||
r.extend_from_slice(&QCLASS_IN.to_be_bytes());
|
||||
// Answer: compressed name pointing at question offset 12
|
||||
r.extend_from_slice(&[0xC0, 0x0C]);
|
||||
r.extend_from_slice(&QTYPE_AAAA.to_be_bytes());
|
||||
r.extend_from_slice(&QCLASS_IN.to_be_bytes());
|
||||
r.extend_from_slice(&300u32.to_be_bytes()); // TTL
|
||||
r.extend_from_slice(&16u16.to_be_bytes()); // RDLENGTH
|
||||
let ip: Ipv6Addr = "fd9d:1192:e800:bad0:eed3:4b0e:b273:8e0e".parse().unwrap();
|
||||
r.extend_from_slice(&ip.octets());
|
||||
let got = decode_response(id, &r, "a").unwrap();
|
||||
assert_eq!(got, ip);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_rejects_id_mismatch() {
|
||||
let r = vec![0u8; 12];
|
||||
let err = decode_response(0x1234, &r, "x").unwrap_err();
|
||||
assert!(err.to_string().contains("id mismatch"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_rejects_rcode() {
|
||||
let mut r = vec![0u8; 12];
|
||||
r[0] = 0xAA;
|
||||
r[1] = 0xBB;
|
||||
r[3] = 3; // NXDOMAIN
|
||||
let err = decode_response(0xAABB, &r, "x").unwrap_err();
|
||||
assert!(err.to_string().contains("rcode 3"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_rejects_empty_answer_section() {
|
||||
let mut r = vec![0u8; 12];
|
||||
r[0] = 0xAA;
|
||||
r[1] = 0xBB;
|
||||
r[4] = 0;
|
||||
r[5] = 0; // QDCOUNT=0
|
||||
r[6] = 0;
|
||||
r[7] = 0; // ANCOUNT=0
|
||||
let err = decode_response(0xAABB, &r, "x").unwrap_err();
|
||||
assert!(err.to_string().contains("no AAAA"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,205 @@
|
||||
//! Last-known-good FIPS peer endpoints (A3.10).
|
||||
//!
|
||||
//! The LAN direct-peering tick (`anchors::lan_fips_anchors`) only helps peers
|
||||
//! we can currently see on the LAN. When a federation peer's LAN path is gone
|
||||
//! (renumbered network, remote site, mDNS blackout) the only route left is the
|
||||
//! anchor spanning tree — the exact hairpin RC2 calls out. But if we were EVER
|
||||
//! connected to that peer directly, the daemon knew a working endpoint for it
|
||||
//! (`fipsctl show peers` → `transport_addr`/`transport_type`, which covers
|
||||
//! LAN, Tailscale, and WAN endpoints alike). This module persists those
|
||||
//! npub-keyed endpoints and re-offers them as dial candidates when the live
|
||||
//! paths disappear: LAN → last-known-good → anchor tree.
|
||||
//!
|
||||
//! Persisted at `<data_dir>/fips-endpoints.json`. Entries are refreshed every
|
||||
//! time the peer is seen connected and dropped after `RETENTION` without a
|
||||
//! sighting, so a peer that genuinely moved doesn't get dialed at a stale
|
||||
//! address forever ( `fipsctl connect` to a dead address is harmless but not
|
||||
//! free).
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::path::Path;
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
use anyhow::Result;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use tokio::fs;
|
||||
|
||||
use super::anchors::SeedAnchor;
|
||||
|
||||
const FILE_NAME: &str = "fips-endpoints.json";
|
||||
/// Forget endpoints not seen connected for this long (seconds) — 30 days.
|
||||
const RETENTION_SECS: u64 = 30 * 24 * 60 * 60;
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
|
||||
pub struct KnownEndpoint {
|
||||
/// "ip:port" as reported by the daemon (`transport_addr`).
|
||||
pub address: String,
|
||||
/// "udp" | "tcp" (`transport_type`).
|
||||
pub transport: String,
|
||||
/// Unix seconds of the last time this peer was seen connected here.
|
||||
pub last_ok_unix: u64,
|
||||
}
|
||||
|
||||
/// A currently-connected peer as parsed from `fipsctl show peers`.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ConnectedPeer {
|
||||
pub npub: String,
|
||||
pub address: String,
|
||||
pub transport: String,
|
||||
}
|
||||
|
||||
fn now_unix() -> u64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.map(|d| d.as_secs())
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
pub async fn load(data_dir: &Path) -> HashMap<String, KnownEndpoint> {
|
||||
let path = data_dir.join(FILE_NAME);
|
||||
match fs::read(&path).await {
|
||||
Ok(bytes) => serde_json::from_slice(&bytes).unwrap_or_default(),
|
||||
Err(_) => HashMap::new(),
|
||||
}
|
||||
}
|
||||
|
||||
async fn save(data_dir: &Path, map: &HashMap<String, KnownEndpoint>) -> Result<()> {
|
||||
let path = data_dir.join(FILE_NAME);
|
||||
let tmp = data_dir.join(format!("{FILE_NAME}.tmp"));
|
||||
fs::write(&tmp, serde_json::to_vec_pretty(map)?).await?;
|
||||
fs::rename(&tmp, &path).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Merge the currently-connected peers into the store (refreshing their
|
||||
/// timestamps), prune expired entries, persist, and return the updated map.
|
||||
/// Persistence failures are non-fatal — the in-memory result is still
|
||||
/// returned so this tick's fallback logic works.
|
||||
pub async fn record_connected(
|
||||
data_dir: &Path,
|
||||
connected: &[ConnectedPeer],
|
||||
) -> HashMap<String, KnownEndpoint> {
|
||||
let mut map = load(data_dir).await;
|
||||
let now = now_unix();
|
||||
let before = map.clone();
|
||||
for p in connected {
|
||||
if p.npub.is_empty() || p.address.is_empty() {
|
||||
continue;
|
||||
}
|
||||
map.insert(
|
||||
p.npub.clone(),
|
||||
KnownEndpoint {
|
||||
address: p.address.clone(),
|
||||
transport: p.transport.clone(),
|
||||
last_ok_unix: now,
|
||||
},
|
||||
);
|
||||
}
|
||||
map.retain(|_, e| now.saturating_sub(e.last_ok_unix) <= RETENTION_SECS);
|
||||
if map != before {
|
||||
if let Err(e) = save(data_dir, &map).await {
|
||||
tracing::debug!("fips endpoint store save failed (non-fatal): {e}");
|
||||
}
|
||||
}
|
||||
map
|
||||
}
|
||||
|
||||
/// Build fallback anchors for federation peers whose live paths are gone:
|
||||
/// every `wanted_npub` that is neither currently connected nor covered by a
|
||||
/// live LAN direct entry, but has a last-known-good endpoint, becomes a dial
|
||||
/// candidate. `fipsctl connect` is idempotent and failure-tolerant, so a
|
||||
/// stale candidate costs one failed dial, bounded by apply()'s per-connect
|
||||
/// timeout.
|
||||
pub fn fallback_anchors(
|
||||
known: &HashMap<String, KnownEndpoint>,
|
||||
wanted_npubs: &[String],
|
||||
connected_npubs: &[String],
|
||||
lan_direct: &[SeedAnchor],
|
||||
) -> Vec<SeedAnchor> {
|
||||
let mut out = Vec::new();
|
||||
for npub in wanted_npubs {
|
||||
if connected_npubs.iter().any(|c| c == npub) {
|
||||
continue;
|
||||
}
|
||||
if lan_direct.iter().any(|a| &a.npub == npub) {
|
||||
continue;
|
||||
}
|
||||
if let Some(e) = known.get(npub) {
|
||||
out.push(SeedAnchor {
|
||||
npub: npub.clone(),
|
||||
address: e.address.clone(),
|
||||
transport: e.transport.clone(),
|
||||
label: "last-known-good endpoint (direct FIPS)".to_string(),
|
||||
});
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn ep(addr: &str) -> KnownEndpoint {
|
||||
KnownEndpoint {
|
||||
address: addr.to_string(),
|
||||
transport: "udp".to_string(),
|
||||
last_ok_unix: now_unix(),
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn record_and_reload_roundtrip() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let connected = vec![ConnectedPeer {
|
||||
npub: "npub1aaa".into(),
|
||||
address: "100.64.0.21:2121".into(),
|
||||
transport: "udp".into(),
|
||||
}];
|
||||
let map = record_connected(dir.path(), &connected).await;
|
||||
assert_eq!(map["npub1aaa"].address, "100.64.0.21:2121");
|
||||
let reloaded = load(dir.path()).await;
|
||||
assert_eq!(reloaded, map);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn expired_entries_are_pruned_on_record() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let mut stale = HashMap::new();
|
||||
stale.insert(
|
||||
"npub1old".to_string(),
|
||||
KnownEndpoint {
|
||||
address: "10.0.0.1:2121".into(),
|
||||
transport: "udp".into(),
|
||||
last_ok_unix: now_unix() - RETENTION_SECS - 60,
|
||||
},
|
||||
);
|
||||
save(dir.path(), &stale).await.unwrap();
|
||||
let map = record_connected(dir.path(), &[]).await;
|
||||
assert!(map.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fallback_skips_connected_and_lan_covered_peers() {
|
||||
let mut known = HashMap::new();
|
||||
known.insert("npub1gone".to_string(), ep("100.1.2.3:2121"));
|
||||
known.insert("npub1conn".to_string(), ep("100.1.2.4:2121"));
|
||||
known.insert("npub1lan".to_string(), ep("100.1.2.5:2121"));
|
||||
let wanted: Vec<String> = ["npub1gone", "npub1conn", "npub1lan", "npub1never"]
|
||||
.iter()
|
||||
.map(|s| s.to_string())
|
||||
.collect();
|
||||
let connected = vec!["npub1conn".to_string()];
|
||||
let lan = vec![SeedAnchor {
|
||||
npub: "npub1lan".into(),
|
||||
address: "192.0.2.198:2121".into(),
|
||||
transport: "udp".into(),
|
||||
label: "LAN".into(),
|
||||
}];
|
||||
let out = fallback_anchors(&known, &wanted, &connected, &lan);
|
||||
assert_eq!(out.len(), 1);
|
||||
assert_eq!(out[0].npub, "npub1gone");
|
||||
assert_eq!(out[0].address, "100.1.2.3:2121");
|
||||
// npub1never has no stored endpoint → nothing to dial.
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
//! Detect the `fips0` TUN interface's ULA (fd00::/8) IPv6 address.
|
||||
//!
|
||||
//! The `fips` daemon configures the TUN device with an address derived
|
||||
//! from the node's identity key. We need that address to bind a
|
||||
//! peer-facing listener that is only reachable from the FIPS overlay —
|
||||
//! WAN IPv6 addresses never carry ULA prefixes, so binding specifically
|
||||
//! to the fips0 address keeps the peer surface off the public internet.
|
||||
//!
|
||||
//! We read `/proc/net/if_inet6` rather than shelling out to `ip` so
|
||||
//! this can run under the `archipelago` service user without extra
|
||||
//! capabilities.
|
||||
#![allow(dead_code)]
|
||||
|
||||
use std::net::Ipv6Addr;
|
||||
|
||||
/// Interface name the FIPS daemon creates (matches upstream default in
|
||||
/// `/etc/fips/fips.yaml: tun.name`).
|
||||
pub const FIPS_IFACE: &str = "fips0";
|
||||
|
||||
/// Return the first ULA (fd00::/8) address assigned to `fips0`, if any.
|
||||
///
|
||||
/// - `None` if the interface is missing, has no address, or only has
|
||||
/// link-local addresses.
|
||||
/// - Link-local (`fe80::/10`) and non-ULA addresses are ignored — we
|
||||
/// only want the mesh-routable ULA that `<npub>.fips` DNS resolves to.
|
||||
pub fn fips0_ula() -> Option<Ipv6Addr> {
|
||||
addresses_on(FIPS_IFACE).into_iter().find(|a| is_ula(a))
|
||||
}
|
||||
|
||||
/// List every IPv6 address bound to a given interface from
|
||||
/// `/proc/net/if_inet6`. Returns empty on any parse failure.
|
||||
pub fn addresses_on(iface: &str) -> Vec<Ipv6Addr> {
|
||||
let contents = match std::fs::read_to_string("/proc/net/if_inet6") {
|
||||
Ok(s) => s,
|
||||
Err(_) => return Vec::new(),
|
||||
};
|
||||
contents
|
||||
.lines()
|
||||
.filter_map(|line| parse_line(line, iface))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// `fd00::/8` test — covers the full ULA range.
|
||||
pub fn is_ula(addr: &Ipv6Addr) -> bool {
|
||||
(addr.octets()[0] & 0xFE) == 0xFC
|
||||
}
|
||||
|
||||
fn parse_line(line: &str, iface: &str) -> Option<Ipv6Addr> {
|
||||
// /proc/net/if_inet6 format (whitespace-separated):
|
||||
// <32 hex chars addr> <idx> <prefixlen> <scope> <flags> <devname>
|
||||
// e.g. "fdd8...cd85 6f 80 00 80 fips0"
|
||||
let mut parts = line.split_whitespace();
|
||||
let hex = parts.next()?;
|
||||
let _idx = parts.next()?;
|
||||
let _prefix = parts.next()?;
|
||||
let _scope = parts.next()?;
|
||||
let _flags = parts.next()?;
|
||||
let name = parts.next()?;
|
||||
if name != iface {
|
||||
return None;
|
||||
}
|
||||
if hex.len() != 32 {
|
||||
return None;
|
||||
}
|
||||
let mut octets = [0u8; 16];
|
||||
for i in 0..16 {
|
||||
octets[i] = u8::from_str_radix(&hex[i * 2..i * 2 + 2], 16).ok()?;
|
||||
}
|
||||
Some(Ipv6Addr::from(octets))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn parse_line_extracts_address() {
|
||||
let line = "fdd83d5aabe08c0ee67f75fcf0d4cd85 6f 80 00 80 fips0";
|
||||
let addr = parse_line(line, "fips0").unwrap();
|
||||
assert_eq!(
|
||||
addr,
|
||||
"fdd8:3d5a:abe0:8c0e:e67f:75fc:f0d4:cd85"
|
||||
.parse::<Ipv6Addr>()
|
||||
.unwrap()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_line_rejects_other_iface() {
|
||||
let line = "fdd83d5aabe08c0ee67f75fcf0d4cd85 6f 80 00 80 eth0";
|
||||
assert!(parse_line(line, "fips0").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_line_ignores_malformed() {
|
||||
assert!(parse_line("garbage", "fips0").is_none());
|
||||
assert!(parse_line("shorthex 6f 80 00 80 fips0", "fips0").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ula_classifier_matches_fd_range() {
|
||||
assert!(is_ula(&"fd00::1".parse().unwrap()));
|
||||
assert!(is_ula(&"fdff::".parse().unwrap()));
|
||||
assert!(is_ula(&"fc00::1".parse().unwrap()));
|
||||
assert!(!is_ula(&"fe80::1".parse().unwrap())); // link-local
|
||||
assert!(!is_ula(&"2001:db8::1".parse().unwrap())); // global
|
||||
assert!(!is_ula(&"::1".parse().unwrap())); // loopback
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,331 @@
|
||||
//! FIPS (Free Internetworking Peering System) daemon integration.
|
||||
//!
|
||||
//! github.com/jmcorgan/fips — a spanning-tree mesh routing protocol that
|
||||
//! uses Nostr secp256k1 keys as native node identity. Archipelago ships
|
||||
//! the daemon as an apt package, feeds it the seed-derived key from
|
||||
//! `/data/identity/fips_key`, and supervises it via
|
||||
//! `archipelago-fips.service`.
|
||||
//!
|
||||
//! This module is the in-process bridge:
|
||||
//! - [`service`]: systemctl status / start / stop / restart / unmask.
|
||||
//! - [`config`]: materialise `/etc/fips/fips.yaml` + install the key.
|
||||
//! - [`update`]: query GitHub (tracking `main`) for a newer build,
|
||||
//! verify SHA256, install via dpkg, restart.
|
||||
//!
|
||||
//! Privileged operations shell out via `sudo systemctl …` and `sudo dpkg …`
|
||||
//! (mirroring the vpn/update patterns already in the codebase); the
|
||||
//! sudoers rule shipped in the ISO whitelists exactly those commands for
|
||||
//! the `archipelago` service user.
|
||||
//!
|
||||
//! FIPS is dark on the wire until onboarding writes the key. Before that,
|
||||
//! `FipsStatus::installed` reports the package state and `service_active`
|
||||
//! returns false; the transport router keeps routing via Tor.
|
||||
|
||||
// Consumers land in the next phase (RPC endpoints + onboarding hookup);
|
||||
// the module is deliberately API-ready ahead of those call-sites.
|
||||
#![allow(dead_code)]
|
||||
|
||||
pub mod anchors;
|
||||
pub mod app_ports;
|
||||
pub mod config;
|
||||
pub mod dial;
|
||||
pub mod endpoints;
|
||||
pub mod iface;
|
||||
pub mod service;
|
||||
pub mod telemetry;
|
||||
pub mod update;
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
/// Auto-activate FIPS with no user interaction. Once seed onboarding has
|
||||
/// materialised the fips key, install the daemon config + start the service if
|
||||
/// it isn't already up. Idempotent and best-effort: FIPS is the preferred
|
||||
/// transport and should come up on its own — the UI "Activate" button is now a
|
||||
/// manual fallback, not a requirement. No-op pre-onboarding (no key yet) or
|
||||
/// when the service is already active.
|
||||
pub async fn ensure_activated(data_dir: &std::path::Path) {
|
||||
let identity_dir = identity_dir_from(data_dir);
|
||||
if !identity_dir.join("fips_key").exists() {
|
||||
return; // pre-onboarding: nothing to activate yet
|
||||
}
|
||||
if dial::is_service_active().await {
|
||||
return; // already up
|
||||
}
|
||||
tracing::info!("FIPS inactive — auto-activating (no user interaction needed)");
|
||||
if let Err(e) = config::install(&identity_dir).await {
|
||||
tracing::warn!("FIPS auto-activate: config install failed: {:#}", e);
|
||||
return;
|
||||
}
|
||||
let unit = service::activation_unit().await;
|
||||
if let Err(e) = service::activate(unit).await {
|
||||
tracing::warn!("FIPS auto-activate: service activate failed: {:#}", e);
|
||||
return;
|
||||
}
|
||||
tracing::info!("FIPS auto-activated");
|
||||
}
|
||||
|
||||
/// Spawn the FIPS supervisor: every 25s it (1) auto-activates FIPS if onboarding
|
||||
/// is done but the service is down — so it comes up with zero user interaction,
|
||||
/// and (2) keeps hole-punched paths to known federation peers warm, so on-demand
|
||||
/// dials land on FIPS instead of falling back to Tor. Warms peers concurrently
|
||||
/// so one slow/offline peer doesn't delay the rest.
|
||||
///
|
||||
/// The interval MUST be shorter than the NAT/hole-punch cold window
|
||||
/// (`warm_path` docs it at ~30-60s). The previous 45s sat at the edge of that
|
||||
/// window: a path that went cold at ~30s stayed cold until the next 45s tick,
|
||||
/// so real peer dials in that gap hit a cold path and fell back to Tor (~18s
|
||||
/// onion latency instead of FIPS's ~2-3s). 25s keeps every path refreshed
|
||||
/// inside the minimum cold window, which is what actually makes FIPS — not Tor —
|
||||
/// the transport peer requests land on. Measured: warm FIPS browse ~2.6s vs a
|
||||
/// cold-path fallback browse ~18-22s over Tor to the same peer.
|
||||
pub fn spawn_fips_supervisor(data_dir: std::path::PathBuf) {
|
||||
tokio::spawn(async move {
|
||||
let mut tick = tokio::time::interval(std::time::Duration::from_secs(25));
|
||||
// Connectivity watcher state: re-apply seed anchors the moment the
|
||||
// anchor link drops (edge) or the data path degrades (dials keep
|
||||
// failing with zero successes), instead of waiting for the 300s
|
||||
// anchor tick. Bounded: at most one re-apply per RE_APPLY_BACKOFF.
|
||||
const RE_APPLY_BACKOFF: std::time::Duration = std::time::Duration::from_secs(60);
|
||||
let mut prev_connected: Option<bool> = None;
|
||||
let mut prev_totals = telemetry::totals();
|
||||
let mut last_apply: Option<std::time::Instant> = None;
|
||||
loop {
|
||||
tick.tick().await;
|
||||
// Bring FIPS up on its own once onboarding has materialised the key.
|
||||
ensure_activated(&data_dir).await;
|
||||
if !dial::is_service_active().await {
|
||||
prev_connected = None; // daemon restart = fresh edge detection
|
||||
continue;
|
||||
}
|
||||
|
||||
// ── Warm the union of federation peers + configured seed
|
||||
// anchors. Warming only federation npubs left the direct
|
||||
// anchors (vps2, LAN peers) to go cold between 300s ticks.
|
||||
let nodes = crate::federation::load_nodes(&data_dir)
|
||||
.await
|
||||
.unwrap_or_default();
|
||||
let seed = anchors::load(&data_dir).await.unwrap_or_default();
|
||||
let mut warm_npubs: std::collections::BTreeSet<String> =
|
||||
nodes.iter().filter_map(|n| n.fips_npub.clone()).collect();
|
||||
warm_npubs.extend(seed.iter().map(|a| a.npub.clone()));
|
||||
let mut handles = Vec::new();
|
||||
for npub in warm_npubs {
|
||||
// Service-active was checked once above for the whole batch.
|
||||
handles.push(tokio::spawn(async move {
|
||||
dial::warm_path_unchecked(&npub).await
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
let _ = h.await;
|
||||
}
|
||||
|
||||
// ── Connectivity watcher: detect anchor-link loss AND silent
|
||||
// data-path death (daemon reports "connected" but every dial
|
||||
// connect-fails — observed live on .198, 2026-07-27, where the
|
||||
// 300s tick never healed it).
|
||||
let mut anchor_npubs = vec![service::PUBLIC_ANCHOR_NPUB.to_string()];
|
||||
anchor_npubs.extend(seed.iter().map(|a| a.npub.clone()));
|
||||
let (_, connected) = service::peer_connectivity_summary(&anchor_npubs).await;
|
||||
let totals = telemetry::totals();
|
||||
let link_dropped = prev_connected == Some(true) && !connected;
|
||||
let never_connected = prev_connected.is_none() && !connected;
|
||||
let data_path_dead =
|
||||
totals.1.saturating_sub(prev_totals.1) >= 5 && totals.0 == prev_totals.0;
|
||||
prev_connected = Some(connected);
|
||||
prev_totals = totals;
|
||||
|
||||
let backoff_ok = last_apply.is_none_or(|t| t.elapsed() >= RE_APPLY_BACKOFF);
|
||||
if (link_dropped || never_connected || data_path_dead) && backoff_ok && !seed.is_empty()
|
||||
{
|
||||
tracing::info!(
|
||||
link_dropped,
|
||||
never_connected,
|
||||
data_path_dead,
|
||||
"FIPS connectivity degraded — re-applying seed anchors now"
|
||||
);
|
||||
last_apply = Some(std::time::Instant::now());
|
||||
let _ = anchors::apply(&seed).await;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
/// Systemd unit name supervised by archipelago.
|
||||
pub const SERVICE_UNIT: &str = "archipelago-fips.service";
|
||||
|
||||
/// Path the FIPS daemon reads its config from (Debian package default).
|
||||
pub const DAEMON_CONFIG_PATH: &str = "/etc/fips/fips.yaml";
|
||||
|
||||
/// Path the FIPS daemon reads its private key from.
|
||||
pub const DAEMON_KEY_PATH: &str = "/etc/fips/fips.key";
|
||||
|
||||
/// Path the FIPS daemon reads its public key from.
|
||||
pub const DAEMON_PUB_PATH: &str = "/etc/fips/fips.pub";
|
||||
|
||||
/// Upstream repository the updater tracks (branch `main`).
|
||||
pub const UPSTREAM_REPO: &str = "jmcorgan/fips";
|
||||
|
||||
/// Default UDP port the daemon listens on.
|
||||
pub const DEFAULT_UDP_PORT: u16 = 8668;
|
||||
|
||||
/// UDP port archipelago actually publishes/binds for the daemon. The
|
||||
/// container publishes `2121:2121/udp` (see `package/config.rs`) and every
|
||||
/// peer roster in the fleet dials `<host>:2121`, but the rendered daemon
|
||||
/// config used to bind upstream's 8668 — leaving inbound UDP dead on
|
||||
/// bridged-network installs and everything silently riding TCP 8443. The
|
||||
/// bind now uses this port so the published mapping, the fleet rosters,
|
||||
/// and the pairing QR all agree.
|
||||
pub const PUBLISHED_UDP_PORT: u16 = 2121;
|
||||
|
||||
/// Default TCP port the daemon listens on. Used as a fallback when a
|
||||
/// peer can't be reached over UDP — common on networks that block UDP
|
||||
/// (corporate/guest wifi) and the path the public fips.v0l.io anchor
|
||||
/// currently accepts. Upstream factory default enables both transports
|
||||
/// and archipelago intentionally matches that baseline so fresh nodes
|
||||
/// can reach the broader FIPS mesh without operator config.
|
||||
pub const DEFAULT_TCP_PORT: u16 = 8443;
|
||||
|
||||
/// Upstream systemd unit shipped by the `fips` debian package. Archipelago
|
||||
/// prefers its own supervision (`archipelago-fips.service`) but respects an
|
||||
/// already-running upstream unit so legacy/dev nodes — where no seed-derived
|
||||
/// key exists — still report FIPS as active in the UI.
|
||||
pub const UPSTREAM_SERVICE_UNIT: &str = "fips.service";
|
||||
|
||||
/// Aggregated runtime status of the FIPS subsystem, surfaced to the dashboard.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct FipsStatus {
|
||||
/// Whether the `fips` debian package is installed on the host.
|
||||
pub installed: bool,
|
||||
/// Installed daemon version string reported by `fipsctl --version`,
|
||||
/// or None if not installed / not queryable.
|
||||
pub version: Option<String>,
|
||||
/// `systemctl is-active archipelago-fips.service` result: "active",
|
||||
/// "inactive", "failed", "masked", "unknown".
|
||||
pub service_state: String,
|
||||
/// State of the upstream `fips.service` (shipped by the debian package).
|
||||
pub upstream_service_state: String,
|
||||
/// True if either the archipelago-managed or upstream unit is active.
|
||||
pub service_active: bool,
|
||||
/// Whether the seed-derived FIPS key has been materialised on disk.
|
||||
/// The archipelago-managed service cannot start meaningfully until
|
||||
/// this is true; legacy nodes may still report FIPS active via the
|
||||
/// upstream unit without this file.
|
||||
pub key_present: bool,
|
||||
/// Local FIPS npub (bech32). Prefers the seed-derived key when
|
||||
/// present; falls back to the upstream daemon's own key on legacy
|
||||
/// nodes where `/etc/fips/fips.pub` is readable.
|
||||
pub npub: Option<String>,
|
||||
/// Number of currently authenticated FIPS peers, per
|
||||
/// `fipsctl show peers`. 0 → isolated / anchor unreachable;
|
||||
/// >0 → DHT routing is viable.
|
||||
#[serde(default)]
|
||||
pub authenticated_peer_count: u32,
|
||||
/// True when at least one peer in the identity cache is a known
|
||||
/// public anchor (currently `fips.v0l.io`). Anchors bootstrap DHT
|
||||
/// routing for general-case deployments, so a red anchor status is
|
||||
/// the top UX indicator of "FIPS traffic will probably degrade to
|
||||
/// Tor until the anchor is reachable."
|
||||
#[serde(default)]
|
||||
pub anchor_connected: bool,
|
||||
}
|
||||
|
||||
impl FipsStatus {
|
||||
/// Snapshot the current state across package, key, and service.
|
||||
///
|
||||
/// `data_dir` is the archipelago data-dir (used to load the
|
||||
/// operator-configured seed-anchor list so "anchor_connected" means
|
||||
/// "at least one authenticated peer matches a public or configured
|
||||
/// seed anchor", not just "fips.v0l.io specifically").
|
||||
pub async fn query(data_dir: &Path) -> Self {
|
||||
let identity_dir = identity_dir_from(data_dir);
|
||||
let installed = service::package_installed().await;
|
||||
let version = if installed {
|
||||
service::daemon_version().await.ok()
|
||||
} else {
|
||||
None
|
||||
};
|
||||
let service_state = service::unit_state(SERVICE_UNIT).await;
|
||||
let upstream_service_state = service::unit_state(UPSTREAM_SERVICE_UNIT).await;
|
||||
let service_active = service_state == "active" || upstream_service_state == "active";
|
||||
let key_present = crate::identity::fips_key_exists(&identity_dir);
|
||||
|
||||
// Prefer the seed-derived npub; otherwise read the daemon's own
|
||||
// key file at /etc/fips/fips.pub (world-readable per debian pkg).
|
||||
let npub = match crate::identity::fips_npub(&identity_dir).await {
|
||||
Ok(Some(n)) => Some(n),
|
||||
_ => service::read_upstream_npub().await.ok().flatten(),
|
||||
};
|
||||
|
||||
let (authenticated_peer_count, anchor_connected) = if service_active {
|
||||
// Build the anchor-candidate list: hardcoded public anchor
|
||||
// plus every entry in the operator's seed-anchors.json.
|
||||
// The card lights up if any of them is authenticated.
|
||||
let mut anchor_npubs = vec![service::PUBLIC_ANCHOR_NPUB.to_string()];
|
||||
if let Ok(seed) = anchors::load(data_dir).await {
|
||||
anchor_npubs.extend(seed.into_iter().map(|a| a.npub));
|
||||
}
|
||||
service::peer_connectivity_summary(&anchor_npubs).await
|
||||
} else {
|
||||
(0, false)
|
||||
};
|
||||
|
||||
Self {
|
||||
installed,
|
||||
version,
|
||||
service_state,
|
||||
upstream_service_state,
|
||||
service_active,
|
||||
key_present,
|
||||
npub,
|
||||
authenticated_peer_count,
|
||||
anchor_connected,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Compose a data-dir–relative identity directory path.
|
||||
/// Mirrors the convention used elsewhere in the codebase so callers don't
|
||||
/// have to repeat the `.join("identity")` each time.
|
||||
pub fn identity_dir_from(data_dir: &Path) -> PathBuf {
|
||||
data_dir.join("identity")
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_status_reports_no_key_pre_onboarding() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
// query() now takes a data_dir (parent) rather than identity_dir,
|
||||
// since it also reads seed-anchors.json for the anchor check.
|
||||
// No identity/ subdir → no key; no seed-anchors.json → public
|
||||
// anchor is the only candidate.
|
||||
let status = FipsStatus::query(dir.path()).await;
|
||||
assert!(!status.key_present, "no key before onboarding");
|
||||
// `npub` falls back to whatever an already-running local fips
|
||||
// daemon advertises, so on a dev machine or node with fips
|
||||
// installed this field can be Some(...) even when the test
|
||||
// data_dir is empty. We only assert that key_present is false.
|
||||
// `installed`, `service_state`, `version` depend on the host and are
|
||||
// not asserted here — query() must return cleanly regardless.
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_identity_dir_from() {
|
||||
let data = Path::new("/var/lib/archipelago");
|
||||
assert_eq!(
|
||||
identity_dir_from(data),
|
||||
Path::new("/var/lib/archipelago/identity")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_constants_have_expected_shape() {
|
||||
assert!(SERVICE_UNIT.ends_with(".service"));
|
||||
assert!(DAEMON_CONFIG_PATH.starts_with('/'));
|
||||
assert!(DAEMON_KEY_PATH.starts_with('/'));
|
||||
assert_eq!(UPSTREAM_REPO, "jmcorgan/fips");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,324 @@
|
||||
//! systemctl + dpkg-query helpers for the FIPS daemon.
|
||||
//!
|
||||
//! Read-only queries (`is-active`, `--version`, `dpkg-query`) run as the
|
||||
//! archipelago service user. Write operations (`unmask`, `start`, `stop`,
|
||||
//! `restart`) go through `sudo`, matching the pattern established in
|
||||
//! `src/vpn.rs` and `src/api/rpc/vpn.rs`. The sudoers rule shipped in the
|
||||
//! ISO whitelists exactly these invocations.
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use nostr_sdk::ToBech32;
|
||||
use tokio::process::Command;
|
||||
|
||||
use super::DAEMON_PUB_PATH;
|
||||
|
||||
/// `systemctl is-active <unit>` → "active" / "inactive" / "failed" / "masked"
|
||||
/// / "unknown". Never errors; returns "unknown" on any failure.
|
||||
pub async fn unit_state(unit: &str) -> String {
|
||||
match Command::new("systemctl")
|
||||
.args(["is-active", unit])
|
||||
.output()
|
||||
.await
|
||||
{
|
||||
Ok(out) => {
|
||||
let s = String::from_utf8_lossy(&out.stdout).trim().to_string();
|
||||
if s.is_empty() {
|
||||
"unknown".to_string()
|
||||
} else {
|
||||
s
|
||||
}
|
||||
}
|
||||
Err(_) => "unknown".to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether systemd knows about `unit`.
|
||||
pub async fn unit_exists(unit: &str) -> bool {
|
||||
Command::new("systemctl")
|
||||
.args(["cat", unit])
|
||||
.output()
|
||||
.await
|
||||
.map(|out| out.status.success())
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
/// Whether the `fips` debian package is installed on the host.
|
||||
pub async fn package_installed() -> bool {
|
||||
// dpkg-query -W -f='${Status}' fips → "install ok installed" when present.
|
||||
let out = Command::new("dpkg-query")
|
||||
.args(["-W", "-f=${Status}", "fips"])
|
||||
.output()
|
||||
.await;
|
||||
match out {
|
||||
Ok(o) if o.status.success() => {
|
||||
String::from_utf8_lossy(&o.stdout).contains("install ok installed")
|
||||
}
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// `fipsctl --version` output stripped of the "fipsctl " prefix if present.
|
||||
pub async fn daemon_version() -> Result<String> {
|
||||
let out = Command::new("fipsctl")
|
||||
.arg("--version")
|
||||
.output()
|
||||
.await
|
||||
.context("fipsctl --version failed to launch")?;
|
||||
if !out.status.success() {
|
||||
anyhow::bail!("fipsctl exited with non-zero status");
|
||||
}
|
||||
let raw = String::from_utf8_lossy(&out.stdout).trim().to_string();
|
||||
Ok(raw
|
||||
.strip_prefix("fipsctl ")
|
||||
.map(|s| s.to_string())
|
||||
.unwrap_or(raw))
|
||||
}
|
||||
|
||||
/// `sudo systemctl <verb> <unit>` — returns stderr on non-zero exit.
|
||||
async fn sudo_systemctl(verb: &str, unit: &str) -> Result<()> {
|
||||
let out = Command::new("sudo")
|
||||
.args(["systemctl", verb, unit])
|
||||
.output()
|
||||
.await
|
||||
.with_context(|| format!("sudo systemctl {} {} failed to launch", verb, unit))?;
|
||||
if !out.status.success() {
|
||||
let stderr = String::from_utf8_lossy(&out.stderr).trim().to_string();
|
||||
anyhow::bail!("systemctl {} {}: {}", verb, unit, stderr);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Unmask + start + enable the FIPS service. Idempotent — safe to call
|
||||
/// on every backend startup once the key is on disk.
|
||||
pub async fn activate(unit: &str) -> Result<()> {
|
||||
kill_stale_daemons().await?;
|
||||
// Order matters: unmask before enable/start, otherwise enable fails
|
||||
// on a masked unit.
|
||||
sudo_systemctl("unmask", unit).await?;
|
||||
sudo_systemctl("enable", unit).await?;
|
||||
sudo_systemctl("start", unit).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn stop(unit: &str) -> Result<()> {
|
||||
sudo_systemctl("stop", unit).await
|
||||
}
|
||||
|
||||
pub async fn restart(unit: &str) -> Result<()> {
|
||||
kill_stale_daemons().await?;
|
||||
sudo_systemctl("restart", unit).await
|
||||
}
|
||||
|
||||
/// Kill orphaned `fips` processes not owned by either known systemd unit.
|
||||
///
|
||||
/// Field failure, 2026-07-24: a stale daemon survived outside systemd and kept
|
||||
/// `0.0.0.0:8443` bound. The supervised daemon then started UDP-only, so every
|
||||
/// TCP seed-anchor connect failed with "no operational transport" and phones
|
||||
/// on 5G could not discover the node. We keep the cleanup narrow: preserve the
|
||||
/// MainPID of both units and terminate only extra `pgrep -x fips` matches.
|
||||
pub async fn kill_stale_daemons() -> Result<()> {
|
||||
let script = format!(
|
||||
r#"keep="$(systemctl show -p MainPID --value {managed} 2>/dev/null; systemctl show -p MainPID --value {upstream} 2>/dev/null)"
|
||||
for pid in $(pgrep -x fips 2>/dev/null || true); do
|
||||
case " $keep " in
|
||||
*" $pid "*) ;;
|
||||
*) kill "$pid" 2>/dev/null || true ;;
|
||||
esac
|
||||
done
|
||||
"#,
|
||||
managed = super::SERVICE_UNIT,
|
||||
upstream = super::UPSTREAM_SERVICE_UNIT,
|
||||
);
|
||||
let out = Command::new("sudo")
|
||||
.args(["sh", "-c", &script])
|
||||
.output()
|
||||
.await
|
||||
.context("sudo stale fips cleanup failed to launch")?;
|
||||
if !out.status.success() {
|
||||
let stderr = String::from_utf8_lossy(&out.stderr).trim().to_string();
|
||||
anyhow::bail!("stale fips cleanup failed: {}", stderr);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Resolve which systemd unit should be started when FIPS is inactive.
|
||||
/// Newer Archipelago images may ship `archipelago-fips.service`; nodes with
|
||||
/// the upstream Debian package may only have `fips.service`. Activation must
|
||||
/// choose a unit systemd can actually load, otherwise the dashboard repeatedly
|
||||
/// offers an "Activate" action that can never succeed.
|
||||
pub async fn activation_unit() -> &'static str {
|
||||
if unit_exists(super::SERVICE_UNIT).await {
|
||||
return super::SERVICE_UNIT;
|
||||
}
|
||||
if unit_exists(super::UPSTREAM_SERVICE_UNIT).await {
|
||||
return super::UPSTREAM_SERVICE_UNIT;
|
||||
}
|
||||
super::SERVICE_UNIT
|
||||
}
|
||||
|
||||
/// Resolve which systemd unit is actually supervising the fips daemon on this
|
||||
/// host. Restart/Reconnect must operate on whichever one is running, otherwise
|
||||
/// the UI button is a silent no-op.
|
||||
///
|
||||
/// Returns the archipelago-managed unit name if it's active,
|
||||
/// else the upstream unit name if that's active,
|
||||
/// else a startable activation unit.
|
||||
pub async fn active_unit() -> &'static str {
|
||||
if unit_state(super::SERVICE_UNIT).await == "active" {
|
||||
return super::SERVICE_UNIT;
|
||||
}
|
||||
if unit_state(super::UPSTREAM_SERVICE_UNIT).await == "active" {
|
||||
return super::UPSTREAM_SERVICE_UNIT;
|
||||
}
|
||||
activation_unit().await
|
||||
}
|
||||
|
||||
pub async fn mask(unit: &str) -> Result<()> {
|
||||
let _ = sudo_systemctl("stop", unit).await;
|
||||
let _ = sudo_systemctl("disable", unit).await;
|
||||
sudo_systemctl("mask", unit).await
|
||||
}
|
||||
|
||||
/// Known public anchor npub (fips.v0l.io as of 2026-04). Used to decide
|
||||
/// whether the `anchor_connected` badge in the dashboard lights up.
|
||||
pub const PUBLIC_ANCHOR_NPUB: &str =
|
||||
"npub1zv58cn7v83mxvttl70w5fwjwuclfmntv9cnmv5wmz2nzz88u5urqvdx96n";
|
||||
|
||||
/// Summarise peer connectivity from `fipsctl show peers`. Returns
|
||||
/// `(authenticated_peer_count, anchor_connected)`.
|
||||
///
|
||||
/// `anchor_candidates` is the operator-controlled list of npubs this
|
||||
/// node considers a valid mesh anchor — always includes the hard-coded
|
||||
/// public anchor, plus any entries from `seed-anchors.json`. A node is
|
||||
/// "anchor connected" when at least one currently-authenticated peer
|
||||
/// matches one of these npubs. We used to check the identity cache
|
||||
/// (which includes transient hearsay from other peers), but a cache
|
||||
/// hit on `fips.v0l.io` didn't mean we could actually route through
|
||||
/// it, and the card lied to users whose mesh was federated through
|
||||
/// their own seed anchors instead.
|
||||
pub async fn peer_connectivity_summary(anchor_candidates: &[String]) -> (u32, bool) {
|
||||
let peers_json = match Command::new("sudo")
|
||||
.args(["-n", "fipsctl", "show", "peers"])
|
||||
.output()
|
||||
.await
|
||||
{
|
||||
Ok(o) if o.status.success() => o.stdout,
|
||||
_ => return (0, false),
|
||||
};
|
||||
let parsed: serde_json::Value = match serde_json::from_slice(&peers_json) {
|
||||
Ok(v) => v,
|
||||
Err(_) => return (0, false),
|
||||
};
|
||||
let peers = parsed
|
||||
.get("peers")
|
||||
.and_then(|p| p.as_array())
|
||||
.cloned()
|
||||
.unwrap_or_default();
|
||||
let authenticated_peer_count = peers.len() as u32;
|
||||
let anchor_connected = peers.iter().any(|p| {
|
||||
let npub = p.get("npub").and_then(|n| n.as_str()).unwrap_or_default();
|
||||
let connected = p
|
||||
.get("connectivity")
|
||||
.and_then(|c| c.as_str())
|
||||
.map(|s| s == "connected")
|
||||
.unwrap_or(true);
|
||||
connected && anchor_candidates.iter().any(|a| a == npub)
|
||||
});
|
||||
(authenticated_peer_count, anchor_connected)
|
||||
}
|
||||
|
||||
/// Currently-connected peers with their live endpoints, from
|
||||
/// `fipsctl show peers` (`transport_addr`/`transport_type`). Feeds the
|
||||
/// last-known-good endpoint store (A3.10); empty on any failure.
|
||||
pub async fn connected_peer_endpoints() -> Vec<crate::fips::endpoints::ConnectedPeer> {
|
||||
let peers_json = match Command::new("sudo")
|
||||
.args(["-n", "fipsctl", "show", "peers"])
|
||||
.output()
|
||||
.await
|
||||
{
|
||||
Ok(o) if o.status.success() => o.stdout,
|
||||
_ => return Vec::new(),
|
||||
};
|
||||
let parsed: serde_json::Value = match serde_json::from_slice(&peers_json) {
|
||||
Ok(v) => v,
|
||||
Err(_) => return Vec::new(),
|
||||
};
|
||||
parsed
|
||||
.get("peers")
|
||||
.and_then(|p| p.as_array())
|
||||
.map(|peers| {
|
||||
peers
|
||||
.iter()
|
||||
.filter(|p| {
|
||||
p.get("connectivity")
|
||||
.and_then(|c| c.as_str())
|
||||
.map(|s| s == "connected")
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.filter_map(|p| {
|
||||
let npub = p.get("npub").and_then(|n| n.as_str())?;
|
||||
let address = p.get("transport_addr").and_then(|a| a.as_str())?;
|
||||
let transport = p
|
||||
.get("transport_type")
|
||||
.and_then(|t| t.as_str())
|
||||
.unwrap_or("udp");
|
||||
Some(crate::fips::endpoints::ConnectedPeer {
|
||||
npub: npub.to_string(),
|
||||
address: address.to_string(),
|
||||
transport: transport.to_string(),
|
||||
})
|
||||
})
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Read the upstream daemon's public key at `/etc/fips/fips.pub` and return
|
||||
/// it as a bech32 npub. Returns `Ok(None)` if the file doesn't exist — used
|
||||
/// as a fallback on legacy/dev nodes where no seed-derived key exists.
|
||||
///
|
||||
/// Upstream writes the key as a bech32 string (`npub1…`); older builds may
|
||||
/// have written 32 raw bytes, so we accept either form.
|
||||
pub async fn read_upstream_npub() -> Result<Option<String>> {
|
||||
let bytes = match tokio::fs::read(DAEMON_PUB_PATH).await {
|
||||
Ok(b) => b,
|
||||
Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(None),
|
||||
Err(e) => return Err(e).context("read /etc/fips/fips.pub"),
|
||||
};
|
||||
if let Ok(s) = std::str::from_utf8(&bytes) {
|
||||
let trimmed = s.trim();
|
||||
if trimmed.starts_with("npub1") {
|
||||
if let Ok(pk) = nostr_sdk::PublicKey::parse(trimmed) {
|
||||
return Ok(pk.to_bech32().ok());
|
||||
}
|
||||
}
|
||||
}
|
||||
let pk = nostr_sdk::PublicKey::from_slice(&bytes)
|
||||
.context("parse /etc/fips/fips.pub as secp256k1 public key")?;
|
||||
Ok(pk.to_bech32().ok())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_unit_state_returns_string_for_bogus_unit() {
|
||||
// Nonexistent unit: systemctl returns "inactive" or "unknown" — we
|
||||
// just care that the helper doesn't panic and returns *something*.
|
||||
let s = unit_state("archipelago-bogus-test.service").await;
|
||||
assert!(!s.is_empty());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_package_installed_is_bool() {
|
||||
// Must not panic regardless of host state.
|
||||
let _ = package_installed().await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_unit_exists_is_bool() {
|
||||
// Must not panic regardless of host state.
|
||||
let _ = unit_exists("archipelago-bogus-test.service").await;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
//! In-process counters for FIPS dial outcomes.
|
||||
//!
|
||||
//! Every peer dial that could have used FIPS either succeeds over FIPS or
|
||||
//! falls back to Tor for one of six reasons (F1–F6). Before these counters
|
||||
//! existed, fallbacks were `debug!`-only and invisible in production, which
|
||||
//! made "FIPS uptime" unfalsifiable — several paths were 100% Tor for months
|
||||
//! (dead ports, firewalled listeners, allowlist 404s) and nothing surfaced
|
||||
//! it. The counters are process-lifetime (reset on restart) and exposed via
|
||||
//! `fips.status` as `dial_stats`, so a fleet-wide fallback regression shows
|
||||
//! up on the dashboard instead of as vague slowness.
|
||||
|
||||
use std::sync::atomic::{AtomicU64, Ordering};
|
||||
|
||||
/// Why a FIPS-capable dial fell back to Tor.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum FallbackReason {
|
||||
/// F1 — no FIPS npub known for the peer (never meshed, or pre-npub
|
||||
/// federation record). Expected for non-FIPS peers; high counts here
|
||||
/// mean npub propagation is broken, not the transport.
|
||||
NoNpub,
|
||||
/// F2 — the local FIPS daemon service isn't active.
|
||||
ServiceInactive,
|
||||
/// F3 — the local FIPS DNS resolver couldn't resolve the peer's npub
|
||||
/// (daemon up but peer not in the identity cache / mesh unreachable).
|
||||
DnsFail,
|
||||
/// F4 — TCP/HTTP dial to the peer's ULA failed or exceeded the FIPS
|
||||
/// attempt budget (firewalled :5679, cold hole-punch, peer down).
|
||||
ConnectFail,
|
||||
/// F5 — peer answered over FIPS with 404: its listener doesn't serve
|
||||
/// this path (older build / stricter allowlist).
|
||||
Http404,
|
||||
/// F6 — peer answered over FIPS with a 5xx server error.
|
||||
Http5xx,
|
||||
}
|
||||
|
||||
impl FallbackReason {
|
||||
pub fn key(self) -> &'static str {
|
||||
match self {
|
||||
Self::NoNpub => "no_npub",
|
||||
Self::ServiceInactive => "service_inactive",
|
||||
Self::DnsFail => "dns_fail",
|
||||
Self::ConnectFail => "connect_fail",
|
||||
Self::Http404 => "http_404",
|
||||
Self::Http5xx => "http_5xx",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static FIPS_OK: AtomicU64 = AtomicU64::new(0);
|
||||
static NO_NPUB: AtomicU64 = AtomicU64::new(0);
|
||||
static SERVICE_INACTIVE: AtomicU64 = AtomicU64::new(0);
|
||||
static DNS_FAIL: AtomicU64 = AtomicU64::new(0);
|
||||
static CONNECT_FAIL: AtomicU64 = AtomicU64::new(0);
|
||||
static HTTP_404: AtomicU64 = AtomicU64::new(0);
|
||||
static HTTP_5XX: AtomicU64 = AtomicU64::new(0);
|
||||
|
||||
fn counter(reason: FallbackReason) -> &'static AtomicU64 {
|
||||
match reason {
|
||||
FallbackReason::NoNpub => &NO_NPUB,
|
||||
FallbackReason::ServiceInactive => &SERVICE_INACTIVE,
|
||||
FallbackReason::DnsFail => &DNS_FAIL,
|
||||
FallbackReason::ConnectFail => &CONNECT_FAIL,
|
||||
FallbackReason::Http404 => &HTTP_404,
|
||||
FallbackReason::Http5xx => &HTTP_5XX,
|
||||
}
|
||||
}
|
||||
|
||||
/// A dial completed over FIPS (any HTTP status that wasn't a fallback
|
||||
/// trigger — the peer was reached on the mesh).
|
||||
pub fn record_fips_ok() {
|
||||
FIPS_OK.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
/// A FIPS-capable dial fell back to Tor.
|
||||
pub fn record_fallback(reason: FallbackReason) {
|
||||
counter(reason).fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
/// `(fips_ok, connect_fail)` totals for the connectivity watcher: a window
|
||||
/// where connect_fail grows while fips_ok doesn't is a degraded data path —
|
||||
/// including the "daemon says connected but packets blackhole" failure the
|
||||
/// link-state check alone can't see (observed live 2026-07-27 on .198).
|
||||
pub fn totals() -> (u64, u64) {
|
||||
(
|
||||
FIPS_OK.load(Ordering::Relaxed),
|
||||
CONNECT_FAIL.load(Ordering::Relaxed),
|
||||
)
|
||||
}
|
||||
|
||||
/// Snapshot for `fips.status` (`dial_stats`). Process-lifetime counts.
|
||||
pub fn snapshot() -> serde_json::Value {
|
||||
let f1 = NO_NPUB.load(Ordering::Relaxed);
|
||||
let f2 = SERVICE_INACTIVE.load(Ordering::Relaxed);
|
||||
let f3 = DNS_FAIL.load(Ordering::Relaxed);
|
||||
let f4 = CONNECT_FAIL.load(Ordering::Relaxed);
|
||||
let f5 = HTTP_404.load(Ordering::Relaxed);
|
||||
let f6 = HTTP_5XX.load(Ordering::Relaxed);
|
||||
serde_json::json!({
|
||||
"fips_ok": FIPS_OK.load(Ordering::Relaxed),
|
||||
"fallbacks": {
|
||||
"no_npub": f1,
|
||||
"service_inactive": f2,
|
||||
"dns_fail": f3,
|
||||
"connect_fail": f4,
|
||||
"http_404": f5,
|
||||
"http_5xx": f6,
|
||||
"total": f1 + f2 + f3 + f4 + f5 + f6,
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn snapshot_counts_recorded_events() {
|
||||
// Counters are global; assert deltas rather than absolutes so this
|
||||
// test stays correct alongside any other test that dials.
|
||||
let before = snapshot();
|
||||
record_fips_ok();
|
||||
record_fallback(FallbackReason::ConnectFail);
|
||||
record_fallback(FallbackReason::Http404);
|
||||
let after = snapshot();
|
||||
let d = |v: &serde_json::Value, path: &[&str]| -> u64 {
|
||||
let mut cur = v;
|
||||
for p in path {
|
||||
cur = &cur[p];
|
||||
}
|
||||
cur.as_u64().unwrap()
|
||||
};
|
||||
assert_eq!(d(&after, &["fips_ok"]) - d(&before, &["fips_ok"]), 1);
|
||||
assert_eq!(
|
||||
d(&after, &["fallbacks", "connect_fail"]) - d(&before, &["fallbacks", "connect_fail"]),
|
||||
1
|
||||
);
|
||||
assert_eq!(
|
||||
d(&after, &["fallbacks", "http_404"]) - d(&before, &["fallbacks", "http_404"]),
|
||||
1
|
||||
);
|
||||
assert!(d(&after, &["fallbacks", "total"]) >= 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reason_keys_are_stable() {
|
||||
// These strings are the fips.status API surface — renaming one is a
|
||||
// breaking change for the UI.
|
||||
assert_eq!(FallbackReason::NoNpub.key(), "no_npub");
|
||||
assert_eq!(FallbackReason::ServiceInactive.key(), "service_inactive");
|
||||
assert_eq!(FallbackReason::DnsFail.key(), "dns_fail");
|
||||
assert_eq!(FallbackReason::ConnectFail.key(), "connect_fail");
|
||||
assert_eq!(FallbackReason::Http404.key(), "http_404");
|
||||
assert_eq!(FallbackReason::Http5xx.key(), "http_5xx");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,401 @@
|
||||
//! User-triggered FIPS upgrade from upstream GitHub releases.
|
||||
//!
|
||||
//! Flow (no auto-update, no background polling — user clicks a button):
|
||||
//! 1. Query GitHub for the latest *stable* release of `jmcorgan/fips`
|
||||
//! (`/releases/latest` returns the newest non-prerelease, non-draft
|
||||
//! tag, so release candidates like `v0.4.0-rc1` are skipped).
|
||||
//! 2. Compare its tag (e.g. `v0.3.0`) with the installed daemon version
|
||||
//! reported by `fipsctl --version`. A dev/pre-release build of the
|
||||
//! same number (`0.3.0-dev`) counts as older than the released tag.
|
||||
//! 3. Pick the Debian package asset matching the host architecture
|
||||
//! (`fips_<ver>_amd64.deb` / `_arm64.deb`) plus `checksums-linux.txt`.
|
||||
//! 4. Download both, SHA256-verify the .deb against the checksums file.
|
||||
//! 5. `sudo dpkg -i` the verified .deb, then restart the active fips unit.
|
||||
//!
|
||||
//! Upstream began publishing tagged releases with `.deb` artefacts and
|
||||
//! `checksums-linux.txt` (verified present as of v0.1.0 → v0.4.0-rc1), so
|
||||
//! the apply path is fully wired against those assets.
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
use super::{service, UPSTREAM_REPO};
|
||||
|
||||
const GITHUB_API: &str = "https://api.github.com";
|
||||
const USER_AGENT: &str = "archipelago-fips-updater";
|
||||
|
||||
/// Result of `check()` — what the dashboard renders.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct UpdateCheck {
|
||||
/// Currently installed daemon version (from `fipsctl --version`).
|
||||
pub current: Option<String>,
|
||||
/// Tag of the latest stable upstream release, e.g. `v0.3.0`.
|
||||
pub latest_version: String,
|
||||
/// True when the installed version is older than `latest_version`.
|
||||
pub update_available: bool,
|
||||
/// Release channel this check tracked. Currently always "stable".
|
||||
pub channel: String,
|
||||
/// Browser download URL of the architecture-matched .deb for the
|
||||
/// latest release, when one exists (informational; apply() re-resolves).
|
||||
pub asset_url: Option<String>,
|
||||
/// Human-readable note for the UI.
|
||||
pub notes: String,
|
||||
}
|
||||
|
||||
/// One GitHub release as we consume it.
|
||||
#[derive(Debug, Clone, Deserialize)]
|
||||
struct Release {
|
||||
tag_name: String,
|
||||
#[serde(default)]
|
||||
prerelease: bool,
|
||||
#[serde(default)]
|
||||
draft: bool,
|
||||
#[serde(default)]
|
||||
assets: Vec<Asset>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Deserialize)]
|
||||
struct Asset {
|
||||
name: String,
|
||||
browser_download_url: String,
|
||||
}
|
||||
|
||||
fn http_client() -> Result<reqwest::Client> {
|
||||
reqwest::Client::builder()
|
||||
.user_agent(USER_AGENT)
|
||||
.timeout(std::time::Duration::from_secs(30))
|
||||
.build()
|
||||
.context("Build HTTP client")
|
||||
}
|
||||
|
||||
/// Debian architecture string for the host (`amd64` / `arm64`). Returns
|
||||
/// the raw `std::env::consts::ARCH` for anything we don't map, so the
|
||||
/// asset lookup simply finds nothing and surfaces a clear error.
|
||||
fn deb_arch() -> &'static str {
|
||||
match std::env::consts::ARCH {
|
||||
"x86_64" => "amd64",
|
||||
"aarch64" => "arm64",
|
||||
other => other,
|
||||
}
|
||||
}
|
||||
|
||||
/// Query GitHub for the latest stable release and compare to the installed
|
||||
/// version. Never errors on "no package installed" — that is itself a valid
|
||||
/// state where an update is available.
|
||||
pub async fn check() -> Result<UpdateCheck> {
|
||||
let current = service::daemon_version().await.ok();
|
||||
let client = http_client()?;
|
||||
let release = fetch_latest_stable(&client).await?;
|
||||
|
||||
let update_available = match ¤t {
|
||||
Some(v) => version_is_older(v, &release.tag_name),
|
||||
None => true,
|
||||
};
|
||||
|
||||
let asset_url = release
|
||||
.assets
|
||||
.iter()
|
||||
.find(|a| is_deb_for_arch(&a.name))
|
||||
.map(|a| a.browser_download_url.clone());
|
||||
|
||||
let notes = if update_available {
|
||||
format!(
|
||||
"Update available: {} (installed: {})",
|
||||
release.tag_name,
|
||||
current.as_deref().unwrap_or("not installed")
|
||||
)
|
||||
} else {
|
||||
format!("Up to date ({})", release.tag_name)
|
||||
};
|
||||
|
||||
Ok(UpdateCheck {
|
||||
current,
|
||||
latest_version: release.tag_name,
|
||||
update_available,
|
||||
channel: "stable".to_string(),
|
||||
asset_url,
|
||||
notes,
|
||||
})
|
||||
}
|
||||
|
||||
/// Download, verify, and install the latest stable FIPS release, then
|
||||
/// restart the daemon. Steps: resolve release → match .deb for this arch
|
||||
/// → download .deb + checksums → SHA256-verify → `sudo dpkg -i` → restart.
|
||||
pub async fn apply() -> Result<()> {
|
||||
let client = http_client()?;
|
||||
let release = fetch_latest_stable(&client).await?;
|
||||
|
||||
let deb = release
|
||||
.assets
|
||||
.iter()
|
||||
.find(|a| is_deb_for_arch(&a.name))
|
||||
.ok_or_else(|| {
|
||||
anyhow::anyhow!(
|
||||
"release {} has no .deb for architecture {}",
|
||||
release.tag_name,
|
||||
deb_arch()
|
||||
)
|
||||
})?;
|
||||
let checksums = release
|
||||
.assets
|
||||
.iter()
|
||||
.find(|a| a.name == "checksums-linux.txt")
|
||||
.ok_or_else(|| {
|
||||
anyhow::anyhow!("release {} has no checksums-linux.txt", release.tag_name)
|
||||
})?;
|
||||
|
||||
// Download the .deb (bytes) and the checksums (text).
|
||||
let deb_bytes = client
|
||||
.get(&deb.browser_download_url)
|
||||
.send()
|
||||
.await
|
||||
.context("download .deb")?
|
||||
.error_for_status()
|
||||
.context(".deb download HTTP error")?
|
||||
.bytes()
|
||||
.await
|
||||
.context("read .deb body")?;
|
||||
let checksums_text = client
|
||||
.get(&checksums.browser_download_url)
|
||||
.send()
|
||||
.await
|
||||
.context("download checksums")?
|
||||
.error_for_status()
|
||||
.context("checksums download HTTP error")?
|
||||
.text()
|
||||
.await
|
||||
.context("read checksums body")?;
|
||||
|
||||
// Verify SHA256 against the checksums manifest (sha256sum format:
|
||||
// "<hex>␠␠<filename>"). The filename column may include a leading
|
||||
// "*" (binary mode) or a path prefix, so match on the basename.
|
||||
let expected = checksums_text
|
||||
.lines()
|
||||
.filter_map(|line| {
|
||||
let mut parts = line.split_whitespace();
|
||||
let hash = parts.next()?;
|
||||
let name = parts.next()?.trim_start_matches('*');
|
||||
let base = name.rsplit('/').next().unwrap_or(name);
|
||||
(base == deb.name).then(|| hash.to_lowercase())
|
||||
})
|
||||
.next()
|
||||
.ok_or_else(|| anyhow::anyhow!("checksums-linux.txt has no entry for {}", deb.name))?;
|
||||
|
||||
let actual = {
|
||||
let mut hasher = Sha256::new();
|
||||
hasher.update(&deb_bytes);
|
||||
hex::encode(hasher.finalize())
|
||||
};
|
||||
if actual != expected {
|
||||
anyhow::bail!(
|
||||
"SHA256 mismatch for {}: expected {}, got {}",
|
||||
deb.name,
|
||||
expected,
|
||||
actual
|
||||
);
|
||||
}
|
||||
|
||||
// Stage the verified .deb in /tmp (shared with the host — the
|
||||
// service runs with PrivateTmp=no) and install it.
|
||||
let dest = std::env::temp_dir().join(&deb.name);
|
||||
tokio::fs::write(&dest, &deb_bytes)
|
||||
.await
|
||||
.with_context(|| format!("write {}", dest.display()))?;
|
||||
|
||||
// Run dpkg via `systemd-run` rather than `sudo dpkg` directly. The
|
||||
// archipelago service runs under `ProtectSystem=strict`, so `/usr`
|
||||
// and `/var/lib/dpkg` are read-only *inside the service's mount
|
||||
// namespace* — and a `sudo` child inherits that namespace, so a
|
||||
// bare `sudo dpkg -i` fails with "Read-only file system" on the
|
||||
// dpkg database. `systemd-run` asks PID 1 to launch the command in
|
||||
// a fresh transient scope outside our sandbox, where the real
|
||||
// (writable) host filesystem is visible. `--wait` blocks until it
|
||||
// finishes and propagates the exit status; `--pipe` forwards
|
||||
// dpkg's output; `--collect` reaps the unit even on failure.
|
||||
//
|
||||
// dpkg flags, both load-bearing for this package specifically:
|
||||
// --force-confold: the fips package ships conffiles under
|
||||
// /etc/fips that archipelago rewrites at install time, so dpkg
|
||||
// hits an interactive "keep/replace?" conffile prompt. With our
|
||||
// closed stdin that aborts the configure step ("EOF on stdin at
|
||||
// conffile prompt") and leaves the package half-unpacked
|
||||
// (status `iU`), which `fips.status` then reports as
|
||||
// `installed:false`. confold = keep our managed config, no prompt.
|
||||
// --force-downgrade: ISO/dev nodes carry `0.3.0-dev-1`, which dpkg
|
||||
// orders as NEWER than the stable tag `0.3.0` (a trailing
|
||||
// `-dev` sorts above the bare release). Moving a dev build onto
|
||||
// the stable line is therefore a dpkg "downgrade"; without this
|
||||
// flag dpkg warns and exits non-zero. Our own version_is_older()
|
||||
// gate already decided this is the wanted direction.
|
||||
// DEBIAN_FRONTEND=noninteractive belt-and-suspenders against any
|
||||
// other maintainer-script prompt.
|
||||
let dpkg = tokio::process::Command::new("sudo")
|
||||
.args([
|
||||
"-n",
|
||||
"systemd-run",
|
||||
"--collect",
|
||||
"--wait",
|
||||
"--quiet",
|
||||
"--pipe",
|
||||
"--",
|
||||
"env",
|
||||
"DEBIAN_FRONTEND=noninteractive",
|
||||
"dpkg",
|
||||
"--force-confold",
|
||||
"--force-downgrade",
|
||||
"-i",
|
||||
])
|
||||
.arg(&dest)
|
||||
.output()
|
||||
.await
|
||||
.context("sudo systemd-run dpkg -i failed to launch")?;
|
||||
// Best-effort cleanup regardless of dpkg result.
|
||||
let _ = tokio::fs::remove_file(&dest).await;
|
||||
if !dpkg.status.success() {
|
||||
anyhow::bail!(
|
||||
"dpkg -i {} exited {}: {}",
|
||||
deb.name,
|
||||
dpkg.status,
|
||||
String::from_utf8_lossy(&dpkg.stderr).trim()
|
||||
);
|
||||
}
|
||||
|
||||
// Restart whichever fips unit is supervising the daemon so the new
|
||||
// binary takes over.
|
||||
let unit = service::active_unit().await;
|
||||
service::restart(unit)
|
||||
.await
|
||||
.with_context(|| format!("restart {} after install", unit))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// `/releases/latest` returns the most recent non-prerelease, non-draft
|
||||
/// release. We still re-check the flags defensively in case the endpoint
|
||||
/// or repo settings change.
|
||||
async fn fetch_latest_stable(client: &reqwest::Client) -> Result<Release> {
|
||||
let url = format!("{}/repos/{}/releases/latest", GITHUB_API, UPSTREAM_REPO);
|
||||
let resp = client
|
||||
.get(&url)
|
||||
.header("Accept", "application/vnd.github+json")
|
||||
.send()
|
||||
.await
|
||||
.context("GitHub releases/latest API")?;
|
||||
if !resp.status().is_success() {
|
||||
anyhow::bail!("GitHub releases/latest API returned {}", resp.status());
|
||||
}
|
||||
let release: Release = resp.json().await.context("Parse release JSON")?;
|
||||
if release.draft || release.prerelease {
|
||||
anyhow::bail!(
|
||||
"releases/latest returned a {} release ({})",
|
||||
if release.draft { "draft" } else { "prerelease" },
|
||||
release.tag_name
|
||||
);
|
||||
}
|
||||
Ok(release)
|
||||
}
|
||||
|
||||
fn is_deb_for_arch(name: &str) -> bool {
|
||||
name.starts_with("fips_") && name.ends_with(&format!("_{}.deb", deb_arch()))
|
||||
}
|
||||
|
||||
/// Parse the leading `MAJOR.MINOR.PATCH` triple from a version string,
|
||||
/// plus whether a pre-release suffix (`-dev`, `-rc1`, …) follows it.
|
||||
fn parse_version(s: &str) -> Option<((u64, u64, u64), bool)> {
|
||||
// Take the first whitespace token, drop a leading 'v'.
|
||||
let tok = s.split_whitespace().next().unwrap_or(s);
|
||||
let tok = tok.strip_prefix('v').unwrap_or(tok);
|
||||
// Split off any pre-release / build suffix.
|
||||
let (core, rest) = match tok.find(|c: char| c == '-' || c == '+') {
|
||||
Some(i) => (&tok[..i], &tok[i..]),
|
||||
None => (tok, ""),
|
||||
};
|
||||
let mut it = core.split('.');
|
||||
let major = it.next()?.parse::<u64>().ok()?;
|
||||
let minor = it.next().unwrap_or("0").parse::<u64>().ok()?;
|
||||
let patch = it.next().unwrap_or("0").parse::<u64>().ok()?;
|
||||
let has_prerelease = rest.starts_with('-');
|
||||
Some(((major, minor, patch), has_prerelease))
|
||||
}
|
||||
|
||||
/// True when `installed` is strictly older than release tag `latest`.
|
||||
/// Same numeric triple but `installed` carries a pre-release suffix while
|
||||
/// `latest` doesn't ⇒ installed is older (e.g. `0.3.0-dev` < `v0.3.0`).
|
||||
/// If either side can't be parsed, fall back to "differs ⇒ update".
|
||||
fn version_is_older(installed: &str, latest: &str) -> bool {
|
||||
match (parse_version(installed), parse_version(latest)) {
|
||||
(Some((ic, ipre)), Some((lc, lpre))) => {
|
||||
if ic != lc {
|
||||
ic < lc
|
||||
} else {
|
||||
// Equal cores: a pre-release is older than the final release.
|
||||
ipre && !lpre
|
||||
}
|
||||
}
|
||||
_ => {
|
||||
// Unparseable: be conservative — offer the update unless the
|
||||
// installed string already mentions the latest tag.
|
||||
!installed.contains(latest.trim_start_matches('v'))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_deb_arch_maps_known() {
|
||||
// On the host running tests this is whatever the test arch is;
|
||||
// just assert it returns a non-empty, lowercase token.
|
||||
let a = deb_arch();
|
||||
assert!(!a.is_empty());
|
||||
assert_eq!(a, a.to_lowercase());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_version_older() {
|
||||
assert!(version_is_older("0.3.0-dev (rev abc123)", "v0.3.0"));
|
||||
assert!(version_is_older("0.2.1", "v0.3.0"));
|
||||
assert!(version_is_older("0.3.0-rc1", "v0.3.0"));
|
||||
assert!(!version_is_older("0.3.0", "v0.3.0"));
|
||||
assert!(!version_is_older("0.4.0", "v0.3.0"));
|
||||
assert!(!version_is_older("0.3.1", "v0.3.0"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_version() {
|
||||
assert_eq!(parse_version("v0.3.0"), Some(((0, 3, 0), false)));
|
||||
assert_eq!(parse_version("0.3.0-dev (rev x)"), Some(((0, 3, 0), true)));
|
||||
assert_eq!(parse_version("0.4.0-rc1"), Some(((0, 4, 0), true)));
|
||||
assert_eq!(parse_version("1.2"), Some(((1, 2, 0), false)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_deb_for_arch() {
|
||||
let arch = deb_arch();
|
||||
assert!(is_deb_for_arch(&format!("fips_0.3.0_{}.deb", arch)));
|
||||
assert!(!is_deb_for_arch("fips_0.3.0_someotherarch.deb"));
|
||||
assert!(!is_deb_for_arch("checksums-linux.txt"));
|
||||
assert!(!is_deb_for_arch(&format!(
|
||||
"fips-0.3.0-linux-{}.tar.gz",
|
||||
arch
|
||||
)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_update_check_serialises() {
|
||||
let uc = UpdateCheck {
|
||||
current: Some("0.3.0-dev".to_string()),
|
||||
latest_version: "v0.3.0".to_string(),
|
||||
update_available: true,
|
||||
channel: "stable".to_string(),
|
||||
asset_url: Some("https://example/fips_0.3.0_amd64.deb".to_string()),
|
||||
notes: "test".to_string(),
|
||||
};
|
||||
let json = serde_json::to_string(&uc).unwrap();
|
||||
assert!(json.contains("latest_version"));
|
||||
assert!(json.contains("update_available"));
|
||||
assert!(json.contains("stable"));
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user