From 99f5d383ed2f4a26c8a13d7c115b3af8dc21b1ab Mon Sep 17 00:00:00 2001 From: Michiel de Jong Date: Fri, 11 Sep 2026 13:42:03 +0200 Subject: [PATCH 1/2] Add full-instance backup capture and offline restore --- CHANGELOG.md | 4 + Cargo.lock | 1 + TESTING_COVERAGE.md | 17 + docs/src/SUMMARY.md | 1 + docs/src/instance-backups.md | 108 +++ lib/src/db.rs | 725 +++++++++--------- lib/src/db/kv_store.rs | 12 + lib/src/db/maintenance.rs | 125 ++++ lib/src/db/redb_store.rs | 91 +++ lib/src/sync/engine.rs | 479 ++++++------ lib/src/sync/ws_apply.rs | 78 +- scripts/backup-instance.sh | 7 + server/Cargo.toml | 7 +- server/build.rs | 36 + server/src/appstate.rs | 4 + server/src/backup.rs | 1123 ++++++++++++++++++++++++++++ server/src/bin.rs | 15 + server/src/config.rs | 23 + server/src/lib.rs | 1 + server/src/routes.rs | 1 + server/src/serve.rs | 30 +- server/tests/it/instance_backup.rs | 197 +++++ server/tests/it/main.rs | 1 + 23 files changed, 2466 insertions(+), 620 deletions(-) create mode 100644 docs/src/instance-backups.md create mode 100644 lib/src/db/maintenance.rs create mode 100755 scripts/backup-instance.sh create mode 100644 server/src/backup.rs create mode 100644 server/tests/it/instance_backup.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 19d96cea17..27e1179b30 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -58,6 +58,10 @@ See [STATUS.md](server/STATUS.md) to learn more about which features will remain switch. `Db::init_redb_file` now owns the 100ms durable-flush tick for every binding, and an idle tick no longer writes anything. - Fix remaining `clippy` warnings in `wasm/src/lib.rs` blocking `develop`'s pre-commit hook ([#1508](https://github.com/ontola/atomic-server/issues/1508)). +- Add opt-in full-instance backups: `--backup-dir`, local authenticated backup + control, `backup`/`restore` CLI commands, checksummed ZIP64 archives and an + offline restore guard. Writes and incoming sync application pause during + capture and resume before compression. See [instance backups](docs/src/instance-backups.md). ## [v0.41.0-beta.7] - 2026-09-12 diff --git a/Cargo.lock b/Cargo.lock index 557994848c..03730e5d35 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1266,6 +1266,7 @@ dependencies = [ "portpicker", "rand 0.8.6", "rcgen 0.14.8", + "redb", "regex", "reqwest 0.13.3", "ring", diff --git a/TESTING_COVERAGE.md b/TESTING_COVERAGE.md index 6bee8ca6b0..a2e3845263 100644 --- a/TESTING_COVERAGE.md +++ b/TESTING_COVERAGE.md @@ -180,6 +180,23 @@ duplicate issues/comments. Live proxy OAuth, GitHub writes and a guided uncertain-write recovery UI remain unverified/unbuilt; proxy v40 CORS and browser OAuth are verified, but its GitHub credential returns 404 for the private sandbox. +Instance backup: `server/src/backup.rs` covers full redb/file round-trip including +Loro historical checkout, blobs, envelopes and identities; future byte tables; +concurrent writer exclusion; buffered-batch refusal; capture error/panic recovery; +loopback/token authorization; unsafe paths, hash mismatch and corrupt-database +refusal. `lib/src/db/maintenance.rs` covers nested admission during drain, +overlapping pause refusal, queued work and cancellation recovery. The shared +sync-engine test proves a paused import is neither acknowledged nor dropped. +`server/tests/it/instance_backup.rs` starts a real server process, replicates a +second node over WebSockets, invokes the backup CLI, replicates a later change, +restores the earlier checkpoint and verifies accidental startup is refused. + +Remaining backup coverage gaps: OS-level disk-full/power-loss injection, +large (>4 GiB) ZIP64 fixtures and pause-duration benchmarks, and a dedicated +Iroh disconnect-during-capture test. The gate is shared by both transports; +these tests do not assert a globally synchronized checkpoint or lifetime history +retention. See `docs/src/instance-backups.md` for the operational limits. + What is tested, at which layer, and — the part that matters — **what is not**. This exists because the protocol is far better tested than the glue around it, diff --git a/docs/src/SUMMARY.md b/docs/src/SUMMARY.md index b7d1ab86ea..62a4873e10 100644 --- a/docs/src/SUMMARY.md +++ b/docs/src/SUMMARY.md @@ -13,6 +13,7 @@ - [When (not) to use it](atomicserver/when-to-use.md) - [Installation](atomicserver/installation.md) - [Browser peer sync](browser-peer-sync.md) + - [Full-instance backups](instance-backups.md) - [Using the GUI](atomicserver/gui.md) - [Tables](atomicserver/gui/tables.md) - [AI and Atomic Assistant](atomicserver/gui/ai-and-atomic-assistant.md) diff --git a/docs/src/instance-backups.md b/docs/src/instance-backups.md new file mode 100644 index 0000000000..d6c25b3adf --- /dev/null +++ b/docs/src/instance-backups.md @@ -0,0 +1,108 @@ +# Full-instance backups + +Run a dedicated replica with an output directory outside its data and config +folders. Data and config must be separate, non-overlapping directories; keep +`--cache-dir` outside both too: + +```sh +atomic-server --data-dir /srv/atomic/data --config-dir /srv/atomic/config \ + --backup-dir /srv/atomic-backups +``` + +Request a backup on that same machine: + +```sh +atomic-server --config-dir /srv/atomic/config backup --server http://127.0.0.1:9883 +``` + +The command prints the archive path after verification, or exits nonzero on +failure. The server creates `backup.token` in its config directory with private +permissions. Backup control requires that token and a loopback connection; drive +write access does not grant instance backup access. Do not share the token. +The `--backup-dir` option also accepts `ATOMIC_BACKUP_DIR`. + +The server drains admitted operations (up to 30 seconds), temporarily gates new +HTTP requests with `503` and `Retry-After`, and pauses incoming sync application. +It copies every redb table through one read transaction into a fresh database, +while a storage write transaction prevents all writers from crossing the capture +boundary. This includes Loro histories, retained signed envelopes, blobs, node +identity and internal metadata. Associated data/config files are captured under +the same barrier. Symlinks and special files fail the backup rather than escaping +the source directories. Operator edits to these directories should wait until the +capture finishes. + +Capture time grows with database and file size; multi-GB pause duration has not +yet been benchmarked. Normal writes and queued sync resume before the staged +files are compressed. +Sync connections need not be disconnected: pending operations wait for admission +and then continue; a timed-out connection uses the existing reconnect/reconcile +path. No maintenance update is acknowledged and discarded. The output is a +ZIP64 archive named `atomic-backup--.zip`. File sizes +and BLAKE3 hashes, package version, Git revision, redb version, capture time, retention policy and source path +mapping are recorded in `manifest.json`. The ZIP is read back and hashes checked +before it receives its final name. Partial work uses private temporary paths. + +This is a checkpoint of **this instance**, not proof that it received every +change from every peer. Replication freshness is explicitly reported as unknown. +It cannot recover Loro history or envelopes that were previously deleted or +never replicated. Cached vector indexes are outside the data/config roots and +are rebuilt when needed. External services, environment variables, separately +mounted storage and an exact executable are not bundled; retain your binary and +deployment configuration alongside the archive. A ZIP contains private data and +credentials; keep an independent protected copy off-machine. + +## Restore + +Use the same server version and a destination that does not exist: + +```sh +atomic-server restore --archive /srv/atomic-backups/atomic-backup-EXAMPLE.zip \ + --target /srv/atomic-restored +``` + +Restore validates the manifest, paths, all file hashes and the redb database +before promoting staged files. It does not boot the node or start networking. +The result contains `data/`, `config/` and `manifest.json`. A marker in `data/` +prevents accidentally starting a normal server with copied node identities. +Inspect the restored files first. For actual recovery, stop the original instance +before explicitly activating the restored one: + +```sh +atomic-server --data-dir /srv/atomic-restored/data \ + --config-dir /srv/atomic-restored/config --activate-restored +``` + +Reapply the original deployment settings, including the envelope-retention +policy recorded in the manifest. Activation can reconnect existing peers and +integrations. Reconnecting an old +checkpoint can merge newer remote changes back into it. Experimental branches +need a separate, network-isolated environment; the restore command deliberately +does not start one automatically. Do not activate both copies with the same node +identity on the same network. + +## Nightly scheduling + +`scripts/backup-instance.sh` accepts `ATOMIC_SERVER_BIN`, `ATOMIC_BACKUP_SERVER` +and the required `ATOMIC_BACKUP_TOKEN_FILE`. For example, a cron entry: + +```cron +0 2 * * * ATOMIC_BACKUP_TOKEN_FILE=/srv/atomic/config/backup.token ATOMIC_SERVER_BIN=/usr/local/bin/atomic-server /path/to/atomic-server/scripts/backup-instance.sh >> /srv/atomic-backup.log 2>&1 +``` + +On macOS, use the same script and environment with a launchd job whose +`StartCalendarInterval` has `Hour=2` and `Minute=0`, and set explicit +`StandardOutPath`/`StandardErrorPath`. The server rejects overlapping jobs. The +CLI waits for its job, detects server restarts or replaced status and exits +nonzero on failure. Check logs and disk capacity; nothing prunes old backups. + +For status, use authenticated `GET /__atomic/backup` from loopback. An +unauthenticated request cannot see filesystem paths or backup errors. `POST` to +that route starts a job and returns its ID. Client disconnect does not cancel a +capture. Process crashes may leave `.atomic-backup-*` staging directories, but +never make a partial ZIP appear complete; remove stale staging only while no +backup job is running. Restore promotion errors may leave a partial destination; +inspect it and choose a fresh destination for retry. + +Test a restore before relying on the first archive, then repeat periodically. +Take an extra checkpoint before risky experiments. Nightly retention accepts up +to a day of data loss; it does not retain every intermediate edit. diff --git a/lib/src/db.rs b/lib/src/db.rs index b20b31988f..6305b8f77e 100644 --- a/lib/src/db.rs +++ b/lib/src/db.rs @@ -8,6 +8,7 @@ mod encoding; #[cfg(feature = "db-redb")] pub mod encrypted_backend; pub mod kv_store; +pub mod maintenance; #[cfg(feature = "db-sled")] mod migrations; #[cfg(all(feature = "db-redb", target_arch = "wasm32"))] @@ -305,6 +306,8 @@ type PendingBlobRequests = HashMap<[u8; 32], (String, web_time::Instant)>; /// `Db` should be easily, cheaply clone-able, as users of this library could have one `Db` per connection. #[derive(Clone)] pub struct Db { + /// Shared admission barrier for coherent instance backup. + pub maintenance: maintenance::Maintenance, /// The key-value store backend. Abstracted behind a trait so different /// backends (sled, BTreeMap, etc.) can be used interchangeably. pub kv: Arc, @@ -623,6 +626,7 @@ impl Db { base_domain, sync_policy: default_sync_policy(), envelope_retention: Arc::new(RwLock::new(Default::default())), + maintenance: maintenance::Maintenance::default(), pending_blob_requests: Arc::new(RwLock::new(HashMap::new())), }; @@ -665,6 +669,7 @@ impl Db { base_domain, sync_policy: default_sync_policy(), envelope_retention: Arc::new(RwLock::new(Default::default())), + maintenance: maintenance::Maintenance::default(), pending_blob_requests: Arc::new(RwLock::new(HashMap::new())), }; @@ -703,6 +708,7 @@ impl Db { base_domain, sync_policy: default_sync_policy(), envelope_retention: Arc::new(RwLock::new(Default::default())), + maintenance: maintenance::Maintenance::default(), pending_blob_requests: Arc::new(RwLock::new(HashMap::new())), }; @@ -802,6 +808,7 @@ impl Db { base_domain, sync_policy: default_sync_policy(), envelope_retention: Arc::new(RwLock::new(Default::default())), + maintenance: maintenance::Maintenance::default(), pending_blob_requests: Arc::new(RwLock::new(HashMap::new())), }; @@ -984,6 +991,7 @@ impl Db { base_domain, sync_policy: default_sync_policy(), envelope_retention: Arc::new(RwLock::new(Default::default())), + maintenance: maintenance::Maintenance::default(), pending_blob_requests: Arc::new(RwLock::new(HashMap::new())), }; @@ -3561,33 +3569,37 @@ impl Storelike for Db { /// Validates datatypes and required props presence. #[instrument(skip_all)] async fn add_atoms(&self, atoms: Vec) -> AtomicResult<()> { - // Start with a nested HashMap, containing only strings. - let mut map: HashMap = HashMap::new(); - for atom in atoms { - match map.get_mut(&atom.subject) { - // Resource exists in map - Some(resource) => { - resource - .set_string(atom.property.clone(), &atom.value.to_string(), self) - .await - .map_err(|e| format!("Failed adding attom {}. {}", atom, e))?; + self.maintenance + .run(async { + // Start with a nested HashMap, containing only strings. + let mut map: HashMap = HashMap::new(); + for atom in atoms { + match map.get_mut(&atom.subject) { + // Resource exists in map + Some(resource) => { + resource + .set_string(atom.property.clone(), &atom.value.to_string(), self) + .await + .map_err(|e| format!("Failed adding attom {}. {}", atom, e))?; + } + // Resource does not exist + None => { + let mut resource = Resource::new(atom.subject.to_string()); + resource + .set_string(atom.property.clone(), &atom.value.to_string(), self) + .await + .map_err(|e| format!("Failed adding attom {}. {}", atom, e))?; + map.insert(atom.subject, resource); + } + } } - // Resource does not exist - None => { - let mut resource = Resource::new(atom.subject.to_string()); - resource - .set_string(atom.property.clone(), &atom.value.to_string(), self) - .await - .map_err(|e| format!("Failed adding attom {}. {}", atom, e))?; - map.insert(atom.subject, resource); + for (_subject, resource) in map.iter() { + self.add_resource(resource).await? } - } - } - for resource in map.values() { - self.add_resource(resource).await? - } - self.kv.flush()?; - Ok(()) + self.kv.flush()?; + Ok(()) + }) + .await } /// Maps a host (domain/subdomain) to a Drive DID. @@ -3642,13 +3654,14 @@ impl Storelike for Db { } else { None }; - self.persist_resource_projection( - resource, - check_required_props, - update_index, - overwrite_existing, - ) - .await + self.maintenance + .run(self.persist_resource_projection( + resource, + check_required_props, + update_index, + overwrite_existing, + )) + .await } /// Apply a single signed Commit to the Db. @@ -3661,6 +3674,7 @@ impl Storelike for Db { commit: Commit, opts: &CommitOpts, ) -> AtomicResult { + self.maintenance.run(async { let store = self; // Persisting a commit is a read-modify-write: `validate_and_build_response` @@ -3993,6 +4007,7 @@ impl Storelike for Db { } } Ok(commit_response) + }).await } fn get_default_agent(&self) -> AtomicResult { @@ -4013,179 +4028,189 @@ impl Storelike for Db { #[instrument(skip_all)] async fn get_resource(&self, subject: &Subject) -> AtomicResult { - let normalized = self.normalize_subject(subject); - let subject_str = normalized.pure_id(); - if let Ok(propvals) = self.get_propvals(&subject_str) { - let mut res_subject = normalized.clone(); - - // If it's a DID and we don't have a hint in the requested subject, - // check if we have one persisted in the did_mapping tree. - if let Subject::Did { - drive_hint: None, .. - } = &res_subject - { - if let Ok(Some(hint_bin)) = self.kv.get(Tree::DidMapping, subject_str.as_bytes()) { - if let Ok(hint) = std::str::from_utf8(&hint_bin) { - res_subject = res_subject.set_drive_hint(hint.to_string()); + self.maintenance + .run(async { + let normalized = self.normalize_subject(subject); + let subject_str = normalized.pure_id(); + if let Ok(propvals) = self.get_propvals(&subject_str) { + let mut res_subject = normalized.clone(); + + // If it's a DID and we don't have a hint in the requested subject, + // check if we have one persisted in the did_mapping tree. + if let Subject::Did { + drive_hint: None, .. + } = &res_subject + { + if let Ok(Some(hint_bin)) = + self.kv.get(Tree::DidMapping, subject_str.as_bytes()) + { + if let Ok(hint) = std::str::from_utf8(&hint_bin) { + res_subject = res_subject.set_drive_hint(hint.to_string()); + } + } } - } - } - let mut resource = Resource::from_propvals(propvals, res_subject); - // Authoritative merged CRDT state (full oplog) lives in LoroSnapshots. - // Propvals may carry a smaller incremental `loroUpdate` from the last commit. - if let Ok(Some(snapshot)) = self.kv.get( - crate::db::trees::Tree::LoroSnapshots, - subject_str.as_bytes(), - ) { - if let Ok(doc) = crate::loro::AtomicLoroDoc::from_snapshot(&snapshot) { - // We already hold the exact bytes `doc` was just imported - // from — reuse them instead of having `apply_state_doc` - // re-export an equivalent snapshot. This is the hot path - // for every resource read (including once per member of - // a collection query), so the saved export is per-read, - // not one-off. - let _ = resource.apply_state_doc_with_snapshot(doc, snapshot); - } - } - Ok(resource) - } else { - // Resolve the subject to a full URL for network operations - let origin = self - .get_base_domain() - .unwrap_or_else(|| "http://localhost".to_string()); - let resolved_url = normalized.resolve(&origin); - - // If the resource is not found, it might be an endpoint. - // This is checking if the subject matches one of the endpoints - if let Ok(url) = url::Url::parse(&resolved_url) { - if self.is_endpoint(&url) { - let agent_opt = self.get_default_agent().ok(); - let for_agent = if let Some(agent) = &agent_opt { - ForAgent::from(agent) - } else { - ForAgent::Public - }; - return Ok(self - .call_endpoint(&resolved_url, &for_agent) - .await? - .to_single()); - } - } - let resolved_url = normalized.resolve(&origin); - - if normalized.is_did() || normalized.path().starts_with("/did") { - // If it's an agent DID and not found locally, return a minimal resource - // instead of an error. This is important for "just-in-time" agent registration. - if normalized.is_agent_did() || normalized.path().starts_with("/did:ad:agent:") { - let lookup = if normalized.path().starts_with('/') { - &normalized.path()[1..] - } else { - &normalized.path() - }; - if let Some(pubkey) = lookup.strip_prefix("did:ad:agent:") { - if let Ok(agent) = crate::agents::Agent::new_from_public_key(pubkey) { - if let Ok(mut resource) = agent.to_resource() { - // A lookup is not creation of an agent. There is - // no known creation date or signed history yet. - // Seed the same fallback ops on every read and - // device, otherwise a refresh replaces the cached - // profile with new ops and grows its vault backup. - resource.remove_propval(crate::urls::CREATED_AT)?; - let doc = crate::loro::AtomicLoroDoc::new(); - doc.set_peer_id(0)?; - let ordered: std::collections::BTreeMap<_, _> = - resource.get_propvals().iter().collect(); - for (property, value) in ordered { - doc.set_property(property, value)?; + let mut resource = Resource::from_propvals(propvals, res_subject); + // Authoritative merged CRDT state (full oplog) lives in LoroSnapshots. + // Propvals may carry a smaller incremental `loroUpdate` from the last commit. + if let Ok(Some(snapshot)) = self.kv.get( + crate::db::trees::Tree::LoroSnapshots, + subject_str.as_bytes(), + ) { + if let Ok(doc) = crate::loro::AtomicLoroDoc::from_snapshot(&snapshot) { + // We already hold the exact bytes `doc` was just imported + // from — reuse them instead of having `apply_state_doc` + // re-export an equivalent snapshot. This is the hot path + // for every resource read (including once per member of + // a collection query), so the saved export is per-read, + // not one-off. + let _ = resource.apply_state_doc_with_snapshot(doc, snapshot); + } + } + Ok(resource) + } else { + // Resolve the subject to a full URL for network operations + let origin = self + .get_base_domain() + .unwrap_or_else(|| "http://localhost".to_string()); + let resolved_url = normalized.resolve(&origin); + + // If the resource is not found, it might be an endpoint. + // This is checking if the subject matches one of the endpoints + if let Ok(url) = url::Url::parse(&resolved_url) { + if self.is_endpoint(&url) { + let agent_opt = self.get_default_agent().ok(); + let for_agent = if let Some(agent) = &agent_opt { + ForAgent::from(agent) + } else { + ForAgent::Public + }; + return Ok(self + .call_endpoint(&resolved_url, &for_agent) + .await? + .to_single()); + } + } + let resolved_url = normalized.resolve(&origin); + + if normalized.is_did() || normalized.path().starts_with("/did") { + // If it's an agent DID and not found locally, return a minimal resource + // instead of an error. This is important for "just-in-time" agent registration. + if normalized.is_agent_did() + || normalized.path().starts_with("/did:ad:agent:") + { + let lookup = if normalized.path().starts_with('/') { + &normalized.path()[1..] + } else { + &normalized.path() + }; + if let Some(pubkey) = lookup.strip_prefix("did:ad:agent:") { + if let Ok(agent) = crate::agents::Agent::new_from_public_key(pubkey) + { + if let Ok(mut resource) = agent.to_resource() { + // A lookup is not creation of an agent. There is + // no known creation date or signed history yet. + // Seed the same fallback ops on every read and + // device, otherwise a refresh replaces the cached + // profile with new ops and grows its vault backup. + resource.remove_propval(crate::urls::CREATED_AT)?; + let doc = crate::loro::AtomicLoroDoc::new(); + doc.set_peer_id(0)?; + let ordered: std::collections::BTreeMap<_, _> = + resource.get_propvals().iter().collect(); + for (property, value) in ordered { + doc.set_property(property, value)?; + } + doc.doc() + .commit_with(loro::CommitOptions::new().timestamp(0)); + resource.apply_state_doc(doc)?; + return Ok(resource); + } } - doc.doc() - .commit_with(loro::CommitOptions::new().timestamp(0)); - resource.apply_state_doc(doc)?; - return Ok(resource); } } - } - } - - if normalized.is_did() || resolved_url.starts_with("/did:") { - return Err(AtomicError::not_found(format!( - "DID Resource {} not found locally", - resolved_url - ))); - } - return self - .handle_not_found( - &resolved_url, - format!("Resource {} not found locally", resolved_url).into(), - self.get_default_agent().ok().as_ref(), - ) - .await; - } + if normalized.is_did() || resolved_url.starts_with("/did:") { + return Err(AtomicError::not_found(format!( + "DID Resource {} not found locally", + resolved_url + ))); + } - // Only attempt a network fetch for external subjects. - // Fetching a local URL would cause the server to request itself, - // creating an infinite loop. - // - // `is_local()` alone is not enough: the canonical atomicdata.dev - // vocabulary is deliberately kept `External` even on its own host - // (see `Subject::CANONICAL_VOCABULARY_PREFIXES`), so on - // atomicdata.dev a miss for `/properties/*` would fall through to a - // network fetch of this very server. Anything served from our own - // authority is ours whether or not it is `Internal`, so compare - // authorities too — and ignore the scheme, since a store migrated - // as `https://` must not self-fetch when served over `http://`. - let base_domain = self.get_base_domain(); - let resolved_subject_obj = Subject::from_raw(&resolved_url, base_domain.as_deref()); - let is_own_authority = base_domain - .as_deref() - .map(|base| { - let strip = |s: &str| { - s.trim_start_matches("https://") - .trim_start_matches("http://") - .trim_end_matches('/') - .to_string() - }; - let base_authority = strip(base); - let resolved = strip(&resolved_url); - resolved == base_authority - || resolved.starts_with(&format!("{}/", base_authority)) - }) - .unwrap_or(false); + return self + .handle_not_found( + &resolved_url, + format!("Resource {} not found locally", resolved_url).into(), + self.get_default_agent().ok().as_ref(), + ) + .await; + } - if resolved_subject_obj.is_local() || is_own_authority { - return self - .handle_not_found( - &resolved_url, - "Not found in DB".into(), - self.get_default_agent().ok().as_ref(), - ) - .await; - } + // Only attempt a network fetch for external subjects. + // Fetching a local URL would cause the server to request itself, + // creating an infinite loop. + // + // `is_local()` alone is not enough: the canonical atomicdata.dev + // vocabulary is deliberately kept `External` even on its own host + // (see `Subject::CANONICAL_VOCABULARY_PREFIXES`), so on + // atomicdata.dev a miss for `/properties/*` would fall through to a + // network fetch of this very server. Anything served from our own + // authority is ours whether or not it is `Internal`, so compare + // authorities too — and ignore the scheme, since a store migrated + // as `https://` must not self-fetch when served over `http://`. + let base_domain = self.get_base_domain(); + let resolved_subject_obj = + Subject::from_raw(&resolved_url, base_domain.as_deref()); + let is_own_authority = base_domain + .as_deref() + .map(|base| { + let strip = |s: &str| { + s.trim_start_matches("https://") + .trim_start_matches("http://") + .trim_end_matches('/') + .to_string() + }; + let base_authority = strip(base); + let resolved = strip(&resolved_url); + resolved == base_authority + || resolved.starts_with(&format!("{}/", base_authority)) + }) + .unwrap_or(false); + + if resolved_subject_obj.is_local() || is_own_authority { + return self + .handle_not_found( + &resolved_url, + "Not found in DB".into(), + self.get_default_agent().ok().as_ref(), + ) + .await; + } - if let Ok(resource) = self - .fetch_resource(&resolved_url, self.get_default_agent().ok().as_ref()) - .await - { - // If the resource is external, it's not present in the store. - // However, we did fetch it (because the user probably requested it). - // So we should add it to the store. - // Note that this logic is also in `Store`'s `get_resource`, but it's slightly different there. - // We should probably unify this. - // Also, this might cause issues if we want to get a resource but NOT save it. - self.add_resource_opts(&resource, false, false, true) - .await?; - Ok(resource) - } else { - self.handle_not_found( - &resolved_url, - "Not found in DB".into(), - self.get_default_agent().ok().as_ref(), - ) - .await - } - } + if let Ok(resource) = self + .fetch_resource(&resolved_url, self.get_default_agent().ok().as_ref()) + .await + { + // If the resource is external, it's not present in the store. + // However, we did fetch it (because the user probably requested it). + // So we should add it to the store. + // Note that this logic is also in `Store`'s `get_resource`, but it's slightly different there. + // We should probably unify this. + // Also, this might cause issues if we want to get a resource but NOT save it. + self.add_resource_opts(&resource, false, false, true) + .await?; + Ok(resource) + } else { + self.handle_not_found( + &resolved_url, + "Not found in DB".into(), + self.get_default_agent().ok().as_ref(), + ) + .await + } + } + }) + .await } fn has_stored_resource(&self, subject: &Subject) -> bool { @@ -4217,98 +4242,106 @@ impl Storelike for Db { skip_dynamic: bool, for_agent: &ForAgent, ) -> AtomicResult { - let subject_without_params = subject.without_params(); - - // Get the inner URL for endpoint checking and extender context - let inner_url = match subject { - Subject::Internal { url, .. } => url, - Subject::External(u) => u, - Subject::Did { url, .. } => url, - }; - - // Check if the subject matches one of the endpoints, if so, call the endpoint. - let is_endpoint = self.is_endpoint(inner_url); + self.maintenance + .run(async { + let subject_without_params = subject.without_params(); + + // Get the inner URL for endpoint checking and extender context + let inner_url = match subject { + Subject::Internal { url, .. } => url, + Subject::External(u) => u, + Subject::Did { url, .. } => url, + }; - if is_endpoint { - return self.call_endpoint(subject.as_str(), for_agent).await; - } + // Check if the subject matches one of the endpoints, if so, call the endpoint. + let is_endpoint = self.is_endpoint(inner_url); - async move { - let mut resource = self.get_resource(&subject_without_params).await?; + if is_endpoint { + return self.call_endpoint(subject.as_str(), for_agent).await; + } - let _explanation = crate::hierarchy::check_read(self, &resource, for_agent).await?; + async move { + let mut resource = self.get_resource(&subject_without_params).await?; - let mut root_subject: Option = None; + let _explanation = + crate::hierarchy::check_read(self, &resource, for_agent).await?; - let extenders = self - .class_extenders - .read() - .map_err(|e| format!("Failed to read class extenders: {}", e))? - .clone(); - for extender in extenders.iter() { - if !extender.can_extend(&resource) { - continue; - } + let mut root_subject: Option = None; - if extender.resource_has_extender(&resource)? { - let (is_in_scope, cached_root) = - extender.check_scope(&resource, self, root_subject).await?; + let extenders = self + .class_extenders + .read() + .map_err(|e| format!("Failed to read class extenders: {}", e))? + .clone(); + for extender in extenders.iter() { + if !extender.can_extend(&resource) { + continue; + } - root_subject = cached_root; + if extender.resource_has_extender(&resource)? { + let (is_in_scope, cached_root) = + extender.check_scope(&resource, self, root_subject).await?; - if !is_in_scope { - continue; - } + root_subject = cached_root; - if skip_dynamic { - // This lets clients know that the resource may have dynamic properties that are currently not included - resource - .set( - crate::urls::INCOMPLETE.into(), - crate::Value::Boolean(true), - self, - ) - .await?; + if !is_in_scope { + continue; + } - return Ok(resource.into()); - } + if skip_dynamic { + // This lets clients know that the resource may have dynamic properties that are currently not included + resource + .set( + crate::urls::INCOMPLETE.into(), + crate::Value::Boolean(true), + self, + ) + .await?; - if let Some(handler) = extender.on_resource_get.as_ref() { - let fut = (handler)(GetExtenderContext { - store: self, - url: inner_url, - db_resource: &mut resource, - for_agent, - }); - let resource_response = fut.await?; - - // TODO: Check if we actually need this - // make sure the actual subject matches the one requested - It should not be changed in the logic above - match resource_response { - ResourceResponse::Resource(mut resource) => { - resource.set_subject(subject.to_string()); return Ok(resource.into()); } - ResourceResponse::ResourceWithReferenced(mut resource, referenced) => { - resource.set_subject(subject.to_string()); - return Ok(ResourceResponse::ResourceWithReferenced( - resource, referenced, - )); - } - ResourceResponse::Redirect(target) => { - return Ok(ResourceResponse::Redirect(target)); + if let Some(handler) = extender.on_resource_get.as_ref() { + let fut = (handler)(GetExtenderContext { + store: self, + url: inner_url, + db_resource: &mut resource, + for_agent, + }); + let resource_response = fut.await?; + + // TODO: Check if we actually need this + // make sure the actual subject matches the one requested - It should not be changed in the logic above + match resource_response { + ResourceResponse::Resource(mut resource) => { + resource.set_subject(subject.to_string()); + return Ok(resource.into()); + } + ResourceResponse::ResourceWithReferenced( + mut resource, + referenced, + ) => { + resource.set_subject(subject.to_string()); + + return Ok(ResourceResponse::ResourceWithReferenced( + resource, referenced, + )); + } + ResourceResponse::Redirect(target) => { + return Ok(ResourceResponse::Redirect(target)); + } + } } } } - } - } - resource.set_subject(subject.to_string()); + resource.set_subject(subject.to_string()); - Ok(resource.into()) - } - .await + Ok(resource.into()) + } + .await + }) + .await } fn handle_commit(&self, commit_response: &CommitResponse) { @@ -4322,27 +4355,31 @@ impl Storelike for Db { /// Tries `query_cache`, which you should implement yourself. #[instrument(skip_all)] async fn query(&self, q: &Query) -> AtomicResult { - // A constraint on a computed value can't come from the index, so it is - // applied to the set the index narrows to — which means paging has to - // happen after it, not in it. - let mut result = if !q.expression_filters.is_empty() { - self.query_with_expression_filters(q).await? - } else if requires_query_index(q) { - self.query_complex(q).await? - } else { - self.query_basic(q).await? - }; + self.maintenance + .run(async { + // A constraint on a computed value can't come from the index, so it is + // applied to the set the index narrows to — which means paging has to + // happen after it, not in it. + let mut result = if !q.expression_filters.is_empty() { + self.query_with_expression_filters(q).await? + } else if requires_query_index(q) { + self.query_complex(q).await? + } else { + self.query_basic(q).await? + }; - // Aggregates run over the whole matching set, so they need their own - // pass — the one above is limited to the requested page. Only when - // asked: a query without aggregates pays nothing for this. - if let Some(aggregation) = &q.aggregation { - if !aggregation.is_empty() { - result.aggregates = self.compute_aggregation(q, aggregation).await?; - } - } + // Aggregates run over the whole matching set, so they need their own + // pass — the one above is limited to the requested page. Only when + // asked: a query without aggregates pays nothing for this. + if let Some(aggregation) = &q.aggregation { + if !aggregation.is_empty() { + result.aggregates = self.compute_aggregation(q, aggregation).await?; + } + } - Ok(result) + Ok(result) + }) + .await } async fn search( @@ -4386,66 +4423,76 @@ impl Storelike for Db { body: Vec, for_agent: &ForAgent, ) -> AtomicResult { - let endpoints = self.endpoints.iter().filter(|e| e.handle_post.is_some()); - let subj_url = url::Url::try_from(subject)?; - for e in endpoints { - if let Some(fun) = &e.handle_post { - if subj_url.path() == e.path { - let handle_post_context = crate::endpoints::HandlePostContext { - store: self, - body: body.clone(), - for_agent, - subject: subj_url.clone(), - }; - let mut resource = fun(handle_post_context).await?.to_single(); - resource.set_subject(subject.into()); + self.maintenance + .run(async { + let endpoints = self.endpoints.iter().filter(|e| e.handle_post.is_some()); + let subj_url = url::Url::try_from(subject)?; + for e in endpoints { + if let Some(fun) = &e.handle_post { + if subj_url.path() == e.path { + let handle_post_context = crate::endpoints::HandlePostContext { + store: self, + body: body.clone(), + for_agent, + subject: subj_url.clone(), + }; + let mut resource = fun(handle_post_context).await?.to_single(); + resource.set_subject(subject.into()); - return Ok(resource); + return Ok(resource); + } + } } - } - } - // If we get Class Handlers with POST, this is where the code goes - // let mut r = self.get_resource(subject)?; - // for class in r.get_classes(self)? { - // match class.subject.as_str() { - // urls::IMPORTER => { - // let query_params = url::Url::try_from(subject)?; - // return crate::plugins::importer::construct_importer( - // self, - // query_params.query_pairs(), - // &mut r, - // for_agent, - // Some(body), - // ); - // } - // _ => {} - // } - // } - Err( - AtomicError::method_not_allowed("Cannot post here - no Endpoint Post handler found") - .set_subject(subject), - ) + // If we get Class Handlers with POST, this is where the code goes + // let mut r = self.get_resource(subject)?; + // for class in r.get_classes(self)? { + // match class.subject.as_str() { + // urls::IMPORTER => { + // let query_params = url::Url::try_from(subject)?; + // return crate::plugins::importer::construct_importer( + // self, + // query_params.query_pairs(), + // &mut r, + // for_agent, + // Some(body), + // ); + // } + // _ => {} + // } + // } + Err(AtomicError::method_not_allowed( + "Cannot post here - no Endpoint Post handler found", + ) + .set_subject(subject)) + }) + .await } async fn populate(&self) -> AtomicResult<()> { - crate::populate::bootstrap(self).await.map(|_| ()) + self.maintenance + .run(async { crate::populate::bootstrap(self).await.map(|_| ()) }) + .await } #[instrument(skip_all)] async fn remove_resource(&self, subject: &Subject) -> AtomicResult<()> { - let mut transaction = Transaction::new(); - let mut removed = Vec::new(); - self.recursive_remove(subject, &mut transaction, &mut removed, None) - .await?; - self.apply_transaction(&mut transaction)?; - // Tombstone every removed subject so bulk sync (Iroh / WS `SYNC`) - // does not resurrect them from a peer that still holds a stale copy. - for s in &removed { - crate::sync::tombstones::record_tombstone(self, s); - } - // TODO: deletion sync — should create a signed destroy commit - // and push it through the normal commit pipeline, not a raw DESTROY frame. - Ok(()) + self.maintenance + .run(async { + let mut transaction = Transaction::new(); + let mut removed = Vec::new(); + self.recursive_remove(subject, &mut transaction, &mut removed, None) + .await?; + self.apply_transaction(&mut transaction)?; + // Tombstone every removed subject so bulk sync (Iroh / WS `SYNC`) + // does not resurrect them from a peer that still holds a stale copy. + for s in &removed { + crate::sync::tombstones::record_tombstone(self, s); + } + // TODO: deletion sync — should create a signed destroy commit + // and push it through the normal commit pipeline, not a raw DESTROY frame. + Ok(()) + }) + .await } fn set_default_agent(&self, agent: crate::agents::Agent) { diff --git a/lib/src/db/kv_store.rs b/lib/src/db/kv_store.rs index 93543c4708..ed71ddfe8f 100644 --- a/lib/src/db/kv_store.rs +++ b/lib/src/db/kv_store.rs @@ -66,6 +66,18 @@ pub trait KvStore: Send + Sync { /// Flush all pending writes to durable storage. No-op for in-memory backends. fn flush(&self) -> AtomicResult<()>; + /// Copy all persisted tables at one transaction boundary while excluding + /// storage writers. The callback captures associated files under that same + /// barrier. It must not access this store. Call from a blocking worker. + #[cfg(not(target_arch = "wasm32"))] + fn backup_snapshot( + &self, + _destination: &std::path::Path, + _capture_files: &mut dyn FnMut() -> AtomicResult<()>, + ) -> AtomicResult<()> { + Err("Instance backup requires the redb backend".into()) + } + /// Start buffering writes. All `insert`, `remove`, and `apply_batch` calls /// will be accumulated until `commit_batch()` is called. fn begin_batch(&self) {} diff --git a/lib/src/db/maintenance.rs b/lib/src/db/maintenance.rs new file mode 100644 index 0000000000..3d84862b01 --- /dev/null +++ b/lib/src/db/maintenance.rs @@ -0,0 +1,125 @@ +//! Per-store async admission, above the storage transaction barrier. +//! +//! Calls already admitted can finish their nested store operations even when a +//! backup is waiting. Spawned tasks do not inherit admission. This avoids the +//! recursive-read deadlock of a plain fair RwLock around nested store methods. +use std::{ + future::Future, + sync::{ + atomic::{AtomicBool, Ordering}, + Arc, + }, +}; +use tokio::sync::{OwnedRwLockWriteGuard, RwLock}; + +tokio::task_local! { static ADMITTED: Vec; } + +#[derive(Clone, Default)] +pub struct Maintenance(Arc); + +#[derive(Default)] +struct Inner { + lock: Arc>, + paused: AtomicBool, +} + +impl Maintenance { + pub fn is_paused(&self) -> bool { + self.0.paused.load(Ordering::Acquire) + } + + pub async fn run(&self, work: F) -> F::Output { + let id = Arc::as_ptr(&self.0) as usize; + let mut admitted = ADMITTED.try_with(Clone::clone).unwrap_or_default(); + if admitted.contains(&id) { + return work.await; + } + let _guard = self.0.lock.read().await; + admitted.push(id); + ADMITTED.scope(admitted, work).await + } + + /// Cancellation while draining also clears the maintenance flag. + pub async fn pause(&self) -> Result { + if self + .0 + .paused + .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) + .is_err() + { + return Err("Backup already in progress"); + } + let mut pause = Pause { + inner: self.0.clone(), + guard: None, + }; + pause.guard = Some(self.0.lock.clone().write_owned().await); + Ok(pause) + } +} + +pub struct Pause { + inner: Arc, + guard: Option>, +} + +impl Drop for Pause { + fn drop(&mut self) { + self.guard.take(); + self.inner.paused.store(false, Ordering::Release); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn drain_allows_nested_operations_and_blocks_new_work() { + let gate = Maintenance::default(); + let (entered_tx, entered_rx) = tokio::sync::oneshot::channel(); + let (finish_tx, finish_rx) = tokio::sync::oneshot::channel(); + let g = gate.clone(); + let writer = tokio::spawn(async move { + g.run(async { + entered_tx.send(()).unwrap(); + finish_rx.await.unwrap(); + g.run(async {}).await; + }) + .await; + }); + entered_rx.await.unwrap(); + let g = gate.clone(); + let backup = tokio::spawn(async move { g.pause().await.unwrap() }); + while !gate.is_paused() { + tokio::task::yield_now().await; + } + finish_tx.send(()).unwrap(); + writer.await.unwrap(); + let pause = backup.await.unwrap(); + assert!(gate.pause().await.is_err()); + let g = gate.clone(); + let waiter = tokio::spawn(async move { g.run(async { 42 }).await }); + tokio::task::yield_now().await; + assert!(!waiter.is_finished()); + drop(pause); + assert_eq!(waiter.await.unwrap(), 42); + assert!(!gate.is_paused()); + } + + #[tokio::test] + async fn cancelling_a_drain_restores_admission() { + let gate = Maintenance::default(); + let guard = gate.0.lock.read().await; + let g = gate.clone(); + let job = tokio::spawn(async move { g.pause().await }); + while !gate.is_paused() { + tokio::task::yield_now().await; + } + job.abort(); + let _ = job.await; + assert!(!gate.is_paused()); + drop(guard); + drop(gate.pause().await.unwrap()); + } +} diff --git a/lib/src/db/redb_store.rs b/lib/src/db/redb_store.rs index a20b5f1013..5db473a885 100644 --- a/lib/src/db/redb_store.rs +++ b/lib/src/db/redb_store.rs @@ -298,6 +298,77 @@ fn prefix_upper_bound(prefix: &[u8]) -> Option> { } impl KvStore for RedbStore { + #[cfg(not(target_arch = "wasm32"))] + fn backup_snapshot( + &self, + destination: &std::path::Path, + capture_files: &mut dyn FnMut() -> AtomicResult<()>, + ) -> AtomicResult<()> { + use redb::TableHandle; + // No raw copy of an open file. A read transaction pins every table to + // the same version; the uncommitted write transaction excludes ALL + // storage writers, including periodic flush and direct metadata writes. + self.flush()?; + let buffer = self + .batch_buffer + .lock() + .map_err(|_| "Poisoned batch buffer")?; + if buffer.is_some() { + return Err("Cannot back up an active buffered batch".into()); + } + // Do not poison the batch mutex if capture code panics. The runtime + // reports the worker failure and must still be able to resume writes. + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let _writers = self + .db + .begin_write() + .map_err(|e| format!("Backup barrier: {e}"))?; + let snapshot = self + .db + .begin_read() + .map_err(|e| format!("Backup snapshot: {e}"))?; + if snapshot + .list_multimap_tables() + .map_err(|e| e.to_string())? + .next() + .is_some() + { + return Err("Backup does not support multimap tables".into()); + } + // create_new prevents overwriting a database, including through a symlink. + let file = std::fs::OpenOptions::new() + .write(true) + .create_new(true) + .open(destination)?; + drop(file); + let target = + Database::create(destination).map_err(|e| format!("Backup destination: {e}"))?; + let mut tx = target.begin_write().map_err(|e| e.to_string())?; + tx.set_quick_repair(true); + for handle in snapshot.list_tables().map_err(|e| e.to_string())? { + let definition: TableDefinition<&[u8], &[u8]> = TableDefinition::new(handle.name()); + // A future table with other types fails explicitly instead of being skipped. + let source = snapshot.open_table(definition).map_err(|e| e.to_string())?; + let mut output = tx.open_table(definition).map_err(|e| e.to_string())?; + for row in source.iter().map_err(|e| e.to_string())? { + let (key, value) = row.map_err(|e| e.to_string())?; + output + .insert(key.value(), value.value()) + .map_err(|e| e.to_string())?; + } + } + tx.commit().map_err(|e| e.to_string())?; + drop(target); + capture_files()?; + Ok(()) + })); + drop(buffer); + match result { + Ok(result) => result, + Err(panic) => std::panic::resume_unwind(panic), + } + } + fn get(&self, tree: Tree, key: &[u8]) -> AtomicResult>> { // Read-your-writes: check the batch buffer first { @@ -666,6 +737,26 @@ impl KvStore for RedbStore { } } +/// Check a closed snapshot without booting a node or contacting any peers. +#[cfg(not(target_arch = "wasm32"))] +pub fn verify_snapshot(path: &std::path::Path) -> AtomicResult<()> { + use redb::TableHandle; + let mut db = Database::open(path).map_err(|e| e.to_string())?; + if !db.check_integrity().map_err(|e| e.to_string())? { + return Err("Snapshot database failed integrity verification".into()); + } + let tx = db.begin_read().map_err(|e| e.to_string())?; + for handle in tx.list_tables().map_err(|e| e.to_string())? { + let table = tx + .open_table(TableDefinition::<&[u8], &[u8]>::new(handle.name())) + .map_err(|e| e.to_string())?; + for row in table.iter().map_err(|e| e.to_string())? { + row.map_err(|e| e.to_string())?; + } + } + Ok(()) +} + #[cfg(all(test, not(target_arch = "wasm32")))] mod tests { use super::*; diff --git a/lib/src/sync/engine.rs b/lib/src/sync/engine.rs index 4a3dd25f8b..29ebdede9a 100644 --- a/lib/src/sync/engine.rs +++ b/lib/src/sync/engine.rs @@ -258,276 +258,281 @@ pub async fn handle_frame_full( store: &Db, agent: &mut crate::agents::ForAgent, ) -> HandleOutput { - if frame.is_empty() { - return HandleOutput::default(); - } + store + .maintenance + .run(async { + if frame.is_empty() { + return HandleOutput::default(); + } - let tag = frame[0]; - let payload = &frame[1..]; + let tag = frame[0]; + let payload = &frame[1..]; - match tag { - protocol::tag::SUB => return handle_sub(payload, store, agent).await, - protocol::tag::UNSUB => return handle_unsub(payload), - _ => {} - } - - let frames = match tag { - protocol::tag::AUTH => { - handle_auth_frame( - payload, - store, - agent, - AuthBinding::Unbound, - AuthChallenge::None, - ) - .await - } - - protocol::tag::GET => { - if let Some(decoded) = protocol::decode_get(payload) { - vec![answer_get(store, agent, decoded.request_id, decoded.subject).await] - } else { - vec![protocol::encode_error( - 0, - protocol::error_code::UNKNOWN, - "Invalid GET frame", - )] + match tag { + protocol::tag::SUB => return handle_sub(payload, store, agent).await, + protocol::tag::UNSUB => return handle_unsub(payload), + _ => {} } - } - protocol::tag::GET_MANY => { - if let Some(decoded) = protocol::decode_get_many(payload) { - // One answer per subject, in request order (join_all keeps it), so - // the client can pair an `ERROR` (which names no subject) with what - // it asked. Evaluated concurrently: the batch is only worth its - // frame if it is not slower than the single GETs it replaces. - let agent: &crate::agents::ForAgent = agent; - let answers = futures::future::join_all( - decoded - .subjects - .iter() - .map(|subject| answer_get(store, agent, decoded.request_id, subject)), - ) - .await; - vec![protocol::encode_get_many_result( - decoded.request_id, - &answers, - )] - } else { - vec![protocol::encode_error( - 0, - protocol::error_code::UNKNOWN, - "Invalid GET_MANY frame", - )] - } - } + let frames = match tag { + protocol::tag::AUTH => { + handle_auth_frame( + payload, + store, + agent, + AuthBinding::Unbound, + AuthChallenge::None, + ) + .await + } - protocol::tag::COMMIT => { - // A signed commit is the unit of authority on every transport: it - // carries its own signature and the signer's rights are checked - // here, so a peer relaying it can only ever apply a change its - // signer was already entitled to make — no escalation from "I - // dialed you." This is what lets a serverless peer apply a `COMMIT` - // exactly like atomic-server's HTTP path does; the connection's own - // AUTH identity is not the gate (the commit's signature is). - // - // Differs from the server's WS `COMMIT` arm in two deliberate ways: - // no `source_id` echo-suppression (peer transports don't fan out - // through the commit monitor), and `validate_loro_causality` is - // OFF because concurrent writes between peers are expected (see the - // field's own docs in `commit.rs`). - match protocol::decode_commit(payload) { - Some(decoded) => { - let request_id = decoded.request_id; - match apply_peer_commit(store, decoded.commit_json).await { - Ok(commit_json) => { - vec![protocol::encode_commit_ok(request_id, &commit_json)] - } - Err(e) => { - let msg = e.to_string(); - vec![protocol::encode_error( - request_id, - protocol::classify_commit_error(&msg), - &msg, - )] - } + protocol::tag::GET => { + if let Some(decoded) = protocol::decode_get(payload) { + vec![answer_get(store, agent, decoded.request_id, decoded.subject).await] + } else { + vec![protocol::encode_error( + 0, + protocol::error_code::UNKNOWN, + "Invalid GET frame", + )] } } - None => vec![protocol::encode_error( - 0, - protocol::error_code::UNKNOWN, - "Invalid COMMIT frame", - )], - } - } - protocol::tag::SYNC => match protocol::decode_sync(payload) { - // Hash-first probe: compare the drive hash over what this - // session may read, without either side exchanging the - // O(drive) version vector. In sync → SYNC_OK; otherwise - // SYNC_RESEND asks the client to reconcile. Hashed over the - // readable subjects both so it can match the client's and so an - // anonymous socket learns nothing about a drive it cannot read. - Some(sync) if sync.probe => { - match drive_sync_hash_for(store, &sync.drive, agent).await { - Ok(server_hash) if server_hash == sync.drive_hash => { - vec![protocol::encode_sync_ok(&sync.drive)] + protocol::tag::GET_MANY => { + if let Some(decoded) = protocol::decode_get_many(payload) { + // One answer per subject, in request order (join_all keeps it), so + // the client can pair an `ERROR` (which names no subject) with what + // it asked. Evaluated concurrently: the batch is only worth its + // frame if it is not slower than the single GETs it replaces. + let agent: &crate::agents::ForAgent = agent; + let answers = futures::future::join_all( + decoded + .subjects + .iter() + .map(|subject| answer_get(store, agent, decoded.request_id, subject)), + ) + .await; + vec![protocol::encode_get_many_result( + decoded.request_id, + &answers, + )] + } else { + vec![protocol::encode_error( + 0, + protocol::error_code::UNKNOWN, + "Invalid GET_MANY frame", + )] } - Ok(_) => vec![protocol::encode_sync_resend(&sync.drive)], - Err(reason) => vec![protocol::encode_error( - 0, - protocol::error_code::UNAUTHORIZED_READ, - &format!("SYNC refused for {}: {reason}", sync.drive), - )], } - } - Some(sync) => { - // `subjects`, when present, is the RBSR-reduced set: build - // version vectors for just those instead of walking the drive. - let filter = sync - .subjects - .as_ref() - .map(|s| s.iter().cloned().collect::>()); - handle_sync_vv_filtered( - &sync.drive, - &sync.drive_hash, - &sync.peers, - &sync.resources, - filter.as_ref(), - store, - agent, - ) - .await - } - None => vec![protocol::encode_error( - 0, - protocol::error_code::UNKNOWN, - "Invalid SYNC frame", - )], - }, - protocol::tag::SYNC_PUSH => { - if let Some(push) = protocol::decode_sync_push(payload) { - // handle_frame serves connections dialed *into* us (accept side, - // WS): no owned-drive relaxation — the sender must itself hold - // write rights. The dial side calls import_sync_push directly - // with trust_owned=true. - match import_sync_push(&push, store, agent, false).await { - Ok((_count, mut blob_requests)) => { - let mut responses = vec![protocol::encode_sync_ok(&push.drive)]; - responses.append(&mut blob_requests); - responses + protocol::tag::COMMIT => { + // A signed commit is the unit of authority on every transport: it + // carries its own signature and the signer's rights are checked + // here, so a peer relaying it can only ever apply a change its + // signer was already entitled to make — no escalation from "I + // dialed you." This is what lets a serverless peer apply a `COMMIT` + // exactly like atomic-server's HTTP path does; the connection's own + // AUTH identity is not the gate (the commit's signature is). + // + // Differs from the server's WS `COMMIT` arm in two deliberate ways: + // no `source_id` echo-suppression (peer transports don't fan out + // through the commit monitor), and `validate_loro_causality` is + // OFF because concurrent writes between peers are expected (see the + // field's own docs in `commit.rs`). + match protocol::decode_commit(payload) { + Some(decoded) => { + let request_id = decoded.request_id; + match apply_peer_commit(store, decoded.commit_json).await { + Ok(commit_json) => { + vec![protocol::encode_commit_ok(request_id, &commit_json)] + } + Err(e) => { + let msg = e.to_string(); + vec![protocol::encode_error( + request_id, + protocol::classify_commit_error(&msg), + &msg, + )] + } + } + } + None => vec![protocol::encode_error( + 0, + protocol::error_code::UNKNOWN, + "Invalid COMMIT frame", + )], } - // A refused import used to be answered with `SYNC_OK` all - // the same, so a sender could never tell "landed" from - // "dropped" (`replicate.rs` re-probed with a second SYNC - // to find out). Say no when the answer is no. - Err(rejected) => vec![rejected.to_error_frame()], } - } else { - vec![protocol::encode_error( - 0, - protocol::error_code::UNKNOWN, - "Invalid SYNC_PUSH frame", - )] - } - } - protocol::tag::BLOB_REQUEST => { - if let Some(hash) = protocol::decode_blob_request(payload) { - match store.get_blob(&hash).await { - Ok(Some(bytes)) => vec![protocol::encode_blob_response(&hash, &bytes)], - _ => vec![protocol::encode_error( + protocol::tag::SYNC => match protocol::decode_sync(payload) { + // Hash-first probe: compare the drive hash over what this + // session may read, without either side exchanging the + // O(drive) version vector. In sync → SYNC_OK; otherwise + // SYNC_RESEND asks the client to reconcile. Hashed over the + // readable subjects both so it can match the client's and so an + // anonymous socket learns nothing about a drive it cannot read. + Some(sync) if sync.probe => { + match drive_sync_hash_for(store, &sync.drive, agent).await { + Ok(server_hash) if server_hash == sync.drive_hash => { + vec![protocol::encode_sync_ok(&sync.drive)] + } + Ok(_) => vec![protocol::encode_sync_resend(&sync.drive)], + Err(reason) => vec![protocol::encode_error( + 0, + protocol::error_code::UNAUTHORIZED_READ, + &format!("SYNC refused for {}: {reason}", sync.drive), + )], + } + } + Some(sync) => { + // `subjects`, when present, is the RBSR-reduced set: build + // version vectors for just those instead of walking the drive. + let filter = sync + .subjects + .as_ref() + .map(|s| s.iter().cloned().collect::>()); + handle_sync_vv_filtered( + &sync.drive, + &sync.drive_hash, + &sync.peers, + &sync.resources, + filter.as_ref(), + store, + agent, + ) + .await + } + None => vec![protocol::encode_error( 0, protocol::error_code::UNKNOWN, - "Blob not found", + "Invalid SYNC frame", )], + }, + + protocol::tag::SYNC_PUSH => { + if let Some(push) = protocol::decode_sync_push(payload) { + // handle_frame serves connections dialed *into* us (accept side, + // WS): no owned-drive relaxation — the sender must itself hold + // write rights. The dial side calls import_sync_push directly + // with trust_owned=true. + match import_sync_push(&push, store, agent, false).await { + Ok((_count, mut blob_requests)) => { + let mut responses = vec![protocol::encode_sync_ok(&push.drive)]; + responses.append(&mut blob_requests); + responses + } + // A refused import used to be answered with `SYNC_OK` all + // the same, so a sender could never tell "landed" from + // "dropped" (`replicate.rs` re-probed with a second SYNC + // to find out). Say no when the answer is no. + Err(rejected) => vec![rejected.to_error_frame()], + } + } else { + vec![protocol::encode_error( + 0, + protocol::error_code::UNKNOWN, + "Invalid SYNC_PUSH frame", + )] + } } - } else { - vec![protocol::encode_error( - 0, - protocol::error_code::UNKNOWN, - "Invalid BLOB_REQUEST frame", - )] - } - } - protocol::tag::BLOB_RESPONSE => { - if let Some(resp) = protocol::decode_blob_response(payload) { - // F4 (planning/unified-sync.md): a `BLOB_RESPONSE` with no - // matching `BLOB_REQUEST` we issued is unsolicited — reject - // it rather than storing arbitrary bytes with no admission - // check at all. A matching entry names the (already- - // admitted at request time) drive; re-check admission here - // too, since enrollment/quota state can change between the - // request and this response. - match store.take_pending_blob_request(&resp.hash) { - // Blobs are content-addressed: the bytes must hash to the - // key they are stored under, or any session that answers - // a pending request poisons what every later reader of - // that hash gets. - Some(_drive) if blake3::hash(&resp.bytes).as_bytes() != &resp.hash => { - tracing::warn!( - "BLOB_RESPONSE: bytes do not hash to {}, dropped", - hex::encode(resp.hash) - ); - vec![] + protocol::tag::BLOB_REQUEST => { + if let Some(hash) = protocol::decode_blob_request(payload) { + match store.get_blob(&hash).await { + Ok(Some(bytes)) => vec![protocol::encode_blob_response(&hash, &bytes)], + _ => vec![protocol::encode_error( + 0, + protocol::error_code::UNKNOWN, + "Blob not found", + )], + } + } else { + vec![protocol::encode_error( + 0, + protocol::error_code::UNKNOWN, + "Invalid BLOB_REQUEST frame", + )] } - Some(drive) if store.sync_policy().admit_drive_write(&drive) => { - match store.put_blob(&resp.hash, &resp.bytes).await { - Ok(()) => vec![], - Err(error) => { - tracing::warn!("BLOB_RESPONSE: storage failed: {error}"); + } + + protocol::tag::BLOB_RESPONSE => { + if let Some(resp) = protocol::decode_blob_response(payload) { + // F4 (planning/unified-sync.md): a `BLOB_RESPONSE` with no + // matching `BLOB_REQUEST` we issued is unsolicited — reject + // it rather than storing arbitrary bytes with no admission + // check at all. A matching entry names the (already- + // admitted at request time) drive; re-check admission here + // too, since enrollment/quota state can change between the + // request and this response. + match store.take_pending_blob_request(&resp.hash) { + // Blobs are content-addressed: the bytes must hash to the + // key they are stored under, or any session that answers + // a pending request poisons what every later reader of + // that hash gets. + Some(_drive) if blake3::hash(&resp.bytes).as_bytes() != &resp.hash => { + tracing::warn!( + "BLOB_RESPONSE: bytes do not hash to {}, dropped", + hex::encode(resp.hash) + ); + vec![] + } + Some(drive) if store.sync_policy().admit_drive_write(&drive) => { + match store.put_blob(&resp.hash, &resp.bytes).await { + Ok(()) => vec![], + Err(error) => { + tracing::warn!("BLOB_RESPONSE: storage failed: {error}"); + vec![protocol::encode_error( + 0, + protocol::error_code::UNKNOWN, + "Blob storage failed", + )] + } + } + } + Some(drive) => { + tracing::warn!( + "BLOB_RESPONSE: drive {} not admitted by sync policy, dropping blob", + drive + ); + vec![protocol::encode_error( + 0, + protocol::error_code::UNKNOWN, + "Drive not admitted for sync", + )] + } + None => { + tracing::warn!( + "BLOB_RESPONSE: no matching pending BLOB_REQUEST, dropping blob" + ); vec![protocol::encode_error( 0, protocol::error_code::UNKNOWN, - "Blob storage failed", + "Unsolicited blob response", )] } } - } - Some(drive) => { - tracing::warn!( - "BLOB_RESPONSE: drive {} not admitted by sync policy, dropping blob", - drive - ); - vec![protocol::encode_error( - 0, - protocol::error_code::UNKNOWN, - "Drive not admitted for sync", - )] - } - None => { - tracing::warn!( - "BLOB_RESPONSE: no matching pending BLOB_REQUEST, dropping blob" - ); + } else { vec![protocol::encode_error( 0, protocol::error_code::UNKNOWN, - "Unsolicited blob response", + "Invalid BLOB_RESPONSE frame", )] } } - } else { - vec![protocol::encode_error( - 0, - protocol::error_code::UNKNOWN, - "Invalid BLOB_RESPONSE frame", - )] - } - } - _ => { - tracing::debug!("Unhandled frame tag: 0x{:02x}", tag); - vec![] - } - }; + _ => { + tracing::debug!("Unhandled frame tag: 0x{:02x}", tag); + vec![] + } + }; - HandleOutput { - frames, - subscribe: None, - unsubscribe: None, - } + HandleOutput { + frames, + subscribe: None, + unsubscribe: None, + } + }) + .await } /// `SUB `: parse, `check_read`, and tell the transport to register. @@ -1429,6 +1434,7 @@ pub async fn import_sync_push( for_agent: &crate::agents::ForAgent, trust_owned: bool, ) -> Result<(usize, Vec>), SyncPushRejected> { + store.maintenance.run(async { let drive_subject = crate::Subject::from_raw(&push.drive, store.get_base_domain().as_deref()); let policy = store.sync_policy(); @@ -1711,6 +1717,7 @@ pub async fn import_sync_push( ); } Ok((count, blob_requests)) + }).await } /// Whether the owner deliberately dialled this node. Peer-to-peer sync only diff --git a/lib/src/sync/ws_apply.rs b/lib/src/sync/ws_apply.rs index 4bf8b0f30f..480b48cfe7 100644 --- a/lib/src/sync/ws_apply.rs +++ b/lib/src/sync/ws_apply.rs @@ -186,33 +186,38 @@ pub async fn persist_update( subject: &str, resolved: ResolvedUpdate, ) -> AtomicResult<()> { - let snapshot_key = - crate::Subject::from_raw(subject, store.get_base_domain().as_deref()).pure_id(); - - // Exclusive for the same reason `apply_commit` is: persistence replaces - // the stored snapshot, so a concurrent commit must not be clobbered. - let _subject_guard = store.subject_locks.lock(&snapshot_key).await; - - // `resolved` was built from a read taken before the lock, so re-merge it - // into whatever is stored *now*. Safe to do here — unlike a commit, a sync - // apply only ever adds a peer's operations, so union is the correct - // outcome. Re-importing already-known operations is a Loro no-op. - let doc = match store.kv.get( - crate::db::trees::Tree::LoroSnapshots, - snapshot_key.as_bytes(), - )? { - Some(current) => { - let doc = crate::loro::AtomicLoroDoc::from_snapshot(¤t)?; - doc.import_update(&resolved.snapshot)?; - doc - } - None => crate::loro::AtomicLoroDoc::from_snapshot(&resolved.snapshot)?, - }; - let mut resource = resolved.resource; - resource.apply_state_doc(doc)?; - // Projection, indexes and snapshot must commit together. Do not acknowledge - // a snapshot whose searchable resource failed to persist. - store.persist_replicated_resource(&resource).await + store + .maintenance + .run(async { + let snapshot_key = + crate::Subject::from_raw(subject, store.get_base_domain().as_deref()).pure_id(); + + // Exclusive for the same reason `apply_commit` is: persistence replaces + // the stored snapshot, so a concurrent commit must not be clobbered. + let _subject_guard = store.subject_locks.lock(&snapshot_key).await; + + // `resolved` was built from a read taken before the lock, so re-merge it + // into whatever is stored *now*. Safe to do here — unlike a commit, a sync + // apply only ever adds a peer's operations, so union is the correct + // outcome. Re-importing already-known operations is a Loro no-op. + let doc = match store.kv.get( + crate::db::trees::Tree::LoroSnapshots, + snapshot_key.as_bytes(), + )? { + Some(current) => { + let doc = crate::loro::AtomicLoroDoc::from_snapshot(¤t)?; + doc.import_update(&resolved.snapshot)?; + doc + } + None => crate::loro::AtomicLoroDoc::from_snapshot(&resolved.snapshot)?, + }; + let mut resource = resolved.resource; + resource.apply_state_doc(doc)?; + // Projection, indexes and snapshot must commit together. Do not acknowledge + // a snapshot whose searchable resource failed to persist. + store.persist_replicated_resource(&resource).await + }) + .await } /// Remove a resource from the local store (a `DESTROY` frame or a @@ -223,14 +228,19 @@ pub async fn persist_update( /// 2026-09 those two callers went through two identically-bodied functions /// (`apply_destroy` and `apply_destroy_checked`) that differed in name only. pub async fn apply_destroy(store: &Db, subject: &str) -> AtomicResult<()> { - if subject.is_empty() { - return Ok(()); - } + store + .maintenance + .run(async { + if subject.is_empty() { + return Ok(()); + } - set_importing(true); - let result = apply_destroy_unchecked(store, subject).await; - set_importing(false); - result + set_importing(true); + let result = apply_destroy_unchecked(store, subject).await; + set_importing(false); + result + }) + .await } async fn apply_destroy_unchecked(store: &Db, subject: &str) -> AtomicResult<()> { diff --git a/scripts/backup-instance.sh b/scripts/backup-instance.sh new file mode 100755 index 0000000000..49ccce2dc6 --- /dev/null +++ b/scripts/backup-instance.sh @@ -0,0 +1,7 @@ +#!/usr/bin/env sh +# Run from cron, launchd or a systemd timer. Archives are retained indefinitely. +set -eu +: "${ATOMIC_BACKUP_TOKEN_FILE:?Set the path to the backup.token in the server config directory}" +exec "${ATOMIC_SERVER_BIN:-atomic-server}" backup \ + --server "${ATOMIC_BACKUP_SERVER:-http://127.0.0.1:9883}" \ + --token-file "$ATOMIC_BACKUP_TOKEN_FILE" diff --git a/server/Cargo.toml b/server/Cargo.toml index 6bc1a891de..05977965ff 100644 --- a/server/Cargo.toml +++ b/server/Cargo.toml @@ -87,7 +87,7 @@ html2md = "0.2.15" kuchikiki = "0.8.2" indexmap = { version = "2.14.0", features = ["std"] } lol_html = "2.9" -zip = { version = "8.6.0", optional = true } +zip = { version = "8.6.0" } reqwest = { version = "0.13.3", default-features = false, features = [ "rustls", "json", @@ -206,6 +206,7 @@ version = "0.3.23" [dev-dependencies] brotli = "8" tempfile = "3.27" +redb = "4.1.0" actix-rt = "2.11.0" assert_cmd = "2.2.2" mainline = "6.2.0" @@ -259,7 +260,7 @@ telemetry = [ ] img = ["webp", "image"] https = ["rustls", "instant-acme", "rcgen", "rustls-pemfile"] -wasm-plugins = ["wasmtime", "wasmtime-wasi", "wasmtime-wasi-http", "zip"] +wasm-plugins = ["wasmtime", "wasmtime-wasi", "wasmtime-wasi-http"] image = ["dep:image"] tonic = ["dep:tonic"] tracing-opentelemetry = ["dep:tracing-opentelemetry"] @@ -267,7 +268,7 @@ wasmtime = ["dep:wasmtime"] wasmtime-wasi = ["dep:wasmtime-wasi"] wasmtime-wasi-http = ["dep:wasmtime-wasi-http"] webp = ["dep:webp"] -zip = ["dep:zip"] +zip = [] vector-search = [ "dep:fastembed", "dep:text-splitter", diff --git a/server/build.rs b/server/build.rs index b77bbbb484..0bf83eb466 100644 --- a/server/build.rs +++ b/server/build.rs @@ -21,6 +21,42 @@ struct Dirs { } fn main() -> std::io::Result<()> { + // Git worktrees keep HEAD/refs outside the checkout. Track their actual + // paths so the manifest revision refreshes after a commit or branch move. + for name in [ + "HEAD".to_owned(), + std::process::Command::new("git") + .args(["symbolic-ref", "-q", "HEAD"]) + .output() + .ok() + .and_then(|o| String::from_utf8(o.stdout).ok()) + .unwrap_or_default() + .trim() + .to_owned(), + ] { + if name.is_empty() { + continue; + } + if let Ok(output) = std::process::Command::new("git") + .args(["rev-parse", "--git-path", &name]) + .output() + { + if output.status.success() { + println!( + "cargo:rerun-if-changed={}", + String::from_utf8_lossy(&output.stdout).trim() + ); + } + } + } + let revision = std::process::Command::new("git") + .args(["rev-parse", "HEAD"]) + .output() + .ok() + .filter(|o| o.status.success()) + .and_then(|o| String::from_utf8(o.stdout).ok()) + .unwrap_or_else(|| "unknown".into()); + println!("cargo:rustc-env=ATOMIC_BACKUP_REVISION={}", revision.trim()); let start_total = Instant::now(); // The embedded component is only needed when the server-side WASM plugin diff --git a/server/src/appstate.rs b/server/src/appstate.rs index c8078ef9fd..86b77c1120 100644 --- a/server/src/appstate.rs +++ b/server/src/appstate.rs @@ -18,6 +18,7 @@ use crate::plugins::wasm; // A good option here is to use Actors for things that can change (e.g. commit_monitor) #[derive(Clone)] pub struct AppState { + pub backup: Arc, /// Contains all the data pub store: atomic_lib::Db, /// App Configuration @@ -50,6 +51,7 @@ impl AppState { /// Initializes or opens a store on disk. /// Creates a new agent, if necessary. pub async fn init(config: Config) -> AtomicServerResult { + crate::backup::check_restore_activation(&config)?; tracing::info!("Initializing AppState"); // We warn over here because tracing needs to be initialized first. @@ -235,7 +237,9 @@ impl AppState { config.opts.write_rate_limit, config.opts.anonymous_write_rate_limit, )); + let backup = crate::backup::BackupService::new(&config)?; Ok(AppState { + backup, store, config, write_rate_limiter, diff --git a/server/src/backup.rs b/server/src/backup.rs new file mode 100644 index 0000000000..86a6e12258 --- /dev/null +++ b/server/src/backup.rs @@ -0,0 +1,1123 @@ +//! Native instance backup. Control is opt-in, loopback-only and token-protected. +//! Capture owns the maintenance guard in the blocking worker, so cancellation +//! of an HTTP request cannot resume writers while files are still being copied. +use crate::{appstate::AppState, config::Config, errors::AtomicServerResult}; +use actix_web::{web, HttpRequest, HttpResponse}; +use atomic_lib::{errors::AtomicResult, Db}; +use serde::{Deserialize, Serialize}; +use std::{ + collections::BTreeMap, + fs::{self, File, OpenOptions}, + io::{Read, Write}, + path::{Component, Path, PathBuf}, + sync::{Arc, Mutex}, +}; + +const FORMAT: u32 = 1; +const MANIFEST: &str = "manifest.json"; +const RESTORED: &str = "RESTORED_OFFLINE"; +const CONTROL: &str = "/__atomic/backup"; + +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub struct Status { + pub id: Option, + pub phase: String, + pub archive: Option, + pub error: Option, +} + +pub struct BackupService { + output: Option, + token: String, + status: Mutex, +} + +impl BackupService { + pub fn new(config: &Config) -> AtomicResult> { + let mut token = String::new(); + let output = if let Some(path) = &config.opts.backup_dir { + if !path.exists() { + let mut builder = fs::DirBuilder::new(); + builder.recursive(true); + #[cfg(unix)] + { + use std::os::unix::fs::DirBuilderExt; + builder.mode(0o700); + } + builder.create(path)?; + } + // Existing destinations may be shared directories. Never chmod + // them; staging directories and final ZIP files are owner-only. + let output = path.canonicalize()?; + let data = data_dir(config)?.canonicalize()?; + let configuration = config.config_dir.canonicalize()?; + if data.starts_with(&configuration) || configuration.starts_with(&data) { + return Err("Instance backup requires separate, non-overlapping data and config directories".into()); + } + for source in [data_dir(config)?, config.config_dir.clone()] { + let source = source.canonicalize()?; + if output.starts_with(&source) || source.starts_with(&output) { + return Err( + "Backup directory must be outside data and config directories".into(), + ); + } + } + // Vector-index flush runs independently and is deliberately not + // backed up. Require cache storage outside the capture roots. + for cache in [&config.vector_search_index_path, &config.plugin_cache_path] { + let cache = resolve_future_path(cache)?; + if cache.starts_with(data_dir(config)?.canonicalize()?) + || cache.starts_with(config.config_dir.canonicalize()?) + { + return Err( + "Instance backup requires --cache-dir outside data and config directories" + .into(), + ); + } + } + let token_path = config.config_dir.join("backup.token"); + if token_path.exists() { + if fs::symlink_metadata(&token_path)?.file_type().is_symlink() { + return Err("Backup token must not be a symlink".into()); + } + token = fs::read_to_string(&token_path)?.trim().to_owned(); + if token.len() != 64 || !token.bytes().all(|b| b.is_ascii_hexdigit()) { + return Err("Invalid backup.token; expected 64 hexadecimal characters".into()); + } + private_file(&token_path)?; + } else { + // Agent creation uses the same cryptographic RNG as server identities. + token = blake3::hash( + atomic_lib::agents::Agent::new(None)? + .private_key + .ok_or("Missing generated private key")? + .as_bytes(), + ) + .to_hex() + .to_string(); + let mut file = create_private(&token_path)?; + file.write_all(token.as_bytes())?; + file.sync_all()?; + } + Some(output) + } else { + None + }; + Ok(Arc::new(Self { + output, + token, + status: Mutex::new(Status { + phase: "idle".into(), + ..Default::default() + }), + })) + } + + fn authorized(&self, request: &HttpRequest) -> bool { + self.output.is_some() + && request.peer_addr().is_some_and(|p| p.ip().is_loopback()) + && request + .headers() + .get("authorization") + .and_then(|v| v.to_str().ok()) + .and_then(|v| v.strip_prefix("Bearer ")) + .is_some_and(|token| { + blake3::hash(token.as_bytes()) == blake3::hash(self.token.as_bytes()) + }) + } + + fn phase(&self, phase: &str) { + self.status.lock().unwrap().phase = phase.into(); + } +} + +pub fn routes(cfg: &mut web::ServiceConfig) { + cfg.service( + web::resource(CONTROL) + .route(web::post().to(start)) + .route(web::get().to(status)), + ); +} + +async fn status(request: HttpRequest, app: web::Data) -> HttpResponse { + if !app.backup.authorized(&request) { + return HttpResponse::Unauthorized().finish(); + } + HttpResponse::Ok().json(app.backup.status.lock().unwrap().clone()) +} + +async fn start(request: HttpRequest, app: web::Data) -> HttpResponse { + if !app.backup.authorized(&request) { + return HttpResponse::Unauthorized().finish(); + } + let service = app.backup.clone(); + let id = format!( + "{}-{}", + chrono::Utc::now().format("%Y-%m-%dT%H%M%S%.3fZ"), + std::process::id() + ); + { + let mut state = service.status.lock().unwrap(); + if !matches!(state.phase.as_str(), "idle" | "complete" | "failed") { + return HttpResponse::Conflict().json(state.clone()); + } + *state = Status { + id: Some(id.clone()), + phase: "draining".into(), + ..Default::default() + }; + } + let store = app.store.clone(); + let config = app.config.clone(); + let job_id = id.clone(); + // Detached from the caller; the worker, not an HTTP connection, owns cleanup. + tokio::spawn(async move { + let result = run_backup(&store, &config, &service, &job_id).await; + let mut state = service.status.lock().unwrap(); + match result { + Ok(path) => { + state.phase = "complete".into(); + state.archive = Some(path); + } + Err(error) => { + state.phase = "failed".into(); + state.error = Some(error.to_string()); + } + } + }); + HttpResponse::Accepted().json(Status { + id: Some(id), + phase: "draining".into(), + ..Default::default() + }) +} + +async fn run_backup( + store: &Db, + config: &Config, + service: &Arc, + id: &str, +) -> AtomicResult { + let pause = tokio::time::timeout( + std::time::Duration::from_secs(30), + store.maintenance.pause(), + ) + .await + .map_err(|_| "Timed out draining operations; server resumed")? + .map_err(|e| e.to_string())?; + let store = store.clone(); + let config = config.clone(); + let service = service.clone(); + let id = id.to_owned(); + tokio::task::spawn_blocking(move || { + service.phase("capturing"); + let output = service.output.as_ref().ok_or("Backups disabled")?; + let staging = tempfile::Builder::new() + .prefix(".atomic-backup-") + .tempdir_in(output)?; + let captured = capture(&store, &config, staging.path()); + // Also runs on capture errors. During a panic RAII unwinds this guard. + drop(pause); + captured?; + service.phase("archiving"); + archive(staging.path(), output, &id) + }) + .await + .map_err(|e| format!("Backup worker failed: {e}"))? +} + +#[derive(Debug, Serialize, Deserialize)] +struct Manifest { + format: u32, + server_version: String, + build_revision: String, + redb_version: String, + snapshot_time: String, + freshness: String, + envelope_retention: String, + source_data: PathBuf, + source_config: PathBuf, + files: BTreeMap, +} + +#[derive(Debug, Serialize, Deserialize)] +struct Entry { + bytes: u64, + blake3: String, +} + +// Canonicalize an existing ancestor, then append not-yet-created components. +fn resolve_future_path(path: &Path) -> AtomicResult { + if path.exists() { + return Ok(path.canonicalize()?); + } + let parent = path + .parent() + .filter(|p| !p.as_os_str().is_empty()) + .unwrap_or(Path::new(".")); + Ok(resolve_future_path(parent)?.join(path.file_name().ok_or("Invalid storage path")?)) +} + +fn data_dir(config: &Config) -> AtomicResult { + Ok(config + .store_path + .parent() + .ok_or("Store has no parent")? + .to_path_buf()) +} + +fn capture(store: &Db, config: &Config, stage: &Path) -> AtomicResult<()> { + let source_data = data_dir(config)?.canonicalize()?; + let source_config = config.config_dir.canonicalize()?; + let db_path = stage.join("data/store/atomic.redb"); + fs::create_dir_all(db_path.parent().unwrap())?; + let mut snapshot_time = String::new(); + store.kv.backup_snapshot(&db_path, &mut || { + snapshot_time = chrono::Utc::now().to_rfc3339(); + copy_tree( + &source_data, + &stage.join("data"), + Some(Path::new("store/atomic.redb")), + )?; + copy_tree(&source_config, &stage.join("config"), None)?; + Ok(()) + })?; + let mut files = BTreeMap::new(); + inventory(stage, stage, &mut files)?; + let manifest = Manifest { + format: FORMAT, + server_version: env!("CARGO_PKG_VERSION").into(), + build_revision: env!("ATOMIC_BACKUP_REVISION").into(), + redb_version: "4.1.0".into(), + snapshot_time, + freshness: "Local persisted checkpoint; remote replication completeness is not asserted" + .into(), + envelope_retention: store.envelope_retention().as_str().into(), + source_data, + source_config, + files, + }; + fs::write(stage.join(MANIFEST), serde_json::to_vec_pretty(&manifest)?)?; + Ok(()) +} + +fn copy_tree(source: &Path, target: &Path, exclude: Option<&Path>) -> AtomicResult<()> { + fs::create_dir_all(target)?; + for item in fs::read_dir(source)? { + let item = item?; + let rel = PathBuf::from(item.file_name()); + if exclude == Some(rel.as_path()) { + continue; + } + let ty = item.file_type()?; + let dest = target.join(&rel); + if ty.is_symlink() { + return Err(format!("Refusing symlink in backup: {}", item.path().display()).into()); + } + if ty.is_dir() { + let nested = exclude.and_then(|p| p.strip_prefix(&rel).ok()); + copy_tree(&item.path(), &dest, nested)?; + } else if ty.is_file() { + let before = item.metadata()?; + fs::copy(item.path(), &dest)?; + let after = item.metadata()?; + if before.len() != after.len() || before.modified()? != after.modified()? { + return Err( + format!("File changed during backup: {}", item.path().display()).into(), + ); + } + private_file(&dest)?; + } else { + return Err(format!("Unsupported backup file: {}", item.path().display()).into()); + } + } + Ok(()) +} + +fn digest(path: &Path) -> AtomicResult { + let mut file = File::open(path)?; + let mut hasher = blake3::Hasher::new(); + let mut buffer = [0u8; 65536]; + let mut bytes = 0; + loop { + let n = file.read(&mut buffer)?; + if n == 0 { + break; + } + hasher.update(&buffer[..n]); + bytes += n as u64; + } + Ok(Entry { + bytes, + blake3: hasher.finalize().to_hex().to_string(), + }) +} + +fn inventory(root: &Path, dir: &Path, files: &mut BTreeMap) -> AtomicResult<()> { + for item in fs::read_dir(dir)? { + let item = item?; + if item.file_type()?.is_dir() { + inventory(root, &item.path(), files)?; + } else { + let path = item.path(); + let name = path + .strip_prefix(root) + .map_err(|e| e.to_string())? + .to_str() + .ok_or("Non-UTF8 backup path")? + .replace('\\', "/"); + files.insert(name, digest(&path)?); + } + } + Ok(()) +} + +fn archive(stage: &Path, output: &Path, id: &str) -> AtomicResult { + use zip::write::SimpleFileOptions; + let manifest: Manifest = serde_json::from_reader(File::open(stage.join(MANIFEST))?)?; + let mut temp = tempfile::NamedTempFile::new_in(output)?; + { + let mut writer = zip::ZipWriter::new(temp.as_file_mut()); + for name in manifest + .files + .keys() + .map(String::as_str) + .chain(std::iter::once(MANIFEST)) + { + let path = stage.join(name); + let options = SimpleFileOptions::default() + .compression_method(zip::CompressionMethod::Deflated) + .unix_permissions(0o600) + .large_file(fs::metadata(&path)?.len() >= u32::MAX as u64); + writer + .start_file(name, options) + .map_err(|e| e.to_string())?; + std::io::copy(&mut File::open(path)?, &mut writer)?; + } + writer.finish().map_err(|e| e.to_string())?; + } + temp.as_file().sync_all()?; + // Read back the ZIP and every hash before publication, without extracting. + verify_archive(temp.path())?; + let path = output.join(format!("atomic-backup-{id}.zip")); + temp.persist_noclobber(&path).map_err(|e| e.to_string())?; + File::open(output)?.sync_all()?; + Ok(path) +} + +fn safe_name(name: &str) -> bool { + !name.contains('\\') + && !name.contains(':') + && !name.contains('\0') + && Path::new(name) + .components() + .all(|p| matches!(p, Component::Normal(_))) + && (name.starts_with("data/") || name.starts_with("config/")) +} + +fn verify_archive(path: &Path) -> AtomicResult { + let mut zip = zip::ZipArchive::new(File::open(path)?).map_err(|e| e.to_string())?; + let manifest: Manifest = { + let mut file = zip.by_name(MANIFEST).map_err(|e| e.to_string())?; + if file.size() > 16 * 1024 * 1024 { + return Err("Oversized backup manifest".into()); + } + let mut json = Vec::new(); + file.read_to_end(&mut json)?; + serde_json::from_slice(&json)? + }; + if manifest.format != FORMAT { + return Err("Unsupported backup format".into()); + } + if manifest.server_version != env!("CARGO_PKG_VERSION") { + return Err(format!("Restore requires atomic-server {}", manifest.server_version).into()); + } + if !manifest.files.contains_key("data/store/atomic.redb") + || !manifest.files.contains_key("config/config.toml") + { + return Err("Backup lacks database or identity configuration".into()); + } + if zip.len() != manifest.files.len() + 1 { + return Err("Unexpected or duplicate archive entries".into()); + } + let mut seen = std::collections::HashSet::new(); + for i in 0..zip.len() { + let mut file = zip.by_index(i).map_err(|e| e.to_string())?; + let name = file.name().to_owned(); + if !seen.insert(name.clone()) { + return Err("Duplicate archive path".into()); + } + if name == MANIFEST { + continue; + } + if !safe_name(&name) + || file.is_dir() + || file.unix_mode().is_some_and(|m| m & 0o170000 == 0o120000) + { + return Err(format!("Unsafe archive path: {name}").into()); + } + let expected = manifest.files.get(&name).ok_or("Unlisted archive entry")?; + if file.size() != expected.bytes { + return Err("Archive size mismatch".into()); + } + let mut hasher = blake3::Hasher::new(); + let mut bytes = 0u64; + let mut buffer = [0u8; 65536]; + loop { + let n = file.read(&mut buffer)?; + if n == 0 { + break; + } + bytes += n as u64; + if bytes > expected.bytes { + return Err("Archive expanded beyond manifest size".into()); + } + hasher.update(&buffer[..n]); + } + if bytes != expected.bytes || hasher.finalize().to_hex().as_str() != expected.blake3 { + return Err(format!("Checksum mismatch for {name}").into()); + } + } + Ok(manifest) +} + +/// Restore never starts a runtime, sync transport, plugin or HTTP client. +/// The marker prevents an accidental normal server boot with copied identities. +pub fn restore(archive: &Path, target: &Path) -> AtomicResult<()> { + let manifest = verify_archive(archive)?; + if target.exists() { + return Err("Restore target must not exist".into()); + } + let parent = target + .parent() + .filter(|p| !p.as_os_str().is_empty()) + .unwrap_or(Path::new(".")); + let stage = tempfile::Builder::new() + .prefix(".atomic-restore-") + .tempdir_in(parent)?; + let mut zip = zip::ZipArchive::new(File::open(archive)?).map_err(|e| e.to_string())?; + for name in manifest.files.keys() { + let dest = stage.path().join(name); + fs::create_dir_all(dest.parent().unwrap())?; + let mut file = create_private(&dest)?; + std::io::copy( + &mut zip.by_name(name).map_err(|e| e.to_string())?, + &mut file, + )?; + file.sync_all()?; + let actual = digest(&dest)?; + let expected = &manifest.files[name]; + if actual.bytes != expected.bytes || actual.blake3 != expected.blake3 { + return Err("Archive changed during restore".into()); + } + } + atomic_lib::db::redb_store::verify_snapshot(&stage.path().join("data/store/atomic.redb"))?; + fs::write( + stage.path().join("data").join(RESTORED), + b"Restored offline. Explicit activation is required before starting a server.\n", + )?; + fs::write( + stage.path().join(MANIFEST), + serde_json::to_vec_pretty(&manifest)?, + )?; + // Claim the target without clobbering even an empty directory created by + // another operation since preflight. Incomplete promotion retains the marker. + fs::create_dir(target)?; + private_dir(target)?; + for item in fs::read_dir(stage.path())? { + let item = item?; + fs::rename(item.path(), target.join(item.file_name()))?; + } + File::open(parent)?.sync_all()?; + Ok(()) +} + +pub fn check_restore_activation(config: &Config) -> AtomicResult<()> { + if data_dir(config)?.join(RESTORED).exists() && !config.opts.activate_restored { + return Err("Restored instance is offline. Use --activate-restored only when ready to reconnect its copied identities and integrations".into()); + } + Ok(()) +} + +fn create_private(path: &Path) -> std::io::Result { + let mut options = OpenOptions::new(); + options.write(true).create_new(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt; + options.mode(0o600); + } + options.open(path) +} +fn private_file(path: &Path) -> std::io::Result<()> { + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + fs::set_permissions(path, fs::Permissions::from_mode(0o600))?; + } + Ok(()) +} +fn private_dir(path: &Path) -> std::io::Result<()> { + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + fs::set_permissions(path, fs::Permissions::from_mode(0o700))?; + } + Ok(()) +} + +/// Scheduler-friendly command: wait for this specific job and exit nonzero on +/// failure, replacement or timeout. Never follow redirects with the bearer token. +pub async fn request_backup(server: &str, token_file: &Path) -> AtomicServerResult<()> { + let url = url::Url::parse(server).map_err(|e| e.to_string())?; + let loopback = url.host_str().is_some_and(|h| { + h == "localhost" + || h.trim_matches(['[', ']']) + .parse::() + .is_ok_and(|ip| ip.is_loopback()) + }); + if !loopback + || !matches!(url.scheme(), "http" | "https") + || !url.username().is_empty() + || url.password().is_some() + { + return Err("Backup control requires a loopback HTTP(S) URL".into()); + } + let endpoint = url.join(CONTROL).map_err(|e| e.to_string())?; + let token = fs::read_to_string(token_file)?; + let client = reqwest::Client::builder() + .no_proxy() + .redirect(reqwest::redirect::Policy::none()) + .timeout(std::time::Duration::from_secs(30)) + .build() + .map_err(|e| e.to_string())?; + let started: Status = client + .post(endpoint.clone()) + .bearer_auth(token.trim()) + .send() + .await + .map_err(|e| e.to_string())? + .error_for_status() + .map_err(|e| e.to_string())? + .json() + .await + .map_err(|e| e.to_string())?; + for _ in 0..7200 { + tokio::time::sleep(std::time::Duration::from_secs(1)).await; + let state: Status = client + .get(endpoint.clone()) + .bearer_auth(token.trim()) + .send() + .await + .map_err(|e| e.to_string())? + .error_for_status() + .map_err(|e| e.to_string())? + .json() + .await + .map_err(|e| e.to_string())?; + if state.id != started.id { + return Err("Backup status replaced by another job or server restart".into()); + } + match state.phase.as_str() { + "complete" => { + println!("{}", state.archive.ok_or("Missing archive path")?.display()); + return Ok(()); + } + "failed" => return Err(state.error.unwrap_or_else(|| "Backup failed".into()).into()), + _ => {} + } + } + Err("Backup wait timed out; inspect status before retrying".into()) +} + +/// Drain complete requests, including GET class extenders and plugin filesystem +/// operations. The control route must bypass the gate or it would drain itself. +pub async fn admission( + request: actix_web::dev::ServiceRequest, + next: actix_web::middleware::Next, +) -> Result, actix_web::Error> { + if request.path() == CONTROL { + return next.call(request).await.map(|r| r.map_into_boxed_body()); + } + let gate = request + .app_data::>() + .map(|a| a.store.maintenance.clone()); + if let Some(gate) = gate { + if gate.is_paused() { + return Ok(request.into_response( + HttpResponse::ServiceUnavailable() + .insert_header(("Retry-After", "5")) + .body("Instance backup capture in progress"), + )); + } + gate.run(next.call(request)) + .await + .map(|r| r.map_into_boxed_body()) + } else { + next.call(request).await.map(|r| r.map_into_boxed_body()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use atomic_lib::{ + db::{ + kv_store::KvStore, + redb_store::RedbStore, + trees::{Method, Operation, Tree}, + }, + loro::AtomicLoroDoc, + Value, + }; + use clap::Parser; + + async fn fixture(root: &Path) -> (Db, Config, Arc) { + let config = crate::config::build_config(crate::config::Opts::parse_from([ + "atomic-server", + "--data-dir", + root.join("source").to_str().unwrap(), + "--config-dir", + root.join("configuration").to_str().unwrap(), + "--cache-dir", + root.join("cache").to_str().unwrap(), + "--backup-dir", + root.join("backups").to_str().unwrap(), + ])) + .unwrap(); + fs::create_dir_all(&config.config_dir).unwrap(); + let agent = atomic_lib::agents::Agent::new(None).unwrap(); + atomic_lib::config::Config { + shared: atomic_lib::config::SharedConfig { + agent_secret: agent.build_secret().unwrap(), + initial_drive: None, + }, + client: None, + } + .save(&config.config_file_path) + .unwrap(); + let store = Db::init_redb_file( + &config.store_path, + Some("http://localhost:9883".into()), + &config.uploads_path, + ) + .await + .unwrap(); + let service = BackupService::new(&config).unwrap(); + (store, config, service) + } + + #[tokio::test] + async fn instance_roundtrip_preserves_history_blobs_metadata_and_files() { + let root = tempfile::tempdir().unwrap(); + let (store, config, service) = fixture(root.path()).await; + let doc = AtomicLoroDoc::new(); + doc.set_property("name", &Value::String("before".into())) + .unwrap(); + let first = doc.export_snapshot(); + let first_version = doc.current_version(); + doc.set_property("name", &Value::String("after".into())) + .unwrap(); + let snapshot = doc.export_snapshot(); + assert_ne!(first, snapshot); + store + .kv + .insert(Tree::LoroSnapshots, b"did:ad:test", &snapshot) + .unwrap(); + for tree in [ + Tree::Blobs, + Tree::Envelopes, + Tree::PluginMeta, + Tree::DriveMapping, + ] { + store.kv.insert(tree, b"fixture", b"original").unwrap(); + } + fs::create_dir_all(&config.uploads_path).unwrap(); + fs::write(config.uploads_path.join("legacy.txt"), b"legacy file").unwrap(); + let path = run_backup(&store, &config, &service, "roundtrip") + .await + .unwrap(); + assert!(!store.maintenance.is_paused()); + store + .kv + .insert(Tree::PluginMeta, b"fixture", b"changed after backup") + .unwrap(); + let target = root.path().join("restored"); + restore(&path, &target).unwrap(); + assert!(restore(&path, &target).is_err()); + let restored = RedbStore::new_file(&target.join("data/store/atomic.redb")).unwrap(); + assert_eq!( + restored.get(Tree::PluginMeta, b"fixture").unwrap().unwrap(), + b"original" + ); + assert_eq!( + restored.get(Tree::Blobs, b"fixture").unwrap().unwrap(), + b"original" + ); + assert_eq!( + restored.get(Tree::Envelopes, b"fixture").unwrap().unwrap(), + b"original" + ); + let recovered = AtomicLoroDoc::from_snapshot( + &restored + .get(Tree::LoroSnapshots, b"did:ad:test") + .unwrap() + .unwrap(), + ) + .unwrap(); + assert_eq!(recovered.get_history().len(), doc.get_history().len()); + recovered.fork_at(&first_version).unwrap(); + assert_eq!( + fs::read(target.join("data/uploads/legacy.txt")).unwrap(), + b"legacy file" + ); + assert_eq!( + fs::read(target.join("config/config.toml")).unwrap(), + fs::read(&config.config_file_path).unwrap() + ); + let mut restored_config = config.clone(); + restored_config.store_path = target.join("data/store"); + assert!(check_restore_activation(&restored_config).is_err()); + restored_config.opts.activate_restored = true; + check_restore_activation(&restored_config).unwrap(); + } + + #[test] + fn snapshot_blocks_writers_and_copies_unknown_byte_tables() { + let root = tempfile::tempdir().unwrap(); + let source = root.path().join("source.redb"); + // Future tables must not silently disappear from the backup. + { + let db = redb_for_test(&source); + let tx = db.begin_write().unwrap(); + { + let mut table = tx + .open_table(redb::TableDefinition::<&[u8], &[u8]>::new("future_table")) + .unwrap(); + table + .insert(b"future".as_slice(), b"value".as_slice()) + .unwrap(); + } + tx.commit().unwrap(); + } + let store = Arc::new(RedbStore::new_file(&source).unwrap()); + store + .apply_batch(&[ + Operation { + tree: Tree::Resources, + method: Method::Insert, + key: b"key".to_vec(), + val: Some(b"old".to_vec()), + }, + Operation { + tree: Tree::LoroSnapshots, + method: Method::Insert, + key: b"key".to_vec(), + val: Some(b"old".to_vec()), + }, + ]) + .unwrap(); + let (entered_tx, entered_rx) = std::sync::mpsc::channel(); + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let s = store.clone(); + let destination = root.path().join("snapshot.redb"); + let path = destination.clone(); + let backup = std::thread::spawn(move || { + s.backup_snapshot(&path, &mut || { + entered_tx.send(()).unwrap(); + release_rx.recv().unwrap(); + Ok(()) + }) + }); + entered_rx + .recv_timeout(std::time::Duration::from_secs(10)) + .unwrap(); + let s = store.clone(); + let (written_tx, written_rx) = std::sync::mpsc::channel(); + let writer = std::thread::spawn(move || { + s.insert(Tree::Resources, b"key", b"new").unwrap(); + written_tx.send(()).unwrap(); + }); + assert!(written_rx + .recv_timeout(std::time::Duration::from_millis(50)) + .is_err()); + release_tx.send(()).unwrap(); + backup.join().unwrap().unwrap(); + writer.join().unwrap(); + let snapshot = RedbStore::new_file(&destination).unwrap(); + assert_eq!( + snapshot.get(Tree::Resources, b"key").unwrap().unwrap(), + b"old" + ); + assert_eq!( + snapshot.get(Tree::LoroSnapshots, b"key").unwrap().unwrap(), + b"old" + ); + drop(snapshot); + use redb::ReadableDatabase; + let db = redb_for_test(&destination); + let tx = db.begin_read().unwrap(); + let table = tx + .open_table(redb::TableDefinition::<&[u8], &[u8]>::new("future_table")) + .unwrap(); + assert_eq!( + table.get(b"future".as_slice()).unwrap().unwrap().value(), + b"value" + ); + } + + fn redb_for_test(path: &Path) -> redb::Database { + redb::Database::create(path).unwrap() + } + + #[tokio::test] + async fn capture_failure_resumes_and_publishes_nothing() { + let root = tempfile::tempdir().unwrap(); + let (store, config, service) = fixture(root.path()).await; + #[cfg(unix)] + std::os::unix::fs::symlink("/outside", config.config_dir.join("unsupported")).unwrap(); + #[cfg(not(unix))] + { + store.kv.begin_batch(); + } + assert!(run_backup(&store, &config, &service, "failure") + .await + .is_err()); + assert!(!store.maintenance.is_paused()); + assert!(!root + .path() + .join("backups/atomic-backup-failure.zip") + .exists()); + store + .kv + .insert(Tree::PluginMeta, b"after-failure", b"works") + .unwrap(); + } + + #[test] + fn snapshot_callback_failure_releases_storage_barrier() { + let root = tempfile::tempdir().unwrap(); + let store = RedbStore::new_file(&root.path().join("source.redb")).unwrap(); + assert!(store + .backup_snapshot(&root.path().join("failed.redb"), &mut || Err( + "injected I/O failure".into() + )) + .is_err()); + store.insert(Tree::PluginMeta, b"after", b"works").unwrap(); + assert!(store + .backup_snapshot(&root.path().join("failed.redb"), &mut || Ok(())) + .is_err()); + store.begin_batch(); + assert!(store + .backup_snapshot(&root.path().join("batch.redb"), &mut || Ok(())) + .is_err()); + store.commit_batch().unwrap(); + let panicked = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let _ = store.backup_snapshot(&root.path().join("panic.redb"), &mut || { + panic!("injected capture panic") + }); + })); + assert!(panicked.is_err()); + store + .insert(Tree::PluginMeta, b"after-panic", b"works") + .unwrap(); + } + + #[tokio::test] + async fn operator_requires_token_and_loopback_even_during_pause() { + let root = tempfile::tempdir().unwrap(); + let (store, _config, service) = fixture(root.path()).await; + use actix_web::test::TestRequest; + let local = "127.0.0.1:1234".parse().unwrap(); + let request = TestRequest::get().peer_addr(local).to_http_request(); + assert!(!service.authorized(&request)); + let authorized = TestRequest::get() + .peer_addr(local) + .insert_header(("Authorization", format!("Bearer {}", service.token))) + .to_http_request(); + let pause = store.maintenance.pause().await.unwrap(); + assert!(service.authorized(&authorized)); + let remote = TestRequest::get() + .peer_addr("192.0.2.1:1234".parse().unwrap()) + .insert_header(("Authorization", format!("Bearer {}", service.token))) + .to_http_request(); + assert!(!service.authorized(&remote)); + drop(pause); + } + + #[cfg(unix)] + #[tokio::test] + async fn existing_destination_permissions_are_not_changed() { + use std::os::unix::fs::PermissionsExt; + let root = tempfile::tempdir().unwrap(); + let (_store, config, _service) = fixture(root.path()).await; + let output = config.opts.backup_dir.as_ref().unwrap(); + fs::set_permissions(output, fs::Permissions::from_mode(0o755)).unwrap(); + BackupService::new(&config).unwrap(); + assert_eq!( + fs::metadata(output).unwrap().permissions().mode() & 0o777, + 0o755 + ); + let mut invalid = config.clone(); + invalid.opts.backup_dir = Some(config.config_dir.clone()); + assert!(BackupService::new(&invalid).is_err()); + } + + #[test] + fn restore_rejects_unsafe_paths() { + for name in [ + "../escape", + "data/../../escape", + "/data/file", + "config/../escape", + "data\\escape", + "data/C:escape", + ] { + assert!(!safe_name(name), "{name}"); + } + assert!(safe_name("data/store/atomic.redb")); + } +} + +#[cfg(test)] +mod sync_tests { + use super::*; + use atomic_lib::{ + agents::ForAgent, + db::trees::Tree, + loro::AtomicLoroDoc, + sync::{engine, protocol}, + Storelike, Value, + }; + + #[tokio::test] + async fn incoming_sync_waits_without_acknowledging_or_losing_the_update() { + let root = tempfile::tempdir().unwrap(); + let db = Db::init_redb_file( + &root.path().join("store"), + Some("http://localhost:9883".into()), + &root.path().join("uploads"), + ) + .await + .unwrap(); + let (agent, drive) = db.setup("Backup sync test").await.unwrap(); + let subject = "did:ad:backup-sync-test"; + let doc = AtomicLoroDoc::new(); + doc.set_property( + atomic_lib::urls::DRIVE_PROP, + &Value::AtomicUrl(drive.clone().into()), + ) + .unwrap(); + doc.set_property( + atomic_lib::urls::NAME, + &Value::String("arrived during backup".into()), + ) + .unwrap(); + let frame = protocol::encode_sync_push(&drive, &[(subject, &doc.export_snapshot())], true); + let push = protocol::decode_sync_push(&frame[1..]).unwrap(); + let pause = db.maintenance.pause().await.unwrap(); + let receiver = db.clone(); + let (started_tx, started_rx) = tokio::sync::oneshot::channel(); + let job = tokio::spawn(async move { + started_tx.send(()).unwrap(); + engine::import_sync_push(&push, &receiver, &ForAgent::from(agent), false).await + }); + started_rx.await.unwrap(); + tokio::task::yield_now().await; + assert!(!job.is_finished(), "must not acknowledge a paused import"); + assert!(db + .kv + .get(Tree::LoroSnapshots, subject.as_bytes()) + .unwrap() + .is_none()); + drop(pause); + let (count, _) = tokio::time::timeout(std::time::Duration::from_secs(10), job) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(count, 1); + assert_eq!( + db.get_resource(&subject.into()) + .await + .unwrap() + .get(atomic_lib::urls::NAME) + .unwrap() + .to_string(), + "arrived during backup" + ); + } +} + +#[cfg(test)] +mod archive_tests { + use super::*; + + fn untrusted(root: &Path, name: &str, content: &[u8], expected_hash: &str) -> PathBuf { + let path = root.join("untrusted.zip"); + let mut files = BTreeMap::new(); + for key in ["data/store/atomic.redb", "config/config.toml", name] { + files.insert( + key.into(), + Entry { + bytes: content.len() as u64, + blake3: expected_hash.into(), + }, + ); + } + let manifest = Manifest { + format: FORMAT, + server_version: env!("CARGO_PKG_VERSION").into(), + build_revision: "test".into(), + redb_version: "4.1.0".into(), + snapshot_time: "test".into(), + freshness: "unknown".into(), + envelope_retention: "latest".into(), + source_data: "source".into(), + source_config: "config".into(), + files, + }; + let mut writer = zip::ZipWriter::new(File::create(&path).unwrap()); + let options = zip::write::SimpleFileOptions::default(); + writer.start_file(MANIFEST, options).unwrap(); + writer + .write_all(&serde_json::to_vec(&manifest).unwrap()) + .unwrap(); + for key in manifest.files.keys() { + writer.start_file(key, options).unwrap(); + writer.write_all(content).unwrap(); + } + writer.finish().unwrap(); + path + } + + #[test] + fn corrupt_or_traversing_archive_never_creates_restore_target() { + let root = tempfile::tempdir().unwrap(); + let target = root.path().join("restore"); + let archive = untrusted(root.path(), "data/note", b"tampered", "wrong checksum"); + assert!(restore(&archive, &target) + .unwrap_err() + .to_string() + .contains("Checksum")); + assert!(!target.exists()); + let hash = blake3::hash(b"valid").to_hex().to_string(); + let archive = untrusted(root.path(), "data/../../escape", b"valid", &hash); + assert!(restore(&archive, &target) + .unwrap_err() + .to_string() + .contains("Unsafe")); + assert!(!target.exists()); + assert!(!root.path().join("escape").exists()); + } + + #[test] + fn valid_hashes_do_not_make_a_corrupt_database_restorable() { + let root = tempfile::tempdir().unwrap(); + let hash = blake3::hash(b"not a redb file").to_hex().to_string(); + let archive = untrusted(root.path(), "data/note", b"not a redb file", &hash); + let target = root.path().join("restore"); + assert!(restore(&archive, &target).is_err()); + assert!(!target.exists()); + } +} diff --git a/server/src/bin.rs b/server/src/bin.rs index 0369dfeefe..6012730cd9 100644 --- a/server/src/bin.rs +++ b/server/src/bin.rs @@ -4,6 +4,7 @@ use std::{fs::File, io::Write}; mod actor_messages; mod appstate; +pub mod backup; mod blob_storage; mod commit_monitor; pub mod config; @@ -46,6 +47,20 @@ async fn main_wrapped() -> errors::AtomicServerResult<()> { .map_err(|e| format!("Initialization failed: {}", e))?; match &config.opts.command { + Some(config::Command::Backup { server, token_file }) => { + backup::request_backup( + server, + &token_file + .clone() + .unwrap_or_else(|| config.config_dir.join("backup.token")), + ) + .await + } + Some(config::Command::Restore { archive, target }) => { + backup::restore(archive, target)?; + println!("Restored offline into {}", target.display()); + Ok(()) + } Some(config::Command::Export(e)) => { let path = match e.path.clone() { Some(p) => std::path::Path::new(&p).to_path_buf(), diff --git a/server/src/config.rs b/server/src/config.rs index 57bb555ec3..6fe7156f30 100644 --- a/server/src/config.rs +++ b/server/src/config.rs @@ -14,6 +14,14 @@ pub struct Opts { #[clap(subcommand)] pub command: Option, + /// Directory for full instance ZIP backups. Enables local token-protected control. + #[clap(long, env = "ATOMIC_BACKUP_DIR")] + pub backup_dir: Option, + + /// Explicitly allow a restored instance to reconnect using its copied identity. + #[clap(long)] + pub activate_restored: bool, + /// Recreates the `/setup` Invite for creating a new Root User. Also re-runs various populate commands, and re-builds the index #[clap(long, env = "ATOMIC_INITIALIZE")] pub initialize: bool, @@ -248,6 +256,21 @@ pub enum RebuildIndexMode { #[derive(Parser, Clone, Debug)] pub enum Command { + /// Request a backup from a running local server and wait for verification. + Backup { + #[clap(long, default_value = "http://127.0.0.1:9883")] + server: String, + /// Defaults to backup.token in --config-dir. + #[clap(long)] + token_file: Option, + }, + /// Verify and restore a full instance ZIP into a new offline directory. + Restore { + #[clap(long)] + archive: PathBuf, + #[clap(long)] + target: PathBuf, + }, /// Create and save a JSON-AD backup of the store. #[clap(name = "export")] Export(ExportOpts), diff --git a/server/src/lib.rs b/server/src/lib.rs index 181ae1ac4a..7aeea55e50 100644 --- a/server/src/lib.rs +++ b/server/src/lib.rs @@ -5,6 +5,7 @@ See https://github.com/atomicdata-dev/atomic-server/tree/master/src-tauri */ mod actor_messages; pub mod appstate; +pub mod backup; pub mod blob_storage; mod commit_monitor; pub mod config; diff --git a/server/src/routes.rs b/server/src/routes.rs index 7aef7352dc..f2b57852b8 100644 --- a/server/src/routes.rs +++ b/server/src/routes.rs @@ -344,6 +344,7 @@ fn configure_wasm_plugin_routes(app: &mut actix_web::web::ServiceConfig) { pub fn config_routes(app: &mut actix_web::web::ServiceConfig) { handlers::website::control_routes(app); + crate::backup::routes(app); app.service( web::resource("/upload") .guard(guard::Method(Method::POST)) diff --git a/server/src/serve.rs b/server/src/serve.rs index 0b9f7270bf..c254f7dc82 100644 --- a/server/src/serve.rs +++ b/server/src/serve.rs @@ -44,12 +44,18 @@ async fn rebuild_indexes( actix_web::rt::spawn(async move { appstate_clone .store - .clear_index() - .expect("Failed to clear value index"); - appstate_clone - .store - .build_index(true) - .expect("Failed to build value index"); + .maintenance + .run(async { + appstate_clone + .store + .clear_index() + .expect("Failed to clear value index"); + appstate_clone + .store + .build_index(true) + .expect("Failed to build value index"); + }) + .await; }); } @@ -381,6 +387,8 @@ where // server passes a no-op (see `serve`), so it never phones home. on_ready(&appstate); + #[cfg(feature = "https")] + let maintenance = appstate.store.maintenance.clone(); let server = HttpServer::new(move || { let cors = Cors::permissive().expose_headers([SERVER_VERSION_HEADER]); @@ -388,6 +396,7 @@ where .app_data(web::PayloadConfig::new(PAYLOAD_MAX)) .app_data(web::Data::new(appstate.clone())) .wrap(cors) + .wrap(middleware::from_fn(crate::backup::admission)) // Attaches the request (method, url, headers) to any Sentry event // raised while handling it, and reports handler panics and 5xx // errors. No-op without a bound Sentry client. @@ -477,7 +486,7 @@ where } let https_config = crate::https::get_https_config(&config) .expect("HTTPS TLS Configuration with Let's Encrypt failed."); - spawn_cert_renewal_task(config.clone()); + spawn_cert_renewal_task(config.clone(), maintenance.clone()); let endpoint = format!("{}:{}", config.opts.ip, config.opts.port_https); tracing::info!("Binding HTTPS server to endpoint {}", endpoint); println!("{}", message); @@ -523,7 +532,10 @@ const TIMEOUT: u64 = 15; /// keeps serving the old one, so the server has to be restarted to pick it /// up. Before this, a long-running server never renewed at all. #[cfg(feature = "https")] -fn spawn_cert_renewal_task(config: crate::config::Config) { +fn spawn_cert_renewal_task( + config: crate::config::Config, + maintenance: atomic_lib::db::maintenance::Maintenance, +) { actix_web::rt::spawn(async move { let mut interval = tokio::time::interval(std::time::Duration::from_secs(24 * 60 * 60)); // The first tick completes at once; the certificates were checked @@ -531,6 +543,7 @@ fn spawn_cert_renewal_task(config: crate::config::Config) { interval.tick().await; loop { interval.tick().await; + maintenance.run(async { match crate::https::should_renew_certs_check(&config) { Ok(false) => {} Ok(true) => match crate::https::request_cert(&config).await { @@ -542,6 +555,7 @@ fn spawn_cert_renewal_task(config: crate::config::Config) { }, Err(e) => tracing::error!("Could not check the HTTPS certificate age: {}", e), } + }).await; } }); } diff --git a/server/tests/it/instance_backup.rs b/server/tests/it/instance_backup.rs new file mode 100644 index 0000000000..3ea1b843f7 --- /dev/null +++ b/server/tests/it/instance_backup.rs @@ -0,0 +1,197 @@ +//! Real process, real operator CLI, real WS replication, offline restore. +use atomic_lib::{ + agents::ForAgent, + sync::replicate::{replicate_drive_to_remote, ReplicateAuth}, + Storelike, +}; +use std::{ + path::Path, + process::{Child, Command, Stdio}, + time::Duration, +}; + +struct Server(Child); +impl Drop for Server { + fn drop(&mut self) { + let _ = self.0.kill(); + let _ = self.0.wait(); + } +} +fn binary() -> &'static str { + env!("CARGO_BIN_EXE_atomic-server") +} +fn command(root: &Path) -> Command { + let mut cmd = Command::new(binary()); + cmd.env_clear().current_dir(root).args([ + "--data-dir", + root.join("data").to_str().unwrap(), + "--config-dir", + root.join("config").to_str().unwrap(), + "--cache-dir", + root.join("cache").to_str().unwrap(), + ]); + cmd +} + +#[tokio::test] +async fn cli_backs_up_a_running_replica_and_restores_offline() { + let root = tempfile::tempdir().unwrap(); + let port = std::net::TcpListener::bind("127.0.0.1:0") + .unwrap() + .local_addr() + .unwrap() + .port(); + let log = std::fs::File::create(root.path().join("server.log")).unwrap(); + let mut child = Server( + command(root.path()) + .args([ + "--ip", + "127.0.0.1", + "--domain", + "127.0.0.1", + "--port", + &port.to_string(), + "--host-mode", + "open", + "--backup-dir", + root.path().join("backups").to_str().unwrap(), + ]) + .stdout(Stdio::from(log.try_clone().unwrap())) + .stderr(Stdio::from(log)) + .spawn() + .unwrap(), + ); + let base = format!("http://127.0.0.1:{port}"); + let client = reqwest::Client::builder() + .timeout(Duration::from_secs(5)) + .build() + .unwrap(); + let deadline = std::time::Instant::now() + Duration::from_secs(60); + loop { + if client + .get(format!("{base}/__atomic/backup")) + .send() + .await + .is_ok() + { + break; + } + assert!( + child.0.try_wait().unwrap().is_none(), + "server exited: {}", + std::fs::read_to_string(root.path().join("server.log")).unwrap() + ); + assert!(std::time::Instant::now() < deadline, "server did not start"); + tokio::time::sleep(Duration::from_millis(50)).await; + } + // A second node feeds this replica through the actual WS sync transport. + let source = atomic_lib::Db::init_redb_file( + &root.path().join("source-store"), + None, + &root.path().join("source-uploads"), + ) + .await + .unwrap(); + let (agent, drive) = source.setup("Source").await.unwrap(); + let note = source + .create_resource(atomic_lib::urls::FOLDER, &drive, "Before backup", None) + .await + .unwrap(); + let outcome = replicate_drive_to_remote( + &source, + &drive, + &format!("ws://127.0.0.1:{port}/ws"), + &ForAgent::AgentSubject(agent.subject.clone()), + ReplicateAuth::Agent(Box::new(agent.clone())), + ) + .await + .unwrap(); + assert!(outcome.in_sync); + let status: serde_json::Value = client + .get(format!("{base}/__atomic/backup")) + .send() + .await + .unwrap() + .status() + .as_u16() + .into(); + assert_eq!(status, 401); + let mut backup = command(root.path()); + backup.args(["backup", "--server", &base]); + let output = tokio::task::spawn_blocking(move || backup.output().unwrap()) + .await + .unwrap(); + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let path = String::from_utf8(output.stdout).unwrap(); + let archive = path.trim(); + assert!(Path::new(archive).exists()); + // The live instance stays usable after capture; a later change must not + // alter the previously published checkpoint. + let mut resource = source.get_resource(¬e.clone().into()).await.unwrap(); + resource + .set_string(atomic_lib::urls::NAME.into(), "After backup", &source) + .await + .unwrap(); + resource.save(&source).await.unwrap(); + let outcome = replicate_drive_to_remote( + &source, + &drive, + &format!("ws://127.0.0.1:{port}/ws"), + &ForAgent::AgentSubject(agent.subject.clone()), + ReplicateAuth::Agent(Box::new(agent)), + ) + .await + .unwrap(); + assert!(outcome.in_sync); + let target = root.path().join("restore"); + let output = command(root.path()) + .args([ + "restore", + "--archive", + archive, + "--target", + target.to_str().unwrap(), + ]) + .output() + .unwrap(); + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert!(target.join("data/RESTORED_OFFLINE").exists()); + let output = Command::new(binary()) + .env_clear() + .current_dir(root.path()) + .args([ + "--data-dir", + target.join("data").to_str().unwrap(), + "--config-dir", + target.join("config").to_str().unwrap(), + ]) + .output() + .unwrap(); + assert!(!output.status.success()); + assert!(String::from_utf8_lossy(&output.stderr).contains("Restored instance is offline")); + let recovered = atomic_lib::Db::init_redb_file( + &target.join("data/store"), + Some(base), + &target.join("data/uploads"), + ) + .await + .unwrap(); + assert_eq!( + recovered + .get_resource(¬e.into()) + .await + .unwrap() + .get(atomic_lib::urls::NAME) + .unwrap() + .to_string(), + "Before backup" + ); +} diff --git a/server/tests/it/main.rs b/server/tests/it/main.rs index 9b0723e088..e51f32cd0b 100644 --- a/server/tests/it/main.rs +++ b/server/tests/it/main.rs @@ -11,6 +11,7 @@ mod drive_presence; mod drive_presence_shared; mod file_search_repro; mod history_attribution; +mod instance_backup; mod iroh_pairing; mod loro_ephemeral_sync; mod multi_client_sync; From c9604a2deebd202bff81944da69b5e950128d839 Mon Sep 17 00:00:00 2001 From: Joep Meindertsma Date: Fri, 11 Sep 2026 15:42:29 +0200 Subject: [PATCH 2/2] Move instance checkpoint capture and restore into atomic_lib --- .dagger/src/index.ts | 2 + CHANGELOG.md | 2 + Cargo.lock | 4 +- TESTING_COVERAGE.md | 17 +- docs/src/instance-backups.md | 33 + lib/Cargo.toml | 7 + lib/src/backup.rs | 858 +++++++++++++++++++++++++ lib/src/lib.rs | 2 + lib/tests/check-instance-checkpoint.sh | 13 + lib/tests/instance_checkpoint.rs | 110 ++++ planning/atomic-lib-runtime.md | 13 + server/Cargo.toml | 3 +- server/src/backup.rs | 765 +--------------------- 13 files changed, 1082 insertions(+), 747 deletions(-) create mode 100644 lib/src/backup.rs create mode 100755 lib/tests/check-instance-checkpoint.sh create mode 100644 lib/tests/instance_checkpoint.rs diff --git a/.dagger/src/index.ts b/.dagger/src/index.ts index 2a2d40e1c7..d5aa425510 100644 --- a/.dagger/src/index.ts +++ b/.dagger/src/index.ts @@ -1505,6 +1505,8 @@ export class AtomicServer { `--test-threads ${this.hostKnobs.nextestTestThreads} ` + `--retries ${this.hostKnobs.nextestRetries}`, ]) + // Select the core alone so server dependencies cannot mask missing features. + .withExec(['sh', 'lib/tests/check-instance-checkpoint.sh']) .stdout() ); } diff --git a/CHANGELOG.md b/CHANGELOG.md index 27e1179b30..896d3e2771 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -58,6 +58,8 @@ See [STATUS.md](server/STATUS.md) to learn more about which features will remain switch. `Db::init_redb_file` now owns the 100ms durable-flush tick for every binding, and an idle tick no longer writes anything. - Fix remaining `clippy` warnings in `wasm/src/lib.rs` blocking `develop`'s pre-commit hook ([#1508](https://github.com/ontola/atomic-server/issues/1508)). +- Instance checkpoint capture, verification and offline restore are available in + `atomic_lib` through the optional native `backup` feature, without server/Actix. - Add opt-in full-instance backups: `--backup-dir`, local authenticated backup control, `backup`/`restore` CLI commands, checksummed ZIP64 archives and an offline restore guard. Writes and incoming sync application pause during diff --git a/Cargo.lock b/Cargo.lock index 03730e5d35..d82fbede70 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1266,7 +1266,6 @@ dependencies = [ "portpicker", "rand 0.8.6", "rcgen 0.14.8", - "redb", "regex", "reqwest 0.13.3", "ring", @@ -1374,6 +1373,7 @@ dependencies = [ "bincode", "blake3", "chacha20poly1305", + "chrono", "criterion", "directories", "ed25519-dalek 2.2.0", @@ -1402,6 +1402,7 @@ dependencies = [ "serde_json", "sha2 0.10.9", "sled", + "tempfile", "tokio", "tokio-tungstenite 0.29.0", "toml 1.1.2+spec-1.1.0", @@ -1415,6 +1416,7 @@ dependencies = [ "wasm-bindgen-futures", "web-sys", "web-time", + "zip 8.6.0", ] [[package]] diff --git a/TESTING_COVERAGE.md b/TESTING_COVERAGE.md index a2e3845263..f9ce83d7d2 100644 --- a/TESTING_COVERAGE.md +++ b/TESTING_COVERAGE.md @@ -180,18 +180,25 @@ duplicate issues/comments. Live proxy OAuth, GitHub writes and a guided uncertain-write recovery UI remain unverified/unbuilt; proxy v40 CORS and browser OAuth are verified, but its GitHub credential returns 404 for the private sandbox. -Instance backup: `server/src/backup.rs` covers full redb/file round-trip including -Loro historical checkout, blobs, envelopes and identities; future byte tables; +Instance backup: `lib/src/backup.rs` covers full redb/file round-trip including +Loro historical checkout, blobs, envelopes and configuration files; future byte tables; concurrent writer exclusion; buffered-batch refusal; capture error/panic recovery; -loopback/token authorization; unsafe paths, hash mismatch and corrupt-database -refusal. `lib/src/db/maintenance.rs` covers nested admission during drain, +unsafe paths, hash mismatch and corrupt-database refusal. +`server/src/backup.rs` covers loopback/token authorization and adapter setup. +`lib/src/db/maintenance.rs` covers nested admission during drain, overlapping pause refusal, queued work and cancellation recovery. The shared sync-engine test proves a paused import is neither acknowledged nor dropped. `server/tests/it/instance_backup.rs` starts a real server process, replicates a second node over WebSockets, invokes the backup CLI, replicates a later change, restores the earlier checkpoint and verifies accidental startup is refused. -Remaining backup coverage gaps: OS-level disk-full/power-loss injection, +`lib/tests/check-instance-checkpoint.sh` selects only `atomic_lib` with +`backup,config`, rejects server/Actix dependencies, runs the core backup tests, +and tests an originless capture/verify/restore/reopen with persisted identity in +`lib/tests/instance_checkpoint.rs`. Dagger runs this separately after workspace tests. + +Remaining backup coverage gaps: desktop VFS staging drain and native UI restore, +OS-level disk-full/power-loss injection, large (>4 GiB) ZIP64 fixtures and pause-duration benchmarks, and a dedicated Iroh disconnect-during-capture test. The gate is shared by both transports; these tests do not assert a globally synchronized checkpoint or lifetime history diff --git a/docs/src/instance-backups.md b/docs/src/instance-backups.md index d6c25b3adf..1c337433fe 100644 --- a/docs/src/instance-backups.md +++ b/docs/src/instance-backups.md @@ -106,3 +106,36 @@ inspect it and choose a fresh destination for retry. Test a restore before relying on the first archive, then repeat periodically. Take an extra checkpoint before risky experiments. Nightly retention accepts up to a day of data loss; it does not retain every intermediate edit. + +## Shared native checkpoint API + +Capture, ZIP verification, and offline restore live in `atomic_lib::backup`, +behind the optional native `backup` feature. `CheckpointOptions` supplies explicit +data, configuration and output roots plus the caller's build revision. The server +provides only operator authentication, HTTP/CLI control, status, and server cache +configuration checks. Native callers invoke the same `create`, `verify`, `restore` +and `check_restore_activation` functions without an HTTP listener or Actix. +The v1 layout remains `data/store/atomic.redb` plus data files and +`config/config.toml`. The legacy manifest key `server_version` now records the +core package version; v1 archives remain readable at that same package version. + +These mechanisms serve different recovery needs: + +| Mechanism | Purpose | Restore behavior | +| --- | --- | --- | +| Desktop virtual filesystem (`desktop/src/vfs.rs`) | Read/write projection of graph folders and files; writes become ordinary signed commits | Filesystem operations edit live graph state; the projection does not contain all instance metadata or history | +| Encrypted vault (`atomic_lib::vault`) | Portable drive-level encrypted segments using `VaultObjectStore`, including its filesystem backend | Imports and merges drive data into a node | +| Instance checkpoint (`atomic_lib::backup`) | Consistent persisted database plus local configuration and files | Restores a complete local instance into a new offline directory | + +Checkpoint creation uses the existing `Db` and its maintenance gate; it does not +introduce a second sync engine. A checkpoint is a streamed full-instance ZIP, so +it does not use the vault's in-memory sealed-object interface or its merge format. +A future destination abstraction should preserve streaming and these distinct +restore semantics. + +The checkpoint boundary covers **persisted state**. Desktop VFS writes are staged +in memory before becoming commits. A desktop adapter must stop new staging, +flush pending writes before `create`, and coordinate external file writers with +`Db::maintenance`. It must check the restored-offline marker before starting sync +or integrations. This PR provides the native API, not a desktop backup UI or a +verified desktop staging/restore integration. diff --git a/lib/Cargo.toml b/lib/Cargo.toml index 06985b2cc0..a44a456aca 100644 --- a/lib/Cargo.toml +++ b/lib/Cargo.toml @@ -72,6 +72,11 @@ wasm-bindgen = { version = "0.2.122", optional = true } wasm-bindgen-futures = { version = "0.4.72", optional = true } web-sys = { version = "0.3.99", optional = true, features = ["DomException", "FileSystemDirectoryHandle", "FileSystemFileHandle", "FileSystemGetFileOptions", "FileSystemSyncAccessHandle", "FileSystemReadWriteOptions", "StorageManager", "WorkerGlobalScope", "WorkerNavigator"] } +[target.'cfg(not(target_arch = "wasm32"))'.dependencies] +chrono = { version = "0.4.44", optional = true } +tempfile = { version = "3", optional = true } +zip = { version = "8.6.0", default-features = false, features = ["deflate"], optional = true } + [dev-dependencies] criterion = { version = "0.8.2", features = ["async_tokio"] } iai = "0.1" @@ -88,6 +93,8 @@ optional = true version = "0.32" [features] +## Native full-instance checkpoints; no hosted server or Actix dependency. +backup = ["db-redb", "dep:chrono", "dep:tempfile", "dep:zip", "tokio/time"] config = ["directories", "toml"] ## Core Db with encoding support (BTreeMapStore included). ## Does NOT pull in sled — works in WASM. diff --git a/lib/src/backup.rs b/lib/src/backup.rs new file mode 100644 index 0000000000..82117a1bfd --- /dev/null +++ b/lib/src/backup.rs @@ -0,0 +1,858 @@ +//! Native persisted-instance checkpoints, independent of HTTP and server configuration. +//! +//! This is a complete local redb/config checkpoint, not a graph filesystem +//! projection or a drive-level encrypted vault. Callers must drain their own +//! buffered writes (for example desktop VFS staging) before calling `create`, +//! and keep external file writers under `Db::maintenance` for the capture. +//! Restore never opens a runtime; adapters must check the offline marker before +//! starting services that reconnect copied identities. +use crate::{errors::AtomicResult, Db}; +use serde::{Deserialize, Serialize}; +use std::{ + collections::BTreeMap, + fs::{self, File, OpenOptions}, + io::Read, + path::{Component, Path, PathBuf}, +}; + +const FORMAT: u32 = 1; +const MANIFEST: &str = "manifest.json"; +const RESTORED: &str = "RESTORED_OFFLINE"; + +/// Explicit roots for the v1 instance layout. Data contains `store/atomic.redb` +/// and optional uploads/other native files; config contains `config.toml`. +/// Cache and external storage must be outside these roots. +#[derive(Clone, Debug)] +pub struct CheckpointOptions { + pub data_dir: PathBuf, + pub config_dir: PathBuf, + pub output_dir: PathBuf, + /// Adapter build revision, for diagnostics; not a dependency on its runtime. + pub build_revision: String, +} + +impl CheckpointOptions { + /// Validate roots and prepare a private output directory. Existing directory + /// permissions are preserved. Called again at capture, not only at startup. + pub fn prepare(&self) -> AtomicResult<()> { + let data = self.data_dir.canonicalize()?; + let config = self.config_dir.canonicalize()?; + let output = resolve_future_path(&self.output_dir)?; + if data.starts_with(&config) || config.starts_with(&data) { + return Err( + "Instance backup requires separate, non-overlapping data and config directories" + .into(), + ); + } + for source in [&data, &config] { + if output.starts_with(source) || source.starts_with(&output) { + return Err("Backup directory must be outside data and config directories".into()); + } + } + if !self.output_dir.exists() { + let mut builder = fs::DirBuilder::new(); + builder.recursive(true); + #[cfg(unix)] + { + use std::os::unix::fs::DirBuilderExt; + builder.mode(0o700); + } + builder.create(&self.output_dir)?; + } + Ok(()) + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Phase { + Capturing, + Archiving, +} + +/// Drain admitted store operations, capture a consistent persisted checkpoint, +/// then resume writers before compression. The blocking worker owns the pause +/// guard even if the caller drops this future. `progress` must not access the +/// store while capturing. IDs are single filename components and never overwrite. +pub async fn create( + store: &Db, + options: &CheckpointOptions, + id: &str, + progress: impl Fn(Phase) + Send + 'static, +) -> AtomicResult { + if id.is_empty() + || !id + .bytes() + .all(|b| b.is_ascii_alphanumeric() || b"-_.".contains(&b)) + { + return Err("Invalid backup id".into()); + } + options.prepare()?; + let pause = tokio::time::timeout( + std::time::Duration::from_secs(30), + store.maintenance.pause(), + ) + .await + .map_err(|_| "Timed out draining operations; node resumed")? + .map_err(|e| e.to_string())?; + let store = store.clone(); + let options = options.clone(); + let id = id.to_owned(); + tokio::task::spawn_blocking(move || { + progress(Phase::Capturing); + let staging = tempfile::Builder::new() + .prefix(".atomic-backup-") + .tempdir_in(&options.output_dir)?; + let captured = capture(&store, &options, staging.path()); + drop(pause); + captured?; + progress(Phase::Archiving); + archive(staging.path(), &options.output_dir, &id) + }) + .await + .map_err(|e| format!("Backup worker failed: {e}"))? +} + +/// Verify the manifest, archive paths, sizes and checksums without restoring. +pub fn verify(archive: &Path) -> AtomicResult<()> { + verify_archive(archive).map(|_| ()) +} + +#[derive(Debug, Serialize, Deserialize)] +struct Manifest { + format: u32, + // Preserve the v1 wire name; compatibility belongs to the core package. + #[serde(rename = "server_version")] + core_version: String, + build_revision: String, + redb_version: String, + snapshot_time: String, + freshness: String, + envelope_retention: String, + source_data: PathBuf, + source_config: PathBuf, + files: BTreeMap, +} + +#[derive(Debug, Serialize, Deserialize)] +struct Entry { + bytes: u64, + blake3: String, +} + +// Canonicalize an existing ancestor, then append not-yet-created components. +fn resolve_future_path(path: &Path) -> AtomicResult { + if path.exists() { + return Ok(path.canonicalize()?); + } + let parent = path + .parent() + .filter(|p| !p.as_os_str().is_empty()) + .unwrap_or(Path::new(".")); + Ok(resolve_future_path(parent)?.join(path.file_name().ok_or("Invalid storage path")?)) +} + +fn capture(store: &Db, options: &CheckpointOptions, stage: &Path) -> AtomicResult<()> { + let source_data = options.data_dir.canonicalize()?; + let source_config = options.config_dir.canonicalize()?; + let db_path = stage.join("data/store/atomic.redb"); + fs::create_dir_all(db_path.parent().unwrap())?; + let mut snapshot_time = String::new(); + store.kv.backup_snapshot(&db_path, &mut || { + snapshot_time = chrono::Utc::now().to_rfc3339(); + copy_tree( + &source_data, + &stage.join("data"), + Some(Path::new("store/atomic.redb")), + )?; + copy_tree(&source_config, &stage.join("config"), None)?; + Ok(()) + })?; + let mut files = BTreeMap::new(); + inventory(stage, stage, &mut files)?; + let manifest = Manifest { + format: FORMAT, + core_version: env!("CARGO_PKG_VERSION").into(), + build_revision: options.build_revision.clone(), + redb_version: "4.1.0".into(), + snapshot_time, + freshness: "Local persisted checkpoint; remote replication completeness is not asserted" + .into(), + envelope_retention: store.envelope_retention().as_str().into(), + source_data, + source_config, + files, + }; + fs::write(stage.join(MANIFEST), serde_json::to_vec_pretty(&manifest)?)?; + Ok(()) +} + +fn copy_tree(source: &Path, target: &Path, exclude: Option<&Path>) -> AtomicResult<()> { + fs::create_dir_all(target)?; + for item in fs::read_dir(source)? { + let item = item?; + let rel = PathBuf::from(item.file_name()); + if exclude == Some(rel.as_path()) { + continue; + } + let ty = item.file_type()?; + let dest = target.join(&rel); + if ty.is_symlink() { + return Err(format!("Refusing symlink in backup: {}", item.path().display()).into()); + } + if ty.is_dir() { + let nested = exclude.and_then(|p| p.strip_prefix(&rel).ok()); + copy_tree(&item.path(), &dest, nested)?; + } else if ty.is_file() { + let before = item.metadata()?; + fs::copy(item.path(), &dest)?; + let after = item.metadata()?; + if before.len() != after.len() || before.modified()? != after.modified()? { + return Err( + format!("File changed during backup: {}", item.path().display()).into(), + ); + } + private_file(&dest)?; + } else { + return Err(format!("Unsupported backup file: {}", item.path().display()).into()); + } + } + Ok(()) +} + +fn digest(path: &Path) -> AtomicResult { + let mut file = File::open(path)?; + let mut hasher = blake3::Hasher::new(); + let mut buffer = [0u8; 65536]; + let mut bytes = 0; + loop { + let n = file.read(&mut buffer)?; + if n == 0 { + break; + } + hasher.update(&buffer[..n]); + bytes += n as u64; + } + Ok(Entry { + bytes, + blake3: hasher.finalize().to_hex().to_string(), + }) +} + +fn inventory(root: &Path, dir: &Path, files: &mut BTreeMap) -> AtomicResult<()> { + for item in fs::read_dir(dir)? { + let item = item?; + if item.file_type()?.is_dir() { + inventory(root, &item.path(), files)?; + } else { + let path = item.path(); + let name = path + .strip_prefix(root) + .map_err(|e| e.to_string())? + .to_str() + .ok_or("Non-UTF8 backup path")? + .replace('\\', "/"); + files.insert(name, digest(&path)?); + } + } + Ok(()) +} + +fn archive(stage: &Path, output: &Path, id: &str) -> AtomicResult { + use zip::write::SimpleFileOptions; + let manifest: Manifest = serde_json::from_reader(File::open(stage.join(MANIFEST))?)?; + let mut temp = tempfile::NamedTempFile::new_in(output)?; + { + let mut writer = zip::ZipWriter::new(temp.as_file_mut()); + for name in manifest + .files + .keys() + .map(String::as_str) + .chain(std::iter::once(MANIFEST)) + { + let path = stage.join(name); + let options = SimpleFileOptions::default() + .compression_method(zip::CompressionMethod::Deflated) + .unix_permissions(0o600) + .large_file(fs::metadata(&path)?.len() >= u32::MAX as u64); + writer + .start_file(name, options) + .map_err(|e| e.to_string())?; + std::io::copy(&mut File::open(path)?, &mut writer)?; + } + writer.finish().map_err(|e| e.to_string())?; + } + temp.as_file().sync_all()?; + // Read back the ZIP and every hash before publication, without extracting. + verify_archive(temp.path())?; + let path = output.join(format!("atomic-backup-{id}.zip")); + temp.persist_noclobber(&path).map_err(|e| e.to_string())?; + File::open(output)?.sync_all()?; + Ok(path) +} + +fn safe_name(name: &str) -> bool { + !name.contains('\\') + && !name.contains(':') + && !name.contains('\0') + && Path::new(name) + .components() + .all(|p| matches!(p, Component::Normal(_))) + && (name.starts_with("data/") || name.starts_with("config/")) +} + +fn verify_archive(path: &Path) -> AtomicResult { + let mut zip = zip::ZipArchive::new(File::open(path)?).map_err(|e| e.to_string())?; + let manifest: Manifest = { + let mut file = zip.by_name(MANIFEST).map_err(|e| e.to_string())?; + if file.size() > 16 * 1024 * 1024 { + return Err("Oversized backup manifest".into()); + } + let mut json = Vec::new(); + file.read_to_end(&mut json)?; + serde_json::from_slice(&json)? + }; + if manifest.format != FORMAT { + return Err("Unsupported backup format".into()); + } + if manifest.core_version != env!("CARGO_PKG_VERSION") { + return Err(format!("Restore requires atomic_lib {}", manifest.core_version).into()); + } + if !manifest.files.contains_key("data/store/atomic.redb") + || !manifest.files.contains_key("config/config.toml") + { + return Err("Backup lacks database or identity configuration".into()); + } + if zip.len() != manifest.files.len() + 1 { + return Err("Unexpected or duplicate archive entries".into()); + } + let mut seen = std::collections::HashSet::new(); + for i in 0..zip.len() { + let mut file = zip.by_index(i).map_err(|e| e.to_string())?; + let name = file.name().to_owned(); + if !seen.insert(name.clone()) { + return Err("Duplicate archive path".into()); + } + if name == MANIFEST { + continue; + } + if !safe_name(&name) + || file.is_dir() + || file.unix_mode().is_some_and(|m| m & 0o170000 == 0o120000) + { + return Err(format!("Unsafe archive path: {name}").into()); + } + let expected = manifest.files.get(&name).ok_or("Unlisted archive entry")?; + if file.size() != expected.bytes { + return Err("Archive size mismatch".into()); + } + let mut hasher = blake3::Hasher::new(); + let mut bytes = 0u64; + let mut buffer = [0u8; 65536]; + loop { + let n = file.read(&mut buffer)?; + if n == 0 { + break; + } + bytes += n as u64; + if bytes > expected.bytes { + return Err("Archive expanded beyond manifest size".into()); + } + hasher.update(&buffer[..n]); + } + if bytes != expected.bytes || hasher.finalize().to_hex().as_str() != expected.blake3 { + return Err(format!("Checksum mismatch for {name}").into()); + } + } + Ok(manifest) +} + +/// Restore never starts a runtime, sync transport, plugin or HTTP client. +/// The marker prevents an accidental normal node boot with copied identities. +pub fn restore(archive: &Path, target: &Path) -> AtomicResult<()> { + let manifest = verify_archive(archive)?; + if target.exists() { + return Err("Restore target must not exist".into()); + } + let parent = target + .parent() + .filter(|p| !p.as_os_str().is_empty()) + .unwrap_or(Path::new(".")); + let stage = tempfile::Builder::new() + .prefix(".atomic-restore-") + .tempdir_in(parent)?; + let mut zip = zip::ZipArchive::new(File::open(archive)?).map_err(|e| e.to_string())?; + for name in manifest.files.keys() { + let dest = stage.path().join(name); + fs::create_dir_all(dest.parent().unwrap())?; + let mut file = create_private(&dest)?; + std::io::copy( + &mut zip.by_name(name).map_err(|e| e.to_string())?, + &mut file, + )?; + file.sync_all()?; + let actual = digest(&dest)?; + let expected = &manifest.files[name]; + if actual.bytes != expected.bytes || actual.blake3 != expected.blake3 { + return Err("Archive changed during restore".into()); + } + } + crate::db::redb_store::verify_snapshot(&stage.path().join("data/store/atomic.redb"))?; + fs::write( + stage.path().join("data").join(RESTORED), + b"Restored offline. Explicit activation is required before starting a node.\n", + )?; + fs::write( + stage.path().join(MANIFEST), + serde_json::to_vec_pretty(&manifest)?, + )?; + // Claim the target without clobbering even an empty directory created by + // another operation since preflight. Incomplete promotion retains the marker. + fs::create_dir(target)?; + private_dir(target)?; + for item in fs::read_dir(stage.path())? { + let item = item?; + fs::rename(item.path(), target.join(item.file_name()))?; + } + File::open(parent)?.sync_all()?; + Ok(()) +} + +pub fn check_restore_activation(data_dir: &Path, activate: bool) -> AtomicResult<()> { + if data_dir.join(RESTORED).exists() && !activate { + return Err("Restored instance is offline. Explicitly activate only when ready to reconnect its copied identities and integrations".into()); + } + Ok(()) +} + +fn create_private(path: &Path) -> std::io::Result { + let mut options = OpenOptions::new(); + options.write(true).create_new(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt; + options.mode(0o600); + } + options.open(path) +} +fn private_file(path: &Path) -> std::io::Result<()> { + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + fs::set_permissions(path, fs::Permissions::from_mode(0o600))?; + } + Ok(()) +} +fn private_dir(path: &Path) -> std::io::Result<()> { + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + fs::set_permissions(path, fs::Permissions::from_mode(0o700))?; + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{ + db::{ + kv_store::KvStore, + redb_store::RedbStore, + trees::{Method, Operation, Tree}, + }, + loro::AtomicLoroDoc, + Value, + }; + use std::sync::Arc; + + async fn fixture(root: &Path) -> (Db, CheckpointOptions) { + let options = CheckpointOptions { + data_dir: root.join("source"), + config_dir: root.join("configuration"), + output_dir: root.join("backups"), + build_revision: "test".into(), + }; + fs::create_dir_all(&options.config_dir).unwrap(); + fs::write(options.config_dir.join("config.toml"), b"identity fixture").unwrap(); + let store = Db::init_redb_file( + &options.data_dir.join("store"), + None, + &options.data_dir.join("uploads"), + ) + .await + .unwrap(); + (store, options) + } + + #[tokio::test] + async fn instance_roundtrip_preserves_history_blobs_metadata_and_files() { + let root = tempfile::tempdir().unwrap(); + let (store, config) = fixture(root.path()).await; + let doc = AtomicLoroDoc::new(); + doc.set_property("name", &Value::String("before".into())) + .unwrap(); + let first = doc.export_snapshot(); + let first_version = doc.current_version(); + doc.set_property("name", &Value::String("after".into())) + .unwrap(); + let snapshot = doc.export_snapshot(); + assert_ne!(first, snapshot); + store + .kv + .insert(Tree::LoroSnapshots, b"did:ad:test", &snapshot) + .unwrap(); + for tree in [ + Tree::Blobs, + Tree::Envelopes, + Tree::PluginMeta, + Tree::DriveMapping, + ] { + store.kv.insert(tree, b"fixture", b"original").unwrap(); + } + fs::create_dir_all(&config.data_dir.join("uploads")).unwrap(); + fs::write( + config.data_dir.join("uploads").join("legacy.txt"), + b"legacy file", + ) + .unwrap(); + let path = create(&store, &config, "roundtrip", |_| {}).await.unwrap(); + assert!(!store.maintenance.is_paused()); + store + .kv + .insert(Tree::PluginMeta, b"fixture", b"changed after backup") + .unwrap(); + let target = root.path().join("restored"); + restore(&path, &target).unwrap(); + assert!(restore(&path, &target).is_err()); + let restored = RedbStore::new_file(&target.join("data/store/atomic.redb")).unwrap(); + assert_eq!( + restored.get(Tree::PluginMeta, b"fixture").unwrap().unwrap(), + b"original" + ); + assert_eq!( + restored.get(Tree::Blobs, b"fixture").unwrap().unwrap(), + b"original" + ); + assert_eq!( + restored.get(Tree::Envelopes, b"fixture").unwrap().unwrap(), + b"original" + ); + let recovered = AtomicLoroDoc::from_snapshot( + &restored + .get(Tree::LoroSnapshots, b"did:ad:test") + .unwrap() + .unwrap(), + ) + .unwrap(); + assert_eq!(recovered.get_history().len(), doc.get_history().len()); + recovered.fork_at(&first_version).unwrap(); + assert_eq!( + fs::read(target.join("data/uploads/legacy.txt")).unwrap(), + b"legacy file" + ); + assert_eq!( + fs::read(target.join("config/config.toml")).unwrap(), + fs::read(&config.config_dir.join("config.toml")).unwrap() + ); + assert!(check_restore_activation(&target.join("data"), false).is_err()); + check_restore_activation(&target.join("data"), true).unwrap(); + } + + #[test] + fn snapshot_blocks_writers_and_copies_unknown_byte_tables() { + let root = tempfile::tempdir().unwrap(); + let source = root.path().join("source.redb"); + // Future tables must not silently disappear from the backup. + { + let db = redb_for_test(&source); + let tx = db.begin_write().unwrap(); + { + let mut table = tx + .open_table(redb::TableDefinition::<&[u8], &[u8]>::new("future_table")) + .unwrap(); + table + .insert(b"future".as_slice(), b"value".as_slice()) + .unwrap(); + } + tx.commit().unwrap(); + } + let store = Arc::new(RedbStore::new_file(&source).unwrap()); + store + .apply_batch(&[ + Operation { + tree: Tree::Resources, + method: Method::Insert, + key: b"key".to_vec(), + val: Some(b"old".to_vec()), + }, + Operation { + tree: Tree::LoroSnapshots, + method: Method::Insert, + key: b"key".to_vec(), + val: Some(b"old".to_vec()), + }, + ]) + .unwrap(); + let (entered_tx, entered_rx) = std::sync::mpsc::channel(); + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let s = store.clone(); + let destination = root.path().join("snapshot.redb"); + let path = destination.clone(); + let backup = std::thread::spawn(move || { + s.backup_snapshot(&path, &mut || { + entered_tx.send(()).unwrap(); + release_rx.recv().unwrap(); + Ok(()) + }) + }); + entered_rx + .recv_timeout(std::time::Duration::from_secs(10)) + .unwrap(); + let s = store.clone(); + let (written_tx, written_rx) = std::sync::mpsc::channel(); + let writer = std::thread::spawn(move || { + s.insert(Tree::Resources, b"key", b"new").unwrap(); + written_tx.send(()).unwrap(); + }); + assert!(written_rx + .recv_timeout(std::time::Duration::from_millis(50)) + .is_err()); + release_tx.send(()).unwrap(); + backup.join().unwrap().unwrap(); + writer.join().unwrap(); + let snapshot = RedbStore::new_file(&destination).unwrap(); + assert_eq!( + snapshot.get(Tree::Resources, b"key").unwrap().unwrap(), + b"old" + ); + assert_eq!( + snapshot.get(Tree::LoroSnapshots, b"key").unwrap().unwrap(), + b"old" + ); + drop(snapshot); + use redb::ReadableDatabase; + let db = redb_for_test(&destination); + let tx = db.begin_read().unwrap(); + let table = tx + .open_table(redb::TableDefinition::<&[u8], &[u8]>::new("future_table")) + .unwrap(); + assert_eq!( + table.get(b"future".as_slice()).unwrap().unwrap().value(), + b"value" + ); + } + + fn redb_for_test(path: &Path) -> redb::Database { + redb::Database::create(path).unwrap() + } + + #[tokio::test] + async fn capture_failure_resumes_and_publishes_nothing() { + let root = tempfile::tempdir().unwrap(); + let (store, config) = fixture(root.path()).await; + #[cfg(unix)] + std::os::unix::fs::symlink("/outside", config.config_dir.join("unsupported")).unwrap(); + #[cfg(not(unix))] + { + store.kv.begin_batch(); + } + assert!(create(&store, &config, "failure", |_| {}).await.is_err()); + assert!(!store.maintenance.is_paused()); + assert!(!root + .path() + .join("backups/atomic-backup-failure.zip") + .exists()); + store + .kv + .insert(Tree::PluginMeta, b"after-failure", b"works") + .unwrap(); + } + + #[test] + fn snapshot_callback_failure_releases_storage_barrier() { + let root = tempfile::tempdir().unwrap(); + let store = RedbStore::new_file(&root.path().join("source.redb")).unwrap(); + assert!(store + .backup_snapshot(&root.path().join("failed.redb"), &mut || Err( + "injected I/O failure".into() + )) + .is_err()); + store.insert(Tree::PluginMeta, b"after", b"works").unwrap(); + assert!(store + .backup_snapshot(&root.path().join("failed.redb"), &mut || Ok(())) + .is_err()); + store.begin_batch(); + assert!(store + .backup_snapshot(&root.path().join("batch.redb"), &mut || Ok(())) + .is_err()); + store.commit_batch().unwrap(); + let panicked = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let _ = store.backup_snapshot(&root.path().join("panic.redb"), &mut || { + panic!("injected capture panic") + }); + })); + assert!(panicked.is_err()); + store + .insert(Tree::PluginMeta, b"after-panic", b"works") + .unwrap(); + } + + #[test] + fn restore_rejects_unsafe_paths() { + for name in [ + "../escape", + "data/../../escape", + "/data/file", + "config/../escape", + "data\\escape", + "data/C:escape", + ] { + assert!(!safe_name(name), "{name}"); + } + assert!(safe_name("data/store/atomic.redb")); + } +} + +#[cfg(test)] +mod sync_tests { + use super::*; + use crate::{ + agents::ForAgent, + db::trees::Tree, + loro::AtomicLoroDoc, + sync::{engine, protocol}, + Storelike, Value, + }; + + #[tokio::test] + async fn incoming_sync_waits_without_acknowledging_or_losing_the_update() { + let root = tempfile::tempdir().unwrap(); + let db = Db::init_redb_file( + &root.path().join("store"), + Some("http://localhost:9883".into()), + &root.path().join("uploads"), + ) + .await + .unwrap(); + let (agent, drive) = db.setup("Backup sync test").await.unwrap(); + let subject = "did:ad:backup-sync-test"; + let doc = AtomicLoroDoc::new(); + doc.set_property( + crate::urls::DRIVE_PROP, + &Value::AtomicUrl(drive.clone().into()), + ) + .unwrap(); + doc.set_property( + crate::urls::NAME, + &Value::String("arrived during backup".into()), + ) + .unwrap(); + let frame = protocol::encode_sync_push(&drive, &[(subject, &doc.export_snapshot())], true); + let push = protocol::decode_sync_push(&frame[1..]).unwrap(); + let pause = db.maintenance.pause().await.unwrap(); + let receiver = db.clone(); + let (started_tx, started_rx) = tokio::sync::oneshot::channel(); + let job = tokio::spawn(async move { + started_tx.send(()).unwrap(); + engine::import_sync_push(&push, &receiver, &ForAgent::from(agent), false).await + }); + started_rx.await.unwrap(); + tokio::task::yield_now().await; + assert!(!job.is_finished(), "must not acknowledge a paused import"); + assert!(db + .kv + .get(Tree::LoroSnapshots, subject.as_bytes()) + .unwrap() + .is_none()); + drop(pause); + let (count, _) = tokio::time::timeout(std::time::Duration::from_secs(10), job) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(count, 1); + assert_eq!( + db.get_resource(&subject.into()) + .await + .unwrap() + .get(crate::urls::NAME) + .unwrap() + .to_string(), + "arrived during backup" + ); + } +} + +#[cfg(test)] +mod archive_tests { + use super::*; + use std::io::Write; + + fn untrusted(root: &Path, name: &str, content: &[u8], expected_hash: &str) -> PathBuf { + let path = root.join("untrusted.zip"); + let mut files = BTreeMap::new(); + for key in ["data/store/atomic.redb", "config/config.toml", name] { + files.insert( + key.into(), + Entry { + bytes: content.len() as u64, + blake3: expected_hash.into(), + }, + ); + } + let manifest = Manifest { + format: FORMAT, + core_version: env!("CARGO_PKG_VERSION").into(), + build_revision: "test".into(), + redb_version: "4.1.0".into(), + snapshot_time: "test".into(), + freshness: "unknown".into(), + envelope_retention: "latest".into(), + source_data: "source".into(), + source_config: "config".into(), + files, + }; + let mut writer = zip::ZipWriter::new(File::create(&path).unwrap()); + let options = zip::write::SimpleFileOptions::default(); + writer.start_file(MANIFEST, options).unwrap(); + writer + .write_all(&serde_json::to_vec(&manifest).unwrap()) + .unwrap(); + for key in manifest.files.keys() { + writer.start_file(key, options).unwrap(); + writer.write_all(content).unwrap(); + } + writer.finish().unwrap(); + path + } + + #[test] + fn corrupt_or_traversing_archive_never_creates_restore_target() { + let root = tempfile::tempdir().unwrap(); + let target = root.path().join("restore"); + let archive = untrusted(root.path(), "data/note", b"tampered", "wrong checksum"); + assert!(restore(&archive, &target) + .unwrap_err() + .to_string() + .contains("Checksum")); + assert!(!target.exists()); + let hash = blake3::hash(b"valid").to_hex().to_string(); + let archive = untrusted(root.path(), "data/../../escape", b"valid", &hash); + assert!(restore(&archive, &target) + .unwrap_err() + .to_string() + .contains("Unsafe")); + assert!(!target.exists()); + assert!(!root.path().join("escape").exists()); + } + + #[test] + fn valid_hashes_do_not_make_a_corrupt_database_restorable() { + let root = tempfile::tempdir().unwrap(); + let hash = blake3::hash(b"not a redb file").to_hex().to_string(); + let archive = untrusted(root.path(), "data/note", b"not a redb file", &hash); + let target = root.path().join("restore"); + assert!(restore(&archive, &target).is_err()); + assert!(!target.exists()); + } +} diff --git a/lib/src/lib.rs b/lib/src/lib.rs index e1d0398af9..40422a58ff 100644 --- a/lib/src/lib.rs +++ b/lib/src/lib.rs @@ -66,6 +66,8 @@ pub mod agents; pub mod aggregate; pub mod atoms; pub mod authentication; +#[cfg(all(feature = "backup", not(target_arch = "wasm32")))] +pub mod backup; #[cfg(feature = "db")] pub mod class_extender; pub mod client; diff --git a/lib/tests/check-instance-checkpoint.sh b/lib/tests/check-instance-checkpoint.sh new file mode 100755 index 0000000000..c2e63b7323 --- /dev/null +++ b/lib/tests/check-instance-checkpoint.sh @@ -0,0 +1,13 @@ +#!/bin/sh +# Run from the workspace root; select only atomic_lib to avoid feature unification. +set -eu +checkpoint_deps=$(mktemp) +trap 'rm -f "$checkpoint_deps"' EXIT HUP INT TERM +cargo tree --locked -p atomic_lib --no-default-features --features backup,config \ + --edges normal --prefix none --format '{p}' > "$checkpoint_deps" +if grep -E '^(atomic-server|actix(-[^ ]+)?) ' "$checkpoint_deps"; then + echo 'Native checkpoints must not depend on atomic-server or Actix.' >&2 + exit 1 +fi +cargo test --locked -p atomic_lib --no-default-features --features backup,config --test instance_checkpoint +cargo test --locked -p atomic_lib --no-default-features --features backup,config --lib backup:: diff --git a/lib/tests/instance_checkpoint.rs b/lib/tests/instance_checkpoint.rs new file mode 100644 index 0000000000..4d203ae34e --- /dev/null +++ b/lib/tests/instance_checkpoint.rs @@ -0,0 +1,110 @@ +//! This test must also pass without workspace/server feature unification. +#![cfg(all(feature = "backup", feature = "config", not(target_arch = "wasm32")))] + +use atomic_lib::{ + backup::{self, CheckpointOptions}, + db::trees::Tree, + runtime::AtomicNode, + Db, +}; + +#[tokio::test] +async fn checkpoint_restores_an_originless_node_without_a_server() { + let root = tempfile::tempdir().unwrap(); + let data = root.path().join("data"); + let config = root.path().join("config"); + std::fs::create_dir_all(&config).unwrap(); + let agent = atomic_lib::agents::Agent::new(None).unwrap(); + let identity = atomic_lib::config::Config { + shared: atomic_lib::config::SharedConfig { + agent_secret: agent.build_secret().unwrap(), + initial_drive: None, + }, + client: None, + }; + identity.save(&config.join("config.toml")).unwrap(); + let db = Db::init_redb_file(&data.join("store"), None, &data.join("uploads")) + .await + .unwrap(); + db.kv + .insert(Tree::PluginMeta, b"checkpoint-test", b"before") + .unwrap(); + let options = CheckpointOptions { + data_dir: data, + config_dir: config.clone(), + output_dir: root.path().join("backups"), + build_revision: "standalone-test".into(), + }; + let phases = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + let observed = phases.clone(); + let observed_store = db.clone(); + let archive = backup::create(&db, &options, "native", move |phase| { + // Compression must not extend the write pause. + assert_eq!( + observed_store.maintenance.is_paused(), + phase == backup::Phase::Capturing + ); + observed.lock().unwrap().push(phase); + }) + .await + .unwrap(); + assert_eq!( + *phases.lock().unwrap(), + [backup::Phase::Capturing, backup::Phase::Archiving] + ); + db.kv + .insert(Tree::PluginMeta, b"checkpoint-test", b"after") + .unwrap(); + let target = root.path().join("restored"); + backup::verify(&archive).unwrap(); + backup::restore(&archive, &target).unwrap(); + assert!(backup::check_restore_activation(&target.join("data"), false).is_err()); + backup::check_restore_activation(&target.join("data"), true).unwrap(); + let restored = Db::init_redb_file( + &target.join("data/store"), + None, + &target.join("data/uploads"), + ) + .await + .unwrap(); + assert_eq!( + restored + .kv + .get(Tree::PluginMeta, b"checkpoint-test") + .unwrap() + .unwrap(), + b"before" + ); + let restored_identity = + atomic_lib::config::read_config(Some(&target.join("config/config.toml"))).unwrap(); + let node = AtomicNode::from_db(restored); + node.set_agent( + atomic_lib::agents::Agent::from_secret(&restored_identity.shared.agent_secret).unwrap(), + ); + assert_eq!(node.agent().unwrap().subject, agent.subject); + assert_eq!( + std::fs::read(target.join("config/config.toml")).unwrap(), + std::fs::read(config.join("config.toml")).unwrap() + ); +} + +#[test] +fn native_options_reject_overlapping_roots_before_creating_output() { + let root = tempfile::tempdir().unwrap(); + let data = root.path().join("data"); + let config = root.path().join("config"); + std::fs::create_dir_all(&data).unwrap(); + std::fs::create_dir_all(&config).unwrap(); + let mut options = CheckpointOptions { + data_dir: data.clone(), + config_dir: config, + output_dir: data.join("backups"), + build_revision: "test".into(), + }; + assert!(options.prepare().is_err()); + assert!(!options.output_dir.exists()); + options.output_dir = root.path().join("backups"); + options.config_dir = data; + assert!(options.prepare().is_err()); + assert!(!options.output_dir.exists()); +} diff --git a/planning/atomic-lib-runtime.md b/planning/atomic-lib-runtime.md index 30a76537e7..fa6d9ba05a 100644 --- a/planning/atomic-lib-runtime.md +++ b/planning/atomic-lib-runtime.md @@ -30,6 +30,19 @@ adapter needs it: `Db::init_*`, `get_resource_extended`, `save_locally` / `Replica` (the hub-relayed text `COMMIT` profile) went with that frame the same day. +Native persisted-instance checkpoint operations now live in `atomic_lib::backup` +behind the optional `backup` feature. The server control adapter calls these with +explicit paths; it owns no archive/restore implementation. The standalone +`lib/tests/check-instance-checkpoint.sh` gate rejects server/Actix dependencies. +This is independent of the native startup extraction in #1416 and can be consumed +by its runtime without importing server configuration. + +Remaining checkpoint adapter work: + +- [ ] Desktop: stop and flush VFS staging before capture; gate external file writers. +- [ ] Native startup: enforce the offline restore marker before reconnecting identities. +- [ ] Verify desktop capture/restore with no HTTP listener. + Next: 1. `server/src/handlers/commit.rs` → `node.apply_commit(_, Hub { source_id, diff --git a/server/Cargo.toml b/server/Cargo.toml index 05977965ff..6f051c6aad 100644 --- a/server/Cargo.toml +++ b/server/Cargo.toml @@ -170,7 +170,7 @@ version = "4.13.0" # `db-sled` is included so an in-place upgrade from a pre-redb (sled) store # auto-migrates to redb on first boot. It is inert unless a legacy sled DB is # present, so it adds no runtime cost for fresh installs. -features = ["config", "db-redb", "db-sled", "rdf", "discovery", "iroh", "ring", "ws"] +features = ["backup", "config", "db-redb", "db-sled", "rdf", "discovery", "iroh", "ring", "ws"] path = "../lib" version = "0.41.0-beta.7" @@ -206,7 +206,6 @@ version = "0.3.23" [dev-dependencies] brotli = "8" tempfile = "3.27" -redb = "4.1.0" actix-rt = "2.11.0" assert_cmd = "2.2.2" mainline = "6.2.0" diff --git a/server/src/backup.rs b/server/src/backup.rs index 86a6e12258..cc0e796c77 100644 --- a/server/src/backup.rs +++ b/server/src/backup.rs @@ -3,19 +3,20 @@ //! of an HTTP request cannot resume writers while files are still being copied. use crate::{appstate::AppState, config::Config, errors::AtomicServerResult}; use actix_web::{web, HttpRequest, HttpResponse}; -use atomic_lib::{errors::AtomicResult, Db}; +pub use atomic_lib::backup::restore; +use atomic_lib::{ + backup::{CheckpointOptions, Phase}, + errors::AtomicResult, + Db, +}; use serde::{Deserialize, Serialize}; use std::{ - collections::BTreeMap, fs::{self, File, OpenOptions}, - io::{Read, Write}, - path::{Component, Path, PathBuf}, + io::Write, + path::{Path, PathBuf}, sync::{Arc, Mutex}, }; -const FORMAT: u32 = 1; -const MANIFEST: &str = "manifest.json"; -const RESTORED: &str = "RESTORED_OFFLINE"; const CONTROL: &str = "/__atomic/backup"; #[derive(Clone, Debug, Default, Serialize, Deserialize)] @@ -36,32 +37,8 @@ impl BackupService { pub fn new(config: &Config) -> AtomicResult> { let mut token = String::new(); let output = if let Some(path) = &config.opts.backup_dir { - if !path.exists() { - let mut builder = fs::DirBuilder::new(); - builder.recursive(true); - #[cfg(unix)] - { - use std::os::unix::fs::DirBuilderExt; - builder.mode(0o700); - } - builder.create(path)?; - } - // Existing destinations may be shared directories. Never chmod - // them; staging directories and final ZIP files are owner-only. + checkpoint_options(config, path.clone())?.prepare()?; let output = path.canonicalize()?; - let data = data_dir(config)?.canonicalize()?; - let configuration = config.config_dir.canonicalize()?; - if data.starts_with(&configuration) || configuration.starts_with(&data) { - return Err("Instance backup requires separate, non-overlapping data and config directories".into()); - } - for source in [data_dir(config)?, config.config_dir.clone()] { - let source = source.canonicalize()?; - if output.starts_with(&source) || source.starts_with(&output) { - return Err( - "Backup directory must be outside data and config directories".into(), - ); - } - } // Vector-index flush runs independently and is deliberately not // backed up. Require cache storage outside the capture roots. for cache in [&config.vector_search_index_path, &config.plugin_cache_path] { @@ -192,58 +169,34 @@ async fn start(request: HttpRequest, app: web::Data) -> HttpResponse { }) } +fn checkpoint_options(config: &Config, output_dir: PathBuf) -> AtomicResult { + Ok(CheckpointOptions { + data_dir: data_dir(config)?, + config_dir: config.config_dir.clone(), + output_dir, + build_revision: env!("ATOMIC_BACKUP_REVISION").into(), + }) +} + async fn run_backup( store: &Db, config: &Config, service: &Arc, id: &str, ) -> AtomicResult { - let pause = tokio::time::timeout( - std::time::Duration::from_secs(30), - store.maintenance.pause(), - ) - .await - .map_err(|_| "Timed out draining operations; server resumed")? - .map_err(|e| e.to_string())?; - let store = store.clone(); - let config = config.clone(); + let options = checkpoint_options(config, service.output.clone().ok_or("Backups disabled")?)?; let service = service.clone(); - let id = id.to_owned(); - tokio::task::spawn_blocking(move || { - service.phase("capturing"); - let output = service.output.as_ref().ok_or("Backups disabled")?; - let staging = tempfile::Builder::new() - .prefix(".atomic-backup-") - .tempdir_in(output)?; - let captured = capture(&store, &config, staging.path()); - // Also runs on capture errors. During a panic RAII unwinds this guard. - drop(pause); - captured?; - service.phase("archiving"); - archive(staging.path(), output, &id) + atomic_lib::backup::create(store, &options, id, move |phase| { + service.phase(match phase { + Phase::Capturing => "capturing", + Phase::Archiving => "archiving", + }); }) .await - .map_err(|e| format!("Backup worker failed: {e}"))? } -#[derive(Debug, Serialize, Deserialize)] -struct Manifest { - format: u32, - server_version: String, - build_revision: String, - redb_version: String, - snapshot_time: String, - freshness: String, - envelope_retention: String, - source_data: PathBuf, - source_config: PathBuf, - files: BTreeMap, -} - -#[derive(Debug, Serialize, Deserialize)] -struct Entry { - bytes: u64, - blake3: String, +pub fn check_restore_activation(config: &Config) -> AtomicResult<()> { + atomic_lib::backup::check_restore_activation(&data_dir(config)?, config.opts.activate_restored) } // Canonicalize an existing ancestor, then append not-yet-created components. @@ -266,279 +219,6 @@ fn data_dir(config: &Config) -> AtomicResult { .to_path_buf()) } -fn capture(store: &Db, config: &Config, stage: &Path) -> AtomicResult<()> { - let source_data = data_dir(config)?.canonicalize()?; - let source_config = config.config_dir.canonicalize()?; - let db_path = stage.join("data/store/atomic.redb"); - fs::create_dir_all(db_path.parent().unwrap())?; - let mut snapshot_time = String::new(); - store.kv.backup_snapshot(&db_path, &mut || { - snapshot_time = chrono::Utc::now().to_rfc3339(); - copy_tree( - &source_data, - &stage.join("data"), - Some(Path::new("store/atomic.redb")), - )?; - copy_tree(&source_config, &stage.join("config"), None)?; - Ok(()) - })?; - let mut files = BTreeMap::new(); - inventory(stage, stage, &mut files)?; - let manifest = Manifest { - format: FORMAT, - server_version: env!("CARGO_PKG_VERSION").into(), - build_revision: env!("ATOMIC_BACKUP_REVISION").into(), - redb_version: "4.1.0".into(), - snapshot_time, - freshness: "Local persisted checkpoint; remote replication completeness is not asserted" - .into(), - envelope_retention: store.envelope_retention().as_str().into(), - source_data, - source_config, - files, - }; - fs::write(stage.join(MANIFEST), serde_json::to_vec_pretty(&manifest)?)?; - Ok(()) -} - -fn copy_tree(source: &Path, target: &Path, exclude: Option<&Path>) -> AtomicResult<()> { - fs::create_dir_all(target)?; - for item in fs::read_dir(source)? { - let item = item?; - let rel = PathBuf::from(item.file_name()); - if exclude == Some(rel.as_path()) { - continue; - } - let ty = item.file_type()?; - let dest = target.join(&rel); - if ty.is_symlink() { - return Err(format!("Refusing symlink in backup: {}", item.path().display()).into()); - } - if ty.is_dir() { - let nested = exclude.and_then(|p| p.strip_prefix(&rel).ok()); - copy_tree(&item.path(), &dest, nested)?; - } else if ty.is_file() { - let before = item.metadata()?; - fs::copy(item.path(), &dest)?; - let after = item.metadata()?; - if before.len() != after.len() || before.modified()? != after.modified()? { - return Err( - format!("File changed during backup: {}", item.path().display()).into(), - ); - } - private_file(&dest)?; - } else { - return Err(format!("Unsupported backup file: {}", item.path().display()).into()); - } - } - Ok(()) -} - -fn digest(path: &Path) -> AtomicResult { - let mut file = File::open(path)?; - let mut hasher = blake3::Hasher::new(); - let mut buffer = [0u8; 65536]; - let mut bytes = 0; - loop { - let n = file.read(&mut buffer)?; - if n == 0 { - break; - } - hasher.update(&buffer[..n]); - bytes += n as u64; - } - Ok(Entry { - bytes, - blake3: hasher.finalize().to_hex().to_string(), - }) -} - -fn inventory(root: &Path, dir: &Path, files: &mut BTreeMap) -> AtomicResult<()> { - for item in fs::read_dir(dir)? { - let item = item?; - if item.file_type()?.is_dir() { - inventory(root, &item.path(), files)?; - } else { - let path = item.path(); - let name = path - .strip_prefix(root) - .map_err(|e| e.to_string())? - .to_str() - .ok_or("Non-UTF8 backup path")? - .replace('\\', "/"); - files.insert(name, digest(&path)?); - } - } - Ok(()) -} - -fn archive(stage: &Path, output: &Path, id: &str) -> AtomicResult { - use zip::write::SimpleFileOptions; - let manifest: Manifest = serde_json::from_reader(File::open(stage.join(MANIFEST))?)?; - let mut temp = tempfile::NamedTempFile::new_in(output)?; - { - let mut writer = zip::ZipWriter::new(temp.as_file_mut()); - for name in manifest - .files - .keys() - .map(String::as_str) - .chain(std::iter::once(MANIFEST)) - { - let path = stage.join(name); - let options = SimpleFileOptions::default() - .compression_method(zip::CompressionMethod::Deflated) - .unix_permissions(0o600) - .large_file(fs::metadata(&path)?.len() >= u32::MAX as u64); - writer - .start_file(name, options) - .map_err(|e| e.to_string())?; - std::io::copy(&mut File::open(path)?, &mut writer)?; - } - writer.finish().map_err(|e| e.to_string())?; - } - temp.as_file().sync_all()?; - // Read back the ZIP and every hash before publication, without extracting. - verify_archive(temp.path())?; - let path = output.join(format!("atomic-backup-{id}.zip")); - temp.persist_noclobber(&path).map_err(|e| e.to_string())?; - File::open(output)?.sync_all()?; - Ok(path) -} - -fn safe_name(name: &str) -> bool { - !name.contains('\\') - && !name.contains(':') - && !name.contains('\0') - && Path::new(name) - .components() - .all(|p| matches!(p, Component::Normal(_))) - && (name.starts_with("data/") || name.starts_with("config/")) -} - -fn verify_archive(path: &Path) -> AtomicResult { - let mut zip = zip::ZipArchive::new(File::open(path)?).map_err(|e| e.to_string())?; - let manifest: Manifest = { - let mut file = zip.by_name(MANIFEST).map_err(|e| e.to_string())?; - if file.size() > 16 * 1024 * 1024 { - return Err("Oversized backup manifest".into()); - } - let mut json = Vec::new(); - file.read_to_end(&mut json)?; - serde_json::from_slice(&json)? - }; - if manifest.format != FORMAT { - return Err("Unsupported backup format".into()); - } - if manifest.server_version != env!("CARGO_PKG_VERSION") { - return Err(format!("Restore requires atomic-server {}", manifest.server_version).into()); - } - if !manifest.files.contains_key("data/store/atomic.redb") - || !manifest.files.contains_key("config/config.toml") - { - return Err("Backup lacks database or identity configuration".into()); - } - if zip.len() != manifest.files.len() + 1 { - return Err("Unexpected or duplicate archive entries".into()); - } - let mut seen = std::collections::HashSet::new(); - for i in 0..zip.len() { - let mut file = zip.by_index(i).map_err(|e| e.to_string())?; - let name = file.name().to_owned(); - if !seen.insert(name.clone()) { - return Err("Duplicate archive path".into()); - } - if name == MANIFEST { - continue; - } - if !safe_name(&name) - || file.is_dir() - || file.unix_mode().is_some_and(|m| m & 0o170000 == 0o120000) - { - return Err(format!("Unsafe archive path: {name}").into()); - } - let expected = manifest.files.get(&name).ok_or("Unlisted archive entry")?; - if file.size() != expected.bytes { - return Err("Archive size mismatch".into()); - } - let mut hasher = blake3::Hasher::new(); - let mut bytes = 0u64; - let mut buffer = [0u8; 65536]; - loop { - let n = file.read(&mut buffer)?; - if n == 0 { - break; - } - bytes += n as u64; - if bytes > expected.bytes { - return Err("Archive expanded beyond manifest size".into()); - } - hasher.update(&buffer[..n]); - } - if bytes != expected.bytes || hasher.finalize().to_hex().as_str() != expected.blake3 { - return Err(format!("Checksum mismatch for {name}").into()); - } - } - Ok(manifest) -} - -/// Restore never starts a runtime, sync transport, plugin or HTTP client. -/// The marker prevents an accidental normal server boot with copied identities. -pub fn restore(archive: &Path, target: &Path) -> AtomicResult<()> { - let manifest = verify_archive(archive)?; - if target.exists() { - return Err("Restore target must not exist".into()); - } - let parent = target - .parent() - .filter(|p| !p.as_os_str().is_empty()) - .unwrap_or(Path::new(".")); - let stage = tempfile::Builder::new() - .prefix(".atomic-restore-") - .tempdir_in(parent)?; - let mut zip = zip::ZipArchive::new(File::open(archive)?).map_err(|e| e.to_string())?; - for name in manifest.files.keys() { - let dest = stage.path().join(name); - fs::create_dir_all(dest.parent().unwrap())?; - let mut file = create_private(&dest)?; - std::io::copy( - &mut zip.by_name(name).map_err(|e| e.to_string())?, - &mut file, - )?; - file.sync_all()?; - let actual = digest(&dest)?; - let expected = &manifest.files[name]; - if actual.bytes != expected.bytes || actual.blake3 != expected.blake3 { - return Err("Archive changed during restore".into()); - } - } - atomic_lib::db::redb_store::verify_snapshot(&stage.path().join("data/store/atomic.redb"))?; - fs::write( - stage.path().join("data").join(RESTORED), - b"Restored offline. Explicit activation is required before starting a server.\n", - )?; - fs::write( - stage.path().join(MANIFEST), - serde_json::to_vec_pretty(&manifest)?, - )?; - // Claim the target without clobbering even an empty directory created by - // another operation since preflight. Incomplete promotion retains the marker. - fs::create_dir(target)?; - private_dir(target)?; - for item in fs::read_dir(stage.path())? { - let item = item?; - fs::rename(item.path(), target.join(item.file_name()))?; - } - File::open(parent)?.sync_all()?; - Ok(()) -} - -pub fn check_restore_activation(config: &Config) -> AtomicResult<()> { - if data_dir(config)?.join(RESTORED).exists() && !config.opts.activate_restored { - return Err("Restored instance is offline. Use --activate-restored only when ready to reconnect its copied identities and integrations".into()); - } - Ok(()) -} - fn create_private(path: &Path) -> std::io::Result { let mut options = OpenOptions::new(); options.write(true).create_new(true); @@ -557,15 +237,6 @@ fn private_file(path: &Path) -> std::io::Result<()> { } Ok(()) } -fn private_dir(path: &Path) -> std::io::Result<()> { - #[cfg(unix)] - { - use std::os::unix::fs::PermissionsExt; - fs::set_permissions(path, fs::Permissions::from_mode(0o700))?; - } - Ok(()) -} - /// Scheduler-friendly command: wait for this specific job and exit nonzero on /// failure, replacement or timeout. Never follow redirects with the bearer token. pub async fn request_backup(server: &str, token_file: &Path) -> AtomicServerResult<()> { @@ -661,17 +332,7 @@ pub async fn admission( #[cfg(test)] mod tests { use super::*; - use atomic_lib::{ - db::{ - kv_store::KvStore, - redb_store::RedbStore, - trees::{Method, Operation, Tree}, - }, - loro::AtomicLoroDoc, - Value, - }; use clap::Parser; - async fn fixture(root: &Path) -> (Db, Config, Arc) { let config = crate::config::build_config(crate::config::Opts::parse_from([ "atomic-server", @@ -707,222 +368,6 @@ mod tests { (store, config, service) } - #[tokio::test] - async fn instance_roundtrip_preserves_history_blobs_metadata_and_files() { - let root = tempfile::tempdir().unwrap(); - let (store, config, service) = fixture(root.path()).await; - let doc = AtomicLoroDoc::new(); - doc.set_property("name", &Value::String("before".into())) - .unwrap(); - let first = doc.export_snapshot(); - let first_version = doc.current_version(); - doc.set_property("name", &Value::String("after".into())) - .unwrap(); - let snapshot = doc.export_snapshot(); - assert_ne!(first, snapshot); - store - .kv - .insert(Tree::LoroSnapshots, b"did:ad:test", &snapshot) - .unwrap(); - for tree in [ - Tree::Blobs, - Tree::Envelopes, - Tree::PluginMeta, - Tree::DriveMapping, - ] { - store.kv.insert(tree, b"fixture", b"original").unwrap(); - } - fs::create_dir_all(&config.uploads_path).unwrap(); - fs::write(config.uploads_path.join("legacy.txt"), b"legacy file").unwrap(); - let path = run_backup(&store, &config, &service, "roundtrip") - .await - .unwrap(); - assert!(!store.maintenance.is_paused()); - store - .kv - .insert(Tree::PluginMeta, b"fixture", b"changed after backup") - .unwrap(); - let target = root.path().join("restored"); - restore(&path, &target).unwrap(); - assert!(restore(&path, &target).is_err()); - let restored = RedbStore::new_file(&target.join("data/store/atomic.redb")).unwrap(); - assert_eq!( - restored.get(Tree::PluginMeta, b"fixture").unwrap().unwrap(), - b"original" - ); - assert_eq!( - restored.get(Tree::Blobs, b"fixture").unwrap().unwrap(), - b"original" - ); - assert_eq!( - restored.get(Tree::Envelopes, b"fixture").unwrap().unwrap(), - b"original" - ); - let recovered = AtomicLoroDoc::from_snapshot( - &restored - .get(Tree::LoroSnapshots, b"did:ad:test") - .unwrap() - .unwrap(), - ) - .unwrap(); - assert_eq!(recovered.get_history().len(), doc.get_history().len()); - recovered.fork_at(&first_version).unwrap(); - assert_eq!( - fs::read(target.join("data/uploads/legacy.txt")).unwrap(), - b"legacy file" - ); - assert_eq!( - fs::read(target.join("config/config.toml")).unwrap(), - fs::read(&config.config_file_path).unwrap() - ); - let mut restored_config = config.clone(); - restored_config.store_path = target.join("data/store"); - assert!(check_restore_activation(&restored_config).is_err()); - restored_config.opts.activate_restored = true; - check_restore_activation(&restored_config).unwrap(); - } - - #[test] - fn snapshot_blocks_writers_and_copies_unknown_byte_tables() { - let root = tempfile::tempdir().unwrap(); - let source = root.path().join("source.redb"); - // Future tables must not silently disappear from the backup. - { - let db = redb_for_test(&source); - let tx = db.begin_write().unwrap(); - { - let mut table = tx - .open_table(redb::TableDefinition::<&[u8], &[u8]>::new("future_table")) - .unwrap(); - table - .insert(b"future".as_slice(), b"value".as_slice()) - .unwrap(); - } - tx.commit().unwrap(); - } - let store = Arc::new(RedbStore::new_file(&source).unwrap()); - store - .apply_batch(&[ - Operation { - tree: Tree::Resources, - method: Method::Insert, - key: b"key".to_vec(), - val: Some(b"old".to_vec()), - }, - Operation { - tree: Tree::LoroSnapshots, - method: Method::Insert, - key: b"key".to_vec(), - val: Some(b"old".to_vec()), - }, - ]) - .unwrap(); - let (entered_tx, entered_rx) = std::sync::mpsc::channel(); - let (release_tx, release_rx) = std::sync::mpsc::channel(); - let s = store.clone(); - let destination = root.path().join("snapshot.redb"); - let path = destination.clone(); - let backup = std::thread::spawn(move || { - s.backup_snapshot(&path, &mut || { - entered_tx.send(()).unwrap(); - release_rx.recv().unwrap(); - Ok(()) - }) - }); - entered_rx - .recv_timeout(std::time::Duration::from_secs(10)) - .unwrap(); - let s = store.clone(); - let (written_tx, written_rx) = std::sync::mpsc::channel(); - let writer = std::thread::spawn(move || { - s.insert(Tree::Resources, b"key", b"new").unwrap(); - written_tx.send(()).unwrap(); - }); - assert!(written_rx - .recv_timeout(std::time::Duration::from_millis(50)) - .is_err()); - release_tx.send(()).unwrap(); - backup.join().unwrap().unwrap(); - writer.join().unwrap(); - let snapshot = RedbStore::new_file(&destination).unwrap(); - assert_eq!( - snapshot.get(Tree::Resources, b"key").unwrap().unwrap(), - b"old" - ); - assert_eq!( - snapshot.get(Tree::LoroSnapshots, b"key").unwrap().unwrap(), - b"old" - ); - drop(snapshot); - use redb::ReadableDatabase; - let db = redb_for_test(&destination); - let tx = db.begin_read().unwrap(); - let table = tx - .open_table(redb::TableDefinition::<&[u8], &[u8]>::new("future_table")) - .unwrap(); - assert_eq!( - table.get(b"future".as_slice()).unwrap().unwrap().value(), - b"value" - ); - } - - fn redb_for_test(path: &Path) -> redb::Database { - redb::Database::create(path).unwrap() - } - - #[tokio::test] - async fn capture_failure_resumes_and_publishes_nothing() { - let root = tempfile::tempdir().unwrap(); - let (store, config, service) = fixture(root.path()).await; - #[cfg(unix)] - std::os::unix::fs::symlink("/outside", config.config_dir.join("unsupported")).unwrap(); - #[cfg(not(unix))] - { - store.kv.begin_batch(); - } - assert!(run_backup(&store, &config, &service, "failure") - .await - .is_err()); - assert!(!store.maintenance.is_paused()); - assert!(!root - .path() - .join("backups/atomic-backup-failure.zip") - .exists()); - store - .kv - .insert(Tree::PluginMeta, b"after-failure", b"works") - .unwrap(); - } - - #[test] - fn snapshot_callback_failure_releases_storage_barrier() { - let root = tempfile::tempdir().unwrap(); - let store = RedbStore::new_file(&root.path().join("source.redb")).unwrap(); - assert!(store - .backup_snapshot(&root.path().join("failed.redb"), &mut || Err( - "injected I/O failure".into() - )) - .is_err()); - store.insert(Tree::PluginMeta, b"after", b"works").unwrap(); - assert!(store - .backup_snapshot(&root.path().join("failed.redb"), &mut || Ok(())) - .is_err()); - store.begin_batch(); - assert!(store - .backup_snapshot(&root.path().join("batch.redb"), &mut || Ok(())) - .is_err()); - store.commit_batch().unwrap(); - let panicked = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { - let _ = store.backup_snapshot(&root.path().join("panic.redb"), &mut || { - panic!("injected capture panic") - }); - })); - assert!(panicked.is_err()); - store - .insert(Tree::PluginMeta, b"after-panic", b"works") - .unwrap(); - } - #[tokio::test] async fn operator_requires_token_and_loopback_even_during_pause() { let root = tempfile::tempdir().unwrap(); @@ -962,162 +407,4 @@ mod tests { invalid.opts.backup_dir = Some(config.config_dir.clone()); assert!(BackupService::new(&invalid).is_err()); } - - #[test] - fn restore_rejects_unsafe_paths() { - for name in [ - "../escape", - "data/../../escape", - "/data/file", - "config/../escape", - "data\\escape", - "data/C:escape", - ] { - assert!(!safe_name(name), "{name}"); - } - assert!(safe_name("data/store/atomic.redb")); - } -} - -#[cfg(test)] -mod sync_tests { - use super::*; - use atomic_lib::{ - agents::ForAgent, - db::trees::Tree, - loro::AtomicLoroDoc, - sync::{engine, protocol}, - Storelike, Value, - }; - - #[tokio::test] - async fn incoming_sync_waits_without_acknowledging_or_losing_the_update() { - let root = tempfile::tempdir().unwrap(); - let db = Db::init_redb_file( - &root.path().join("store"), - Some("http://localhost:9883".into()), - &root.path().join("uploads"), - ) - .await - .unwrap(); - let (agent, drive) = db.setup("Backup sync test").await.unwrap(); - let subject = "did:ad:backup-sync-test"; - let doc = AtomicLoroDoc::new(); - doc.set_property( - atomic_lib::urls::DRIVE_PROP, - &Value::AtomicUrl(drive.clone().into()), - ) - .unwrap(); - doc.set_property( - atomic_lib::urls::NAME, - &Value::String("arrived during backup".into()), - ) - .unwrap(); - let frame = protocol::encode_sync_push(&drive, &[(subject, &doc.export_snapshot())], true); - let push = protocol::decode_sync_push(&frame[1..]).unwrap(); - let pause = db.maintenance.pause().await.unwrap(); - let receiver = db.clone(); - let (started_tx, started_rx) = tokio::sync::oneshot::channel(); - let job = tokio::spawn(async move { - started_tx.send(()).unwrap(); - engine::import_sync_push(&push, &receiver, &ForAgent::from(agent), false).await - }); - started_rx.await.unwrap(); - tokio::task::yield_now().await; - assert!(!job.is_finished(), "must not acknowledge a paused import"); - assert!(db - .kv - .get(Tree::LoroSnapshots, subject.as_bytes()) - .unwrap() - .is_none()); - drop(pause); - let (count, _) = tokio::time::timeout(std::time::Duration::from_secs(10), job) - .await - .unwrap() - .unwrap() - .unwrap(); - assert_eq!(count, 1); - assert_eq!( - db.get_resource(&subject.into()) - .await - .unwrap() - .get(atomic_lib::urls::NAME) - .unwrap() - .to_string(), - "arrived during backup" - ); - } -} - -#[cfg(test)] -mod archive_tests { - use super::*; - - fn untrusted(root: &Path, name: &str, content: &[u8], expected_hash: &str) -> PathBuf { - let path = root.join("untrusted.zip"); - let mut files = BTreeMap::new(); - for key in ["data/store/atomic.redb", "config/config.toml", name] { - files.insert( - key.into(), - Entry { - bytes: content.len() as u64, - blake3: expected_hash.into(), - }, - ); - } - let manifest = Manifest { - format: FORMAT, - server_version: env!("CARGO_PKG_VERSION").into(), - build_revision: "test".into(), - redb_version: "4.1.0".into(), - snapshot_time: "test".into(), - freshness: "unknown".into(), - envelope_retention: "latest".into(), - source_data: "source".into(), - source_config: "config".into(), - files, - }; - let mut writer = zip::ZipWriter::new(File::create(&path).unwrap()); - let options = zip::write::SimpleFileOptions::default(); - writer.start_file(MANIFEST, options).unwrap(); - writer - .write_all(&serde_json::to_vec(&manifest).unwrap()) - .unwrap(); - for key in manifest.files.keys() { - writer.start_file(key, options).unwrap(); - writer.write_all(content).unwrap(); - } - writer.finish().unwrap(); - path - } - - #[test] - fn corrupt_or_traversing_archive_never_creates_restore_target() { - let root = tempfile::tempdir().unwrap(); - let target = root.path().join("restore"); - let archive = untrusted(root.path(), "data/note", b"tampered", "wrong checksum"); - assert!(restore(&archive, &target) - .unwrap_err() - .to_string() - .contains("Checksum")); - assert!(!target.exists()); - let hash = blake3::hash(b"valid").to_hex().to_string(); - let archive = untrusted(root.path(), "data/../../escape", b"valid", &hash); - assert!(restore(&archive, &target) - .unwrap_err() - .to_string() - .contains("Unsafe")); - assert!(!target.exists()); - assert!(!root.path().join("escape").exists()); - } - - #[test] - fn valid_hashes_do_not_make_a_corrupt_database_restorable() { - let root = tempfile::tempdir().unwrap(); - let hash = blake3::hash(b"not a redb file").to_hex().to_string(); - let archive = untrusted(root.path(), "data/note", b"not a redb file", &hash); - let target = root.path().join("restore"); - assert!(restore(&archive, &target).is_err()); - assert!(!target.exists()); - } }