From 1ad9a16b18be51ac35fd855b4a82e96a99b99513 Mon Sep 17 00:00:00 2001 From: AJ Roetker Date: Thu, 8 Oct 2026 10:40:04 -0700 Subject: [PATCH 1/4] perf(lite): bound decoded cache and reuse hot read payloads --- .../baselines/lite-decoded-cache-budget.json | 54 +++ .../src/storage/lsm_backend.zig | 422 ++++++++++++++++-- .../src/storage/lsm_backend/recovery.zig | 1 + .../src/storage/lsm_backend/runtime.zig | 63 ++- .../src/storage/lsm_backend/shared_bytes.zig | 6 + 5 files changed, 511 insertions(+), 35 deletions(-) create mode 100644 zig/bench/baselines/lite-decoded-cache-budget.json diff --git a/zig/bench/baselines/lite-decoded-cache-budget.json b/zig/bench/baselines/lite-decoded-cache-budget.json new file mode 100644 index 0000000000..9034e09c18 --- /dev/null +++ b/zig/bench/baselines/lite-decoded-cache-budget.json @@ -0,0 +1,54 @@ +{ + "schema_version": 1, + "base_commit": "84dfbf5a953443cd89d2ddec477b8cda3ba922ce", + "reference_commit": "4cd4e471e7d82d9574e03a4a0f1b350ea1f1c8d7", + "platform": "aarch64-macos", + "optimization": "Debug", + "scope": "Deterministic allocation and lifetime fixtures; not a query-throughput or 2 GiB soak benchmark", + "hot_point_reads": { + "queries_after_initial_cold_read": 100, + "prefix": { + "reference_backend_allocator_calls": 500, + "updated_backend_allocator_calls": 8 + }, + "prefix_snappy": { + "reference_backend_allocator_calls": 604, + "updated_backend_allocator_calls": 9 + }, + "reference_encoded_block_loads": 100, + "updated_encoded_block_loads": 1, + "updated_retained_decoded_blocks": 1, + "subsequent_100_warm_reads_backend_allocator_calls": 0, + "subsequent_100_warm_reads_encoded_block_loads": 0 + }, + "repeated_batch_results": { + "two_key_batches": 101, + "reference_retained_pins": 64, + "reference_distinct_pinned_blocks": 1, + "reference_owned_values": 74, + "updated_retained_pins": 1, + "updated_distinct_pinned_blocks": 1, + "updated_owned_values": 0 + }, + "cache_policy": { + "default_payload_residency_bytes": 2097152, + "default_max_admitted_block_bytes": 262144, + "max_cache_entries": 64, + "max_result_owner_pinned_bytes": 1048576, + "max_result_owner_distinct_pins": 64, + "promotion_misses": 2, + "denied_promotion_cooldown_misses": 64, + "resource_credit_covers_payload_and_shared_header": true, + "evicted_pinned_payload_credit_survives_until_final_release": true, + "host_pressure_reclaims_unpinned_cache_without_blocking_backend": true, + "denied_admission_preserves_read_results": true, + "shutdown_fences_pending_reclaimer_registration": true + }, + "notes": [ + "Allocation counts cover the backend allocator; owned point results still allocate in the caller's result allocator.", + "Encoded block loads may hit the provider memory cache; these counts are not physical disk reads.", + "Cold decoding and first promotion still allocate. Zero allocation applies to the measured warm backend path.", + "Resource accounting follows admitted payloads through eviction and result leases. Unadmitted decoded scratch remains transient, and resource-managed result owners copy its values.", + "The on-disk format and public C ABI are unchanged." + ] +} diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend.zig index 4bda1ad88b..396d57d583 100644 --- a/zig/pkg/antfly-embedded/src/storage/lsm_backend.zig +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend.zig @@ -454,6 +454,10 @@ pub const Options = struct { storage: ?storage_io.Storage = null, cache: ?*cache_mod.Cache = null, local_block_cache_enabled: bool = true, + /// Payload residency, independent of the 64-entry metadata bound. Zero disables retention. + local_block_cache_bytes: usize = 2 * 1024 * 1024, + /// Large one-value blocks stay on the direct/ephemeral path. + local_block_cache_max_block_bytes: usize = 256 * 1024, max_concurrent_point_block_reads: usize = 16, // Share sparse point-read fan-out across concurrent batches. This keeps // single-query cold latency low without allowing N public queries to each @@ -1419,6 +1423,18 @@ pub const Backend = struct { result: ?*RunSourceLease = null, failure: ?anyerror = null, }; + const LocalBlockHeat = struct { + valid: bool = false, + run_id: u64 = 0, + path_hash: u64 = 0, + offset: u64 = 0, + len: u32 = 0, + misses: u8 = 0, + promoting: bool = false, + cooldown: u8 = 0, + access: u64 = 0, + }; + const CachedRunBlock = struct { run_id: u64, path: []u8, @@ -1648,6 +1664,9 @@ pub const Backend = struct { run_state_cache: std.ArrayListUnmanaged(CachedRunState) = .empty, run_index_cache: std.ArrayListUnmanaged(CachedRunIndex) = .empty, run_block_cache: std.ArrayListUnmanaged(CachedRunBlock) = .empty, + run_block_cache_bytes: usize = 0, + local_block_cache_reclaimer: ?u64 = null, + local_block_heat: [64]LocalBlockHeat = @splat(.{}), // Independent from the backend lock, which some range-read callers own. run_source_mutex: std.atomic.Mutex = .unlocked, run_sources: std.ArrayListUnmanaged(RunSource) = .empty, @@ -1918,6 +1937,7 @@ pub const Backend = struct { pub fn close(self: *Backend) void { self.closing.store(true, .release); + self.detachLocalBlockCacheReclaimer(); self.background_executor.drain(); self.waitForGenerationReadersToDrain(); self.releaseTrackedResourceUsage(); @@ -1926,6 +1946,7 @@ pub const Backend = struct { pub fn abandonAfterCrash(self: *Backend) void { self.closing.store(true, .release); + self.detachLocalBlockCacheReclaimer(); self.background_executor.drain(); self.releaseTrackedResourceUsage(); recovery_mod.abandon(Backend, self); @@ -6813,7 +6834,7 @@ pub const Backend = struct { } pub fn localBlockCacheEnabled(self: *const Backend) bool { - return self.options.local_block_cache_enabled; + return self.options.local_block_cache_enabled and self.options.local_block_cache_bytes != 0; } pub fn recordCursorBlockReuse(self: *Backend) void { @@ -7111,7 +7132,7 @@ pub const Backend = struct { block_offset: u64, block_len: u32, ) ?[]const u8 { - if (!self.options.local_block_cache_enabled) return null; + if (!self.localBlockCacheEnabled()) return null; for (self.run_block_cache.items) |*cached| { if (cached.run_id != run_id or cached.block_offset != block_offset or @@ -7124,7 +7145,7 @@ pub const Backend = struct { } pub fn retainCachedRunBlock(self: *Backend, path: []const u8, run_id: u64, offset: u64, len: u32) ?*SharedBytes { - if (!self.options.local_block_cache_enabled) return null; + if (!self.localBlockCacheEnabled()) return null; for (self.run_block_cache.items) |*cached| { if (cached.run_id == run_id and cached.block_offset == offset and cached.block_len == len and std.mem.eql(u8, cached.path, path)) { cached.last_access = self.nextLocalCacheAccess(); @@ -7134,13 +7155,111 @@ pub const Backend = struct { return null; } - /// Owns one reference on success; the caller keeps its original reference. - pub fn cacheRunBlockLease(self: *Backend, path: []const u8, run_id: u64, offset: u64, len: u32, payload: *SharedBytes) !void { - const cached_path = try self.allocator.dupe(u8, path); - errdefer self.allocator.free(cached_path); - try self.run_block_cache.append(self.allocator, .{ .run_id = run_id, .path = cached_path, .block_offset = offset, .block_len = len, .payload = payload, .last_access = self.nextLocalCacheAccess() }); + pub fn localBlockCacheEligible(self: *const Backend, bytes: usize) bool { + return self.localBlockCacheEnabled() and bytes <= self.options.local_block_cache_bytes and bytes <= self.options.local_block_cache_max_block_bytes; + } + + /// Called under the backend mutex. Fixed metadata records two misses before + /// promotion; one promoter owns the attempt and budget denials back off. + pub fn beginLocalBlockPromotion(self: *Backend, path: []const u8, run_id: u64, offset: u64, len: u32, decoded_len: usize) bool { + if (!self.localBlockCacheEligible(decoded_len)) return false; + const path_hash = std.hash.Wyhash.hash(0, path); + const access = self.nextLocalCacheAccess(); + var victim: ?usize = null; + for (&self.local_block_heat, 0..) |*heat, i| { + if (heat.valid and heat.run_id == run_id and heat.path_hash == path_hash and heat.offset == offset and heat.len == len) { + heat.access = access; + if (heat.promoting) return false; + if (heat.cooldown != 0) { + heat.cooldown -= 1; + return false; + } + heat.misses +|= 1; + if (heat.misses < 2) return false; + heat.promoting = true; + return true; + } + if (!heat.promoting and (victim == null or !heat.valid or heat.access < self.local_block_heat[victim.?].access)) victim = i; + } + const slot = victim orelse return false; + self.local_block_heat[slot] = .{ .valid = true, .run_id = run_id, .path_hash = path_hash, .offset = offset, .len = len, .misses = 1, .access = access }; + return false; + } + + pub fn finishLocalBlockPromotion(self: *Backend, path: []const u8, run_id: u64, offset: u64, len: u32, admitted: bool) void { + const path_hash = std.hash.Wyhash.hash(0, path); + for (&self.local_block_heat) |*heat| { + if (!heat.valid or heat.run_id != run_id or heat.path_hash != path_hash or heat.offset != offset or heat.len != len) continue; + heat.promoting = false; + heat.misses = 0; + heat.cooldown = if (admitted) 0 else 64; + return; + } + } + + fn detachLocalBlockCacheReclaimer(self: *Backend) void { + // Fence the last admission before taking its registration. Unregister + // outside the backend lock so an already selected callback can finish. + platform.sync.lockYielding(&self.mu); + const identity = self.local_block_cache_reclaimer; + self.local_block_cache_reclaimer = null; + self.mu.unlock(); + if (identity) |owned| self.options.resource_manager.?.unregisterReclaimer(owned); + } + + fn reclaimLocalBlocks(raw: *anyopaque, target: u64) u64 { + const self: *Backend = @ptrCast(@alignCast(raw)); + // Resource reclamation may originate under a caller's subsystem lock. + // Never wait for the backend, and never perform storage I/O here. + if (self.closing.load(.acquire) or !self.mu.tryLock()) return 0; + defer self.mu.unlock(); + if (self.closing.load(.acquire)) return 0; + var released: u64 = 0; + while (released < target and self.run_block_cache.items.len != 0) { + released +|= self.evictOldestLocalBlock(); + } + return released; + } + + /// Optional cache admission never fails the mandatory read. Only admitted + /// payloads gain cache ownership; resource credit follows their last lease. + pub fn cacheRunBlockLease(self: *Backend, path: []const u8, run_id: u64, offset: u64, len: u32, payload: *SharedBytes) bool { + if (!self.localBlockCacheEligible(payload.bytes.len) or self.closing.load(.acquire) or payload.cache_admitted) return false; + for (self.run_block_cache.items) |cached| { + if (cached.run_id == run_id and cached.block_offset == offset and cached.block_len == len and std.mem.eql(u8, cached.path, path)) return false; + } + if (self.options.resource_manager) |manager| { + // First read/cache admission occurs at the backend's stable address. + if (self.local_block_cache_reclaimer == null) self.local_block_cache_reclaimer = manager.registerReclaimer(.lsm_block_table_cache, self, reclaimLocalBlocks) catch return false; + } + const cached_path = self.allocator.dupe(u8, path) catch return false; + var admitted = false; + defer if (!admitted) self.allocator.free(cached_path); + self.run_block_cache.ensureUnusedCapacity(self.allocator, 1) catch return false; + while (self.run_block_cache.items.len >= max_local_cached_run_blocks or self.run_block_cache_bytes > self.options.local_block_cache_bytes - payload.bytes.len) { + _ = self.evictOldestLocalBlock(); + } + if (self.options.resource_manager) |manager| { + const bytes = payload.bytes.len + @sizeOf(SharedBytes); + const credit = manager.reserveWithoutReclaim(.lsm_block_table_cache, bytes) catch blk: { + // Reclaim our own cache only, never arbitrary callbacks under + // the backend lock. Pinned victims retain their resource credit. + while (self.run_block_cache.items.len != 0) { + _ = self.evictOldestLocalBlock(); + break :blk manager.reserveWithoutReclaim(.lsm_block_table_cache, bytes) catch continue; + } + return false; + }; + std.debug.assert(payload.reservation == null); + payload.reservation = credit; + } + self.run_block_cache.appendAssumeCapacity(.{ .run_id = run_id, .path = cached_path, .block_offset = offset, .block_len = len, .payload = payload, .last_access = self.nextLocalCacheAccess() }); _ = payload.retain(); - self.evictCachedRunBlocksToBudget(); + payload.cache_admitted = true; + payload.result_pins_allowed = true; + self.run_block_cache_bytes += payload.bytes.len; + admitted = true; + return true; } /// Lease exactly the immutable artifact roots, not the whole checkpoint. @@ -7320,19 +7439,16 @@ pub const Backend = struct { block_len: u32, block: []u8, ) ![]const u8 { - if (!self.options.local_block_cache_enabled) { + if (!self.localBlockCacheEnabled()) { self.allocator.free(block); return &.{}; } errdefer self.allocator.free(block); const payload = try SharedBytes.create(self.allocator, block); - // On error the outer errdefer still owns the byte allocation. - self.cacheRunBlockLease(path, run_id, block_offset, block_len, payload) catch |err| { - self.allocator.destroy(payload); - return err; - }; + const admitted = self.cacheRunBlockLease(path, run_id, block_offset, block_len, payload); + const result = if (admitted) payload.bytes else &.{}; payload.release(); - return self.run_block_cache.items[self.run_block_cache.items.len - 1].payload.bytes; + return result; } fn drainObsoleteRuns(self: *Backend) void { @@ -8754,23 +8870,25 @@ pub const Backend = struct { continue; } var removed = self.run_block_cache.orderedRemove(i); + self.run_block_cache_bytes -= removed.payload.bytes.len; removed.deinit(self.allocator); } } - fn evictCachedRunBlocksToBudget(self: *Backend) void { - while (self.run_block_cache.items.len > max_local_cached_run_blocks) { - var victim_index: usize = 0; - var victim_access = self.run_block_cache.items[0].last_access; - for (self.run_block_cache.items[1..], 1..) |cached, i| { - if (cached.last_access < victim_access) { - victim_access = cached.last_access; - victim_index = i; - } - } - var victim = self.run_block_cache.orderedRemove(victim_index); - victim.deinit(self.allocator); + fn evictOldestLocalBlock(self: *Backend) u64 { + std.debug.assert(self.run_block_cache.items.len != 0); + var victim_index: usize = 0; + for (self.run_block_cache.items[1..], 1..) |cached, i| { + if (cached.last_access < self.run_block_cache.items[victim_index].last_access) victim_index = i; } + var victim = self.run_block_cache.orderedRemove(victim_index); + self.run_block_cache_bytes -= victim.payload.bytes.len; + const released = if (victim.payload.refs.load(.acquire) == 1) + (if (victim.payload.reservation) |credit| credit.reservedBytes() else 0) + else + 0; + victim.deinit(self.allocator); + return released; } fn evictCachedRunTableForRun(self: *Backend, path: []const u8, run_id: u64) void { @@ -25667,3 +25785,251 @@ test "lsm local compressed point reads borrow warm decoded blocks" { try std.testing.expectEqual(@as(u64, 0), extra_loads); } } + +test "lsm local hot compressed point reads promote and repeated batches deduplicate pins" { + const cases = [_]struct { value: []const u8, compression: lsm_table_file.BlockCompression }{ + .{ .value = "other", .compression = .prefix }, + .{ .value = z17RepeatString("compressible-value:", 128), .compression = .prefix_snappy }, + }; + for (cases) |case| { + const a = std.testing.allocator; + var backing = storage_io.MemoryStorage.init(a); + defer backing.deinit(); + var backend_budget = @import("lite/test_allocator.zig").BudgetAllocator{ .backing = a }; + var backend = try Backend.open(backend_budget.allocator(), "/local-result-lifetimes", .{ .storage = backing.storage(), .flush_threshold = 1, .table_block_compression = .snappy_adaptive }); + defer backend.close(); + var runtime = try backend.runtimeStore(a, .{ .name = "docs" }); + defer runtime.deinit(); + var write = try runtime.beginWrite(); + try write.put("document:long-shared-prefix-for-compression:00", "value"); + try write.put("document:long-shared-prefix-for-compression:01", case.value); + try write.commit(); + var read = try runtime_mod.BoundReadTxn(Backend).open(&backend, .{ .name = "docs" }); + defer read.abort(); + var result_budget = @import("lite/test_allocator.zig").BudgetAllocator{ .backing = a }; + var point = try read.openReadScope(result_budget.allocator()); + defer point.close(); + try std.testing.expectEqualStrings("value", try point.get("document:long-shared-prefix-for-compression:00")); + try std.testing.expectEqual(case.compression, backend.run_index_cache.items[0].index.blockWindow(0).compression); + const pure_calls = backend_budget.alloc_calls; + const pure_loads = backend.read_stats.table_block_loads.load(.monotonic); + for (0..100) |_| try std.testing.expectEqualStrings(case.value, try point.get("document:long-shared-prefix-for-compression:01")); + std.debug.print("lite hot point probe: compression={s}, backend allocations={d}, encoded block loads={d}, decoded blocks={d}\n", .{ @tagName(case.compression), backend_budget.alloc_calls - pure_calls, backend.read_stats.table_block_loads.load(.monotonic) - pure_loads, backend.run_block_cache.items.len }); + try std.testing.expect(backend_budget.alloc_calls - pure_calls < 20); + try std.testing.expectEqual(@as(u64, 1), backend.read_stats.table_block_loads.load(.monotonic) - pure_loads); + // Cold compressed point lookup has not populated the decoded cache. + try std.testing.expectEqual(@as(usize, 1), backend.run_block_cache.items.len); + var batch = try read.openReadScope(a); + defer batch.close(); + const keys = [_][]const u8{ "document:long-shared-prefix-for-compression:00", "document:long-shared-prefix-for-compression:01" }; + var values: [2]?[]const u8 = undefined; + try batch.getManySorted(&keys, &values); + const calls = backend_budget.alloc_calls; + const loads = backend.read_stats.table_block_loads.load(.monotonic); + backend_budget.limit = backend_budget.live; + defer backend_budget.limit = std.math.maxInt(usize); + for (0..100) |_| try std.testing.expectEqualStrings(case.value, try point.get("document:long-shared-prefix-for-compression:01")); + const extra_calls = backend_budget.alloc_calls - calls; + const extra_loads = backend.read_stats.table_block_loads.load(.monotonic) - loads; + std.debug.print("lite compressed warm point probe: compression={s}, 100 queries, backend allocations={d}, physical block loads={d}, decoded cache blocks={d}\n", .{ @tagName(case.compression), extra_calls, extra_loads, backend.run_block_cache.items.len }); + try std.testing.expectEqual(@as(usize, 0), extra_calls); + try std.testing.expectEqual(@as(u64, 0), extra_loads); + backend_budget.limit = std.math.maxInt(usize); + for (0..100) |_| try batch.getManySorted(&keys, &values); + var distinct_pins: usize = 0; + for (batch.held_blocks.items, 0..) |pin, pi| { + var seen = false; + for (batch.held_blocks.items[0..pi]) |previous| { + if (pin == .local and previous == .local and pin.local == previous.local) seen = true; + } + if (!seen) distinct_pins += 1; + } + std.debug.print("lite repeated batch probe: compression={s}, pins={d}, distinct pins={d}, owned values={d}\n", .{ @tagName(case.compression), batch.held_blocks.items.len, distinct_pins, batch.held_values.items.len }); + try std.testing.expectEqual(@as(usize, 1), distinct_pins); + try std.testing.expectEqual(@as(usize, 1), batch.held_blocks.items.len); + try std.testing.expectEqual(@as(usize, 0), batch.held_values.items.len); + } +} + +test "lsm local byte budget evicts cache ownership but charges pinned payload until final release" { + const a = std.testing.allocator; + var manager = resource_manager_mod.ResourceManager.init(.{}); + defer manager.deinit(a); + var backend = Backend.init(a, .{ .resource_manager = &manager, .local_block_cache_bytes = 64, .local_block_cache_max_block_bytes = 64 }); + var open = true; + defer if (open) backend.close(); + const first = try a.alloc(u8, 40); + @memset(first, 'a'); + _ = try backend.putCachedRunBlock("/first", 1, 0, 40, first); + const pinned = backend.retainCachedRunBlock("/first", 1, 0, 40).?; + var pinned_open = true; + defer if (pinned_open) pinned.release(); + const charge = 40 + @sizeOf(SharedBytes); + try std.testing.expectEqual(@as(u64, charge), manager.sliceStats(.lsm_block_table_cache).used_bytes); + const second = try a.alloc(u8, 40); + @memset(second, 'b'); + _ = try backend.putCachedRunBlock("/second", 2, 0, 40, second); + try std.testing.expectEqual(@as(usize, 1), backend.run_block_cache.items.len); + try std.testing.expectEqual(@as(usize, 40), backend.run_block_cache_bytes); + try std.testing.expectEqual(@as(u64, 2 * charge), manager.sliceStats(.lsm_block_table_cache).used_bytes); + try std.testing.expectEqual(@as(u8, 'a'), pinned.bytes[0]); + pinned.release(); + pinned_open = false; + try std.testing.expectEqual(@as(u64, charge), manager.sliceStats(.lsm_block_table_cache).used_bytes); + // Oversized blocks never acquire residency or a cache resource charge. + const oversized = try a.alloc(u8, 65); + try std.testing.expectEqual(@as(usize, 0), (try backend.putCachedRunBlock("/large", 3, 0, 65, oversized)).len); + try std.testing.expectEqual(@as(usize, 40), backend.run_block_cache_bytes); + backend.close(); + open = false; + try std.testing.expectEqual(@as(u64, 0), manager.sliceStats(.lsm_block_table_cache).used_bytes); +} + +test "lsm local denied cache admission keeps reads valid and backs off promotion" { + const a = std.testing.allocator; + var budgets = resource_manager_mod.Options.defaultBudgets(); + budgets[@backingInt(resource_manager_mod.Slice.lsm_block_table_cache)] = .{ .hard_limit_bytes = 1 }; + var manager = resource_manager_mod.ResourceManager.init(.{ .budgets = budgets }); + defer manager.deinit(a); + var backing = storage_io.MemoryStorage.init(a); + defer backing.deinit(); + var backend = try Backend.open(a, "/denied-local-cache", .{ .storage = backing.storage(), .resource_manager = &manager, .flush_threshold = 1 }); + defer backend.close(); + var runtime = try backend.runtimeStore(a, .{ .name = "docs" }); + defer runtime.deinit(); + var write = try runtime.beginWrite(); + try write.put("document:long-shared-prefix-for-compression:00", "value"); + try write.put("document:long-shared-prefix-for-compression:01", "other"); + try write.commit(); + var read = try runtime_mod.BoundReadTxn(Backend).open(&backend, .{ .name = "docs" }); + defer read.abort(); + var scope = try read.openReadScope(a); + defer scope.close(); + for (0..100) |_| try std.testing.expectEqualStrings("value", try scope.get("document:long-shared-prefix-for-compression:00")); + const keys = [_][]const u8{ "document:long-shared-prefix-for-compression:00", "document:long-shared-prefix-for-compression:01" }; + var values: [2]?[]const u8 = undefined; + try scope.getManySorted(&keys, &values); + try std.testing.expectEqualStrings("value", values[0].?); + try std.testing.expectEqualStrings("other", values[1].?); + try std.testing.expectEqual(@as(usize, 0), scope.held_blocks.items.len); + try std.testing.expectEqual(@as(usize, 0), backend.run_block_cache.items.len); + try std.testing.expectEqual(@as(u64, 0), manager.sliceStats(.lsm_block_table_cache).used_bytes); + try std.testing.expect(!backend.beginLocalBlockPromotion("/cooldown", 100, 0, 40, 40)); + try std.testing.expect(backend.beginLocalBlockPromotion("/cooldown", 100, 0, 40, 40)); + try std.testing.expect(!backend.beginLocalBlockPromotion("/cooldown", 100, 0, 40, 40)); + backend.finishLocalBlockPromotion("/cooldown", 100, 0, 40, false); + for (0..65) |_| try std.testing.expect(!backend.beginLocalBlockPromotion("/cooldown", 100, 0, 40, 40)); + try std.testing.expect(backend.beginLocalBlockPromotion("/cooldown", 100, 0, 40, 40)); + backend.finishLocalBlockPromotion("/cooldown", 100, 0, 40, true); +} + +test "lsm local bounded cache releases every failed allocation" { + const Fixture = struct { + fn run(a: Allocator) !void { + var backend = Backend.init(a, .{ .local_block_cache_bytes = 64, .local_block_cache_max_block_bytes = 64 }); + defer backend.close(); + _ = try backend.putCachedRunBlock("/one", 1, 0, 40, try a.alloc(u8, 40)); + const pinned = backend.retainCachedRunBlock("/one", 1, 0, 40); + defer if (pinned) |payload| payload.release(); + _ = try backend.putCachedRunBlock("/two", 2, 0, 40, try a.alloc(u8, 40)); + _ = try backend.putCachedRunBlock("/large", 3, 0, 65, try a.alloc(u8, 65)); + } + }; + try std.testing.checkAllAllocationFailures(std.testing.allocator, Fixture.run, .{}); +} + +test "lsm local cache participates in host reclamation without releasing pinned credit" { + const a = std.testing.allocator; + var manager = resource_manager_mod.ResourceManager.init(.{ .memory_budget = .{ .hard_limit_bytes = 512 } }); + defer manager.deinit(a); + var backend = Backend.init(a, .{ .resource_manager = &manager }); + defer backend.close(); + _ = try backend.putCachedRunBlock("/first", 1, 0, 80, try a.alloc(u8, 80)); + const pinned = backend.retainCachedRunBlock("/first", 1, 0, 80).?; + defer pinned.release(); + _ = try backend.putCachedRunBlock("/second", 2, 0, 80, try a.alloc(u8, 80)); + // Holding the backend never deadlocks an arbitrary resource requester. + platform.sync.lockYielding(&backend.mu); + const no_progress = manager.reclaimForAllocation(.dense_apply_working_set, 300); + backend.mu.unlock(); + try std.testing.expectEqual(@as(u64, 0), no_progress); + var foreground = try manager.reserve(.dense_apply_working_set, 300); + defer foreground.release(); + try std.testing.expectEqual(@as(usize, 0), backend.run_block_cache.items.len); + try std.testing.expectEqual(@as(usize, 0), backend.run_block_cache_bytes); + try std.testing.expectEqual(@as(u64, 80 + @sizeOf(SharedBytes)), manager.sliceStats(.lsm_block_table_cache).used_bytes); + try std.testing.expectEqual(@as(usize, 80), pinned.bytes.len); +} + +test "lsm local shutdown fences a pending reclaimer registration" { + const Hook = struct { + var entered: std.Io.Event = .unset; + var release: std.Io.Event = .unset; + var once = std.atomic.Value(bool).init(true); + fn alloc(_: *anyopaque, n: usize, alignment: std.mem.Alignment, ra: usize) ?[*]u8 { + if (once.swap(false, .acq_rel)) { + entered.set(std.testing.io); + release.waitUncancelable(std.testing.io); + } + return std.testing.allocator.rawAlloc(n, alignment, ra); + } + fn resize(_: *anyopaque, memory: []u8, alignment: std.mem.Alignment, n: usize, ra: usize) bool { + return std.testing.allocator.rawResize(memory, alignment, n, ra); + } + fn remap(_: *anyopaque, memory: []u8, alignment: std.mem.Alignment, n: usize, ra: usize) ?[*]u8 { + return std.testing.allocator.rawRemap(memory, alignment, n, ra); + } + fn free(_: *anyopaque, memory: []u8, alignment: std.mem.Alignment, ra: usize) void { + std.testing.allocator.rawFree(memory, alignment, ra); + } + }; + const Worker = struct { + backend: *Backend, + bytes: []u8, + result: anyerror!void = {}, + fn run(self: *@This()) void { + self.result = self.admit(); + } + fn admit(self: *@This()) !void { + platform.sync.lockYielding(&self.backend.mu); + defer self.backend.mu.unlock(); + _ = try self.backend.putCachedRunBlock("/late", 1, 0, 40, self.bytes); + } + }; + const a = std.testing.allocator; + var manager = resource_manager_mod.ResourceManager.init(.{}); + defer manager.deinit(a); + var dummy: u8 = 0; + manager.identity_allocator = .{ .ptr = &dummy, .vtable = &.{ .alloc = Hook.alloc, .resize = Hook.resize, .remap = Hook.remap, .free = Hook.free } }; + var backend = Backend.init(a, .{ .resource_manager = &manager }); + var open = true; + defer if (open) backend.close(); + var worker = Worker{ .backend = &backend, .bytes = try a.alloc(u8, 40) }; + Hook.entered.reset(); + Hook.release.reset(); + Hook.once.store(true, .release); + const admission = try std.Thread.spawn(.{}, Worker.run, .{&worker}); + var closer: ?std.Thread = null; + var joined = false; + defer if (!joined) { + Hook.release.set(std.testing.io); + admission.join(); + if (closer) |thread| thread.join(); + }; + try Hook.entered.waitTimeout(std.testing.io, .{ .duration = .{ .raw = .fromSeconds(5), .clock = .awake } }); + closer = try std.Thread.spawn(.{}, Backend.close, .{&backend}); + open = false; + const deadline = std.Io.Clock.awake.now(std.testing.io).addDuration(.fromSeconds(5)); + while (!backend.closing.load(.acquire) and std.Io.Clock.awake.now(std.testing.io).nanoseconds < deadline.nanoseconds) { + std.testing.io.sleep(.fromMilliseconds(1), .awake) catch {}; + } + const closing = backend.closing.load(.acquire); + Hook.release.set(std.testing.io); + admission.join(); + closer.?.join(); + joined = true; + try std.testing.expect(closing); + try worker.result; + for (manager.reclaimers.items) |slot| try std.testing.expectEqual(@as(u64, 0), slot.identity); + try std.testing.expectEqual(@as(u64, 0), manager.sliceStats(.lsm_block_table_cache).used_bytes); +} diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend/recovery.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend/recovery.zig index 2ea5babf50..60049dc23e 100644 --- a/zig/pkg/antfly-embedded/src/storage/lsm_backend/recovery.zig +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend/recovery.zig @@ -390,6 +390,7 @@ fn cleanup(comptime BackendType: type, backend: *BackendType, finalize_deferred: if (@hasField(BackendType, "run_block_cache")) { for (backend.run_block_cache.items) |*cached| cached.deinit(backend.allocator); backend.run_block_cache.deinit(backend.allocator); + if (@hasField(BackendType, "run_block_cache_bytes")) backend.run_block_cache_bytes = 0; } if (@hasField(BackendType, "run_table_cache")) { for (backend.run_table_cache.items) |*cached| cached.deinit(backend.allocator); diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend/runtime.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend/runtime.zig index 4cf1f7cec0..cb74919018 100644 --- a/zig/pkg/antfly-embedded/src/storage/lsm_backend/runtime.zig +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend/runtime.zig @@ -1330,11 +1330,22 @@ pub fn MergeCursor(comptime BackendType: type, comptime MutableType: type) type // budget. Bound retained amplification per result owner, then let // the caller copy values. The count cap also bounds this scan. if (self.source_blocks[source_index] == .local) { + const payload = self.source_blocks[source_index].local; + // Unadmitted payloads are transient read scratch under resource + // pressure, not an uncharged long-lived result pin. + if (!payload.result_pins_allowed) { + if (self.source_result_retention.len != 0) self.source_result_retention[source_index] = .copy; + return false; + } const max_bytes: usize = 1024 * 1024; const max_blocks: usize = 64; var bytes: usize = 0; var count: usize = 0; for (held_blocks.items) |pin| if (pin == .local) { + if (pin.local == payload) { + if (self.source_result_retention.len != 0) self.source_result_retention[source_index] = .pinned; + return true; + } bytes +|= pin.local.bytes.len; count += 1; }; @@ -7325,6 +7336,28 @@ fn findExactEntryWithLocalIndex( return try findExactEntryWithLocalIndexBlockMeta(backend, run, index, namespace, key); } +fn localBlockCacheEligible(backend: anytype, bytes: usize) bool { + if (comptime @hasDecl(@TypeOf(backend.*), "localBlockCacheEligible")) return backend.localBlockCacheEligible(bytes); + return localBlockCacheEnabled(backend); +} + +fn beginLocalBlockPromotion(backend: anytype, run: *Run, index: *const lsm_table_file.TableIndex, window: lsm_table_file.EntryDataWindow, locked: bool) bool { + if (comptime !@hasDecl(@TypeOf(backend.*), "beginLocalBlockPromotion")) return false; + if (window.compression != .prefix and window.compression != .prefix_snappy) return false; + const path = run.path orelse return false; + const held = if (locked) false else lockBackend(@TypeOf(backend.*), backend); + defer unlockBackend(@TypeOf(backend.*), backend, held); + return backend.beginLocalBlockPromotion(path, run.id, @as(u64, @intCast(index.entry_data_start)) + window.physicalRelativeOffset(), window.physicalLen(), window.len); +} + +fn finishLocalBlockPromotion(backend: anytype, run: *Run, index: *const lsm_table_file.TableIndex, window: lsm_table_file.EntryDataWindow, locked: bool, admitted: bool) void { + if (comptime !@hasDecl(@TypeOf(backend.*), "finishLocalBlockPromotion")) return; + const path = run.path orelse return; + const held = if (locked) false else lockBackend(@TypeOf(backend.*), backend); + defer unlockBackend(@TypeOf(backend.*), backend, held); + backend.finishLocalBlockPromotion(path, run.id, @as(u64, @intCast(index.entry_data_start)) + window.physicalRelativeOffset(), window.physicalLen(), admitted); +} + fn retainLocalCachedBlock(backend: anytype, run: *Run, index: *const lsm_table_file.TableIndex, window: lsm_table_file.EntryDataWindow, backend_locked: bool) ?*SharedBytes { const path = run.path orelse return null; const offset = @as(u64, @intCast(index.entry_data_start)) + window.physicalRelativeOffset(); @@ -7346,6 +7379,7 @@ fn loadLocalBlockLease(backend: anytype, run: *Run, index: *const lsm_table_file const bytes = try loadRunTableDecodedBlockWithStats(backend, backend.allocator, path, offset, len, window.compression, window.len, window.checksum); errdefer backend.allocator.free(bytes); const lease = try SharedBytes.create(backend.allocator, bytes); + lease.result_pins_allowed = backend.options.resource_manager == null; const locked = if (backend_locked) false else lockBackend(@TypeOf(backend.*), backend); defer unlockBackend(@TypeOf(backend.*), backend, locked); // A racing reader can populate the cache while decoding. Drop only our @@ -7354,10 +7388,7 @@ fn loadLocalBlockLease(backend: anytype, run: *Run, index: *const lsm_table_file lease.release(); return winner; } - backend.cacheRunBlockLease(path, run.id, offset, len, lease) catch |err| { - backend.allocator.destroy(lease); - return err; - }; + _ = backend.cacheRunBlockLease(path, run.id, offset, len, lease); return lease; } @@ -7414,7 +7445,7 @@ fn loadOwnedBlockForWindowAlloc( { const locked = lockBackend(@TypeOf(backend.*), backend); defer unlockBackend(@TypeOf(backend.*), backend, locked); - if (@hasField(@TypeOf(backend.*), "run_block_cache") and localBlockCacheEnabled(backend)) { + if (@hasField(@TypeOf(backend.*), "run_block_cache") and localBlockCacheEnabled(backend) and localBlockCacheEligible(backend, bytes.len)) { _ = try backend.putCachedRunBlock(path, run.id, absolute_offset, physical_len, try backend.allocator.dupe(u8, bytes)); } } @@ -7470,7 +7501,7 @@ fn loadOwnedBlockForWindowAllocMaybeLocked( window.checksum, ); errdefer allocator.free(bytes); - if (@hasField(@TypeOf(backend.*), "run_block_cache") and localBlockCacheEnabled(backend)) { + if (@hasField(@TypeOf(backend.*), "run_block_cache") and localBlockCacheEnabled(backend) and localBlockCacheEligible(backend, bytes.len)) { _ = try backend.putCachedRunBlock(path, run.id, absolute_offset, physical_len, try backend.allocator.dupe(u8, bytes)); } return bytes; @@ -7555,6 +7586,13 @@ fn findExactEntryWithLocalIndexBlockMeta( if (retainLocalCachedBlock(backend, run, index, window, false)) |lease| { return findExactEntryInBlockLease(lease, index, block_index, namespace, key); } + if (beginLocalBlockPromotion(backend, run, index, window, false)) { + var admitted = false; + defer finishLocalBlockPromotion(backend, run, index, window, false, admitted); + const lease = try loadLocalBlockLease(backend, run, index, window, false); + admitted = lease.cache_admitted; + return findExactEntryInBlockLease(lease, index, block_index, namespace, key); + } } if (try findExactEntryInCompressedPrefixBlock(backend, run, index, block_index, namespace, key)) |entry| { return entry; @@ -7607,6 +7645,13 @@ fn findExactEntryWithLocalIndexBlockMetaMaybeLocked( if (retainLocalCachedBlock(backend, run, index, window, true)) |lease| { return findExactEntryInBlockLease(lease, index, block_index, namespace, key); } + if (beginLocalBlockPromotion(backend, run, index, window, true)) { + var admitted = false; + defer finishLocalBlockPromotion(backend, run, index, window, true, admitted); + const lease = try loadLocalBlockLease(backend, run, index, window, true); + admitted = lease.cache_admitted; + return findExactEntryInBlockLease(lease, index, block_index, namespace, key); + } } if (try findExactEntryInCompressedPrefixBlock(backend, run, index, block_index, namespace, key)) |entry| { return entry; @@ -9027,7 +9072,11 @@ test "lsm local result block retention bounds bytes and pin metadata" { try std.testing.expect(!try cursor.retainCurrentValueForTxn(&held)); try std.testing.expectEqual(@as(usize, 0), held.items.len); blocks[0] = .{ .local = small }; - for (0..64) |_| try held.append(a, .{ .local = small.retain() }); + try held.ensureTotalCapacity(a, 64); + for (0..64) |_| { + const distinct = try SharedBytes.create(a, try a.alloc(u8, 1024)); + held.appendAssumeCapacity(.{ .local = distinct }); + } try std.testing.expect(!try cursor.retainCurrentValueForTxn(&held)); try std.testing.expectEqual(@as(usize, 64), held.items.len); } diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend/shared_bytes.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend/shared_bytes.zig index 768a9b24ac..96c5343380 100644 --- a/zig/pkg/antfly-embedded/src/storage/lsm_backend/shared_bytes.zig +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend/shared_bytes.zig @@ -22,6 +22,10 @@ pub const SharedBytes = struct { allocator: std.mem.Allocator, refs: platform.atomic.Value(usize) = .init(1), bytes: []u8, + /// Assigned before publication; credit follows the payload, not cache membership. + reservation: ?@import("../resource_manager.zig").Reservation = null, + cache_admitted: bool = false, + result_pins_allowed: bool = true, pub fn create(allocator: std.mem.Allocator, bytes: []u8) !*SharedBytes { const self = try allocator.create(SharedBytes); @@ -36,7 +40,9 @@ pub const SharedBytes = struct { if (self.refs.fetchSub(1, .acq_rel) == 1) { const allocator = self.allocator; allocator.free(self.bytes); + var reservation = self.reservation; allocator.destroy(self); + if (reservation) |*credit| credit.release(); } } }; From a3a246a4e0ee3d1ea490eae56a33fcdacff554c2 Mon Sep 17 00:00:00 2001 From: AJ Roetker Date: Thu, 8 Oct 2026 11:52:31 -0700 Subject: [PATCH 2/4] perf(lite): share bounded decode scratch and borrow write batches --- .../baselines/lite-decoded-cache-budget.json | 57 ++- .../src/storage/backend_types.zig | 3 + .../src/storage/lsm/table_file.zig | 82 +++-- .../src/storage/lsm_backend.zig | 271 ++++++++++++++- .../src/storage/lsm_backend/local_reader.zig | 325 ++++++++++++++++++ .../src/storage/lsm_backend/recovery.zig | 1 + .../src/storage/lsm_backend/runtime.zig | 320 +++++++++++------ .../lsm_backend/write_batch_scratch.zig | 63 ++++ .../src/storage/resource_manager.zig | 68 +++- 9 files changed, 1051 insertions(+), 139 deletions(-) create mode 100644 zig/pkg/antfly-embedded/src/storage/lsm_backend/local_reader.zig create mode 100644 zig/pkg/antfly-embedded/src/storage/lsm_backend/write_batch_scratch.zig diff --git a/zig/bench/baselines/lite-decoded-cache-budget.json b/zig/bench/baselines/lite-decoded-cache-budget.json index 9034e09c18..78c365b80d 100644 --- a/zig/bench/baselines/lite-decoded-cache-budget.json +++ b/zig/bench/baselines/lite-decoded-cache-budget.json @@ -9,11 +9,11 @@ "queries_after_initial_cold_read": 100, "prefix": { "reference_backend_allocator_calls": 500, - "updated_backend_allocator_calls": 8 + "updated_backend_allocator_calls": 5 }, "prefix_snappy": { "reference_backend_allocator_calls": 604, - "updated_backend_allocator_calls": 9 + "updated_backend_allocator_calls": 5 }, "reference_encoded_block_loads": 100, "updated_encoded_block_loads": 1, @@ -42,13 +42,58 @@ "evicted_pinned_payload_credit_survives_until_final_release": true, "host_pressure_reclaims_unpinned_cache_without_blocking_backend": true, "denied_admission_preserves_read_results": true, - "shutdown_fences_pending_reclaimer_registration": true + "shutdown_fences_pending_reclaimer_registration": true, + "impossible_admission_preserves_existing_cache": true }, "notes": [ "Allocation counts cover the backend allocator; owned point results still allocate in the caller's result allocator.", "Encoded block loads may hit the provider memory cache; these counts are not physical disk reads.", "Cold decoding and first promotion still allocate. Zero allocation applies to the measured warm backend path.", - "Resource accounting follows admitted payloads through eviction and result leases. Unadmitted decoded scratch remains transient, and resource-managed result owners copy its values.", - "The on-disk format and public C ABI are unchanged." - ] + "Resource accounting follows decoded scratch and uncached payload leases in the read-working-set slice, then transfers admitted payload credit to the cache slice through eviction and result leases. Resource-managed result owners copy unadmitted values.", + "The on-disk format and public C ABI are unchanged.", + "Workspace-retention capacity excludes arena bookkeeping. Backend allocation counts exclude runtime metadata supplied by its separate allocator; writer scratch reuse is checked separately." + ], + "previous_pr_commit": "1ad9a16b18be51ac35fd855b4a82e96a99b99513", + "warm_write_batches": { + "reference_commit": "1ad9a16b18be51ac35fd855b4a82e96a99b99513", + "two_key_batches_after_warmup": 100, + "reference_backend_allocator_calls": 206, + "updated_backend_allocator_calls": 0, + "reference_owned_values": 200, + "updated_owned_values": 0, + "updated_retained_pins": 1, + "writer_metadata_allocations_after_first_warm_batch": 0, + "mutable_results_preserved_after_later_publication": true, + "borrowed_results_preserved_after_cache_eviction": true + }, + "concurrent_cold_batches": { + "readers": 8, + "joined_readers": 7, + "encoded_block_loads": 1, + "shared_failure_and_success_cleanup_verified": true + }, + "transient_compressed_point_reads": { + "queries_after_workspace_warmup": 100, + "prefix_backend_allocator_calls": 100, + "prefix_snappy_backend_allocator_calls": 100, + "new_cache_admissions": 0, + "hot_cached_block_preserved": true + }, + "compressed_negative_lookup": { + "bloom_false_positive": true, + "reference_encoded_block_loads": 2, + "updated_encoded_block_loads": 1, + "updated_cache_admissions": 0 + }, + "decoder_policy": { + "workspaces": 4, + "in_flight_records": 16, + "default_active_working_byte_gate": 8388608, + "max_retained_arena_capacity_per_workspace": 65536, + "oversized_operations_run_alone": true, + "scratch_backing_allocation_byte_cap_enforced": true, + "scratch_and_uncached_lease_resource_slice": "lsm.read_working_set", + "cache_admission_transfers_host_credit_without_double_charge": true, + "max_retained_writer_batch_keys": 1024 + } } diff --git a/zig/pkg/antfly-embedded/src/storage/backend_types.zig b/zig/pkg/antfly-embedded/src/storage/backend_types.zig index c2478ee91a..c7225a244c 100644 --- a/zig/pkg/antfly-embedded/src/storage/backend_types.zig +++ b/zig/pkg/antfly-embedded/src/storage/backend_types.zig @@ -64,6 +64,9 @@ pub const Namespace = struct { /// always retained normally. This is a read hint and is deliberately not /// part of namespace identity or persisted state. block_cache_admission: BlockCacheAdmission = .retain, + /// Opt-in owner-bounded local point borrowing; ordinary point APIs keep + /// their owned-result contract. This hint is not namespace identity. + borrow_local_point_results: bool = false, pub const BlockCacheAdmission = enum { retain, diff --git a/zig/pkg/antfly-embedded/src/storage/lsm/table_file.zig b/zig/pkg/antfly-embedded/src/storage/lsm/table_file.zig index 284c11caf0..be742e4f94 100644 --- a/zig/pkg/antfly-embedded/src/storage/lsm/table_file.zig +++ b/zig/pkg/antfly-embedded/src/storage/lsm/table_file.zig @@ -2015,18 +2015,19 @@ fn encodePrefixCompressedBlockAlloc(allocator: std.mem.Allocator, block_bytes: [ fn decodePrefixCompressedBlockAlloc( allocator: std.mem.Allocator, + scratch: std.mem.Allocator, payload: []const u8, expected_len: usize, ) ![]u8 { const view = try parsePrefixBlockPayload(payload); var previous_key = std.ArrayListUnmanaged(u8).empty; - defer previous_key.deinit(allocator); + defer previous_key.deinit(scratch); var current_key = std.ArrayListUnmanaged(u8).empty; - defer current_key.deinit(allocator); + defer current_key.deinit(scratch); var out = std.ArrayListUnmanaged(u8).empty; errdefer out.deinit(allocator); - try out.ensureTotalCapacity(allocator, expected_len); + try out.ensureTotalCapacityPrecise(allocator, expected_len); var entries_cursor: usize = 0; for (0..view.entry_count) |entry_index| { @@ -2034,7 +2035,8 @@ fn decodePrefixCompressedBlockAlloc( const expected_restart = try view.restartOffset(entry_index / view.restart_interval); if (expected_restart != entries_cursor) return error.InvalidTableFile; } - const entry = try readPrefixBlockEntry(allocator, view.encoded_entries, &entries_cursor, previous_key.items, ¤t_key); + const entry = try readPrefixBlockEntry(scratch, view.encoded_entries, &entries_cursor, previous_key.items, ¤t_key); + if (try tableEntryEncodedLen(entry) > expected_len -| out.items.len) return error.InvalidTableFile; try appendEntryBytesToList(allocator, &out, .{ .namespace_name = entry.namespace_name, .key = entry.key, @@ -2043,12 +2045,12 @@ fn decodePrefixCompressedBlockAlloc( }); previous_key.clearRetainingCapacity(); - try previous_key.ensureTotalCapacity(allocator, entry.key.len); + try previous_key.ensureTotalCapacity(scratch, entry.key.len); previous_key.appendSliceAssumeCapacity(entry.key); } if (entries_cursor != view.encoded_entries.len) return error.InvalidTableFile; if (out.items.len != expected_len) return error.InvalidTableFile; - return try out.toOwnedSlice(allocator); + return out.toOwnedSliceAssert(); } const PrefixBlockView = struct { @@ -2134,6 +2136,8 @@ fn prefixRestartEntry( fn findExactEntryInPrefixPayloadAlloc( allocator: std.mem.Allocator, + scratch: std.mem.Allocator, + max_result_bytes: usize, payload: []const u8, first_entry_index: usize, namespace_name: ?[]const u8, @@ -2143,12 +2147,12 @@ fn findExactEntryInPrefixPayloadAlloc( if (view.entry_count == 0) return null; var restart_key = std.ArrayListUnmanaged(u8).empty; - defer restart_key.deinit(allocator); + defer restart_key.deinit(scratch); var lo: usize = 0; var hi: usize = view.restart_count; while (lo < hi) { const mid = lo + (hi - lo) / 2; - const entry = try prefixRestartEntry(allocator, view, mid, &restart_key); + const entry = try prefixRestartEntry(scratch, view, mid, &restart_key); if (compareEntryTo(entry, namespace_name, key) != .gt) { lo = mid + 1; } else { @@ -2164,18 +2168,21 @@ fn findExactEntryInPrefixPayloadAlloc( view.encoded_entries.len; var previous_key = std.ArrayListUnmanaged(u8).empty; - defer previous_key.deinit(allocator); + defer previous_key.deinit(scratch); var current_key = std.ArrayListUnmanaged(u8).empty; - defer current_key.deinit(allocator); + defer current_key.deinit(scratch); var entry_index = first_entry_index + restart_index * view.restart_interval; while (entries_cursor < end_cursor and entry_index < first_entry_index + view.entry_count) : (entry_index += 1) { - const entry = try readPrefixBlockEntry(allocator, view.encoded_entries, &entries_cursor, previous_key.items, ¤t_key); + const entry = try readPrefixBlockEntry(scratch, view.encoded_entries, &entries_cursor, previous_key.items, ¤t_key); const order = compareEntryTo(entry, namespace_name, key); if (order == .eq) { + const size = try tableEntryEncodedLen(entry); + if (size > max_result_bytes) return error.InvalidTableFile; var out = std.ArrayListUnmanaged(u8).empty; errdefer out.deinit(allocator); + try out.ensureTotalCapacityPrecise(allocator, size); try appendEntryBytesToList(allocator, &out, entry); - const bytes = try out.toOwnedSlice(allocator); + const bytes = out.toOwnedSliceAssert(); errdefer allocator.free(bytes); return .{ .index = entry_index, @@ -2186,7 +2193,7 @@ fn findExactEntryInPrefixPayloadAlloc( if (order == .gt) return null; previous_key.clearRetainingCapacity(); - try previous_key.ensureTotalCapacity(allocator, entry.key.len); + try previous_key.ensureTotalCapacity(scratch, entry.key.len); previous_key.appendSliceAssumeCapacity(entry.key); } if (entries_cursor != end_cursor) return error.InvalidTableFile; @@ -2201,19 +2208,39 @@ pub fn findExactEntryInCompressedBlockPayloadAlloc( first_entry_index: usize, namespace_name: ?[]const u8, key: []const u8, +) !?OwnedPositionedEntry { + return findExactEntryInCompressedBlockPayloadWithScratchAlloc(allocator, allocator, compression, payload, expected_checksum, first_entry_index, namespace_name, key, std.math.maxInt(usize)); +} + +pub fn findExactEntryInCompressedBlockPayloadWithScratchAlloc( + allocator: std.mem.Allocator, + scratch: std.mem.Allocator, + compression: BlockCompression, + payload: []const u8, + expected_checksum: u32, + first_entry_index: usize, + namespace_name: ?[]const u8, + key: []const u8, + max_result_bytes: usize, ) !?OwnedPositionedEntry { try validateBlockPayload(payload, expected_checksum); return switch (compression) { - .prefix => try findExactEntryInPrefixPayloadAlloc(allocator, payload, first_entry_index, namespace_name, key), + .prefix => try findExactEntryInPrefixPayloadAlloc(allocator, scratch, max_result_bytes, payload, first_entry_index, namespace_name, key), .prefix_snappy => blk: { - const prefix_payload = try snappy.decode(allocator, payload); - defer allocator.free(prefix_payload); - break :blk try findExactEntryInPrefixPayloadAlloc(allocator, prefix_payload, first_entry_index, namespace_name, key); + if (max_result_bytes != std.math.maxInt(usize)) try validatePrefixDecodedSize(payload, max_result_bytes); + const prefix_payload = try snappy.decode(scratch, payload); + defer scratch.free(prefix_payload); + break :blk try findExactEntryInPrefixPayloadAlloc(allocator, scratch, max_result_bytes, prefix_payload, first_entry_index, namespace_name, key); }, .none, .snappy => null, }; } +pub fn validatePrefixDecodedSize(payload: []const u8, logical_len: usize) !void { + const ceiling = std.math.add(usize, std.math.mul(usize, logical_len, 4) catch return error.InvalidTableFile, 256) catch return error.InvalidTableFile; + if (try snappy.decodedLen(payload) > ceiling) return error.InvalidTableFile; +} + fn commonPrefixLen(lhs: []const u8, rhs: []const u8) usize { const limit = @min(lhs.len, rhs.len); var index: usize = 0; @@ -2227,6 +2254,17 @@ pub fn decodeBlockPayloadAlloc( payload: []const u8, expected_len: usize, expected_checksum: u32, +) ![]u8 { + return decodeBlockPayloadWithScratchAlloc(allocator, allocator, compression, payload, expected_len, expected_checksum); +} + +pub fn decodeBlockPayloadWithScratchAlloc( + allocator: std.mem.Allocator, + scratch: std.mem.Allocator, + compression: BlockCompression, + payload: []const u8, + expected_len: usize, + expected_checksum: u32, ) ![]u8 { try validateBlockPayload(payload, expected_checksum); return switch (compression) { @@ -2235,16 +2273,18 @@ pub fn decodeBlockPayloadAlloc( break :blk try allocator.dupe(u8, payload); }, .snappy => blk: { + if (try snappy.decodedLen(payload) != expected_len) return error.InvalidTableFile; const decoded = try snappy.decode(allocator, payload); errdefer allocator.free(decoded); if (decoded.len != expected_len) return error.InvalidTableFile; break :blk decoded; }, - .prefix => try decodePrefixCompressedBlockAlloc(allocator, payload, expected_len), + .prefix => try decodePrefixCompressedBlockAlloc(allocator, scratch, payload, expected_len), .prefix_snappy => blk: { - const prefix_payload = try snappy.decode(allocator, payload); - defer allocator.free(prefix_payload); - break :blk try decodePrefixCompressedBlockAlloc(allocator, prefix_payload, expected_len); + try validatePrefixDecodedSize(payload, expected_len); + const prefix_payload = try snappy.decode(scratch, payload); + defer scratch.free(prefix_payload); + break :blk try decodePrefixCompressedBlockAlloc(allocator, scratch, prefix_payload, expected_len); }, }; } diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend.zig index 396d57d583..2ccc171457 100644 --- a/zig/pkg/antfly-embedded/src/storage/lsm_backend.zig +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend.zig @@ -46,6 +46,7 @@ const platform_time = @import("antfly_platform").time; const fs_paths = @import("antfly_runtime_fs").fs_paths; const CancellationToken = @import("antfly_cancellation").CancellationToken; const SharedBytes = @import("lsm_backend/shared_bytes.zig").SharedBytes; +const LocalReader = @import("lsm_backend/local_reader.zig").Pool; const RunSourceLease = @import("lsm_backend/source_lease.zig").Lease; const native_artifact_sink = @import("native_artifact_sink.zig"); @@ -458,6 +459,8 @@ pub const Options = struct { local_block_cache_bytes: usize = 2 * 1024 * 1024, /// Large one-value blocks stay on the direct/ephemeral path. local_block_cache_max_block_bytes: usize = 256 * 1024, + /// Cold local decoding, including one oversized operation admitted alone. + local_decode_working_bytes: usize = 8 * 1024 * 1024, max_concurrent_point_block_reads: usize = 16, // Share sparse point-read fan-out across concurrent batches. This keeps // single-query cold latency low without allowing N public queries to each @@ -1664,6 +1667,7 @@ pub const Backend = struct { run_state_cache: std.ArrayListUnmanaged(CachedRunState) = .empty, run_index_cache: std.ArrayListUnmanaged(CachedRunIndex) = .empty, run_block_cache: std.ArrayListUnmanaged(CachedRunBlock) = .empty, + local_reader: LocalReader = .{}, run_block_cache_bytes: usize = 0, local_block_cache_reclaimer: ?u64 = null, local_block_heat: [64]LocalBlockHeat = @splat(.{}), @@ -7232,26 +7236,50 @@ pub const Backend = struct { // First read/cache admission occurs at the backend's stable address. if (self.local_block_cache_reclaimer == null) self.local_block_cache_reclaimer = manager.registerReclaimer(.lsm_block_table_cache, self, reclaimLocalBlocks) catch return false; } + // Only reclaim when this owner can actually make admission succeed. + // An impossible candidate or metadata OOM must preserve useful blocks. + if (self.options.resource_manager) |manager| { + const required = payload.bytes.len + @sizeOf(SharedBytes); + var reclaimable: u64 = 0; + for (self.run_block_cache.items) |cached| { + if (cached.payload.refs.load(.acquire) == 1) { + if (cached.payload.reservation) |credit| reclaimable +|= credit.reservedBytes(); + } + } + const available = if (payload.reservation != null) blk: { + // The read payload already owns its host charge. Reclassification + // needs destination slice headroom, not another aggregate charge. + const stats = manager.sliceStats(.lsm_block_table_cache); + break :blk if (stats.hard_limit_bytes == 0) std.math.maxInt(u64) else stats.hard_limit_bytes -| stats.used_bytes; + } else manager.availableAdmissionBytes(.lsm_block_table_cache); + if (required > available +| reclaimable) return false; + } const cached_path = self.allocator.dupe(u8, path) catch return false; var admitted = false; defer if (!admitted) self.allocator.free(cached_path); self.run_block_cache.ensureUnusedCapacity(self.allocator, 1) catch return false; - while (self.run_block_cache.items.len >= max_local_cached_run_blocks or self.run_block_cache_bytes > self.options.local_block_cache_bytes - payload.bytes.len) { - _ = self.evictOldestLocalBlock(); - } if (self.options.resource_manager) |manager| { const bytes = payload.bytes.len + @sizeOf(SharedBytes); - const credit = manager.reserveWithoutReclaim(.lsm_block_table_cache, bytes) catch blk: { - // Reclaim our own cache only, never arbitrary callbacks under - // the backend lock. Pinned victims retain their resource credit. - while (self.run_block_cache.items.len != 0) { - _ = self.evictOldestLocalBlock(); - break :blk manager.reserveWithoutReclaim(.lsm_block_table_cache, bytes) catch continue; + while (true) { + if (payload.reservation) |*credit| { + credit.reclassify(.lsm_block_table_cache) catch |err| { + if (err != error.ResourceBudgetExceeded or self.run_block_cache.items.len == 0) return false; + _ = self.evictOldestLocalBlock(); + continue; + }; + } else { + const credit = manager.reserveWithoutReclaim(.lsm_block_table_cache, bytes) catch |err| { + if (err != error.ResourceBudgetExceeded or self.run_block_cache.items.len == 0) return false; + _ = self.evictOldestLocalBlock(); + continue; + }; + payload.reservation = credit; } - return false; - }; - std.debug.assert(payload.reservation == null); - payload.reservation = credit; + break; + } + } + while (self.run_block_cache.items.len >= max_local_cached_run_blocks or self.run_block_cache_bytes > self.options.local_block_cache_bytes - payload.bytes.len) { + _ = self.evictOldestLocalBlock(); } self.run_block_cache.appendAssumeCapacity(.{ .run_id = run_id, .path = cached_path, .block_offset = offset, .block_len = len, .payload = payload, .last_access = self.nextLocalCacheAccess() }); _ = payload.retain(); @@ -26033,3 +26061,220 @@ test "lsm local shutdown fences a pending reclaimer registration" { for (manager.reclaimers.items) |slot| try std.testing.expectEqual(@as(u64, 0), slot.identity); try std.testing.expectEqual(@as(u64, 0), manager.sliceStats(.lsm_block_table_cache).used_bytes); } + +test "lsm local impossible admission preserves useful cache" { + const a = std.testing.allocator; + var budgets = resource_manager_mod.Options.defaultBudgets(); + budgets[@backingInt(resource_manager_mod.Slice.lsm_block_table_cache)] = .{ .hard_limit_bytes = 256 }; + var manager = resource_manager_mod.ResourceManager.init(.{ .budgets = budgets }); + defer manager.deinit(a); + var backend = Backend.init(a, .{ .resource_manager = &manager }); + defer backend.close(); + _ = try backend.putCachedRunBlock("/useful", 1, 0, 40, try a.alloc(u8, 40)); + const rejected = try backend.putCachedRunBlock("/impossible", 2, 0, 257, try a.alloc(u8, 257)); + try std.testing.expectEqual(@as(usize, 0), rejected.len); + try std.testing.expectEqual(@as(usize, 1), backend.run_block_cache.items.len); + const useful = backend.retainCachedRunBlock("/useful", 1, 0, 40); + try std.testing.expect(useful != null); + useful.?.release(); +} + +test "lsm local writer batches borrow within owner limits and preserve mutable results" { + const a = std.testing.allocator; + var backing = storage_io.MemoryStorage.init(a); + defer backing.deinit(); + var budget = @import("lite/test_allocator.zig").BudgetAllocator{ .backing = a }; + var backend = try Backend.open(budget.allocator(), "/writer-borrow", .{ .storage = backing.storage(), .flush_threshold = 1 }); + defer backend.close(); + var runtime = try backend.runtimeStore(a, .{ .name = "docs" }); + defer runtime.deinit(); + var initial = try runtime.beginWrite(); + try initial.put("document:long-shared-prefix-for-compression:00", "value"); + try initial.put("document:long-shared-prefix-for-compression:01", "other"); + try initial.commit(); + const keys = [_][]const u8{ "document:long-shared-prefix-for-compression:00", "document:long-shared-prefix-for-compression:01" }; + var values: [2]?[]const u8 = undefined; + var read = try runtime_mod.BoundReadTxn(Backend).open(&backend, .{ .name = "docs" }); + defer read.abort(); + var scope = try read.openReadScope(a); + defer scope.close(); + try scope.getManySorted(&keys, &values); + var writer = try runtime_mod.NamespaceWriteTxn(Backend).open(&backend); + defer writer.abort(); + var metadata = @import("lite/test_allocator.zig").BudgetAllocator{ .backing = a }; + writer.metadata_allocator = metadata.allocator(); + try writer.getManySorted(.{ .name = "docs" }, &keys, &values); + const calls = budget.alloc_calls; + const owned = writer.held_values.items.len; + const scratch_keys = writer.batch_scratch.keys.items.ptr; + for (0..100) |_| { + try writer.getManySorted(.{ .name = "docs" }, &keys, &values); + try std.testing.expectEqualStrings("value", values[0].?); + try std.testing.expectEqualStrings("other", values[1].?); + } + std.debug.print("lite writer batch probe: 100 batches, backend allocations={d}, owned values={d}, pins={d}\n", .{ budget.alloc_calls - calls, writer.held_values.items.len - owned, writer.held_blocks.items.len }); + try std.testing.expectEqual(@as(usize, 0), budget.alloc_calls - calls); + try std.testing.expectEqual(owned, writer.held_values.items.len); + try std.testing.expectEqual(@as(usize, 1), writer.held_blocks.items.len); + try std.testing.expect(scratch_keys == writer.batch_scratch.keys.items.ptr); + try std.testing.expectEqual(@as(usize, 3), metadata.alloc_calls); + const borrowed = values[0].?; + const cached = backend.run_block_cache.items[0]; + backend.evictCachedRunBlocksForRun(cached.path, cached.run_id); + try std.testing.expectEqualStrings("value", borrowed); + // A current mutable hit must be copied before its captured tip is released. + backend.options.flush_threshold = std.math.maxInt(usize); + var mutation = try runtime.beginWrite(); + try mutation.put(keys[0], "new-value"); + try mutation.commit(); + try writer.getManySorted(.{ .name = "docs" }, &keys, &values); + const mutable_value = values[0].?; + try std.testing.expectEqualStrings("new-value", mutable_value); + var later = try runtime.beginWrite(); + try later.put(keys[0], "later-value"); + try later.commit(); + try std.testing.expectEqualStrings("new-value", mutable_value); +} + +test "lsm local concurrent cold batches share success and failure decodes" { + const Hook = struct { + var original: *const storage_io.Storage.VTable = undefined; + var offset: u64 = 0; + var enabled = false; + var fail = false; + var entered: std.Io.Event = .unset; + var release: std.Io.Event = .unset; + fn read(ptr: *anyopaque, a: Allocator, path: []const u8, at: u64, len: usize) anyerror![]u8 { + if (enabled and at == offset) { + entered.set(std.testing.io); + release.waitUncancelable(std.testing.io); + if (fail) return error.InjectedDecodeReadFailure; + } + return original.read_file_range_alloc(ptr, a, path, at, len); + } + fn worker(backend: *Backend) anyerror!void { + var read_tx = try runtime_mod.BoundReadTxn(Backend).open(backend, .{ .name = "docs" }); + defer read_tx.abort(); + var scope = try read_tx.openReadScope(std.testing.allocator); + defer scope.close(); + const keys = [_][]const u8{ "document:long-shared-prefix-for-compression:00", "document:long-shared-prefix-for-compression:01" }; + var values: [2]?[]const u8 = undefined; + if (fail) { + try std.testing.expectError(error.InjectedDecodeReadFailure, scope.getManySorted(&keys, &values)); + } else { + try scope.getManySorted(&keys, &values); + try std.testing.expectEqualStrings("value", values[0].?); + try std.testing.expectEqualStrings("other", values[1].?); + } + } + }; + const a = std.testing.allocator; + var backing = storage_io.MemoryStorage.init(a); + defer backing.deinit(); + var vtable = backing.storage().vtable.*; + Hook.original = backing.storage().vtable; + vtable.read_file_range_alloc = Hook.read; + Hook.enabled = false; + var backend = try Backend.open(a, "/cold-shared", .{ .storage = .{ .ptr = backing.storage().ptr, .vtable = &vtable }, .flush_threshold = 1 }); + defer backend.close(); + var runtime = try backend.runtimeStore(a, .{ .name = "docs" }); + defer runtime.deinit(); + var write = try runtime.beginWrite(); + try write.put("document:long-shared-prefix-for-compression:00", "value"); + try write.put("document:long-shared-prefix-for-compression:01", "other"); + try write.commit(); + const run = backend.runs.at(0); + const cached_index = try backend.getCachedRunIndexIndex(run.path.?, run.id); + const index = backend.getCachedRunIndexByIndex(cached_index); + Hook.offset = @intCast(index.entry_data_start); + Hook.enabled = true; + defer Hook.enabled = false; + for ([_]bool{ true, false }) |fail| { + Hook.fail = fail; + Hook.entered.reset(); + Hook.release.reset(); + const loads = backend.read_stats.table_block_loads.load(.monotonic); + const joined = backend.local_reader.joined; + var workers: [8]std.Io.Future(anyerror!void) = undefined; + var started: usize = 0; + var done = false; + defer if (!done) { + Hook.release.set(std.testing.io); + for (workers[0..started]) |*worker| worker.await(std.testing.io) catch {}; + }; + workers[0] = try std.testing.io.concurrent(Hook.worker, .{&backend}); + started = 1; + try Hook.entered.waitTimeout(std.testing.io, .{ .duration = .{ .raw = .fromSeconds(5), .clock = .awake } }); + for (1..8) |i| { + workers[i] = try std.testing.io.concurrent(Hook.worker, .{&backend}); + started += 1; + } + const deadline = std.Io.Clock.awake.now(std.testing.io).addDuration(.fromSeconds(5)); + var joined_now: usize = 0; + while (std.Io.Clock.awake.now(std.testing.io).nanoseconds < deadline.nanoseconds) { + @import("antfly_platform").sync.lockYielding(&backend.local_reader.mutex); + joined_now = backend.local_reader.joined - joined; + backend.local_reader.mutex.unlock(); + if (joined_now == 7) break; + platform_time.yieldBriefly(); + } + Hook.release.set(std.testing.io); + var failure: ?anyerror = null; + for (&workers) |*worker| worker.await(std.testing.io) catch |err| { + if (failure == null) failure = err; + }; + done = true; + if (failure) |err| return err; + try std.testing.expectEqual(@as(usize, 7), joined_now); + const extra = backend.read_stats.table_block_loads.load(.monotonic) - loads; + std.debug.print("lite shared cold decode probe: failure={any}, readers=8, joined={d}, encoded loads={d}\n", .{ fail, joined_now, extra }); + try std.testing.expectEqual(@as(u64, 1), extra); + try std.testing.expectEqual(@as(usize, 0), backend.local_reader.active); + } +} + +test "lsm local transient reads preserve hot cache and reuse compressed point scratch" { + const cases = [_][]const u8{ "other", z17RepeatString("compressible-value:", 128) }; + for (cases) |value| { + const a = std.testing.allocator; + var storage = storage_io.MemoryStorage.init(a); + defer storage.deinit(); + var budget = @import("lite/test_allocator.zig").BudgetAllocator{ .backing = a }; + var backend = try Backend.open(budget.allocator(), "/scan-admission", .{ .storage = storage.storage(), .flush_threshold = 1 }); + defer backend.close(); + var runtime = try backend.runtimeStore(a, .{ .name = "scan" }); + defer runtime.deinit(); + var write = try runtime.beginWrite(); + try write.put("document:long-shared-prefix-for-compression:00", "value"); + try write.put("document:long-shared-prefix-for-compression:01", value); + try write.commit(); + // An unrelated useful payload must not be displaced by this scan. + _ = try backend.putCachedRunBlock("/hot", 999, 0, 32, try backend.allocator.alloc(u8, 32)); + var read = try runtime_mod.BoundReadTxn(Backend).open(&backend, .{ .name = "scan", .block_cache_admission = .transient }); + defer read.abort(); + var point = try read.openReadScope(a); + defer point.close(); + try std.testing.expectEqualStrings(value, try point.get("document:long-shared-prefix-for-compression:01")); + const calls = budget.alloc_calls; + const loads = backend.read_stats.table_block_loads.load(.monotonic); + for (0..100) |_| try std.testing.expectEqualStrings(value, try point.get("document:long-shared-prefix-for-compression:01")); + const extra = budget.alloc_calls - calls; + std.debug.print("lite transient point scratch probe: 100 reads, backend allocations={d}, encoded loads={d}\n", .{ extra, backend.read_stats.table_block_loads.load(.monotonic) - loads }); + // Only compact owned point entries allocate; encoded input and prefix + // reconstruction/snappy intermediates reuse the bounded workspace. + try std.testing.expectEqual(@as(usize, 100), extra); + const keys = [_][]const u8{ "document:long-shared-prefix-for-compression:00", "document:long-shared-prefix-for-compression:01" }; + var values: [2]?[]const u8 = undefined; + var scan = try read.openReadScope(a); + defer scan.close(); + try scan.getManySorted(&keys, &values); + try std.testing.expectEqualStrings("value", values[0].?); + try std.testing.expectEqualStrings(value, values[1].?); + try std.testing.expectEqual(@as(usize, 1), backend.run_block_cache.items.len); + try std.testing.expectEqual(@as(u64, 999), backend.run_block_cache.items[0].run_id); + try std.testing.expectEqual(@as(usize, 32), backend.run_block_cache_bytes); + for (backend.local_reader.slots) |slot| if (slot.initialized) { + try std.testing.expect(slot.cap.live <= LocalReader.retained_bytes_per_workspace + @sizeOf(usize) * 4); + }; + } +} diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend/local_reader.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend/local_reader.zig new file mode 100644 index 0000000000..3e6810fb36 --- /dev/null +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend/local_reader.zig @@ -0,0 +1,325 @@ +// Copyright 2026 Antfly, Inc. +// SPDX-License-Identifier: Apache-2.0 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +const std = @import("std"); +const platform = @import("antfly_platform"); +const resources = @import("../resource_manager.zig"); +const SharedBytes = @import("shared_bytes.zig").SharedBytes; + +/// A fixed number of decoder workspaces and flight records. Workspace storage +/// is independent of result storage; releasing scratch never invalidates a row. +pub const Pool = struct { + pub const workspace_count = 4; + pub const flight_count = 16; + pub const retained_bytes_per_workspace = 64 * 1024; + const Waiter = struct { + next: ?*Waiter = null, + io: ?std.Io, + done: std.Io.Event = .unset, + notified: std.atomic.Value(bool) = .init(false), + }; + const Capped = struct { + backing: std.mem.Allocator = undefined, + live: usize = 0, + limit: usize = 0, + fn allocator(self: *Capped) std.mem.Allocator { + return .{ .ptr = self, .vtable = &.{ .alloc = alloc, .resize = resize, .remap = remap, .free = free } }; + } + fn alloc(raw: *anyopaque, len: usize, alignment: std.mem.Alignment, ra: usize) ?[*]u8 { + const self: *Capped = @ptrCast(@alignCast(raw)); + if (len > self.limit -| self.live) return null; + const ptr = self.backing.rawAlloc(len, alignment, ra) orelse return null; + self.live += len; + return ptr; + } + fn resize(raw: *anyopaque, memory: []u8, alignment: std.mem.Alignment, len: usize, ra: usize) bool { + const self: *Capped = @ptrCast(@alignCast(raw)); + if (len > memory.len and len - memory.len > self.limit -| self.live) return false; + if (!self.backing.rawResize(memory, alignment, len, ra)) return false; + self.live = self.live - memory.len + len; + return true; + } + fn remap(raw: *anyopaque, memory: []u8, alignment: std.mem.Alignment, len: usize, ra: usize) ?[*]u8 { + const self: *Capped = @ptrCast(@alignCast(raw)); + if (len > memory.len and len - memory.len > self.limit -| self.live) return null; + const ptr = self.backing.rawRemap(memory, alignment, len, ra) orelse return null; + self.live = self.live - memory.len + len; + return ptr; + } + fn free(raw: *anyopaque, memory: []u8, alignment: std.mem.Alignment, ra: usize) void { + const self: *Capped = @ptrCast(@alignCast(raw)); + self.backing.rawFree(memory, alignment, ra); + self.live -= memory.len; + } + }; + const Slot = struct { + busy: bool = false, + initialized: bool = false, + arena: std.heap.ArenaAllocator = undefined, + budget: ?resources.BudgetedAllocator = null, + cap: Capped = .{}, + }; + pub const Flight = struct { + refs: usize = 0, + path: []u8 = &.{}, + run_id: u64 = 0, + offset: u64 = 0, + len: u32 = 0, + admit: bool = false, + completed: bool = false, + io: ?std.Io = null, + done: std.Io.Event = .unset, + result: ?*SharedBytes = null, + failure: ?anyerror = null, + }; + mutex: std.atomic.Mutex = .unlocked, + slots: [workspace_count]Slot = @splat(.{}), + flights: [flight_count]Flight = @splat(.{}), + waiters: ?*Waiter = null, + active_bytes: usize = 0, + active: usize = 0, + peak_active_bytes: usize = 0, + joined: usize = 0, + allocator: ?std.mem.Allocator = null, + + fn notifyLocked(self: *Pool) void { + var next = self.waiters; + self.waiters = null; + while (next) |waiter| { + next = waiter.next; + waiter.notified.store(true, .release); + if (waiter.io) |io| waiter.done.set(io); + } + } + /// Caller owns mutex; this returns with it unlocked. The stack waiter is + /// detached before notification, so neither reuse nor teardown races it. + fn waitLocked(self: *Pool, io: ?std.Io) void { + var waiter = Waiter{ .next = self.waiters, .io = io }; + self.waiters = &waiter; + self.mutex.unlock(); + if (io) |owned| waiter.done.waitUncancelable(owned) else { + while (!waiter.notified.load(.acquire)) platform.time.yieldBriefly(); + } + } + + pub const Workspace = struct { + pool: *Pool, + slot: *Slot, + bytes: usize, + pub fn allocator(self: *const Workspace) std.mem.Allocator { + return self.slot.arena.allocator(); + } + pub fn release(self: *Workspace) void { + _ = self.slot.arena.reset(.{ .retain_with_limit = retained_bytes_per_workspace }); + if (self.slot.budget) |*budget| _ = budget.releaseUnusedCredit(); + platform.sync.lockYielding(&self.pool.mutex); + self.slot.busy = false; + self.pool.active_bytes -= self.bytes; + self.pool.active -= 1; + self.pool.notifyLocked(); + self.pool.mutex.unlock(); + self.* = undefined; + } + }; + + /// Ordinary operations share the byte ceiling. One operation larger than + /// the ceiling runs alone, preserving support for legitimate large rows. + pub fn acquire(self: *Pool, backing: std.mem.Allocator, manager: ?*resources.ResourceManager, io: ?std.Io, bytes: usize, limit: usize, output_bytes: usize) Workspace { + while (true) { + platform.sync.lockYielding(&self.mutex); + if (self.active == 0 or bytes <= limit -| self.active_bytes) { + for (&self.slots) |*slot| if (!slot.busy) { + slot.busy = true; + self.active += 1; + self.active_bytes += bytes; + self.peak_active_bytes = @max(self.peak_active_bytes, self.active_bytes); + if (self.allocator == null) self.allocator = backing; + if (!slot.initialized) { + if (manager) |host| { + slot.budget = resources.BudgetedAllocator.init(host, .lsm_read_working_set, backing, 1); + slot.budget.?.credit_quantum = 4096; + } + slot.cap.backing = if (slot.budget) |*budget| budget.allocator() else backing; + slot.arena = .init(slot.cap.allocator()); + slot.initialized = true; + } + slot.cap.limit = @max(slot.cap.live, bytes -| output_bytes); + self.mutex.unlock(); + return .{ .pool = self, .slot = slot, .bytes = bytes }; + }; + } + self.waitLocked(io); + } + } + + pub const Ticket = struct { flight: *Flight, leader: bool }; + /// Called without the writer mutex. Flight exhaustion waits for a bounded + /// record instead of allocating an unbounded coordination structure. + pub fn begin(self: *Pool, backing: std.mem.Allocator, io: ?std.Io, path: []const u8, run_id: u64, offset: u64, len: u32, admit: bool) !Ticket { + while (true) { + platform.sync.lockYielding(&self.mutex); + for (&self.flights) |*flight| { + if (flight.refs != 0 and flight.run_id == run_id and flight.offset == offset and flight.len == len and flight.admit == admit and std.mem.eql(u8, flight.path, path)) { + flight.refs += 1; + self.joined += 1; + self.mutex.unlock(); + return .{ .flight = flight, .leader = false }; + } + } + for (&self.flights) |*flight| if (flight.refs == 0) { + const owned = backing.dupe(u8, path) catch |err| { + self.mutex.unlock(); + return err; + }; + if (self.allocator == null) self.allocator = backing; + flight.* = .{ .refs = 1, .path = owned, .run_id = run_id, .offset = offset, .len = len, .admit = admit, .io = io }; + self.mutex.unlock(); + return .{ .flight = flight, .leader = true }; + }; + self.waitLocked(io); + } + } + pub fn join(self: *Pool, path: []const u8, run_id: u64, offset: u64, len: u32, admit: bool) ?*Flight { + platform.sync.lockYielding(&self.mutex); + defer self.mutex.unlock(); + for (&self.flights) |*flight| { + if (flight.refs != 0 and flight.run_id == run_id and flight.offset == offset and flight.len == len and flight.admit == admit and std.mem.eql(u8, flight.path, path)) { + flight.refs += 1; + self.joined += 1; + return flight; + } + } + return null; + } + pub fn finish(self: *Pool, flight: *Flight, result: anyerror!*SharedBytes) void { + platform.sync.lockYielding(&self.mutex); + if (result) |payload| flight.result = payload.retain() else |err| flight.failure = err; + flight.completed = true; + if (flight.io) |io| flight.done.set(io); + self.mutex.unlock(); + } + pub fn wait(self: *Pool, flight: *Flight) !*SharedBytes { + if (flight.io) |io| flight.done.waitUncancelable(io) else { + while (true) { + platform.sync.lockYielding(&self.mutex); + const completed = flight.completed; + self.mutex.unlock(); + if (completed) break; + platform.time.yieldBriefly(); + } + } + if (flight.failure) |err| return err; + return flight.result.?.retain(); + } + pub fn releaseFlight(self: *Pool, flight: *Flight) void { + platform.sync.lockYielding(&self.mutex); + flight.refs -= 1; + if (flight.refs != 0) { + self.mutex.unlock(); + return; + } + const result = flight.result; + const path = flight.path; + flight.result = null; + flight.path = &.{}; + self.notifyLocked(); + self.mutex.unlock(); + // Final result/free callbacks never run under the coordination lock. + if (result) |payload| payload.release(); + self.allocator.?.free(path); + } + pub fn deinit(self: *Pool) void { + std.debug.assert(self.active == 0 and self.waiters == null); + for (&self.flights) |*flight| std.debug.assert(flight.refs == 0); + for (&self.slots) |*slot| if (slot.initialized) { + slot.arena.deinit(); + if (slot.budget) |*budget| budget.deinit(); + }; + self.* = .{}; + } +}; + +test "lsm local decoder scratch reuse is bounded charged and failure-safe" { + const Fixture = struct { + fn run(a: std.mem.Allocator) !void { + var manager = resources.ResourceManager.init(.{}); + defer manager.deinit(std.testing.allocator); + var pool: Pool = .{}; + var open = true; + defer if (open) pool.deinit(); + for (0..2) |_| { + var work = pool.acquire(a, &manager, std.testing.io, 256 * 1024, 1024 * 1024, 0); + defer work.release(); + const bytes = try work.allocator().alloc(u8, 1024); + @memset(bytes, 123); + try std.testing.expectEqual(@as(u8, 123), bytes[0]); + } + try std.testing.expectEqual(@as(u64, pool.slots[0].cap.live), manager.sliceStats(.lsm_read_working_set).used_bytes); + try std.testing.expect(pool.slots[0].cap.live <= Pool.retained_bytes_per_workspace + @sizeOf(usize) * 4); + { + var work = pool.acquire(a, &manager, std.testing.io, 256 * 1024, 1024 * 1024, 0); + defer work.release(); + try std.testing.expectError(error.OutOfMemory, work.allocator().alloc(u8, 512 * 1024)); + } + pool.deinit(); + open = false; + try std.testing.expectEqual(@as(u64, 0), manager.sliceStats(.lsm_read_working_set).used_bytes); + } + }; + try std.testing.checkAllAllocationFailures(std.testing.allocator, Fixture.run, .{}); +} + +test "lsm local decoder byte gate admits oversized work alone and wakes waiters" { + const Worker = struct { + pool: *Pool, + entered: std.Io.Event = .unset, + fn run(self: *@This()) void { + var work = self.pool.acquire(std.testing.allocator, null, std.testing.io, 64 * 1024, 64 * 1024, 0); + self.entered.set(std.testing.io); + work.release(); + } + }; + var pool: Pool = .{}; + defer pool.deinit(); + var big = pool.acquire(std.testing.allocator, null, std.testing.io, 128 * 1024, 64 * 1024, 0); + var released = false; + defer if (!released) big.release(); + var worker = Worker{ .pool = &pool }; + const thread = try std.Thread.spawn(.{}, Worker.run, .{&worker}); + var joined = false; + defer if (!joined) { + if (!released) { + big.release(); + released = true; + } + thread.join(); + }; + const deadline = std.Io.Clock.awake.now(std.testing.io).addDuration(.fromSeconds(5)); + var waiting = false; + while (std.Io.Clock.awake.now(std.testing.io).nanoseconds < deadline.nanoseconds) { + platform.sync.lockYielding(&pool.mutex); + waiting = pool.waiters != null; + pool.mutex.unlock(); + if (waiting) break; + platform.time.yieldBriefly(); + } + big.release(); + released = true; + thread.join(); + joined = true; + try std.testing.expect(waiting); + try std.testing.expectEqual(@as(usize, 128 * 1024), pool.peak_active_bytes); + try std.testing.expectEqual(@as(usize, 0), pool.active); +} diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend/recovery.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend/recovery.zig index 60049dc23e..af51ac573b 100644 --- a/zig/pkg/antfly-embedded/src/storage/lsm_backend/recovery.zig +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend/recovery.zig @@ -387,6 +387,7 @@ fn cleanup(comptime BackendType: type, backend: *BackendType, finalize_deferred: backend.run_index_cache.deinit(backend.allocator); } if (@hasDecl(BackendType, "deinitRunSources")) backend.deinitRunSources(); + if (@hasField(BackendType, "local_reader")) backend.local_reader.deinit(); if (@hasField(BackendType, "run_block_cache")) { for (backend.run_block_cache.items) |*cached| cached.deinit(backend.allocator); backend.run_block_cache.deinit(backend.allocator); diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend/runtime.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend/runtime.zig index cb74919018..b491e25144 100644 --- a/zig/pkg/antfly-embedded/src/storage/lsm_backend/runtime.zig +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend/runtime.zig @@ -88,6 +88,7 @@ const VisibleBytes = union(enum) { }; const SharedBytes = @import("shared_bytes.zig").SharedBytes; +const LocalReader = @import("local_reader.zig").Pool; const RunSourceLease = @import("source_lease.zig").Lease; /// Transaction result pins share the same lifetime contract for both caches. @@ -103,6 +104,23 @@ const BlockPin = union(enum) { } }; +/// One owner-wide bound for cursor and point-result borrowing. +fn retainLocalResultPin(backend: anytype, payload: *SharedBytes, held: *std.ArrayListUnmanaged(BlockPin)) !bool { + if (!payload.result_pins_allowed) return false; + var bytes: usize = 0; + var count: usize = 0; + for (held.items) |pin| if (pin == .local) { + if (pin.local == payload) return true; + bytes +|= pin.local.bytes.len; + count += 1; + }; + if (count >= 64 or payload.bytes.len > (1024 * 1024) -| bytes) return false; + const owned = payload.retain(); + errdefer owned.release(); + try held.append(backend.allocator, .{ .local = owned }); + return true; +} + const ResultBlockRetention = enum { unknown, pinned, copy }; const SourceBlockLease = union(enum) { @@ -1330,29 +1348,9 @@ pub fn MergeCursor(comptime BackendType: type, comptime MutableType: type) type // budget. Bound retained amplification per result owner, then let // the caller copy values. The count cap also bounds this scan. if (self.source_blocks[source_index] == .local) { - const payload = self.source_blocks[source_index].local; - // Unadmitted payloads are transient read scratch under resource - // pressure, not an uncharged long-lived result pin. - if (!payload.result_pins_allowed) { - if (self.source_result_retention.len != 0) self.source_result_retention[source_index] = .copy; - return false; - } - const max_bytes: usize = 1024 * 1024; - const max_blocks: usize = 64; - var bytes: usize = 0; - var count: usize = 0; - for (held_blocks.items) |pin| if (pin == .local) { - if (pin.local == payload) { - if (self.source_result_retention.len != 0) self.source_result_retention[source_index] = .pinned; - return true; - } - bytes +|= pin.local.bytes.len; - count += 1; - }; - if (count >= max_blocks or self.source_blocks[source_index].local.bytes.len > max_bytes -| bytes) { - if (self.source_result_retention.len != 0) self.source_result_retention[source_index] = .copy; - return false; - } + const pinned = try retainLocalResultPin(self.backend, self.source_blocks[source_index].local, held_blocks); + if (self.source_result_retention.len != 0) self.source_result_retention[source_index] = if (pinned) .pinned else .copy; + return pinned; } if (self.source_blocks[source_index].retainPin()) |retained_block| { var retained = retained_block; @@ -2225,7 +2223,7 @@ pub fn MergeCursor(comptime BackendType: type, comptime MutableType: type) type self.source_blocks[source_index] = .{ .cached = handle }; break :blk block; } else if (localBlockCacheEnabled(self.backend)) blk: { - const payload = try loadLocalBlockLease(self.backend, run, index, window, self.backend_locked); + const payload = try loadLocalBlockLease(self.backend, run, index, window, self.backend_locked, self.namespace.retainDataBlocks()); self.source_blocks[source_index] = .{ .local = payload }; break :blk payload.bytes; } else blk: { @@ -2945,7 +2943,7 @@ fn getFromRunPointRetainedLocked( return located.entry.value; } } - const value = try getFromRunWithLocalIndex(backend, run, held_values, value_allocator, namespace, key, true) orelse return null; + const value = try getFromRunWithLocalIndex(backend, run, held_blocks, held_values, value_allocator, namespace, key, true) orelse return null; recordPointValueCopy(backend); if (run.level == 0) backend.recordL0Hit() else backend.recordLevelHit(); return value; @@ -3076,7 +3074,7 @@ fn readManyCurrentSortedPointByRunLocked( continue; } break :blk located.entry.value; - } else getFromRunWithLocalIndex(backend, run, held_values, allocator, namespace, keys[key_index], true) catch |err| switch (err) { + } else getFromRunWithLocalIndex(backend, run, held_blocks, held_values, allocator, namespace, keys[key_index], true) catch |err| switch (err) { error.NotFound => { resolved[key_index] = true; result.misses += 1; @@ -3857,10 +3855,48 @@ pub fn BoundProbeTxn(comptime BackendType: type) type { return owned; } - fn ownValues(self: *@This(), values: []?[]const u8) !void { - for (values) |*value| { - const present = value.* orelse continue; - value.* = try self.ownValue(present); + fn ownUnpinnedValues(self: *@This(), values: []?[]const u8, first_owned: usize) !void { + const Range = struct { + base: usize, + len: usize, + fn less(_: void, lhs: @This(), rhs: @This()) bool { + return lhs.base < rhs.base; + } + }; + const count = self.held_values.items.len - first_owned; + var inline_ranges: [128]Range = undefined; + const ranges = if (count <= inline_ranges.len) inline_ranges[0..count] else try self.metadata_allocator.alloc(Range, count); + defer if (count > inline_ranges.len) self.metadata_allocator.free(ranges); + for (self.held_values.items[first_owned..], ranges) |owned, *range| range.* = .{ .base = @intFromPtr(owned.ptr), .len = owned.len }; + std.mem.sort(Range, ranges, {}, Range.less); + for (values) |*slot| { + const value = slot.* orelse continue; + const address = @intFromPtr(value.ptr); + var retained = false; + for (self.held_blocks.items) |*pin| { + const bytes = switch (pin.*) { + .local => |payload| payload.bytes, + .cached => |*handle| handle.runTableBlock(), + }; + const base = @intFromPtr(bytes.ptr); + if (address >= base and address - base <= bytes.len and value.len <= bytes.len - (address - base)) { + retained = true; + break; + } + } + if (retained) continue; + var lo: usize = 0; + var hi = ranges.len; + while (lo < hi) { + const mid = lo + (hi - lo) / 2; + if (ranges[mid].base <= address) lo = mid + 1 else hi = mid; + } + if (lo != 0) { + const range = ranges[lo - 1]; + if (address - range.base <= range.len and value.len <= range.len - (address - range.base)) continue; + } + slot.* = try self.ownValue(value); + recordPointValueCopy(self.backend); } } @@ -4003,6 +4039,7 @@ pub fn BoundProbeTxn(comptime BackendType: type) type { self.backend.recordGetManySortedLocality(keys); } + const first_owned = self.held_values.items.len; var result: BatchCursorReadResult = .{}; if (self.stable_point_view) { var offset: usize = 0; @@ -4141,10 +4178,7 @@ pub fn BoundProbeTxn(comptime BackendType: type) type { false, ), }; - try self.ownValues(unresolved_values); - for (unresolved_values) |value| { - if (value != null) recordPointValueCopy(self.backend); - } + try self.ownUnpinnedValues(unresolved_values, first_owned); for (unresolved_values, unresolved_indexes) |value, index| values[index] = value; result.add(unresolved_result); } @@ -5925,7 +5959,7 @@ fn readPointRunCandidate( }; return located.entry.value; } - if (try getFromRunWithLocalIndex(backend, run, held_values, value_allocator, namespace, key, backend_locked)) |value| { + if (try getFromRunWithLocalIndex(backend, run, held_blocks, held_values, value_allocator, namespace, key, backend_locked)) |value| { read_hint.* = null; return value; } @@ -6610,7 +6644,7 @@ fn getFromRunIndices( }; return located.entry.value; } - if (try getFromRunWithLocalIndex(backend, run, held_values, value_allocator, namespace, key, backend_locked)) |value| { + if (try getFromRunWithLocalIndex(backend, run, held_blocks, held_values, value_allocator, namespace, key, backend_locked)) |value| { read_hint.* = null; return value; } @@ -7311,7 +7345,7 @@ fn copyTableEntry(allocator: Allocator, entry: lsm_table_file.Entry) !OwnedTable } fn findExactEntryInLocalLease(backend: anytype, run: *Run, index: *const lsm_table_file.TableIndex, window: lsm_table_file.EntryDataWindow, block_index: usize, namespace: backend_types.Namespace, key: []const u8, locked: bool) !?OwnedTableEntry { - const lease = try loadLocalBlockLease(backend, run, index, window, locked); + const lease = try loadLocalBlockLease(backend, run, index, window, locked, namespace.retainDataBlocks()); return findExactEntryInBlockLease(lease, index, block_index, namespace, key); } @@ -7370,28 +7404,88 @@ fn retainLocalCachedBlock(backend: anytype, run: *Run, index: *const lsm_table_f return null; } -fn loadLocalBlockLease(backend: anytype, run: *Run, index: *const lsm_table_file.TableIndex, window: lsm_table_file.EntryDataWindow, backend_locked: bool) !*SharedBytes { +fn localDecodeWorkingBytes(window: lsm_table_file.EntryDataWindow) !usize { + const logical = std.math.mul(usize, window.len, 24) catch return error.InvalidTableFile; + const physical = std.math.mul(usize, window.physicalLen(), 4) catch return error.InvalidTableFile; + return std.math.add(usize, std.math.add(usize, logical, physical) catch return error.InvalidTableFile, LocalReader.retained_bytes_per_workspace + @sizeOf(SharedBytes)) catch return error.InvalidTableFile; +} + +fn localWorkspace(backend: anytype, window: lsm_table_file.EntryDataWindow) !LocalReader.Workspace { + return backend.local_reader.acquire(backend.allocator, backend.options.resource_manager, backend.manifestCoordinationIo(), try localDecodeWorkingBytes(window), backend.options.local_decode_working_bytes, window.len + @sizeOf(SharedBytes)); +} + +fn loadDecodedLocalBlock(backend: anytype, allocator: Allocator, path: []const u8, offset: u64, window: lsm_table_file.EntryDataWindow) ![]u8 { + if (comptime @hasField(@TypeOf(backend.*), "local_reader")) { + var work = try localWorkspace(backend, window); + defer work.release(); + const scratch = work.allocator(); + const payload = try loadRunTableBlockWithStats(backend, scratch, path, offset, window.physicalLen()); + return lsm_table_file.decodeBlockPayloadWithScratchAlloc(allocator, scratch, window.compression, payload, window.len, window.checksum); + } + return loadRunTableDecodedBlockWithStats(backend, allocator, path, offset, window.physicalLen(), window.compression, window.len, window.checksum); +} + +fn loadLocalBlockCandidate(backend: anytype, run: *Run, index: *const lsm_table_file.TableIndex, window: lsm_table_file.EntryDataWindow, backend_locked: bool, admit: bool) !*SharedBytes { const path = run.path orelse return error.RunStateUnavailable; const offset = @as(u64, @intCast(index.entry_data_start)) + window.physicalRelativeOffset(); const len = window.physicalLen(); if (retainLocalCachedBlock(backend, run, index, window, backend_locked)) |lease| return lease; backend.recordLocalBlockCacheMiss(); - const bytes = try loadRunTableDecodedBlockWithStats(backend, backend.allocator, path, offset, len, window.compression, window.len, window.checksum); + // Release decoder admission before taking the writer mutex to publish. + // A writer may wait for a workspace, but never for a cache publisher. + var read_credit: ?@import("../resource_manager.zig").Reservation = if (backend.options.resource_manager) |manager| + try manager.reserveWithoutReclaim(.lsm_read_working_set, @as(usize, window.len) + @sizeOf(SharedBytes)) + else + null; + defer if (read_credit) |*credit| credit.release(); + const bytes = try loadDecodedLocalBlock(backend, backend.allocator, path, offset, window); errdefer backend.allocator.free(bytes); const lease = try SharedBytes.create(backend.allocator, bytes); + lease.reservation = read_credit; + read_credit = null; lease.result_pins_allowed = backend.options.resource_manager == null; const locked = if (backend_locked) false else lockBackend(@TypeOf(backend.*), backend); defer unlockBackend(@TypeOf(backend.*), backend, locked); - // A racing reader can populate the cache while decoding. Drop only our - // candidate, retaining the winner rather than growing duplicate entries. if (backend.retainCachedRunBlock(path, run.id, offset, len)) |winner| { lease.release(); return winner; } - _ = backend.cacheRunBlockLease(path, run.id, offset, len, lease); + if (admit) _ = backend.cacheRunBlockLease(path, run.id, offset, len, lease); return lease; } +fn joinLocalBlockDecode(backend: anytype, run: *Run, index: *const lsm_table_file.TableIndex, window: lsm_table_file.EntryDataWindow) !?*SharedBytes { + if (comptime @hasField(@TypeOf(backend.*), "local_reader")) { + const path = run.path orelse return null; + const offset = @as(u64, @intCast(index.entry_data_start)) + window.physicalRelativeOffset(); + const flight = backend.local_reader.join(path, run.id, offset, window.physicalLen(), true) orelse return null; + defer backend.local_reader.releaseFlight(flight); + return try backend.local_reader.wait(flight); + } + return null; +} + +fn loadLocalBlockLease(backend: anytype, run: *Run, index: *const lsm_table_file.TableIndex, window: lsm_table_file.EntryDataWindow, backend_locked: bool, admit: bool) !*SharedBytes { + if (retainLocalCachedBlock(backend, run, index, window, backend_locked)) |lease| return lease; + if (comptime @hasField(@TypeOf(backend.*), "local_reader")) { + // Joining a publisher while holding the writer mutex would deadlock. + // Locked callers decode independently through the same bounded pool. + if (!backend_locked) { + const path = run.path orelse return error.RunStateUnavailable; + const offset = @as(u64, @intCast(index.entry_data_start)) + window.physicalRelativeOffset(); + const ticket = backend.local_reader.begin(backend.allocator, backend.manifestCoordinationIo(), path, run.id, offset, window.physicalLen(), admit) catch null; + if (ticket) |owned| { + defer backend.local_reader.releaseFlight(owned.flight); + if (!owned.leader) return backend.local_reader.wait(owned.flight); + const payload = loadLocalBlockCandidate(backend, run, index, window, false, admit); + backend.local_reader.finish(owned.flight, payload); + return payload; + } + } + } + return loadLocalBlockCandidate(backend, run, index, window, backend_locked, admit); +} + fn loadOwnedBlockForWindowAlloc( backend: anytype, allocator: Allocator, @@ -7431,16 +7525,7 @@ fn loadOwnedBlockForWindowAlloc( backend.recordLocalBlockCacheMiss(); } - const bytes = try loadRunTableDecodedBlockWithStats( - backend, - allocator, - path, - absolute_offset, - physical_len, - window.compression, - window.len, - window.checksum, - ); + const bytes = try loadDecodedLocalBlock(backend, allocator, path, absolute_offset, window); errdefer allocator.free(bytes); { const locked = lockBackend(@TypeOf(backend.*), backend); @@ -7490,16 +7575,7 @@ fn loadOwnedBlockForWindowAllocMaybeLocked( backend.recordLocalBlockCacheMiss(); } - const bytes = try loadRunTableDecodedBlockWithStats( - backend, - allocator, - path, - absolute_offset, - physical_len, - window.compression, - window.len, - window.checksum, - ); + const bytes = try loadDecodedLocalBlock(backend, allocator, path, absolute_offset, window); errdefer allocator.free(bytes); if (@hasField(@TypeOf(backend.*), "run_block_cache") and localBlockCacheEnabled(backend) and localBlockCacheEligible(backend, bytes.len)) { _ = try backend.putCachedRunBlock(path, run.id, absolute_offset, physical_len, try backend.allocator.dupe(u8, bytes)); @@ -7542,22 +7618,21 @@ fn findExactEntryInCompressedPrefixBlock( } const path = run.path orelse return error.RunStateUnavailable; const absolute_offset = @as(u64, @intCast(index.entry_data_start)) + window.physicalRelativeOffset(); - const payload = try loadRunTableBlockWithStats( - backend, - backend.allocator, - path, - absolute_offset, - window.physicalLen(), - ); - defer backend.allocator.free(payload); - const positioned = try lsm_table_file.findExactEntryInCompressedBlockPayloadAlloc( + var workspace: ?LocalReader.Workspace = if (comptime @hasField(@TypeOf(backend.*), "local_reader")) try localWorkspace(backend, window) else null; + defer if (workspace) |*work| work.release(); + const scratch = if (workspace) |*work| work.allocator() else backend.allocator; + const payload = try loadRunTableBlockWithStats(backend, scratch, path, absolute_offset, window.physicalLen()); + defer scratch.free(payload); + const positioned = try lsm_table_file.findExactEntryInCompressedBlockPayloadWithScratchAlloc( backend.allocator, + scratch, window.compression, payload, window.checksum, block.first_entry_index, namespace.name, key, + window.len, ) orelse return null; return .{ .entry = positioned.entry, @@ -7586,17 +7661,19 @@ fn findExactEntryWithLocalIndexBlockMeta( if (retainLocalCachedBlock(backend, run, index, window, false)) |lease| { return findExactEntryInBlockLease(lease, index, block_index, namespace, key); } - if (beginLocalBlockPromotion(backend, run, index, window, false)) { + if (namespace.retainDataBlocks()) if (try joinLocalBlockDecode(backend, run, index, window)) |lease| { + return findExactEntryInBlockLease(lease, index, block_index, namespace, key); + }; + if (namespace.retainDataBlocks() and beginLocalBlockPromotion(backend, run, index, window, false)) { var admitted = false; defer finishLocalBlockPromotion(backend, run, index, window, false, admitted); - const lease = try loadLocalBlockLease(backend, run, index, window, false); + const lease = try loadLocalBlockLease(backend, run, index, window, false, namespace.retainDataBlocks()); admitted = lease.cache_admitted; return findExactEntryInBlockLease(lease, index, block_index, namespace, key); } } - if (try findExactEntryInCompressedPrefixBlock(backend, run, index, block_index, namespace, key)) |entry| { - return entry; - } + if (window.compression == .prefix or window.compression == .prefix_snappy) + return try findExactEntryInCompressedPrefixBlock(backend, run, index, block_index, namespace, key); if (localBlockCacheEnabled(backend)) return try findExactEntryInLocalLease(backend, run, index, window, block_index, namespace, key, false); const bytes = try loadOwnedBlockForWindow( backend, @@ -7645,17 +7722,16 @@ fn findExactEntryWithLocalIndexBlockMetaMaybeLocked( if (retainLocalCachedBlock(backend, run, index, window, true)) |lease| { return findExactEntryInBlockLease(lease, index, block_index, namespace, key); } - if (beginLocalBlockPromotion(backend, run, index, window, true)) { + if (namespace.retainDataBlocks() and beginLocalBlockPromotion(backend, run, index, window, true)) { var admitted = false; defer finishLocalBlockPromotion(backend, run, index, window, true, admitted); - const lease = try loadLocalBlockLease(backend, run, index, window, true); + const lease = try loadLocalBlockLease(backend, run, index, window, true, namespace.retainDataBlocks()); admitted = lease.cache_admitted; return findExactEntryInBlockLease(lease, index, block_index, namespace, key); } } - if (try findExactEntryInCompressedPrefixBlock(backend, run, index, block_index, namespace, key)) |entry| { - return entry; - } + if (window.compression == .prefix or window.compression == .prefix_snappy) + return try findExactEntryInCompressedPrefixBlock(backend, run, index, block_index, namespace, key); if (localBlockCacheEnabled(backend)) return try findExactEntryInLocalLease(backend, run, index, window, block_index, namespace, key, true); const bytes = try loadOwnedBlockForWindowMaybeLocked( backend, @@ -7717,6 +7793,7 @@ fn loadVisibleEntryFromPathRunMaybeLocked( fn getFromRunWithLocalIndex( backend: anytype, run: *Run, + held_blocks: ?*std.ArrayListUnmanaged(BlockPin), held_values: *std.ArrayListUnmanaged([]u8), value_allocator: Allocator, namespace: backend_types.Namespace, @@ -7727,6 +7804,9 @@ fn getFromRunWithLocalIndex( var transferred = false; defer if (!transferred) loaded.deinit(backend.allocator); if (loaded.entry.tombstone) return error.NotFound; + if (namespace.borrow_local_point_results) if (held_blocks) |pins| if (loaded.local) |payload| { + if (try retainLocalResultPin(backend, payload, pins)) return loaded.entry.value; + }; // Wide values dominate their block. Transfer the decoded allocation when // its owner matches rather than copying the row out and immediately @@ -8015,7 +8095,9 @@ pub fn NamespaceWriteTxn(comptime BackendType: type) type { cursor_runs: []Run = &.{}, cursor_l0_groups: []RunGroup = &.{}, cursor_levels: []RunLevel = &.{}, + held_blocks: std.ArrayListUnmanaged(BlockPin) = .empty, held_values: std.ArrayListUnmanaged([]u8) = .empty, + batch_scratch: @import("write_batch_scratch.zig").Scratch = .{}, batch_options: backend_types.BatchOptions = .{}, cursor_reader_retained: bool = false, closed: bool = false, @@ -8048,6 +8130,8 @@ pub fn NamespaceWriteTxn(comptime BackendType: type) type { self.mutable.deinit(self.allocator); self.bulk_appends.deinit(self.allocator); self.invalidateCursorSnapshot(); + self.batch_scratch.deinit(self.metadata_allocator); + releaseHeldBlocks(&self.held_blocks, self.allocator); releaseHeldValues(&self.held_values, self.allocator); const locked = lockBackend(BackendType, backend); defer unlockBackend(BackendType, backend, locked); @@ -8062,6 +8146,7 @@ pub fn NamespaceWriteTxn(comptime BackendType: type) type { defer if (self.closed) { self.bulk_index.deinit(self.allocator); self.prefix_index.deinit(self.allocator); + self.batch_scratch.deinit(self.metadata_allocator); }; const wire_credit = if (comptime @hasDecl(BackendType, "prepareManifestCredit")) try self.backend.prepareManifestCredit(&self.mutable, &self.bulk_appends) else 0; const locked = lockBackend(BackendType, self.backend); @@ -8073,6 +8158,7 @@ pub fn NamespaceWriteTxn(comptime BackendType: type) type { self.bulk_appends.deinit(self.allocator); self.bulk_appends = .{}; self.invalidateCursorSnapshotLocked(); + releaseHeldBlocks(&self.held_blocks, self.allocator); releaseHeldValues(&self.held_values, self.allocator); self.backend.finishBatchMode(self.batch_options); if (self.cursor_reader_retained) releaseWriteReader(BackendType, self.backend, .current_scan); @@ -8125,6 +8211,7 @@ pub fn NamespaceWriteTxn(comptime BackendType: type) type { release_on_error = false; self.closed = true; self.invalidateCursorSnapshotLocked(); + releaseHeldBlocks(&self.held_blocks, self.allocator); releaseHeldValues(&self.held_values, self.allocator); var finalize_err: ?anyerror = null; if (self.cursor_reader_retained) { @@ -8352,10 +8439,13 @@ pub fn NamespaceWriteTxn(comptime BackendType: type) type { if (keys.len != values.len) return error.InvalidBatch; @memset(values, null); - const miss_keys = try self.metadata_allocator.alloc([]const u8, keys.len); - defer self.metadata_allocator.free(miss_keys); - const miss_indexes = try self.metadata_allocator.alloc(usize, keys.len); - defer self.metadata_allocator.free(miss_indexes); + const Scratch = @import("write_batch_scratch.zig").Scratch; + var oversized: Scratch = .{}; + defer oversized.deinit(self.metadata_allocator); + const scratch = if (keys.len <= Scratch.max_retained_keys) &self.batch_scratch else &oversized; + try scratch.prepareKeys(self.metadata_allocator, keys.len); + const miss_keys = scratch.keys.items; + const miss_indexes = scratch.indexes.items; var miss_count: usize = 0; var overlay_point_gets: usize = 0; @@ -8376,18 +8466,26 @@ pub fn NamespaceWriteTxn(comptime BackendType: type) type { self.backend.recordPointGets(overlay_point_gets); if (miss_count == 0) return; - const miss_values = try self.metadata_allocator.alloc(?[]const u8, miss_count); - defer self.metadata_allocator.free(miss_values); - var probe = NamespaceProbeTxn(BackendType).open(self.backend); - defer probe.abort(); - try probe.getManySorted(namespace, miss_keys[0..miss_count], miss_values); - for (miss_values, 0..) |maybe_value, miss_index| { - const value = maybe_value orelse continue; - const owned = try self.allocator.dupe(u8, value); - errdefer self.allocator.free(owned); - try self.held_values.append(self.allocator, owned); - values[miss_indexes[miss_index]] = owned; - } + try scratch.prepareValues(self.metadata_allocator, miss_count); + const miss_values = scratch.values.items[0..miss_count]; + var probe = try BoundProbeTxn(BackendType).open(self.backend, namespace); + // Preserve the transaction-wide unique pin budget across probes. + // All owned probe results use the writer's allocator, so their + // allocations can transfer directly without a namespace copy. + probe.allocator = self.allocator; + probe.namespace.borrow_local_point_results = true; + probe.held_blocks = self.held_blocks; + self.held_blocks = .empty; + defer { + self.held_blocks = probe.held_blocks; + probe.held_blocks = .empty; + probe.abort(); + } + try probe.getManySorted(miss_keys[0..miss_count], miss_values); + try self.held_values.appendSlice(self.allocator, probe.held_values.items); + probe.held_values.deinit(self.allocator); + probe.held_values = .empty; + for (miss_values, 0..) |value, miss_index| values[miss_indexes[miss_index]] = value; } pub fn put(self: *@This(), namespace: backend_types.Namespace, key: []const u8, value: []const u8) !void { @@ -9080,3 +9178,31 @@ test "lsm local result block retention bounds bytes and pin metadata" { try std.testing.expect(!try cursor.retainCurrentValueForTxn(&held)); try std.testing.expectEqual(@as(usize, 64), held.items.len); } + +test "lsm local compressed absence reads once without promotion" { + const B = @import("../lsm_backend.zig").Backend; + const a = std.testing.allocator; + var storage = storage_io.MemoryStorage.init(a); + defer storage.deinit(); + var backend = try B.open(a, "/review-negative", .{ .storage = storage.storage(), .flush_threshold = 1 }); + defer backend.close(); + var runtime = try backend.runtimeStore(a, .{ .name = "docs" }); + defer runtime.deinit(); + var write = try runtime.beginWrite(); + try write.put("document:long-shared-prefix-for-compression:00", "value"); + try write.put("document:long-shared-prefix-for-compression:02", "other"); + try write.commit(); + const run = backend.runs.at(0); + const index = try indexForRunNoCache(&backend, run); + try std.testing.expectEqual(lsm_table_file.BlockCompression.prefix, index.blockWindow(0).compression); + // Model a Bloom false positive without depending on hash collision luck. + if (index.blocks[0].filter) |filter| @memset(filter.bytes, 255); + const loads = backend.read_stats.table_block_loads.load(.monotonic); + const absent = try findExactEntryWithLocalIndexBlockMeta(&backend, run, index, .{ .name = "docs" }, "document:long-shared-prefix-for-compression:01"); + defer if (absent) |entry| entry.deinit(a); + const extra = backend.read_stats.table_block_loads.load(.monotonic) - loads; + std.debug.print("lite cold compressed absence: loads={d}, decoded cache blocks={d}\n", .{ extra, backend.run_block_cache.items.len }); + try std.testing.expect(absent == null); + try std.testing.expectEqual(@as(u64, 1), extra); + try std.testing.expectEqual(@as(usize, 0), backend.run_block_cache.items.len); +} diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend/write_batch_scratch.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend/write_batch_scratch.zig new file mode 100644 index 0000000000..2dc55433e4 --- /dev/null +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend/write_batch_scratch.zig @@ -0,0 +1,63 @@ +// Copyright 2026 Antfly, Inc. +// SPDX-License-Identifier: Apache-2.0 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +const std = @import("std"); + +/// Reuse small writer batches without retaining an exceptional batch's peak. +pub const Scratch = struct { + pub const max_retained_keys = 1024; + keys: std.ArrayListUnmanaged([]const u8) = .empty, + indexes: std.ArrayListUnmanaged(usize) = .empty, + values: std.ArrayListUnmanaged(?[]const u8) = .empty, + pub fn prepareKeys(self: *Scratch, a: std.mem.Allocator, n: usize) !void { + try self.keys.ensureTotalCapacityPrecise(a, n); + try self.indexes.ensureTotalCapacityPrecise(a, n); + self.keys.items.len = n; + self.indexes.items.len = n; + } + pub fn prepareValues(self: *Scratch, a: std.mem.Allocator, n: usize) !void { + try self.values.ensureTotalCapacityPrecise(a, n); + self.values.items.len = n; + } + pub fn prepare(self: *Scratch, a: std.mem.Allocator, n: usize) !void { + try self.prepareKeys(a, n); + try self.prepareValues(a, n); + } + pub fn deinit(self: *Scratch, a: std.mem.Allocator) void { + self.keys.deinit(a); + self.indexes.deinit(a); + self.values.deinit(a); + self.* = .{}; + } +}; + +test "lsm writer batch scratch reuses allocations and unwinds failures" { + const Fixture = struct { + fn run(a: std.mem.Allocator) !void { + var scratch: Scratch = .{}; + defer scratch.deinit(a); + try scratch.prepare(a, 16); + const keys = scratch.keys.items.ptr; + const indexes = scratch.indexes.items.ptr; + const values = scratch.values.items.ptr; + try scratch.prepare(a, 16); + try std.testing.expect(keys == scratch.keys.items.ptr); + try std.testing.expect(indexes == scratch.indexes.items.ptr); + try std.testing.expect(values == scratch.values.items.ptr); + try scratch.prepare(a, 32); + } + }; + try std.testing.checkAllAllocationFailures(std.testing.allocator, Fixture.run, .{}); +} diff --git a/zig/pkg/antfly-embedded/src/storage/resource_manager.zig b/zig/pkg/antfly-embedded/src/storage/resource_manager.zig index c829cf1e88..2e869a75c9 100644 --- a/zig/pkg/antfly-embedded/src/storage/resource_manager.zig +++ b/zig/pkg/antfly-embedded/src/storage/resource_manager.zig @@ -195,6 +195,9 @@ pub const Slice = enum(u8) { /// the capacity-domain ledger; this slice owns only queued key/payload /// memory until the cache worker completes or drops the write. lake_range_cache_queue, + /// Decoder scratch and uncached decoded payload leases. Appended to keep + /// existing resource IDs stable and separate reads from mutable admission. + lsm_read_working_set, pub fn name(self: Slice) []const u8 { return switch (self) { @@ -232,6 +235,7 @@ pub const Slice = enum(u8) { .dense_vector_block_build_working_set => "dense.vector_block_build_working_set", .dense_source_payload_state => "dense.source_payload_state", .lake_range_cache_queue => "lake.range_cache_queue", + .lsm_read_working_set => "lsm.read_working_set", }; } }; @@ -469,6 +473,7 @@ pub const Options = struct { .dense_source_payload_state = .{ .soft_limit_bytes = 192 * 1024 * 1024, .hard_limit_bytes = 384 * 1024 * 1024 }, .relational_preparation_working_set = .{ .soft_limit_bytes = 128 * 1024 * 1024, .hard_limit_bytes = 256 * 1024 * 1024 }, .lake_range_cache_queue = .{ .soft_limit_bytes = 384 * 1024 * 1024, .hard_limit_bytes = 512 * 1024 * 1024 }, + .lsm_read_working_set = .{ .soft_limit_bytes = 64 * 1024 * 1024, .hard_limit_bytes = 128 * 1024 * 1024 }, }).values; } @@ -509,6 +514,7 @@ pub const Options = struct { .dense_source_payload_state = .{ .soft_action = .report, .hard_action = .throttle_writes }, .relational_preparation_working_set = .{ .soft_action = .report, .hard_action = .reject_work }, .lake_range_cache_queue = .{ .soft_action = .report, .hard_action = .reject_work }, + .lsm_read_working_set = .{ .soft_action = .report, .hard_action = .reject_work }, }).values; } }; @@ -2806,6 +2812,29 @@ pub const ResourceManager = struct { return true; } + /// Reclassify an existing owner atomically without a second host charge. + /// No allocation or callback occurs; destination slice admission still applies. + fn reclassifyReservation(self: *ResourceManager, reservation: *Reservation, destination: Slice) !void { + lockAtomic(&self.mutex); + defer self.mutex.unlock(); + const owned = self.reservation_identities.getPtr(reservation.identity) orelse return error.ReservationReleased; + if (reservation.manager != self or reservation.released or owned.slice != reservation.slice or owned.bytes != reservation.bytes) + return error.ResourceAccountingMismatch; + if (destination == reservation.slice) return; + const source = &self.slices[sliceIndex(reservation.slice)]; + const target = &self.slices[sliceIndex(destination)]; + const next = std.math.add(u64, target.used_bytes, reservation.bytes) catch return error.ResourceBudgetExceeded; + if (target.budget.hard_limit_bytes != 0 and next > target.budget.hard_limit_bytes) return error.ResourceBudgetExceeded; + if (source.used_bytes < reservation.bytes) return error.ResourceAccountingMismatch; + source.used_bytes -= reservation.bytes; + target.used_bytes = next; + target.peak_bytes = @max(target.peak_bytes, next); + if (target.budget.soft_limit_bytes != 0 and next > target.budget.soft_limit_bytes) target.soft_limit_events +|= 1; + owned.slice = destination; + reservation.slice = destination; + self.pressure_change.advance(); + } + /// Move already-accounted credit between two live reservations without /// changing slice or host usage. This is the ownership handoff used when /// operation admission pre-reserves allocator headroom before the @@ -3633,6 +3662,15 @@ pub const Reservation = struct { _ = self.manager.shrinkReservation(self, bytes); } + pub fn reclassify(self: *Reservation, destination: Slice) !void { + if (self.released) return error.ReservationReleased; + if (self.bytes == 0 and self.identity == 0) { + self.slice = destination; + return; + } + try self.manager.reclassifyReservation(self, destination); + } + pub fn transferCreditTo(self: *Reservation, destination: *Reservation, bytes: u64) !void { if (self.released or destination.released) return error.ReservationReleased; try self.manager.transferReservationCredit(self, destination, bytes); @@ -4166,11 +4204,11 @@ test "default tokenizer cache budget is aligned with its resource slice" { ); } -test "default lake range cache queue budget is aligned with its terminal resource slice" { +test "default lake range cache queue budget keeps its stable resource identity" { const budgets = Options.defaultBudgets(); const policies = Options.defaultPolicies(); const index = @backingInt(Slice.lake_range_cache_queue); - try std.testing.expectEqual(slice_count - 1, index); + try std.testing.expectEqual(@as(usize, 33), index); try std.testing.expectEqual(@as(u64, 384 * 1024 * 1024), budgets[index].soft_limit_bytes); try std.testing.expectEqual(@as(u64, 512 * 1024 * 1024), budgets[index].hard_limit_bytes); try std.testing.expectEqual(PressureAction.report, policies[index].soft_action); @@ -6287,3 +6325,29 @@ test "source vector payloads scoped allocator receipts distinguish admission and budget.threadSafeAllocator().free(memory); try std.testing.expect(first.last_failure == null); } + +test "resource manager reclassifies owned read credit without a second host charge" { + const a = std.testing.allocator; + for ([_]u64{ 128, 256 }) |limit| { + var budgets = Options.defaultBudgets(); + budgets[@backingInt(Slice.lsm_block_table_cache)] = .{ .hard_limit_bytes = limit }; + var manager = ResourceManager.init(.{ .budgets = budgets, .memory_budget = .{ .hard_limit_bytes = 150 } }); + defer manager.deinit(a); + var credit = try manager.reserveWithoutReclaim(.lsm_read_working_set, 150); + defer credit.release(); + if (limit < 150) { + try std.testing.expectError(error.ResourceBudgetExceeded, credit.reclassify(.lsm_block_table_cache)); + try std.testing.expectEqual(Slice.lsm_read_working_set, credit.slice); + try std.testing.expectEqual(@as(u64, 150), manager.sliceStats(.lsm_read_working_set).used_bytes); + try std.testing.expectEqual(@as(u64, 0), manager.sliceStats(.lsm_block_table_cache).used_bytes); + } else { + try credit.reclassify(.lsm_block_table_cache); + try std.testing.expectEqual(@as(u64, 0), manager.sliceStats(.lsm_read_working_set).used_bytes); + try std.testing.expectEqual(@as(u64, 150), manager.sliceStats(.lsm_block_table_cache).used_bytes); + } + try std.testing.expectEqual(@as(u64, 150), manager.snapshot().memory.peak_bytes); + credit.release(); + try std.testing.expectEqual(@as(u64, 0), manager.snapshot().memory.used_bytes); + try std.testing.expectEqual(@as(u64, 0), manager.snapshot().memory.accounting_errors); + } +} From 70537b629a1755b32d2bd92531c18db4b8219887 Mon Sep 17 00:00:00 2001 From: AJ Roetker Date: Thu, 8 Oct 2026 12:56:01 -0700 Subject: [PATCH 3/4] fix(capi): export only the public shared-library ABI --- .github/workflows/antfly-artifact-build.yml | 2 +- .github/workflows/zig-tests.yml | 2 + zig/build.zig | 4 +- zig/build_support/embedded/embedded.zig | 39 ++-- zig/build_support/embedded/shared_library.zig | 146 ++++++++++++++ zig/embedded.build.zig | 4 +- .../tests/capi_link_consumer.c | 50 +++++ zig/tools/capi_exports.py | 181 ++++++++++++++++++ zig/tools/test_capi_exports.py | 113 +++++++++++ 9 files changed, 513 insertions(+), 28 deletions(-) create mode 100644 zig/build_support/embedded/shared_library.zig create mode 100644 zig/pkg/antfly-embedded/tests/capi_link_consumer.c create mode 100644 zig/tools/capi_exports.py create mode 100644 zig/tools/test_capi_exports.py diff --git a/.github/workflows/antfly-artifact-build.yml b/.github/workflows/antfly-artifact-build.yml index 1c2e110a8a..d0f11398d4 100644 --- a/.github/workflows/antfly-artifact-build.yml +++ b/.github/workflows/antfly-artifact-build.yml @@ -136,7 +136,7 @@ jobs: if: ${{ matrix.os == 'linux' }} run: | sudo apt-get update - sudo apt-get install -y build-essential clang lld curl xz-utils git perl python3 linux-libc-dev ca-certificates file + sudo apt-get install -y build-essential clang lld llvm curl xz-utils git perl python3 linux-libc-dev ca-certificates file - name: Install Nix if: ${{ matrix.os == 'linux' }} diff --git a/.github/workflows/zig-tests.yml b/.github/workflows/zig-tests.yml index 1934f45641..bad0cf8b94 100644 --- a/.github/workflows/zig-tests.yml +++ b/.github/workflows/zig-tests.yml @@ -587,6 +587,7 @@ jobs: tools/test_audit_test_selection.py tools/test_audit_unit_test_ownership.py tools/test_check_storage_compilation.py + tools/test_capi_exports.py tools/test_run_build_cache_contracts.py tools/test_run_test_partitions.py tools/test_retain_test_process.py @@ -999,6 +1000,7 @@ jobs: tools/test_audit_test_selection.py tools/test_audit_unit_test_ownership.py tools/test_check_storage_compilation.py + tools/test_capi_exports.py tools/test_run_build_cache_contracts.py tools/test_run_test_partitions.py tools/test_retain_test_process.py diff --git a/zig/build.zig b/zig/build.zig index aab1d39dc6..c9fb458b87 100644 --- a/zig/build.zig +++ b/zig/build.zig @@ -250,7 +250,7 @@ pub fn create(b: *std.Build) ?Artifacts { const capi_mod = embedded.capi_mod; const libantfly_link_mod = embedded.libantfly_link_mod; const install_libantfly = embedded.install_libantfly; - if (install_apple_bridge) |install| install_libantfly.step.dependOn(&install.step); + if (install_apple_bridge) |install| install_libantfly.dependOn(&install.step); const install_capi_header = embedded.install_capi_header; const run_capi_smoke = embedded.run_capi_smoke; const run_capi_conformance = embedded.run_capi_conformance; @@ -1020,7 +1020,7 @@ pub fn create(b: *std.Build) ?Artifacts { const lite_step = b.step("lite", "Build and install the Antfly Lite CLI and libantfly C ABI"); lite_step.dependOn(&b.top_level_steps.get("licenses-antfly-lite").?.step); lite_step.dependOn(&install_lite_main.step); - lite_step.dependOn(&install_libantfly.step); + lite_step.dependOn(install_libantfly); lite_step.dependOn(&install_capi_header.step); const lite_test_step = b.step("lite-test", "Run Lite backend, CLI, bindings, examples, and C ABI packaging checks"); diff --git a/zig/build_support/embedded/embedded.zig b/zig/build_support/embedded/embedded.zig index 76d669dbf3..ef50a01656 100644 --- a/zig/build_support/embedded/embedded.zig +++ b/zig/build_support/embedded/embedded.zig @@ -106,7 +106,7 @@ pub const AddEmbeddedResult = struct { capi_root_mod: *std.Build.Module, capi_mod: *std.Build.Module, libantfly_link_mod: *std.Build.Module, - install_libantfly: *std.Build.Step.InstallArtifact, + install_libantfly: *std.Build.Step, install_capi_header: *std.Build.Step.InstallFile, run_capi_smoke: *std.Build.Step.Run, run_capi_conformance: *std.Build.Step.Run, @@ -289,25 +289,15 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult libantfly_link_mod.linkLibrary(native_inference); libantfly_link_mod.linkLibrary(native_enrichment); addMacosSdkPaths(b, libantfly_link_mod, target); - const libantfly = b.addLibrary(.{ - .linkage = .dynamic, - .name = "antfly", - .root_module = libantfly_link_mod, - .max_rss = 12 * 1024 * 1024 * 1024, - }); - libantfly.link_gc_sections = true; - // Homebrew rewrites the dylib ID to its absolute opt/lib path on install. - if (target.result.os.tag == .macos) { - libantfly.headerpad_max_install_names = true; - } - const install_libantfly = b.addInstallArtifact(libantfly, .{}); + const libantfly = @import("shared_library.zig").add(b, libantfly_link_mod); + const install_libantfly = libantfly.install; b.dependOnFileContents(b.path("pkg/antfly-embedded/libantfly.pc.in")); const pc_path = b.root.join(b.allocator, "pkg/antfly-embedded/libantfly.pc.in") catch @panic("OOM"); const pc_template = pc_path.root_dir.handle.readFileAlloc(b.graph.io, pc_path.sub_path, b.allocator, .limited(16 * 1024)) catch @panic("unable to read libantfly.pc.in"); const pc_contents = std.mem.replaceOwned(u8, b.allocator, pc_template, "@VERSION@", options.version) catch @panic("OOM"); const pc_file = b.addWriteFiles().add("libantfly.pc", pc_contents); const install_pkg_config = b.addInstallFileWithDir(pc_file, .lib, "pkgconfig/libantfly.pc"); - install_libantfly.step.dependOn(&install_pkg_config.step); + install_libantfly.dependOn(&install_pkg_config.step); b.step("pkgconfig", "Install relocatable libantfly pkg-config metadata").dependOn(&install_pkg_config.step); const install_capi_header = b.addInstallFileWithDir( @@ -332,7 +322,7 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult b.getInstallStep().dependOn(install_licenses); const capi_step = b.step("capi", "Build the public libantfly C ABI shared library"); capi_step.dependOn(install_licenses); - capi_step.dependOn(&install_libantfly.step); + capi_step.dependOn(install_libantfly); capi_step.dependOn(&install_capi_header.step); const capi_smoke_mod = b.createModule(.{ @@ -349,10 +339,13 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult .name = "antfly-c-smoke", .root_module = capi_smoke_mod, }); - capi_smoke.root_module.linkLibrary(libantfly); + libantfly.link(capi_smoke.root_module); const run_capi_smoke = b.addRunArtifact(capi_smoke); + if (target.result.os.tag == .macos or target.result.ofmt == .elf) b.step("capi-exports-check", "Verify that libantfly exports exactly the public C header functions").dependOn(&libantfly.check.step); const capi_smoke_step = b.step("capi-smoke", "Compile and run a C consumer smoke test for libantfly"); capi_smoke_step.dependOn(&run_capi_smoke.step); + if (target.result.os.tag == .macos or target.result.ofmt == .elf) capi_smoke_step.dependOn(&libantfly.check.step); + if (b.top_level_steps.get("capi-linkage-test")) |linkage| capi_smoke_step.dependOn(&linkage.step); // Reference runner for the shared conformance cases every binding runs // (pkg/antfly-embedded/capi-conformance/README.md). It calls libantfly only @@ -376,7 +369,7 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult .name = "antfly-capi-conformance", .root_module = capi_conformance_mod, }); - capi_conformance.root_module.linkLibrary(libantfly); + libantfly.link(capi_conformance.root_module); const run_capi_conformance = b.addRunArtifact(capi_conformance); run_capi_conformance.addDirectoryArg2(b.path("pkg/antfly-embedded/capi-conformance/cases"), .{ .make_absolute = true }); _ = run_capi_conformance.addOutputDirectoryArg2("capi-conformance-work", .{ .make_absolute = true }); @@ -395,7 +388,7 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult }); run_lite_go_tests.argv.insert(b.allocator, 2, .{ .decorated_directory = .{ .lazy_path = b.graph.path(.install_lib, "pkgconfig"), .prefix = "PKG_CONFIG_PATH=", .suffix = "", .make_absolute = true } }) catch @panic("OOM"); run_lite_go_tests.setCwd(b.path("../go/pkg/embedded")); - run_lite_go_tests.step.dependOn(&install_libantfly.step); + run_lite_go_tests.step.dependOn(install_libantfly); run_lite_go_tests.step.dependOn(&install_capi_header.step); const lite_go_test_step = b.step("lite-go-test", "Run Go Antfly Lite binding tests against libantfly"); lite_go_test_step.dependOn(&run_lite_go_tests.step); @@ -414,7 +407,7 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult }); run_lite_py_tests.argv.insert(b.allocator, 2, .{ .decorated_directory = .{ .lazy_path = b.graph.path(.install_lib, ""), .prefix = "ANTFLY_LIB_DIR=", .suffix = "", .make_absolute = true } }) catch @panic("OOM"); run_lite_py_tests.setCwd(b.path("../py/packages/embedded")); - run_lite_py_tests.step.dependOn(&install_libantfly.step); + run_lite_py_tests.step.dependOn(install_libantfly); const lite_py_test_step = b.step("lite-py-test", "Run Python Antfly Lite binding tests against libantfly"); lite_py_test_step.dependOn(&run_lite_py_tests.step); @@ -432,7 +425,7 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult }); run_lite_rs_tests.argv.insert(b.allocator, 1, .{ .decorated_directory = .{ .lazy_path = b.graph.path(.install_lib, ""), .prefix = "ANTFLY_LIB_DIR=", .suffix = "", .make_absolute = true } }) catch @panic("OOM"); run_lite_rs_tests.setCwd(b.path(".")); - run_lite_rs_tests.step.dependOn(&install_libantfly.step); + run_lite_rs_tests.step.dependOn(install_libantfly); const lite_rs_test_step = b.step("lite-rs-test", "Run Rust Antfly Lite binding tests against libantfly"); lite_rs_test_step.dependOn(&run_lite_rs_tests.step); @@ -445,7 +438,7 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult }); run_lite_ts_tests.argv.insert(b.allocator, 2, .{ .decorated_directory = .{ .lazy_path = b.graph.path(.install_lib, ""), .prefix = "ANTFLY_LIB_DIR=", .suffix = "", .make_absolute = true } }) catch @panic("OOM"); run_lite_ts_tests.setCwd(b.path("../ts/packages/embedded")); - run_lite_ts_tests.step.dependOn(&install_libantfly.step); + run_lite_ts_tests.step.dependOn(install_libantfly); const lite_ts_test_step = b.step("lite-ts-test", "Run TypeScript Antfly Lite binding tests against libantfly (needs pnpm install in ts/)"); lite_ts_test_step.dependOn(&run_lite_ts_tests.step); @@ -463,7 +456,7 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult }); run_lite_go_example.argv.insert(b.allocator, 2, .{ .decorated_directory = .{ .lazy_path = b.graph.path(.install_lib, "pkgconfig"), .prefix = "PKG_CONFIG_PATH=", .suffix = "", .make_absolute = true } }) catch @panic("OOM"); run_lite_go_example.setCwd(b.path("../examples/antfly-lite-go")); - run_lite_go_example.step.dependOn(&install_libantfly.step); + run_lite_go_example.step.dependOn(install_libantfly); run_lite_go_example.step.dependOn(&install_capi_header.step); const lite_go_example_step = b.step("lite-go-example", "Run the embedded Go Antfly Lite example app"); lite_go_example_step.dependOn(&run_lite_go_example.step); @@ -482,7 +475,7 @@ pub fn addEmbedded(b: *std.Build, options: AddEmbeddedOptions) AddEmbeddedResult }); run_lite_go_retrieval_template.argv.insert(b.allocator, 2, .{ .decorated_directory = .{ .lazy_path = b.graph.path(.install_lib, "pkgconfig"), .prefix = "PKG_CONFIG_PATH=", .suffix = "", .make_absolute = true } }) catch @panic("OOM"); run_lite_go_retrieval_template.setCwd(b.path("../examples/antfly-lite-retrieval-go")); - run_lite_go_retrieval_template.step.dependOn(&install_libantfly.step); + run_lite_go_retrieval_template.step.dependOn(install_libantfly); run_lite_go_retrieval_template.step.dependOn(&install_capi_header.step); const lite_go_retrieval_template_step = b.step("lite-go-retrieval-template", "Run the embedded Go Antfly Lite retrieval template"); lite_go_retrieval_template_step.dependOn(&run_lite_go_retrieval_template.step); diff --git a/zig/build_support/embedded/shared_library.zig b/zig/build_support/embedded/shared_library.zig new file mode 100644 index 0000000000..40214a8811 --- /dev/null +++ b/zig/build_support/embedded/shared_library.zig @@ -0,0 +1,146 @@ +// Copyright 2026 Antfly, Inc. +// SPDX-License-Identifier: Apache-2.0 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +//! Final shared-library exports belong to the public header, never to linked archives. +const std = @import("std"); + +pub const Library = struct { + binary: std.Build.LazyPath, + compile: ?*std.Build.Step.Compile, + install: *std.Build.Step, + check: *std.Build.Step.Run, + + pub fn link(self: Library, module: *std.Build.Module) void { + if (self.compile) |compile| module.linkLibrary(compile) else { + module.addObjectFile(self.binary); + module.addRPath(self.binary.dirname()); + } + } +}; + +pub fn add(b: *std.Build, module: *std.Build.Module) Library { + const target = module.resolved_target.?.result; + const macho = target.os.tag == .macos; + const elf = target.ofmt == .elf; + const header = b.path("pkg/antfly-embedded/include/antfly.h"); + var exports: ?std.Build.LazyPath = null; + if (macho or elf) { + const generate = command(b, "manifest"); + generate.addArgs(&.{ "--format", if (macho) "macho" else "elf", "--header" }); + generate.addFileArg(header); + generate.addArg("--output"); + exports = generate.addOutputFileArg(if (macho) "antfly.exports" else "antfly.map"); + } + const compile = b.addLibrary(.{ + .linkage = if (macho) .static else .dynamic, + .name = if (macho) "antfly-capi-objects" else "antfly", + .root_module = module, + .max_rss = 12 * 1024 * 1024 * 1024, + }); + const binary = if (macho) blk: { + module.pic = true; + compile.bundle_compiler_rt = true; + const linker = b.option([]const u8, "macos-linker", "Mach-O linker executable (ld on macOS, ld64.lld for cross builds)") orelse + if (b.graph.host.result.os.tag == .macos) "/usr/bin/ld" else "ld64.lld"; + const run = command(b, "link-macho"); + const deployment = target.os.version_range.semver.min; + run.addArgs(&.{ "--linker", linker, "--arch", if (target.cpu.arch == .aarch64) "arm64" else "x86_64", "--deployment", b.fmt("{d}.{d}.{d}", .{ deployment.major, deployment.minor, deployment.patch }), "--sdk" }); + run.addDirectoryArg(b.named_lazy_paths.get("antfly_macos_sdk_root") orelse @panic("macOS C API requires a selected SDK")); + run.addArg("--exports"); + run.addFileArg(exports.?); + run.addArg("--archive"); + run.addFileArg(compile.getEmittedBin()); + run.addArg("--output"); + const output = run.addOutputFileArg("libantfly.dylib"); + run.addArg("--"); + var visited: std.AutoHashMap(*std.Build.Module, void) = .init(b.allocator); + var libraries: std.AutoHashMap(*std.Build.Step.Compile, void) = .init(b.allocator); + // Traverse imports without caching Module.getGraph: composition owners + // still add build-info imports after this helper returns. + addLinks(b, run, module, &visited, &libraries); + break :blk output; + } else blk: { + compile.link_gc_sections = true; + if (exports) |map| compile.setVersionScript(map); + break :blk compile.getEmittedBin(); + }; + const check = command(b, "check"); + if (macho or elf) { + check.addArgs(&.{ "--format", if (macho) "macho" else "elf", "--header" }); + check.addFileArg(header); + check.addArg("--library"); + check.addFileArg(binary); + } + if ((macho or elf) and target.os.tag == b.graph.host.result.os.tag and target.cpu.arch == b.graph.host.result.cpu.arch and target.abi == b.graph.host.result.abi) { + const consumer = command(b, "consumer-test"); + consumer.addArgs(&.{ "--format", if (macho) "macho" else "elf", "--header" }); + consumer.addFileArg(header); + consumer.addArg("--library"); + consumer.addFileArg(binary); + consumer.addArg("--source"); + consumer.addFileArg(b.path("pkg/antfly-embedded/tests/capi_link_consumer.c")); + consumer.addArg("--work"); + _ = consumer.addOutputDirectoryArg("consumer-linkage"); + // Also run under ordinary C API smoke validation, without making the + // release library require a host executable or host compiler. + b.step("capi-linkage-test", "Link executable and shared-library consumers with local constructor handles").dependOn(&consumer.step); + } + const install = if (macho) &b.addInstallFileWithDir(binary, .lib, "libantfly.dylib").step else &b.addInstallArtifact(compile, .{}).step; + if (macho or elf) install.dependOn(&check.step); + return .{ .binary = binary, .compile = if (macho) null else compile, .install = install, .check = check }; +} + +fn command(b: *std.Build, operation: []const u8) *std.Build.Step.Run { + const run = b.addSystemCommand(&.{"python3"}); + run.addFileArg(b.path("tools/capi_exports.py")); + run.addArg(operation); + return run; +} + +fn addLinks(b: *std.Build, run: *std.Build.Step.Run, module: *std.Build.Module, visited: *std.AutoHashMap(*std.Build.Module, void), libraries: *std.AutoHashMap(*std.Build.Step.Compile, void)) void { + if ((visited.getOrPut(module) catch @panic("OOM")).found_existing) return; + for (module.lib_paths.items) |path| { + run.addArg("-L"); + run.addDirectoryArg(path); + } + for (module.include_dirs.items) |dir| switch (dir) { + .framework_path, .framework_path_system => |path| { + run.addArg("-F"); + run.addDirectoryArg(path); + }, + else => {}, + }; + for (module.rpaths.items) |rpath| { + run.addArg("-rpath"); + switch (rpath) { + .lazy_path => |path| run.addDirectoryArg(path), + .special => |path| run.addArg(path), + } + } + var frameworks = module.frameworks.iterator(); + while (frameworks.next()) |entry| run.addArgs(&.{ if (entry.value_ptr.weak) "-weak_framework" else "-framework", entry.key_ptr.* }); + if (module.link_libcpp orelse false) run.addArg("-lc++"); + for (module.link_objects.items) |object| switch (object) { + .other_step => |library| { + if (!(libraries.getOrPut(library) catch @panic("OOM")).found_existing) { + run.addFileArg(library.getEmittedBin()); + addLinks(b, run, library.root_module, visited, libraries); + } + }, + .static_path => |path| run.addFileArg(path), + .system_lib => |lib| run.addArg(b.fmt("{s}{s}", .{ if (lib.weak) "-weak-l" else "-l", lib.name })), + else => {}, // C/ObjC sources are compiled into their owning archive. + }; + for (module.import_table.values()) |import| addLinks(b, run, import, visited, libraries); +} diff --git a/zig/embedded.build.zig b/zig/embedded.build.zig index 1f560c0dea..bfe92622a5 100644 --- a/zig/embedded.build.zig +++ b/zig/embedded.build.zig @@ -63,12 +63,12 @@ pub fn buildDependency(b: *std.Build, comptime asking_build_zig: type) void { const install_lite = b.addInstallArtifact(lite, .{}); const lite_step = b.step("lite", "Build the independent file-oriented Lite CLI and public C ABI"); lite_step.dependOn(&install_lite.step); - lite_step.dependOn(&embedded.install_libantfly.step); + lite_step.dependOn(embedded.install_libantfly); lite_step.dependOn(&embedded.install_capi_header.step); lite_step.dependOn(&b.top_level_steps.get("licenses-antfly-lite").?.step); b.getInstallStep().dependOn(&install_lite.step); - b.getInstallStep().dependOn(&embedded.install_libantfly.step); + b.getInstallStep().dependOn(embedded.install_libantfly); b.getInstallStep().dependOn(&embedded.install_capi_header.step); b.default_step = lite_step; diff --git a/zig/pkg/antfly-embedded/tests/capi_link_consumer.c b/zig/pkg/antfly-embedded/tests/capi_link_consumer.c new file mode 100644 index 0000000000..ec6b6bd545 --- /dev/null +++ b/zig/pkg/antfly-embedded/tests/capi_link_consumer.c @@ -0,0 +1,50 @@ +// Copyright 2026 Antfly, Inc. +// SPDX-License-Identifier: Apache-2.0 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +#include "antfly.h" +#include + +#if defined(__APPLE__) && defined(__aarch64__) +// Match aws-lc's direct local relocation, rather than a GOT-indirect reference. +extern void *consumer_dso_address(void); +__asm__(".text\n.globl _consumer_dso_address\n.p2align 2\n" + "_consumer_dso_address:\n" + "adrp x0, ___dso_handle@PAGE\n" + "add x0, x0, ___dso_handle@PAGEOFF\nret\n"); +#elif defined(__APPLE__) && defined(__x86_64__) +extern void *consumer_dso_address(void); +__asm__(".text\n.globl _consumer_dso_address\n" + "_consumer_dso_address:\nleaq ___dso_handle(%rip), %rax\nret\n"); +#else +extern void *__dso_handle; +static void *consumer_dso_address(void) { return &__dso_handle; } +#endif + +static void *constructor_handle; +__attribute__((constructor)) static void init(void) { + constructor_handle = consumer_dso_address(); +} + +int consumer_check(void) { + antfly_open_options options; + assert(constructor_handle != 0); + assert(constructor_handle == consumer_dso_address()); + assert(antfly_abi_version() != 0); + assert(antfly_open_options_init(&options) == ANTFLY_OK); + return 0; +} + +#ifndef CONSUMER_SHARED +int main(void) { return consumer_check(); } +#endif diff --git a/zig/tools/capi_exports.py b/zig/tools/capi_exports.py new file mode 100644 index 0000000000..24a7cf18ea --- /dev/null +++ b/zig/tools/capi_exports.py @@ -0,0 +1,181 @@ +#!/usr/bin/env python3 +# Copyright 2026 Antfly, Inc. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +"""Generate and verify libantfly's export boundary from its public C header.""" +from __future__ import annotations + +import argparse +import json +import ctypes +from pathlib import Path +import re +import shutil +import subprocess + + +def public_symbols(header: Path) -> set[str]: + source = re.sub(r'/\*.*?\*/|//[^\n]*', '', header.read_text(), flags=re.S) + names = set(re.findall(r'\b(antfly_[A-Za-z0-9_]+)\s*\(', source)) + if not names: + raise ValueError(f'no public C functions found in {header}') + return names + + +def manifest(header: Path, output: Path, object_format: str) -> None: + names = sorted(public_symbols(header)) + if object_format == 'macho': + text = ''.join(f'_{name}\n' for name in names) + else: + text = '{\n global:\n' + ''.join(f' {name};\n' for name in names) + ' local: *;\n};\n' + output.write_text(text) + + +def exported_symbols(library: Path, object_format: str, nm: str | None = None) -> set[str]: + nm = nm or shutil.which('llvm-nm') or shutil.which('nm') + if not nm: + raise ValueError('export verification requires nm (llvm-nm for Mach-O cross builds)') + if object_format == 'macho': + args = ['-P', '-g', '-U'] + else: + args = ['-P', '-D', '--defined-only', '--extern-only'] + result = subprocess.run([nm, *args, str(library)], check=True, text=True, capture_output=True) + names = set() + for line in result.stdout.splitlines(): + fields = line.split() + if len(fields) >= 2: + name = fields[0] + if object_format == 'macho' and name.startswith('_'): + name = name[1:] + names.add(name) + return names + + +def check(header: Path, library: Path, object_format: str, nm: str | None = None) -> None: + expected = public_symbols(header) + actual = exported_symbols(library, object_format, nm) + extra, missing = actual - expected, expected - actual + if extra or missing: + raise ValueError(f'libantfly exports differ from antfly.h: unexpected={sorted(extra)}, missing={sorted(missing)}') + print(f'libantfly export boundary passed: {len(actual)} public functions, no internal exports') + + +def bundled_macho_object(argument: str) -> bool: + # Zig's static archive already contains direct object inputs, including + # objects owned by its imported modules. Linking them again would define + # their symbols twice. Archive/dylib inputs still belong on the final link. + if argument.startswith('-'): + return False + try: + with open(argument, 'rb') as source: + header = source.read(16) + except OSError: + return False + if len(header) != 16: + return False + if header[:4] in (b'\xcf\xfa\xed\xfe', b'\xce\xfa\xed\xfe'): + return int.from_bytes(header[12:16], 'little') == 1 + if header[:4] in (b'\xfe\xed\xfa\xcf', b'\xfe\xed\xfa\xce'): + return int.from_bytes(header[12:16], 'big') == 1 + return False + + +def link_macho(args: argparse.Namespace) -> None: + sdk_settings = json.loads((args.sdk / 'SDKSettings.json').read_text()) + sdk_version = sdk_settings['Version'] + link_args = [arg for arg in args.link_args if not bundled_macho_object(arg)] + command = [args.linker, '-dylib', '-arch', args.arch, '-platform_version', 'macos', + args.deployment, sdk_version, '-syslibroot', str(args.sdk), + '-install_name', '@rpath/libantfly.dylib', '-headerpad_max_install_names', + '-dead_strip', '-adhoc_codesign', '-exported_symbols_list', str(args.exports), + '-o', str(args.output), '-force_load', str(args.archive), *link_args, '-lSystem'] + # Each declared API is a link root, including APIs owned by a provider + # archive that otherwise has no undefined reference from the C API object. + for name in args.exports.read_text().splitlines(): + command.extend(['-u', name]) + subprocess.run(command, check=True) + + +def consumer_test(header: Path, library: Path, source: Path, work: Path, object_format: str) -> None: + cc = shutil.which('clang') or shutil.which('cc') + if not cc: + raise ValueError('consumer linkage regression requires clang or cc') + work.mkdir(parents=True, exist_ok=True) + object_file = work / 'consumer.o' + shared_object = work / 'consumer-shared.o' + common = [cc, '-std=c11', '-Wall', '-Wextra', '-Werror', '-fPIC', '-I', str(header.parent)] + subprocess.run([*common, '-c', str(source), '-o', str(object_file)], check=True) + subprocess.run([*common, '-DCONSUMER_SHARED', '-c', str(source), '-o', str(shared_object)], check=True) + executable = work / 'consumer' + shared = work / ('consumer.dylib' if object_format == 'macho' else 'consumer.so') + # Put the dependency first: a leaked __dso_handle must not bind a later + # consumer object's direct relocation to libantfly's copy. + linking = [cc, '-Xlinker', '-rpath', '-Xlinker', str(library.parent)] + inputs = lambda obj: [str(library), str(obj)] if object_format == 'macho' else [str(obj), str(library)] + subprocess.run([*linking, *inputs(object_file), '-o', str(executable)], check=True) + subprocess.run([*linking, '-dynamiclib' if object_format == 'macho' else '-shared', + *inputs(shared_object), '-o', str(shared)], check=True) + subprocess.run([str(executable)], check=True) + consumer = ctypes.CDLL(str(shared)) + consumer.consumer_check.restype = ctypes.c_int + if consumer.consumer_check() != 0: + raise ValueError('shared consumer check failed') + print('libantfly executable and shared-library consumer linkage passed') + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + commands = parser.add_subparsers(dest='command', required=True) + generate = commands.add_parser('manifest') + generate.add_argument('--header', type=Path, required=True) + generate.add_argument('--output', type=Path, required=True) + generate.add_argument('--format', choices=['macho', 'elf'], required=True) + verify = commands.add_parser('check') + verify.add_argument('--header', type=Path, required=True) + verify.add_argument('--library', type=Path, required=True) + verify.add_argument('--format', choices=['macho', 'elf'], required=True) + verify.add_argument('--nm') + consumer = commands.add_parser('consumer-test') + consumer.add_argument('--header', type=Path, required=True) + consumer.add_argument('--library', type=Path, required=True) + consumer.add_argument('--source', type=Path, required=True) + consumer.add_argument('--work', type=Path, required=True) + consumer.add_argument('--format', choices=['macho', 'elf'], required=True) + link = commands.add_parser('link-macho') + link.add_argument('--linker', required=True) + link.add_argument('--sdk', type=Path, required=True) + link.add_argument('--deployment', required=True) + link.add_argument('--arch', choices=['arm64', 'x86_64'], required=True) + link.add_argument('--exports', type=Path, required=True) + link.add_argument('--archive', type=Path, required=True) + link.add_argument('--output', type=Path, required=True) + link.add_argument('link_args', nargs=argparse.REMAINDER) + args = parser.parse_args() + try: + if args.command == 'manifest': + manifest(args.header, args.output, args.format) + elif args.command == 'check': + check(args.header, args.library, args.format, args.nm) + elif args.command == 'consumer-test': + consumer_test(args.header, args.library, args.source, args.work, args.format) + else: + if args.link_args[:1] == ['--']: + args.link_args = args.link_args[1:] + link_macho(args) + except (ValueError, OSError, subprocess.CalledProcessError) as error: + parser.exit(1, f'{error}\n') + + +if __name__ == '__main__': + main() diff --git a/zig/tools/test_capi_exports.py b/zig/tools/test_capi_exports.py new file mode 100644 index 0000000000..64ff07f1b4 --- /dev/null +++ b/zig/tools/test_capi_exports.py @@ -0,0 +1,113 @@ +#!/usr/bin/env python3 +# Copyright 2026 Antfly, Inc. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +from __future__ import annotations + +import argparse +import importlib.util +from pathlib import Path +import platform +import shutil +import subprocess +import tempfile +import unittest +from unittest.mock import patch + +spec = importlib.util.spec_from_file_location('capi_exports', Path(__file__).with_name('capi_exports.py')) +exports = importlib.util.module_from_spec(spec) +spec.loader.exec_module(exports) + + +class ExportPolicyTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.root = Path(self.tmp.name) + self.header = self.root / 'antfly.h' + self.header.write_text(''' +/* Calling antfly_comment_only() is not an API declaration. */ +// antfly_line_comment() +typedef void (*antfly_callback)(void *); +int antfly_first(void); +void antfly_second( + int argument); +''') + + def test_manifest_is_exact_and_excludes_callbacks_and_comments(self): + self.assertEqual(exports.public_symbols(self.header), {'antfly_first', 'antfly_second'}) + macho, elf = self.root / 'exports', self.root / 'map' + exports.manifest(self.header, macho, 'macho') + exports.manifest(self.header, elf, 'elf') + self.assertEqual(macho.read_text(), '_antfly_first\n_antfly_second\n') + self.assertIn('antfly_first;', elf.read_text()) + self.assertIn('local: *;', elf.read_text()) + self.assertNotIn('antfly_*', elf.read_text()) + + def test_checker_rejects_internal_exports_even_with_antfly_prefix(self): + for name in ['antfly_private', 'termite_metal_buffer_alloc', '__dso_handle', 'memcpy']: + with self.subTest(symbol=name), patch.object(exports, 'exported_symbols', return_value={'antfly_first', 'antfly_second', name}): + with self.assertRaisesRegex(ValueError, 'unexpected='): + exports.check(self.header, self.root / 'lib', 'macho') + + def test_checker_rejects_missing_public_functions(self): + with patch.object(exports, 'exported_symbols', return_value={'antfly_first'}): + with self.assertRaisesRegex(ValueError, 'antfly_second'): + exports.check(self.header, self.root / 'lib', 'elf') + + def test_nm_normalization_preserves_internal_leaks(self): + output = '_antfly_first T 100 0\n___dso_handle D 200 0\n' + with patch.object(exports.subprocess, 'run', return_value=subprocess.CompletedProcess([], 0, output, '')): + self.assertEqual(exports.exported_symbols(self.root / 'lib', 'macho', 'nm'), {'antfly_first', '__dso_handle'}) + + @unittest.skipUnless(platform.system() in {'Darwin', 'Linux'} and (shutil.which('clang') or shutil.which('cc')) and shutil.which('ar'), 'native compiler and archiver required') + def test_native_link_policy_and_executable_and_shared_consumers(self): + self.header.write_text(''' +#include +typedef int antfly_error_code; +typedef struct { uint32_t size; } antfly_open_options; +#define ANTFLY_OK 0 +uint32_t antfly_abi_version(void); +antfly_error_code antfly_open_options_init(antfly_open_options *); +''') + source = self.root / 'api.c' + source.write_text(''' +#include "antfly.h" +uint32_t antfly_abi_version(void) { return 1; } +int antfly_open_options_init(antfly_open_options *options) { options->size = sizeof(*options); return 0; } +int antfly_private(void) { return 7; } +int termite_metal_buffer_alloc(void) { return 8; } +''') + object_file = self.root / 'api.o' + cc = shutil.which('clang') or shutil.which('cc') + subprocess.run([cc, '-fPIC', '-c', str(source), '-o', str(object_file)], check=True) + macho = platform.system() == 'Darwin' + fmt = 'macho' if macho else 'elf' + manifest = self.root / 'exports' + exports.manifest(self.header, manifest, fmt) + library = self.root / ('libantfly.dylib' if macho else 'libantfly.so') + if macho: + sdk = Path(subprocess.check_output(['xcrun', '--show-sdk-path'], text=True).strip()) + archive = self.root / 'api.a' + subprocess.run(['ar', 'rcs', str(archive), str(object_file)], check=True) + exports.link_macho(argparse.Namespace(sdk=sdk, linker='/usr/bin/ld', arch='arm64' if platform.machine() == 'arm64' else 'x86_64', deployment='11.0', exports=manifest, output=library, archive=archive, link_args=[str(object_file)])) + else: + subprocess.run([cc, '-shared', str(object_file), '-Xlinker', '--version-script', '-Xlinker', str(manifest), '-o', str(library)], check=True) + exports.check(self.header, library, fmt) + consumer = Path(__file__).parents[1] / 'pkg/antfly-embedded/tests/capi_link_consumer.c' + exports.consumer_test(self.header, library, consumer, self.root / 'consumer', fmt) + + +if __name__ == '__main__': + unittest.main() From 7798002373d62a56f4b02e2482880c72eed0f7c9 Mon Sep 17 00:00:00 2001 From: AJ Roetker Date: Thu, 8 Oct 2026 13:24:09 -0700 Subject: [PATCH 4/4] fix(lite): fence decoder waiter lifetime through notification --- zig/build_support/embedded/shared_library.zig | 1 + .../src/storage/lsm_backend/local_reader.zig | 45 ++++++++++++++++++- .../tests/capi_link_consumer.c | 1 + zig/tools/capi_exports.py | 1 + zig/tools/test_capi_exports.py | 1 + 5 files changed, 48 insertions(+), 1 deletion(-) diff --git a/zig/build_support/embedded/shared_library.zig b/zig/build_support/embedded/shared_library.zig index 40214a8811..581e8214f6 100644 --- a/zig/build_support/embedded/shared_library.zig +++ b/zig/build_support/embedded/shared_library.zig @@ -12,6 +12,7 @@ // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. // See the License for the specific language governing permissions and // limitations under the License. + //! Final shared-library exports belong to the public header, never to linked archives. const std = @import("std"); diff --git a/zig/pkg/antfly-embedded/src/storage/lsm_backend/local_reader.zig b/zig/pkg/antfly-embedded/src/storage/lsm_backend/local_reader.zig index 3e6810fb36..f36fb0c6fe 100644 --- a/zig/pkg/antfly-embedded/src/storage/lsm_backend/local_reader.zig +++ b/zig/pkg/antfly-embedded/src/storage/lsm_backend/local_reader.zig @@ -104,7 +104,8 @@ pub const Pool = struct { } } /// Caller owns mutex; this returns with it unlocked. The stack waiter is - /// detached before notification, so neither reuse nor teardown races it. + /// detached before notification. Reacquiring the mutex below fences the + /// notifier's final access, including Event.set's wake operation. fn waitLocked(self: *Pool, io: ?std.Io) void { var waiter = Waiter{ .next = self.waiters, .io = io }; self.waiters = &waiter; @@ -112,6 +113,10 @@ pub const Pool = struct { if (io) |owned| waiter.done.waitUncancelable(owned) else { while (!waiter.notified.load(.acquire)) platform.time.yieldBriefly(); } + // Waking is not proof that notifyLocked has finished accessing this + // stack frame. Keep it alive until the notifier releases the mutex. + platform.sync.lockYielding(&self.mutex); + self.mutex.unlock(); } pub const Workspace = struct { @@ -323,3 +328,41 @@ test "lsm local decoder byte gate admits oversized work alone and wakes waiters" try std.testing.expectEqual(@as(usize, 128 * 1024), pool.peak_active_bytes); try std.testing.expectEqual(@as(usize, 0), pool.active); } + +test "lsm local waiter keeps stack alive until notifier unlocks with and without io" { + const Worker = struct { + pool: *Pool, + io: ?std.Io, + returned: std.Io.Event = .unset, + fn run(self: *@This()) void { + platform.sync.lockYielding(&self.pool.mutex); + self.pool.waitLocked(self.io); + self.returned.set(std.testing.io); + } + }; + for ([_]?std.Io{ null, std.testing.io }) |io| { + var pool: Pool = .{}; + defer pool.deinit(); + var worker = Worker{ .pool = &pool, .io = io }; + const thread = try std.Thread.spawn(.{}, Worker.run, .{&worker}); + defer thread.join(); + const deadline = std.Io.Clock.awake.now(std.testing.io).addDuration(.fromSeconds(5)); + while (true) { + platform.sync.lockYielding(&pool.mutex); + if (pool.waiters != null) break; + pool.mutex.unlock(); + if (std.Io.Clock.awake.now(std.testing.io).nanoseconds >= deadline.nanoseconds) return error.WaiterNotRegistered; + platform.time.yieldBriefly(); + } + pool.notifyLocked(); + const returned_early = blk: { + defer pool.mutex.unlock(); + // Give the awakened thread an opportunity to return while the + // notifier still owns the stack-lifetime fence. + std.testing.io.sleep(.fromMilliseconds(100), .awake) catch {}; + break :blk worker.returned.isSet(); + }; + worker.returned.waitUncancelable(std.testing.io); + try std.testing.expect(!returned_early); + } +} diff --git a/zig/pkg/antfly-embedded/tests/capi_link_consumer.c b/zig/pkg/antfly-embedded/tests/capi_link_consumer.c index ec6b6bd545..f7b999b21c 100644 --- a/zig/pkg/antfly-embedded/tests/capi_link_consumer.c +++ b/zig/pkg/antfly-embedded/tests/capi_link_consumer.c @@ -12,6 +12,7 @@ // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. // See the License for the specific language governing permissions and // limitations under the License. + #include "antfly.h" #include diff --git a/zig/tools/capi_exports.py b/zig/tools/capi_exports.py index 24a7cf18ea..ca32c2ea87 100644 --- a/zig/tools/capi_exports.py +++ b/zig/tools/capi_exports.py @@ -13,6 +13,7 @@ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. + """Generate and verify libantfly's export boundary from its public C header.""" from __future__ import annotations diff --git a/zig/tools/test_capi_exports.py b/zig/tools/test_capi_exports.py index 64ff07f1b4..276f884418 100644 --- a/zig/tools/test_capi_exports.py +++ b/zig/tools/test_capi_exports.py @@ -13,6 +13,7 @@ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. + from __future__ import annotations import argparse