-
Notifications
You must be signed in to change notification settings - Fork 248
perf(table): run compaction groups concurrently in RewriteDataFiles #2041
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
base: main
Are you sure you want to change the base?
Changes from all commits
bc37781
23ef481
e625382
18958a9
2b706fb
7657da6
bae7b26
a283bb2
6c9a36e
1116187
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change | ||||||||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
|
@@ -28,6 +28,7 @@ import ( | |||||||||||
| "github.com/apache/iceberg-go" | ||||||||||||
| iceberginternal "github.com/apache/iceberg-go/internal" | ||||||||||||
| iceio "github.com/apache/iceberg-go/io" | ||||||||||||
| "golang.org/x/sync/errgroup" | ||||||||||||
| ) | ||||||||||||
|
|
||||||||||||
| // RewriteResult summarizes a completed compaction. | ||||||||||||
|
|
@@ -217,6 +218,30 @@ type RewriteDataFilesOptions struct { | |||||||||||
| // size, scan concurrency). See the With* helpers returning | ||||||||||||
| // [CompactionGroupOption]. | ||||||||||||
| GroupOptions []CompactionGroupOption | ||||||||||||
|
|
||||||||||||
| // MaxConcurrentGroups bounds how many compaction groups run at once. | ||||||||||||
| // Zero and one both mean sequential execution, which is the default. | ||||||||||||
| // Larger values run [ExecuteCompactionGroup] calls under a bounded | ||||||||||||
| // errgroup and apply their results in the original group order, so | ||||||||||||
| // manifests and [RewriteResult] are identical to a sequential run. | ||||||||||||
| // A failure cancels the groups still running; the returned error | ||||||||||||
| // is the lowest-index failure not caused by that cancellation, or | ||||||||||||
| // the lowest-index failure when every group was canceled, so | ||||||||||||
| // failures match a sequential run. Every group error is logged. | ||||||||||||
| // Peak record-pipeline memory is MaxConcurrentGroups times the | ||||||||||||
| // per-group bound stated on [WithCompactionArrowBatchSize]: | ||||||||||||
| // | ||||||||||||
| // MaxConcurrentGroups x (workers x (rows in the largest task + n) + (recordBatchBufferSize + 2) x n) | ||||||||||||
| // | ||||||||||||
| // rows, where workers, n and recordBatchBufferSize are the per-group | ||||||||||||
| // values. Multiply rows by the average row width in bytes for a byte | ||||||||||||
| // estimate. Delete-side memory is outside this bound. File-open | ||||||||||||
| // fan-out multiplies too: every group scans with up to | ||||||||||||
| // [WithCompactionScanConcurrency] workers, so N groups open about N | ||||||||||||
| // times the scan worker count in files at once. Size the two knobs | ||||||||||||
| // together against connection and file descriptor limits. Negative | ||||||||||||
| // values are rejected with [ErrInvalidOperation]. | ||||||||||||
| MaxConcurrentGroups int | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| // CompactionGroupOption configures a single [ExecuteCompactionGroup] | ||||||||||||
|
|
@@ -252,7 +277,10 @@ func WithCompactionTargetFileSize(size int64) CompactionGroupOption { | |||||||||||
| // The scan runs min(n, number of tasks) workers, and each worker holds | ||||||||||||
| // the decoded batches of at most one task until the writer has taken | ||||||||||||
| // them, so the worker count multiplies the read-side term of the | ||||||||||||
| // memory bound stated on [WithCompactionArrowBatchSize]. | ||||||||||||
| // memory bound stated on [WithCompactionArrowBatchSize]. It also | ||||||||||||
| // multiplies [RewriteDataFilesOptions.MaxConcurrentGroups] for file | ||||||||||||
| // opens: N groups scan with up to n workers each, so about N times n | ||||||||||||
| // files are open at once. | ||||||||||||
| func WithCompactionScanConcurrency(n int) CompactionGroupOption { | ||||||||||||
| return func(c *compactionGroupConfig) { | ||||||||||||
| c.scanConcurrency = n | ||||||||||||
|
|
@@ -337,6 +365,9 @@ func (t *Transaction) RewriteDataFiles(ctx context.Context, groups []CompactionT | |||||||||||
| if _, err := t.txnMeta(); err != nil { | ||||||||||||
| return nil, err | ||||||||||||
| } | ||||||||||||
| if opts.MaxConcurrentGroups < 0 { | ||||||||||||
| return nil, fmt.Errorf("%w: MaxConcurrentGroups must be non-negative", ErrInvalidOperation) | ||||||||||||
| } | ||||||||||||
| if opts.PartialProgress { | ||||||||||||
| return t.rewriteDataFilesPartial(ctx, groups, opts) | ||||||||||||
| } | ||||||||||||
|
|
@@ -348,31 +379,38 @@ func (t *Transaction) RewriteDataFiles(ctx context.Context, groups []CompactionT | |||||||||||
| rewrite := t.NewRewrite(opts.SnapshotProps) | ||||||||||||
| stagedDeleteFiles := make(map[string]struct{}) | ||||||||||||
|
|
||||||||||||
| for _, group := range groups { | ||||||||||||
| if err := ctx.Err(); err != nil { | ||||||||||||
| return result, err | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| if len(group.Tasks) == 0 { | ||||||||||||
| continue | ||||||||||||
| } | ||||||||||||
| fs, err := t.tbl.fsF(ctx) | ||||||||||||
|
Member
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This doesn't fully close r4108885244. The handle is opened with the caller's Open the cleanup handle detached from cancellation. The partial path has the same pattern at :746 (outside this diff); please change it there too.
Suggested change
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Opening fs once up front closes the canceled-open symptom for local backends, but not the leak on cloud. I'd open the cleanup FS with |
||||||||||||
| if err != nil { | ||||||||||||
| return result, fmt.Errorf("open table IO for atomic rewrite: %w", err) | ||||||||||||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This open is now unconditional, so a rewrite with no non-empty groups, which used to return |
||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| gr, err := ExecuteCompactionGroup(ctx, t.tbl, group, opts.GroupOptions...) | ||||||||||||
| var applied []CompactionGroupResult | ||||||||||||
| if opts.MaxConcurrentGroups > 1 { | ||||||||||||
| results, err := executeCompactionGroups(ctx, t.tbl, groups, opts.GroupOptions, opts.MaxConcurrentGroups) | ||||||||||||
| if err != nil { | ||||||||||||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. The concurrent atomic path leaks output files on failure, and the concurrency makes it strictly worse than the sequential loop. When a group fails here we return without touching The partial-progress branch already handles this via
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Done in 30ffe36. The concurrent branch passes its results to cleanupAtomicRewriteOutputs at table/rewrite_data_files.go:382, and the sequential branch tracks applied results and cleans them at :391 and :400. The helper at :453 opens the IO through t.tbl.fsF and joins a cleanup failure with the cause instead of replacing it. |
||||||||||||
| return result, err | ||||||||||||
| return result, cleanupAtomicRewriteOutputs(fs, results, err) | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| if len(gr.OldDataFiles) == 0 && len(gr.NewDataFiles) == 0 { | ||||||||||||
| continue | ||||||||||||
| applied = results | ||||||||||||
| for _, gr := range results { | ||||||||||||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This per-group apply block (skip empty, Could we factor the per-group apply into a small helper shared by both branches? func applyGroupResult(rewrite *Rewrite, result *RewriteResult, staged map[string]struct{}, gr CompactionGroupResult) { ... }Same shape for the partial accumulation. Thoughts?
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Done in 30ffe36. Both atomic branches call applyAtomicGroupResult at table/rewrite_data_files.go:439, and both partial branches call appendPartialGroupResult at :843. |
||||||||||||
| applyAtomicGroupResult(rewrite, result, stagedDeleteFiles, gr) | ||||||||||||
| } | ||||||||||||
| } else { | ||||||||||||
| for _, group := range groups { | ||||||||||||
| if err := ctx.Err(); err != nil { | ||||||||||||
| return result, cleanupAtomicRewriteOutputs(fs, applied, err) | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| rewrite.ApplyResult(gr) | ||||||||||||
| accumulateGroupMetrics(result, gr) | ||||||||||||
| for _, df := range gr.SafePosDeletes { | ||||||||||||
| stagedDeleteFiles[df.FilePath()] = struct{}{} | ||||||||||||
| } | ||||||||||||
| for _, df := range gr.SafeDeletionVectors { | ||||||||||||
| stagedDeleteFiles[df.FilePath()] = struct{}{} | ||||||||||||
| if len(group.Tasks) == 0 { | ||||||||||||
| continue | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| gr, err := ExecuteCompactionGroup(ctx, t.tbl, group, opts.GroupOptions...) | ||||||||||||
| if err != nil { | ||||||||||||
| return result, cleanupAtomicRewriteOutputs(fs, append(applied, gr), err) | ||||||||||||
| } | ||||||||||||
| applied = append(applied, gr) | ||||||||||||
|
|
||||||||||||
| applyAtomicGroupResult(rewrite, result, stagedDeleteFiles, gr) | ||||||||||||
| } | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
|
|
@@ -401,12 +439,96 @@ func (t *Transaction) RewriteDataFiles(ctx context.Context, groups []CompactionT | |||||||||||
| } | ||||||||||||
|
|
||||||||||||
| if err := rewrite.Commit(ctx); err != nil { | ||||||||||||
| return result, fmt.Errorf("commit compaction: %w", err) | ||||||||||||
| return result, cleanupAtomicRewriteOutputs(fs, applied, fmt.Errorf("commit compaction: %w", err)) | ||||||||||||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Wiring cleanup onto |
||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| return result, nil | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| func applyAtomicGroupResult(rewrite *RewriteFiles, result *RewriteResult, stagedDeleteFiles map[string]struct{}, gr CompactionGroupResult) { | ||||||||||||
| if len(gr.OldDataFiles) == 0 && len(gr.NewDataFiles) == 0 { | ||||||||||||
| return | ||||||||||||
| } | ||||||||||||
| rewrite.ApplyResult(gr) | ||||||||||||
| accumulateGroupMetrics(result, gr) | ||||||||||||
| for _, df := range gr.SafePosDeletes { | ||||||||||||
| stagedDeleteFiles[df.FilePath()] = struct{}{} | ||||||||||||
| } | ||||||||||||
| for _, df := range gr.SafeDeletionVectors { | ||||||||||||
| stagedDeleteFiles[df.FilePath()] = struct{}{} | ||||||||||||
| } | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| func cleanupAtomicRewriteOutputs(fs iceio.IO, results []CompactionGroupResult, cause error) error { | ||||||||||||
| if err := cleanupCompactionOutputs(fs, results); err != nil { | ||||||||||||
| return errors.Join(cause, fmt.Errorf("clean up atomic rewrite outputs: %w", err)) | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| return cause | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| func executeCompactionGroups(ctx context.Context, tbl *Table, groups []CompactionTaskGroup, groupOpts []CompactionGroupOption, maxConcurrentGroups int) ([]CompactionGroupResult, error) { | ||||||||||||
| if err := ctx.Err(); err != nil { | ||||||||||||
| return nil, err | ||||||||||||
| } | ||||||||||||
| limit := min(maxConcurrentGroups, len(groups)) | ||||||||||||
| if limit < 1 { | ||||||||||||
| limit = 1 | ||||||||||||
| } | ||||||||||||
|
Comment on lines
+474
to
+477
Member
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. main enabled the
Suggested change
|
||||||||||||
| var g errgroup.Group | ||||||||||||
| g.SetLimit(limit) | ||||||||||||
| runCtx, cancelRuns := context.WithCancel(ctx) | ||||||||||||
| defer cancelRuns() | ||||||||||||
| results := make([]CompactionGroupResult, len(groups)) | ||||||||||||
| groupErrs := make([]error, len(groups)) | ||||||||||||
| for i, group := range groups { | ||||||||||||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. After |
||||||||||||
| if len(group.Tasks) == 0 { | ||||||||||||
| continue | ||||||||||||
| } | ||||||||||||
| g.Go(func() error { | ||||||||||||
| gr, err := ExecuteCompactionGroup(runCtx, tbl, group, groupOpts...) | ||||||||||||
| results[i] = gr | ||||||||||||
| groupErrs[i] = err | ||||||||||||
| if err != nil { | ||||||||||||
| slog.Warn("compaction group failed", "index", i, "err", err) | ||||||||||||
| cancelRuns() | ||||||||||||
|
Comment on lines
+492
to
+494
Member
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Nit: every group that gets cancelled because a sibling failed also logs here, so one real failure with N groups in flight produces N |
||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| return err | ||||||||||||
| }) | ||||||||||||
| } | ||||||||||||
| if err := g.Wait(); err != nil { | ||||||||||||
| var firstErr, firstNonContextErr error | ||||||||||||
| for _, groupErr := range groupErrs { | ||||||||||||
| if groupErr == nil { | ||||||||||||
| continue | ||||||||||||
| } | ||||||||||||
| if firstErr == nil { | ||||||||||||
| firstErr = groupErr | ||||||||||||
| } | ||||||||||||
| if !errors.Is(groupErr, context.Canceled) && !errors.Is(groupErr, context.DeadlineExceeded) { | ||||||||||||
|
Member
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
Suggested change
|
||||||||||||
| firstNonContextErr = groupErr | ||||||||||||
|
|
||||||||||||
| break | ||||||||||||
| } | ||||||||||||
| } | ||||||||||||
| selected := firstNonContextErr | ||||||||||||
| if selected == nil { | ||||||||||||
| selected = firstErr | ||||||||||||
| } | ||||||||||||
| if selected == nil { | ||||||||||||
| selected = err | ||||||||||||
| } | ||||||||||||
| if ctxErr := ctx.Err(); ctxErr != nil && (errors.Is(selected, context.Canceled) || errors.Is(selected, context.DeadlineExceeded)) { | ||||||||||||
| return results, ctxErr | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| return results, selected | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| return results, nil | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| // ExecuteCompactionGroup reads a compaction group's tasks (with | ||||||||||||
| // deletes applied), writes consolidated output files via | ||||||||||||
| // [WriteRecords], and computes the position-delete files safe to | ||||||||||||
|
|
@@ -639,26 +761,26 @@ func (t *Transaction) rewriteDataFilesPartial(ctx context.Context, groups []Comp | |||||||||||
| return cause | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| for _, group := range batchGroups { | ||||||||||||
| if err := ctx.Err(); err != nil { | ||||||||||||
| return result, cleanupBatch(err) | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| gr, err := ExecuteCompactionGroup(ctx, current, group, opts.GroupOptions...) | ||||||||||||
| if opts.MaxConcurrentGroups > 1 { | ||||||||||||
| results, err := executeCompactionGroups(ctx, current, batchGroups, opts.GroupOptions, opts.MaxConcurrentGroups) | ||||||||||||
|
Comment on lines
+764
to
+765
Member
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. In partial-progress mode this only runs groups concurrently inside one commit batch, and batches still run one after another with a catalog commit in between. A batch is |
||||||||||||
| if err != nil { | ||||||||||||
| return result, cleanupBatch(err, gr) | ||||||||||||
| return result, cleanupBatch(err, results...) | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| if len(gr.OldDataFiles) == 0 && len(gr.NewDataFiles) == 0 { | ||||||||||||
| continue | ||||||||||||
| for _, gr := range results { | ||||||||||||
| batchResults, rewrittenFiles = appendPartialGroupResult(batchResults, rewrittenPaths, rewrittenFiles, gr) | ||||||||||||
| } | ||||||||||||
| batchResults = append(batchResults, gr) | ||||||||||||
| for _, df := range gr.OldDataFiles { | ||||||||||||
| if _, ok := rewrittenPaths[df.FilePath()]; ok { | ||||||||||||
| continue | ||||||||||||
| } else { | ||||||||||||
| for _, group := range batchGroups { | ||||||||||||
| if err := ctx.Err(); err != nil { | ||||||||||||
| return result, cleanupBatch(err) | ||||||||||||
| } | ||||||||||||
| rewrittenPaths[df.FilePath()] = struct{}{} | ||||||||||||
| rewrittenFiles = append(rewrittenFiles, df) | ||||||||||||
|
|
||||||||||||
| gr, err := ExecuteCompactionGroup(ctx, current, group, opts.GroupOptions...) | ||||||||||||
| if err != nil { | ||||||||||||
| return result, cleanupBatch(err, gr) | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| batchResults, rewrittenFiles = appendPartialGroupResult(batchResults, rewrittenPaths, rewrittenFiles, gr) | ||||||||||||
| } | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
|
|
@@ -752,6 +874,22 @@ func (t *Transaction) rewriteDataFilesPartial(ctx context.Context, groups []Comp | |||||||||||
| return result, nil | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| func appendPartialGroupResult(batchResults []CompactionGroupResult, rewrittenPaths map[string]struct{}, rewrittenFiles []iceberg.DataFile, gr CompactionGroupResult) ([]CompactionGroupResult, []iceberg.DataFile) { | ||||||||||||
| if len(gr.OldDataFiles) == 0 && len(gr.NewDataFiles) == 0 { | ||||||||||||
| return batchResults, rewrittenFiles | ||||||||||||
| } | ||||||||||||
| batchResults = append(batchResults, gr) | ||||||||||||
| for _, df := range gr.OldDataFiles { | ||||||||||||
| if _, ok := rewrittenPaths[df.FilePath()]; ok { | ||||||||||||
| continue | ||||||||||||
| } | ||||||||||||
| rewrittenPaths[df.FilePath()] = struct{}{} | ||||||||||||
| rewrittenFiles = append(rewrittenFiles, df) | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| return batchResults, rewrittenFiles | ||||||||||||
| } | ||||||||||||
|
|
||||||||||||
| func recordCommittedRewriteBatch( | ||||||||||||
| result *RewriteResult, | ||||||||||||
| next *Table, | ||||||||||||
|
|
||||||||||||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Worth documenting here that under PartialProgress the groups are split into
ceil(len(groups)/MaxCommits)batches and MaxConcurrentGroups only fans out within a batch. With the default MaxCommits=10, any run of 10 or fewer groups gets one group per batch and this knob does nothing. I'd call out that effective concurrency ismin(MaxConcurrentGroups, ceil(groups/MaxCommits))so partial-mode callers aren't surprised when the speedup disappears.