diff --git a/src/iceberg/CMakeLists.txt b/src/iceberg/CMakeLists.txt index 8a98274ff..7782ba164 100644 --- a/src/iceberg/CMakeLists.txt +++ b/src/iceberg/CMakeLists.txt @@ -26,6 +26,7 @@ set(ICEBERG_SOURCES catalog/memory/in_memory_catalog.cc catalog/session_catalog.cc catalog/session_context.cc + compaction_planner.cc delete_file_index.cc deletes/dv_util.cc deletes/dv_writer.cc @@ -237,11 +238,13 @@ if(MSVC_TOOLCHAIN) endif() set(ICEBERG_DATA_SOURCES + data/compaction_executor.cc data/data_writer.cc data/delete_filter.cc data/delete_loader.cc data/equality_delete_writer.cc data/file_scan_task_reader.cc + data/position_delete_update.cc data/position_delete_writer.cc data/writer.cc) diff --git a/src/iceberg/arrow/arrow_io.cc b/src/iceberg/arrow/arrow_io.cc index 4c795badf..11f905da6 100644 --- a/src/iceberg/arrow/arrow_io.cc +++ b/src/iceberg/arrow/arrow_io.cc @@ -591,7 +591,14 @@ Result> ArrowFileSystemFileIO::NewOutputFile( /// \brief Delete a file at the given location. Status ArrowFileSystemFileIO::DeleteFile(const std::string& file_location) { ICEBERG_ASSIGN_OR_RAISE(auto path, ResolvePath(file_location)); - ICEBERG_ARROW_RETURN_NOT_OK(arrow_fs_->DeleteFile(path)); + auto status = arrow_fs_->DeleteFile(path); + if (!status.ok()) { + auto info = arrow_fs_->GetFileInfo(path); + if (info.ok() && info->type() == ::arrow::fs::FileType::NotFound) { + return {}; + } + } + ICEBERG_ARROW_RETURN_NOT_OK(status); return {}; } diff --git a/src/iceberg/compaction_planner.cc b/src/iceberg/compaction_planner.cc new file mode 100644 index 000000000..e0837d257 --- /dev/null +++ b/src/iceberg/compaction_planner.cc @@ -0,0 +1,316 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#include "iceberg/compaction_planner.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "iceberg/expression/literal.h" +#include "iceberg/manifest/manifest_entry.h" +#include "iceberg/table_scan.h" +#include "iceberg/util/content_file_util.h" +#include "iceberg/util/data_file_set.h" +#include "iceberg/util/macros.h" + +namespace iceberg { +namespace { + +struct CanonicalPartitionKey { + int32_t spec_id; + std::string values; + + auto operator<=>(const CanonicalPartitionKey&) const = default; +}; + +struct Candidate { + CompactionFile file; + bool has_delete_pressure; +}; + +struct PartitionCandidates { + int32_t spec_id; + std::vector files; +}; + +Status ValidateConfig(const CompactionPlannerConfig& config) { + ICEBERG_PRECHECK(config.target_file_size_bytes > 0, + "target_file_size_bytes must be greater than zero"); + ICEBERG_PRECHECK(std::isfinite(config.min_file_size_ratio) && + config.min_file_size_ratio >= 0 && config.min_file_size_ratio <= 1, + "min_file_size_ratio must be finite and in [0, 1]"); + ICEBERG_PRECHECK(config.min_input_files > 0, + "min_input_files must be greater than zero"); + ICEBERG_PRECHECK(config.delete_file_threshold >= 0, + "delete_file_threshold must not be negative"); + ICEBERG_PRECHECK(std::isfinite(config.delete_ratio_threshold) && + config.delete_ratio_threshold >= 0 && + config.delete_ratio_threshold <= 1, + "delete_ratio_threshold must be finite and in [0, 1]"); + return {}; +} + +Result CheckedAddNonNegative(int64_t lhs, int64_t rhs, + std::string_view description) { + ICEBERG_PRECHECK(lhs >= 0 && rhs >= 0, "{} must not be negative", description); + ICEBERG_PRECHECK(lhs <= std::numeric_limits::max() - rhs, "{} overflow", + description); + return lhs + rhs; +} + +void AppendFramed(std::string& output, std::string_view value) { + output.append(std::to_string(value.size())); + output.push_back(':'); + output.append(value); +} + +Result MakePartitionKey(const DataFile& data_file) { + ICEBERG_PRECHECK(data_file.partition_spec_id.has_value(), + "Data file '{}' is missing partition_spec_id", data_file.file_path); + + std::string encoded; + encoded.append(std::to_string(data_file.partition.num_fields())); + encoded.push_back(':'); + for (const auto& literal : data_file.partition.values()) { + AppendFramed(encoded, literal.type()->ToString()); + if (literal.IsNull()) { + encoded.push_back('N'); + continue; + } + + encoded.push_back('V'); + if (literal.IsNaN()) { + encoded.push_back('N'); + continue; + } + + encoded.push_back('B'); + ICEBERG_ASSIGN_OR_RAISE(auto bytes, literal.Serialize()); + std::string_view serialized; + if (!bytes.empty()) { + serialized = + std::string_view(reinterpret_cast(bytes.data()), bytes.size()); + } + AppendFramed(encoded, serialized); + } + return CanonicalPartitionKey{.spec_id = *data_file.partition_spec_id, + .values = std::move(encoded)}; +} + +Result BuildCompactionFile( + const std::shared_ptr& scan_task) { + CompactionFile result{.scan_task = scan_task}; + const auto& data_file = scan_task->data_file(); + ICEBERG_PRECHECK(data_file->record_count >= 0, + "Data file '{}' has negative record count", data_file->file_path); + ICEBERG_PRECHECK(data_file->file_size_in_bytes >= 0, + "Data file '{}' has negative file size", data_file->file_path); + + DeleteFileSet applicable_deletes; + for (const auto& delete_file : scan_task->delete_files()) { + ICEBERG_PRECHECK(delete_file != nullptr, "Data file '{}' has a null delete file", + data_file->file_path); + if (delete_file->content != DataFile::Content::kPositionDeletes) { + continue; + } + + ICEBERG_ASSIGN_OR_RAISE(auto referenced_file, + ContentFileUtil::ReferencedDataFile(*delete_file)); + if (!referenced_file.has_value() || *referenced_file != data_file->file_path) { + continue; + } + + if (!applicable_deletes.insert(delete_file).second) { + continue; + } + + ICEBERG_PRECHECK(delete_file->record_count >= 0, + "Delete file '{}' has negative record count", + delete_file->file_path); + if (delete_file->IsDeletionVector()) { + ICEBERG_PRECHECK( + delete_file->record_count <= data_file->record_count, + "Deletion vector '{}' cardinality {} exceeds data file '{}' record count {}", + delete_file->file_path, delete_file->record_count, data_file->file_path, + data_file->record_count); + } + + ICEBERG_ASSIGN_OR_RAISE(result.file_scoped_delete_count, + CheckedAddNonNegative(result.file_scoped_delete_count, 1, + "File-scoped delete-file count")); + ICEBERG_ASSIGN_OR_RAISE(result.file_scoped_delete_record_count, + CheckedAddNonNegative(result.file_scoped_delete_record_count, + delete_file->record_count, + "File-scoped delete-record count")); + } + + result.file_scoped_delete_record_count = + std::min(result.file_scoped_delete_record_count, data_file->record_count); + return result; +} + +bool CompactionFileLess(const CompactionFile& lhs, const CompactionFile& rhs) { + return lhs.scan_task->data_file()->file_path < rhs.scan_task->data_file()->file_path; +} + +bool CandidateLess(const Candidate& lhs, const Candidate& rhs) { + return CompactionFileLess(lhs.file, rhs.file); +} + +bool PackingCandidateLess(const Candidate& lhs, const Candidate& rhs) { + const int64_t lhs_size = lhs.file.scan_task->data_file()->file_size_in_bytes; + const int64_t rhs_size = rhs.file.scan_task->data_file()->file_size_in_bytes; + return lhs_size != rhs_size ? lhs_size > rhs_size : CandidateLess(lhs, rhs); +} + +Result AddToGroup(CompactionGroup& group, Candidate candidate, + bool& has_delete_pressure) { + const auto& data_file = *candidate.file.scan_task->data_file(); + ICEBERG_ASSIGN_OR_RAISE( + group.data_file_size_bytes, + CheckedAddNonNegative(group.data_file_size_bytes, data_file.file_size_in_bytes, + "Compaction group data-file size")); + ICEBERG_ASSIGN_OR_RAISE( + group.file_scoped_delete_record_count, + CheckedAddNonNegative(group.file_scoped_delete_record_count, + candidate.file.file_scoped_delete_record_count, + "Compaction group delete-record count")); + has_delete_pressure |= candidate.has_delete_pressure; + group.files.push_back(std::move(candidate.file)); + return {}; +} + +struct PendingGroup { + CompactionGroup group; + bool has_delete_pressure = false; +}; + +Result> PackPartition( + PartitionCandidates partition, const CompactionPlannerConfig& config) { + const auto representative = std::ranges::min_element(partition.files, CandidateLess) + ->file.scan_task->data_file() + ->partition; + std::ranges::sort(partition.files, PackingCandidateLess); + + std::vector bins; + for (auto& candidate : partition.files) { + const int64_t file_size = candidate.file.scan_task->data_file()->file_size_in_bytes; + size_t best_bin = bins.size(); + int64_t best_remaining = std::numeric_limits::max(); + for (size_t i = 0; i < bins.size(); ++i) { + const int64_t bin_size = bins[i].group.data_file_size_bytes; + if (bin_size > config.target_file_size_bytes || + file_size > config.target_file_size_bytes - bin_size) { + continue; + } + + const int64_t remaining = config.target_file_size_bytes - bin_size - file_size; + if (remaining < best_remaining) { + best_bin = i; + best_remaining = remaining; + } + } + + if (best_bin == bins.size()) { + bins.push_back(PendingGroup{ + .group = CompactionGroup{.partition_spec_id = partition.spec_id, + .partition = representative}, + }); + best_bin = bins.size() - 1; + } + ICEBERG_RETURN_UNEXPECTED(AddToGroup(bins[best_bin].group, std::move(candidate), + bins[best_bin].has_delete_pressure)); + } + + std::vector groups; + for (auto& bin : bins) { + if (!bin.has_delete_pressure && bin.group.files.size() < config.min_input_files) { + continue; + } + std::ranges::sort(bin.group.files, CompactionFileLess); + groups.push_back(std::move(bin.group)); + } + std::ranges::sort(groups, [](const CompactionGroup& lhs, const CompactionGroup& rhs) { + return CompactionFileLess(lhs.files.front(), rhs.files.front()); + }); + return groups; +} + +} // namespace + +Result CompactionPlanner::Plan( + int64_t source_snapshot_id, std::span> scan_tasks, + const CompactionPlannerConfig& config) { + ICEBERG_PRECHECK(source_snapshot_id >= 0, "source_snapshot_id must not be negative"); + ICEBERG_RETURN_UNEXPECTED(ValidateConfig(config)); + std::map partitions; + DataFileSet seen_data_files; + + for (const auto& scan_task : scan_tasks) { + ICEBERG_PRECHECK(scan_task != nullptr, "File scan task must not be null"); + ICEBERG_PRECHECK(scan_task->data_file() != nullptr, + "File scan task data file must not be null"); + const auto& data_file = *scan_task->data_file(); + ICEBERG_PRECHECK(seen_data_files.insert(scan_task->data_file()).second, + "Duplicate scan task for data file '{}'", data_file.file_path); + ICEBERG_ASSIGN_OR_RAISE(auto partition_key, MakePartitionKey(data_file)); + ICEBERG_ASSIGN_OR_RAISE(auto file, BuildCompactionFile(scan_task)); + + const bool is_small = + static_cast(data_file.file_size_in_bytes) < + static_cast(config.target_file_size_bytes) * config.min_file_size_ratio; + const bool has_many_delete_files = + config.delete_file_threshold > 0 && + file.file_scoped_delete_count >= config.delete_file_threshold; + const bool has_high_delete_ratio = + config.delete_ratio_threshold > 0 && data_file.record_count > 0 && + static_cast(file.file_scoped_delete_record_count) / + static_cast(data_file.record_count) >= + config.delete_ratio_threshold; + const bool has_delete_pressure = has_many_delete_files || has_high_delete_ratio; + if (!is_small && !has_delete_pressure) { + continue; + } + + auto partition = + partitions + .try_emplace(std::move(partition_key), + PartitionCandidates{.spec_id = *data_file.partition_spec_id}) + .first; + partition->second.files.push_back( + Candidate{.file = std::move(file), .has_delete_pressure = has_delete_pressure}); + } + + CompactionPlan plan{.source_snapshot_id = source_snapshot_id}; + for (auto& [_, partition] : partitions) { + ICEBERG_ASSIGN_OR_RAISE(auto groups, PackPartition(std::move(partition), config)); + plan.groups.insert(plan.groups.end(), std::make_move_iterator(groups.begin()), + std::make_move_iterator(groups.end())); + } + return plan; +} + +} // namespace iceberg diff --git a/src/iceberg/compaction_planner.h b/src/iceberg/compaction_planner.h new file mode 100644 index 000000000..ac71fdf46 --- /dev/null +++ b/src/iceberg/compaction_planner.h @@ -0,0 +1,121 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#pragma once + +/// \file iceberg/compaction_planner.h +/// Plan data-file compaction without executing rewrites. + +#include +#include +#include +#include +#include + +#include "iceberg/iceberg_export.h" +#include "iceberg/result.h" +#include "iceberg/row/partition_values.h" +#include "iceberg/type_fwd.h" + +namespace iceberg { + +/// \brief Thresholds used to select data files for compaction. +/// Ratio thresholds use double-precision heuristic comparisons. +struct ICEBERG_EXPORT CompactionPlannerConfig { + /// Desired aggregate data-file size for each compaction group. + /// + /// A single selected file may exceed this size. + int64_t target_file_size_bytes = int64_t{512} * 1024 * 1024; + + /// Files smaller than this ratio of the target size are small-file candidates. + double min_file_size_ratio = 0.75; + + /// Minimum number of files required for a group containing only small-file candidates. + size_t min_input_files = 5; + + /// Minimum number of applicable file-scoped position delete files. + /// + /// A value of zero disables this selection criterion. + int64_t delete_file_threshold = 2; + + /// Minimum ratio of deleted records to data-file records. + /// + /// A value of zero disables this selection criterion. + double delete_ratio_threshold = 0.3; +}; + +/// \brief A selected data file and its file-scoped delete pressure. +struct ICEBERG_EXPORT CompactionFile { + /// Scan task for the selected data file and its applicable delete files. + std::shared_ptr scan_task; + + /// Number of distinct applicable file-scoped position delete files. + int64_t file_scoped_delete_count = 0; + + /// Number of deleted records, capped at the data file's record count. + int64_t file_scoped_delete_record_count = 0; +}; + +/// \brief Selected files from one partition that can be rewritten together. +struct ICEBERG_EXPORT CompactionGroup { + /// Partition spec ID shared by every file in this group. + int32_t partition_spec_id; + + /// Partition tuple shared by every file in this group. + PartitionValues partition; + + /// Selected files in canonical file-path order. + std::vector files; + + /// Aggregate size of the selected data files. + int64_t data_file_size_bytes = 0; + + /// Aggregate file-scoped deleted-record count. + int64_t file_scoped_delete_record_count = 0; +}; + +/// \brief Result of compaction planning. +struct ICEBERG_EXPORT CompactionPlan { + /// Snapshot whose scan tasks were used to produce this plan. + /// + /// An executor must verify this snapshot is still valid before rewriting files. + int64_t source_snapshot_id; + + /// Compaction groups in canonical partition and file order. + std::vector groups; +}; + +/// \brief Select and group scan tasks for data-file compaction. +class ICEBERG_EXPORT CompactionPlanner { + public: + /// \brief Plan deterministic, partition-isolated compaction groups. + /// + /// \param source_snapshot_id Non-negative snapshot whose scan tasks are being planned. + /// \param scan_tasks Data-file scan tasks with applicable delete files. + /// \param config Candidate selection and group sizing thresholds. + /// \return A metadata-only compaction plan, or an error for invalid configuration, + /// snapshot ID, duplicate data-file tasks, metadata, partition keys, delete + /// cardinality, or aggregate overflow. + static Result Plan( + int64_t source_snapshot_id, + std::span> scan_tasks, + const CompactionPlannerConfig& config = {}); +}; + +} // namespace iceberg diff --git a/src/iceberg/data/compaction_executor.cc b/src/iceberg/data/compaction_executor.cc new file mode 100644 index 000000000..c39b35b56 --- /dev/null +++ b/src/iceberg/data/compaction_executor.cc @@ -0,0 +1,294 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#include "iceberg/data/compaction_executor.h" + +#include +#include +#include +#include +#include +#include +#include + +#include + +#include "iceberg/arrow_c_data_guard_internal.h" +#include "iceberg/compaction_planner.h" +#include "iceberg/data/data_writer.h" +#include "iceberg/data/file_scan_task_reader.h" +#include "iceberg/data/output_file_cleanup_internal.h" +#include "iceberg/file_format.h" +#include "iceberg/file_io.h" +#include "iceberg/location_provider.h" +#include "iceberg/manifest/manifest_entry.h" +#include "iceberg/metadata_columns.h" +#include "iceberg/partition_spec.h" +#include "iceberg/schema.h" +#include "iceberg/snapshot.h" +#include "iceberg/table.h" +#include "iceberg/table_metadata.h" +#include "iceberg/table_properties.h" +#include "iceberg/table_scan.h" +#include "iceberg/update/rewrite_files.h" +#include "iceberg/util/content_file_util.h" +#include "iceberg/util/data_file_set.h" +#include "iceberg/util/macros.h" +#include "iceberg/util/struct_like_set.h" +#include "iceberg/util/uuid.h" + +namespace iceberg { +namespace { + +Result> Next(ArrowArrayStream& stream) { + ArrowArray batch{}; + if (stream.get_next(&stream, &batch) != 0) { + const char* detail = + stream.get_last_error == nullptr ? nullptr : stream.get_last_error(&stream); + return IOError("Failed to read compaction input: {}", + detail == nullptr ? "unknown stream error" : detail); + } + if (batch.release == nullptr) { + return std::nullopt; + } + return batch; +} + +Result> RewriteSchema(const Table& table) { + ICEBERG_ASSIGN_OR_RAISE(auto table_schema, table.schema()); + if (table.metadata()->format_version < 3) { + return table_schema; + } + + std::vector fields(table_schema->fields().begin(), + table_schema->fields().end()); + fields.push_back(MetadataColumns::kRowId); + fields.push_back(MetadataColumns::kLastUpdatedSequenceNumber); + ICEBERG_ASSIGN_OR_RAISE(auto schema, + Schema::Make(std::move(fields), table_schema->schema_id(), + table_schema->IdentifierFieldIds())); + return std::shared_ptr(std::move(schema)); +} + +} // namespace + +class CompactionExecutor::Impl { + public: + explicit Impl(std::shared_ptr table) : table_(std::move(table)) {} + + Status Execute(const CompactionPlan& plan) { + ICEBERG_PRECHECK(!terminal_, "Compaction executor is no longer usable"); + ICEBERG_RETURN_UNEXPECTED(Cleanup()); + if (plan.groups.empty()) return {}; + auto status = ExecutePlan(plan); + if (!status.has_value()) { + if (status.error().kind != ErrorKind::kCommitStateUnknown) { + return internal::FailWithOutputCleanup(std::move(status.error()), *table_->io(), + output_paths_); + } + terminal_ = true; + } + output_paths_.clear(); + return status; + } + + Status Cleanup() { return internal::CleanupOutputFiles(*table_->io(), output_paths_); } + + private: + Status ExecutePlan(const CompactionPlan& plan) { + ICEBERG_ASSIGN_OR_RAISE(auto inputs, ValidatePlan(plan)); + + ICEBERG_RETURN_UNEXPECTED(table_->Refresh()); + ICEBERG_ASSIGN_OR_RAISE(auto snapshot, table_->current_snapshot()); + if (snapshot->snapshot_id != plan.source_snapshot_id) { + return ValidationFailed( + "Compaction plan snapshot {} is stale; current snapshot is {}", + plan.source_snapshot_id, snapshot->snapshot_id); + } + ICEBERG_ASSIGN_OR_RAISE(auto table_schema, table_->schema()); + ICEBERG_ASSIGN_OR_RAISE(auto rewrite_schema, RewriteSchema(*table_)); + ICEBERG_ASSIGN_OR_RAISE( + auto format, FileFormatTypeFromString( + table_->properties().Get(TableProperties::kDefaultFileFormat))); + ICEBERG_PRECHECK(format != FileFormatType::kPuffin, + "Puffin is not a data file format"); + ICEBERG_ASSIGN_OR_RAISE(auto location_provider, table_->location_provider()); + + std::vector> schemas(table_->metadata()->schemas.begin(), + table_->metadata()->schemas.end()); + ICEBERG_ASSIGN_OR_RAISE(auto reader, FileScanTaskReader::Make({ + .io = table_->io(), + .table_schema = std::move(table_schema), + .schemas = std::move(schemas), + .projected_schema = rewrite_schema, + .properties = table_->properties().configs(), + })); + + std::vector> added_files; + for (size_t group_index = 0; group_index < plan.groups.size(); ++group_index) { + ICEBERG_ASSIGN_OR_RAISE(auto file, + RewriteGroup(plan.groups[group_index], group_index, format, + rewrite_schema, *reader, *location_provider)); + if (file != nullptr) added_files.push_back(std::move(file)); + } + + ICEBERG_ASSIGN_OR_RAISE(auto rewrite, table_->NewRewriteFiles()); + rewrite->ValidateFromSnapshot(plan.source_snapshot_id) + .SetDataSequenceNumber(snapshot->sequence_number) + .Rewrite(inputs.data_files, inputs.delete_files, added_files, {}); + + return rewrite->Commit(); + } + + struct PlanInputs { + std::vector> data_files; + std::vector> delete_files; + }; + + Result ValidatePlan(const CompactionPlan& plan) { + ICEBERG_PRECHECK(plan.source_snapshot_id >= 0, + "Compaction plan snapshot ID must be non-negative"); + + DataFileSet data_files; + DeleteFileSet delete_files; + for (const auto& group : plan.groups) { + ICEBERG_PRECHECK(!group.files.empty(), "Compaction group must contain a file"); + for (const auto& compaction_file : group.files) { + ICEBERG_PRECHECK(compaction_file.scan_task != nullptr, + "Compaction file is missing its scan task"); + const auto& data_file = compaction_file.scan_task->data_file(); + ICEBERG_PRECHECK(data_file != nullptr, + "Compaction scan task is missing data file"); + ICEBERG_PRECHECK(data_file->content == DataFile::Content::kData, + "Compaction input is not a data file: {}", data_file->file_path); + ICEBERG_ASSIGN_OR_RAISE(auto partitions_match, + StructLikeEqual(data_file->partition, group.partition)); + ICEBERG_PRECHECK( + data_file->partition_spec_id == group.partition_spec_id && partitions_match, + "Compaction input does not match its group: {}", data_file->file_path); + ICEBERG_PRECHECK(data_files.insert(data_file).second, + "Duplicate compaction input data file: {}", + data_file->file_path); + + for (const auto& delete_file : compaction_file.scan_task->delete_files()) { + ICEBERG_PRECHECK(delete_file != nullptr, + "Compaction scan task contains a null delete file"); + if (delete_file->content != DataFile::Content::kPositionDeletes) { + continue; + } + ICEBERG_ASSIGN_OR_RAISE(auto referenced_file, + ContentFileUtil::ReferencedDataFile(*delete_file)); + if (referenced_file == data_file->file_path) { + delete_files.insert(delete_file); + } + } + } + } + + return PlanInputs{ + .data_files = + std::vector>(data_files.begin(), data_files.end()), + .delete_files = std::vector>(delete_files.begin(), + delete_files.end()), + }; + } + + Result> RewriteGroup( + const CompactionGroup& group, size_t group_index, FileFormatType format, + const std::shared_ptr& rewrite_schema, FileScanTaskReader& reader, + LocationProvider& location_provider) { + ICEBERG_ASSIGN_OR_RAISE( + auto spec, table_->metadata()->PartitionSpecById(group.partition_spec_id)); + + std::unique_ptr writer; + for (const auto& compaction_file : group.files) { + const auto& data_file = compaction_file.scan_task->data_file(); + FileScanTask task(data_file, compaction_file.scan_task->delete_files()); + ICEBERG_ASSIGN_OR_RAISE(auto stream, reader.Open(task)); + nanoarrow::UniqueArrayStream stream_guard; + ArrowArrayStreamMove(&stream, stream_guard.get()); + while (true) { + ICEBERG_ASSIGN_OR_RAISE(auto batch, Next(*stream_guard.get())); + if (!batch.has_value()) { + break; + } + + internal::ArrowArrayGuard batch_guard(&batch.value()); + if (batch->length == 0) { + continue; + } + if (writer == nullptr) { + const auto filename = + std::format("compacted-{}-{}.{}", Uuid::GenerateV7().ToString(), + group_index, ToString(format)); + ICEBERG_ASSIGN_OR_RAISE( + auto output_path, + location_provider.NewDataLocation(*spec, group.partition, filename)); + output_paths_.insert(output_path); + ICEBERG_ASSIGN_OR_RAISE(writer, + DataWriter::Make({ + .path = output_path, + .schema = rewrite_schema, + .spec = spec, + .partition = group.partition, + .format = format, + .io = table_->io(), + .properties = table_->properties().configs(), + })); + } + batch_guard.Release(); + ICEBERG_RETURN_UNEXPECTED(writer->Write(&batch.value())); + } + } + + if (writer != nullptr) { + ICEBERG_RETURN_UNEXPECTED(writer->Close()); + ICEBERG_ASSIGN_OR_RAISE(auto metadata, writer->Metadata()); + ICEBERG_PRECHECK(metadata.data_files.size() == 1, + "Compaction writer produced {} data files", + metadata.data_files.size()); + return std::move(metadata.data_files.front()); + } + return nullptr; + } + + std::shared_ptr
table_; + std::set output_paths_; + bool terminal_ = false; +}; + +CompactionExecutor::CompactionExecutor(std::unique_ptr impl) + : impl_(std::move(impl)) {} + +CompactionExecutor::~CompactionExecutor() = default; + +Result> CompactionExecutor::Make( + std::shared_ptr
table) { + ICEBERG_PRECHECK(table != nullptr, "Cannot create compaction executor without table"); + return std::unique_ptr( + new CompactionExecutor(std::make_unique(std::move(table)))); +} + +Status CompactionExecutor::Execute(const CompactionPlan& plan) { + return impl_->Execute(plan); +} + +Status CompactionExecutor::Cleanup() { return impl_->Cleanup(); } + +} // namespace iceberg diff --git a/src/iceberg/data/compaction_executor.h b/src/iceberg/data/compaction_executor.h new file mode 100644 index 000000000..7a2fe9006 --- /dev/null +++ b/src/iceberg/data/compaction_executor.h @@ -0,0 +1,82 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#pragma once + +/// \file iceberg/data/compaction_executor.h +/// Execute planned data-file compaction. + +#include + +#include "iceberg/iceberg_data_export.h" +#include "iceberg/result.h" +#include "iceberg/type_fwd.h" + +namespace iceberg { + +struct CompactionPlan; + +/// \brief Executes a snapshot-bound data-file compaction plan. +/// +/// The executor reads every planned source file through the delete-aware scan reader, +/// writes replacement data files, and commits all data and file-scoped position-delete +/// replacements in one table update. Generated files remain owned by the executor until +/// commit succeeds, commit state becomes unknown, or cleanup deletes them. +class ICEBERG_DATA_EXPORT CompactionExecutor { + public: + /// \brief Destroy the executor. + ~CompactionExecutor(); + + /// \brief Create an executor for a table. + /// + /// \param table Table whose files and metadata will be rewritten. + /// \return A new executor, or an error if table is null. + static Result> Make(std::shared_ptr
table); + + /// \brief Rewrite and atomically commit a snapshot-bound compaction plan. + /// + /// An empty plan is a no-op without writing or committing. + /// Execution refreshes the table and rejects non-empty plans whose snapshot is no + /// longer current before writing. Commit validation also rejects deletes added after + /// the plan snapshot. A definite failure attempts to clean generated files; cleanup + /// failures are appended to the original error and undeleted files remain owned by + /// this executor. + /// + /// \param plan Planner output containing the source snapshot and compaction groups. + /// \return Success after commit, or the original planning, IO, validation, or commit + /// error, including cleanup details when cleanup also fails. + Status Execute(const CompactionPlan& plan); + + /// \brief Retry deletion of uncommitted output files owned by this executor. + /// + /// Successfully deleted or absent paths are released. Paths whose deletion fails + /// remain owned and may be retried by calling Cleanup again. + /// + /// \return Success when every owned output is deleted, otherwise the first cleanup + /// error. + Status Cleanup(); + + private: + class Impl; + std::unique_ptr impl_; + + explicit CompactionExecutor(std::unique_ptr impl); +}; + +} // namespace iceberg diff --git a/src/iceberg/data/output_file_cleanup_internal.h b/src/iceberg/data/output_file_cleanup_internal.h new file mode 100644 index 000000000..c84b5f91e --- /dev/null +++ b/src/iceberg/data/output_file_cleanup_internal.h @@ -0,0 +1,56 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#pragma once + +#include +#include +#include + +#include "iceberg/file_io.h" +#include "iceberg/result.h" + +namespace iceberg::internal { + +// Try every owned output, keeping failed paths for a subsequent cleanup attempt. +inline Status CleanupOutputFiles(FileIO& io, std::set& paths) { + Status first_error; + for (auto it = paths.begin(); it != paths.end();) { + auto status = io.DeleteFile(*it); + if (status.has_value()) { + it = paths.erase(it); + } else { + if (first_error.has_value()) first_error = std::move(status); + ++it; + } + } + return first_error; +} + +// Preserve the operation's error kind while reporting a cleanup failure, if any. +inline Status FailWithOutputCleanup(Error error, FileIO& io, + std::set& paths) { + auto cleanup = CleanupOutputFiles(io, paths); + if (!cleanup.has_value()) { + error.message += "; output cleanup failed: " + cleanup.error().message; + } + return std::unexpected(std::move(error)); +} + +} // namespace iceberg::internal diff --git a/src/iceberg/data/position_delete_update.cc b/src/iceberg/data/position_delete_update.cc new file mode 100644 index 000000000..5b8182361 --- /dev/null +++ b/src/iceberg/data/position_delete_update.cc @@ -0,0 +1,265 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#include "iceberg/data/position_delete_update.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "iceberg/data/delete_loader.h" +#include "iceberg/data/output_file_cleanup_internal.h" +#include "iceberg/data/position_delete_writer.h" +#include "iceberg/deletes/dv_writer.h" +#include "iceberg/file_format.h" +#include "iceberg/file_io.h" +#include "iceberg/location_provider.h" +#include "iceberg/manifest/manifest_entry.h" +#include "iceberg/partition_spec.h" +#include "iceberg/schema.h" +#include "iceberg/table.h" +#include "iceberg/table_metadata.h" +#include "iceberg/table_properties.h" +#include "iceberg/table_scan.h" +#include "iceberg/update/row_delta.h" +#include "iceberg/util/macros.h" +#include "iceberg/util/string_util.h" +#include "iceberg/util/uuid.h" + +namespace iceberg { + +namespace { + +struct TargetFile { + std::shared_ptr data_file; + std::shared_ptr spec; + std::vector> position_delete_files; +}; + +} // namespace + +class PositionDeleteUpdate::Impl { + public: + explicit Impl(std::shared_ptr
table) : table_(std::move(table)) {} + + Status Delete(std::string_view data_file_path, int64_t pos) { + ICEBERG_PRECHECK(!terminal_, "Position delete update is no longer usable"); + ICEBERG_PRECHECK(!data_file_path.empty(), "Data file path cannot be empty"); + ICEBERG_PRECHECK(pos >= 0, "Position delete must be non-negative: {}", pos); + deletes_[std::string(data_file_path)].push_back(pos); + return {}; + } + + Status Commit() { + ICEBERG_PRECHECK(!terminal_, "Position delete update is no longer usable"); + ICEBERG_PRECHECK(!deletes_.empty(), "Position delete update is empty"); + ICEBERG_PRECHECK(table_->metadata()->format_version >= 2, + "Position deletes require table format version 2 or later"); + ICEBERG_RETURN_UNEXPECTED(internal::CleanupOutputFiles(*table_->io(), output_paths_)); + auto status = CommitDeletes(); + if (!status.has_value() && status.error().kind != ErrorKind::kCommitStateUnknown) { + return internal::FailWithOutputCleanup(std::move(status.error()), *table_->io(), + output_paths_); + } + terminal_ = true; + output_paths_.clear(); + return status; + } + + private: + Status CommitDeletes() { + ICEBERG_ASSIGN_OR_RAISE(auto snapshot, table_->current_snapshot()); + ICEBERG_ASSIGN_OR_RAISE(auto targets, ResolveTargets()); + + ICEBERG_ASSIGN_OR_RAISE(auto written, table_->metadata()->format_version >= 3 + ? WriteDeletionVectors(targets) + : WriteParquetDeletes(targets)); + ICEBERG_ASSIGN_OR_RAISE(auto row_delta, table_->NewRowDelta()); + row_delta->ValidateFromSnapshot(snapshot->snapshot_id) + .ValidateDataFilesExist(written.referenced_data_files) + .ValidateDeletedFiles(); + for (const auto& file : written.data_files) { + row_delta->AddDeletes(file); + } + for (const auto& file : written.rewritten_delete_files) { + row_delta->RemoveDeletes(file); + } + + return row_delta->Commit(); + } + + Result> ResolveTargets() const { + ICEBERG_ASSIGN_OR_RAISE(auto scan_builder, table_->NewScan()); + ICEBERG_ASSIGN_OR_RAISE(auto scan, scan_builder->Build()); + ICEBERG_ASSIGN_OR_RAISE(auto tasks, scan->PlanFilesStream()); + + std::unordered_map targets; + while (targets.size() < deletes_.size()) { + ICEBERG_ASSIGN_OR_RAISE(auto task, tasks->Next()); + if (!task.has_value()) break; + const auto& data_file = (*task)->data_file(); + if (!deletes_.contains(data_file->file_path)) { + continue; + } + + ICEBERG_PRECHECK(data_file->partition_spec_id.has_value(), + "Data file is missing partition spec ID: {}", + data_file->file_path); + ICEBERG_ASSIGN_OR_RAISE(auto spec, table_->metadata()->PartitionSpecById( + *data_file->partition_spec_id)); + + TargetFile target{.data_file = data_file, .spec = std::move(spec)}; + for (const auto& delete_file : (*task)->delete_files()) { + if (delete_file->content == DataFile::Content::kPositionDeletes) { + target.position_delete_files.push_back(delete_file); + } + } + + targets.emplace(data_file->file_path, std::move(target)); + } + + for (const auto& [path, _] : deletes_) { + ICEBERG_PRECHECK(targets.contains(path), "Cannot find live data file: {}", path); + } + return targets; + } + + Result WriteDeletionVectors( + const std::unordered_map& targets) { + ICEBERG_ASSIGN_OR_RAISE(auto location_provider, table_->location_provider()); + auto output_path = location_provider->NewDataLocation( + std::format("position-deletes-{}.puffin", Uuid::GenerateV7().ToString())); + output_paths_.insert(output_path); + + DeleteLoader loader(table_->io()); + ICEBERG_ASSIGN_OR_RAISE( + auto writer, + DVWriter::Make(DVWriterOptions{ + .path = output_path, + .io = table_->io(), + .load_previous_deletes = [&targets, &loader](std::string_view path) + -> Result> { + const auto& previous = targets.at(std::string(path)).position_delete_files; + if (previous.empty()) return std::nullopt; + ICEBERG_ASSIGN_OR_RAISE(auto index, + loader.LoadPositionDeletes(previous, path)); + return std::optional(std::move(index)); + }, + })); + + for (const auto& [path, positions] : deletes_) { + const auto& target = targets.at(path); + for (int64_t pos : positions) { + ICEBERG_RETURN_UNEXPECTED( + writer->Delete(path, pos, target.spec, target.data_file->partition)); + } + } + ICEBERG_RETURN_UNEXPECTED(writer->Close()); + return writer->Metadata(); + } + + Result WriteParquetDeletes( + const std::unordered_map& targets) { + ICEBERG_ASSIGN_OR_RAISE(auto location_provider, table_->location_provider()); + ICEBERG_ASSIGN_OR_RAISE(auto schema, table_->schema()); + + DeleteWriteResult result; + result.data_files.reserve(deletes_.size()); + result.referenced_data_files.reserve(deletes_.size()); + auto properties = table_->properties().configs(); + properties[TableProperties::kParquetCompression.key()] = + table_->properties().Get(TableProperties::kDeleteParquetCompression); + properties[TableProperties::kParquetCompressionLevel.key()] = + table_->properties().Get(TableProperties::kDeleteParquetCompressionLevel); + const auto write_uuid = Uuid::GenerateV7().ToString(); + size_t file_number = 0; + + for (const auto& [path, positions] : deletes_) { + auto sorted_positions = positions; + std::ranges::sort(sorted_positions); + + const auto& target = targets.at(path); + const auto filename = + std::format("position-deletes-{}-{}.parquet", write_uuid, file_number++); + auto output_path = location_provider->NewDataLocation(filename); + output_paths_.insert(output_path); + + ICEBERG_ASSIGN_OR_RAISE(auto writer, + PositionDeleteWriter::Make(PositionDeleteWriterOptions{ + .path = output_path, + .schema = schema, + .spec = target.spec, + .partition = target.data_file->partition, + .format = FileFormatType::kParquet, + .io = table_->io(), + .properties = properties, + })); + for (int64_t pos : sorted_positions) { + ICEBERG_RETURN_UNEXPECTED(writer->WriteDelete(path, pos)); + } + ICEBERG_RETURN_UNEXPECTED(writer->Close()); + ICEBERG_ASSIGN_OR_RAISE(auto metadata, writer->Metadata()); + result.data_files.insert(result.data_files.end(), + std::make_move_iterator(metadata.data_files.begin()), + std::make_move_iterator(metadata.data_files.end())); + result.referenced_data_files.push_back(path); + } + return result; + } + + std::shared_ptr
table_; + std::map, StringLess> deletes_; + std::set output_paths_; + bool terminal_ = false; +}; + +PositionDeleteUpdate::PositionDeleteUpdate(std::unique_ptr impl) + : impl_(std::move(impl)) {} + +PositionDeleteUpdate::~PositionDeleteUpdate() = default; + +Result> PositionDeleteUpdate::Make( + std::shared_ptr
table) { + ICEBERG_PRECHECK(table != nullptr, + "Cannot create position delete update without table"); + return std::unique_ptr( + new PositionDeleteUpdate(std::make_unique(std::move(table)))); +} + +PositionDeleteUpdate& PositionDeleteUpdate::Delete(std::string_view data_file_path, + int64_t pos) { + ICEBERG_BUILDER_RETURN_IF_ERROR(impl_->Delete(data_file_path, pos)); + return *this; +} + +Status PositionDeleteUpdate::Commit() { + ICEBERG_RETURN_UNEXPECTED(CheckErrors()); + return impl_->Commit(); +} + +} // namespace iceberg diff --git a/src/iceberg/data/position_delete_update.h b/src/iceberg/data/position_delete_update.h new file mode 100644 index 000000000..3b531d8fa --- /dev/null +++ b/src/iceberg/data/position_delete_update.h @@ -0,0 +1,60 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#pragma once + +/// \file iceberg/data/position_delete_update.h +/// Table-aware position delete writing and commit. + +#include +#include +#include + +#include "iceberg/iceberg_data_export.h" +#include "iceberg/result.h" +#include "iceberg/type_fwd.h" +#include "iceberg/util/error_collector.h" + +namespace iceberg { + +/// \brief Writes and commits position deletes using the table format. +/// +/// Format v2 tables produce Parquet position delete files. Format v3 tables +/// produce deletion vectors and merge previous file-scoped position deletes. +class ICEBERG_DATA_EXPORT PositionDeleteUpdate : public ErrorCollector { + public: + ~PositionDeleteUpdate() override; + + /// \brief Create a position delete update for a table. + static Result> Make(std::shared_ptr
table); + + /// \brief Add a deleted row position for a live data file. + PositionDeleteUpdate& Delete(std::string_view data_file_path, int64_t pos); + + /// \brief Write and commit all added position deletes. + Status Commit(); + + private: + class Impl; + std::unique_ptr impl_; + + explicit PositionDeleteUpdate(std::unique_ptr impl); +}; + +} // namespace iceberg diff --git a/src/iceberg/delete_file_index.cc b/src/iceberg/delete_file_index.cc index a8c4ef126..2bd6fc733 100644 --- a/src/iceberg/delete_file_index.cc +++ b/src/iceberg/delete_file_index.cc @@ -41,6 +41,7 @@ #include "iceberg/util/content_file_util.h" #include "iceberg/util/executor_util_internal.h" #include "iceberg/util/macros.h" +#include "iceberg/util/struct_like_set.h" namespace iceberg { @@ -458,6 +459,15 @@ Result> DeleteFileIndex::FindDV( "sequence number {}", it->second.sequence_number.value(), seq); + const auto& dv = *it->second.data_file; + ICEBERG_CHECK(dv.partition_spec_id == data_file.partition_spec_id, + "DV and data file have mismatched partition specs: {}", + data_file.file_path); + ICEBERG_ASSIGN_OR_RAISE(auto partitions_match, + StructLikeEqual(dv.partition, data_file.partition)); + ICEBERG_CHECK(partitions_match, "DV and data file have mismatched partitions: {}", + data_file.file_path); + return it->second.data_file; } diff --git a/src/iceberg/file_io.h b/src/iceberg/file_io.h index e22e5cf21..791133fcf 100644 --- a/src/iceberg/file_io.h +++ b/src/iceberg/file_io.h @@ -162,6 +162,9 @@ class ICEBERG_EXPORT FileIO { /// \brief Delete a file at the given location. /// + /// Deletion is idempotent: implementations must return success when the file does + /// not exist. + /// /// \param file_location The location of the file to delete. /// \return void if the delete succeeded, an error code if the delete failed. virtual Status DeleteFile(const std::string& file_location) { diff --git a/src/iceberg/test/CMakeLists.txt b/src/iceberg/test/CMakeLists.txt index 6f6ff7603..a00fe3029 100644 --- a/src/iceberg/test/CMakeLists.txt +++ b/src/iceberg/test/CMakeLists.txt @@ -163,6 +163,8 @@ add_iceberg_test(puffin_test puffin_json_test.cc puffin_reader_writer_test.cc) +add_iceberg_test(compaction_planner_test SOURCES compaction_planner_test.cc) + if(ICEBERG_BUILD_BUNDLE) add_iceberg_test(avro_test USE_BUNDLE @@ -259,12 +261,14 @@ if(ICEBERG_BUILD_BUNDLE) SOURCES arrow_c_data_util_test.cc arrow_row_builder_test.cc + compaction_executor_test.cc data_writer_test.cc default_value_test.cc delete_filter_test.cc delete_loader_test.cc dv_writer_test.cc file_scan_task_reader_test.cc + position_delete_update_test.cc literal_util_test.cc) endif() diff --git a/src/iceberg/test/arrow_io_test.cc b/src/iceberg/test/arrow_io_test.cc index 7bc9ebba5..6b3f6e4dd 100644 --- a/src/iceberg/test/arrow_io_test.cc +++ b/src/iceberg/test/arrow_io_test.cc @@ -364,8 +364,7 @@ TEST_F(LocalFileIOTest, DeleteFile) { EXPECT_THAT(del_res, IsOk()); del_res = file_io_->DeleteFile(temp_filepath_); - EXPECT_THAT(del_res, IsError(ErrorKind::kIOError)); - EXPECT_THAT(del_res, HasErrorMessage("Cannot delete file")); + EXPECT_THAT(del_res, IsOk()); } TEST_F(LocalFileIOTest, DeleteFiles) { @@ -414,6 +413,13 @@ TEST_F(LocalFileIOTest, StdReadFullyReadsFromAbsolutePosition) { VerifyReadFullyReadsFromAbsolutePosition(file_io, temp_filepath_)); } +TEST_F(LocalFileIOTest, StdDeleteFileIsIdempotent) { + auto file_io = std::make_shared(); + ASSERT_THAT(file_io->WriteFile(temp_filepath_, "abc"), IsOk()); + EXPECT_THAT(file_io->DeleteFile(temp_filepath_), IsOk()); + EXPECT_THAT(file_io->DeleteFile(temp_filepath_), IsOk()); +} + TEST_F(LocalFileIOTest, StdReadKeepsPositionAvailableAtEof) { auto file_io = std::make_shared(); ASSERT_THAT(file_io->WriteFile(temp_filepath_, "abc"), IsOk()); diff --git a/src/iceberg/test/compaction_executor_test.cc b/src/iceberg/test/compaction_executor_test.cc new file mode 100644 index 000000000..d0265a58b --- /dev/null +++ b/src/iceberg/test/compaction_executor_test.cc @@ -0,0 +1,761 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#include "iceberg/data/compaction_executor.h" + +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +#include "iceberg/arrow/arrow_io_internal.h" +#include "iceberg/avro/avro_register.h" +#include "iceberg/compaction_planner.h" +#include "iceberg/data/data_writer.h" +#include "iceberg/data/file_scan_task_reader.h" +#include "iceberg/data/position_delete_update.h" +#include "iceberg/file_writer.h" +#include "iceberg/manifest/manifest_entry.h" +#include "iceberg/manifest/manifest_reader.h" +#include "iceberg/metadata_columns.h" +#include "iceberg/parquet/parquet_register.h" +#include "iceberg/partition_spec.h" +#include "iceberg/schema.h" +#include "iceberg/schema_internal.h" +#include "iceberg/snapshot.h" +#include "iceberg/table.h" +#include "iceberg/table_metadata.h" +#include "iceberg/table_properties.h" +#include "iceberg/table_scan.h" +#include "iceberg/test/matchers.h" +#include "iceberg/test/mock_catalog.h" +#include "iceberg/test/update_test_base.h" +#include "iceberg/update/fast_append.h" +#include "iceberg/update/update_properties.h" +#include "iceberg/update/update_schema.h" +#include "iceberg/util/macros.h" + +namespace iceberg { +namespace { + +using Row = std::array; + +enum class FailurePoint { kBeforeCreate, kAfterCreate, kDelete }; + +class FailingFileIO : public FileIO { + public: + FailingFileIO(std::shared_ptr delegate, FailurePoint point) + : delegate_(std::move(delegate)), + failure_(std::make_shared>(point)) {} + + Result> NewInputFile(std::string path) override { + return delegate_->NewInputFile(std::move(path)); + } + Result> NewInputFile(std::string path, + size_t length) override { + return delegate_->NewInputFile(std::move(path), length); + } + Result> NewOutputFile(std::string path) override { + ICEBERG_ASSIGN_OR_RAISE(auto output, delegate_->NewOutputFile(path)); + if (!IsCompacted(path) || *failure_ == FailurePoint::kDelete) return output; + return std::unique_ptr( + new FailingOutputFile(std::move(output), failure_)); + } + Status DeleteFile(const std::string& path) override { + if (IsCompacted(path) && *failure_ == FailurePoint::kDelete) { + return IOError("injected cleanup failure: {}", path); + } + return delegate_->DeleteFile(path); + } + void DisableFailure() { failure_->reset(); } + + private: + class FailingOutputFile : public OutputFile { + public: + FailingOutputFile(std::unique_ptr delegate, + std::shared_ptr> failure) + : delegate_(std::move(delegate)), failure_(std::move(failure)) {} + std::string_view location() const override { return delegate_->location(); } + Result> Create() override { + return Open(false); + } + Result> CreateOrOverwrite() override { + return Open(true); + } + + private: + Result> Open(bool overwrite) { + if (*failure_ == FailurePoint::kBeforeCreate) { + failure_->reset(); + return IOError("injected output failure before creation"); + } + ICEBERG_ASSIGN_OR_RAISE( + auto stream, overwrite ? delegate_->CreateOrOverwrite() : delegate_->Create()); + if (*failure_ == FailurePoint::kAfterCreate) { + failure_->reset(); + return IOError("injected output failure after creation"); + } + return stream; + } + std::unique_ptr delegate_; + std::shared_ptr> failure_; + }; + static bool IsCompacted(std::string_view path) { + return path.find("compacted-") != std::string_view::npos; + } + std::shared_ptr delegate_; + std::shared_ptr> failure_; +}; + +class CompactionExecutorTest : public MinimalUpdateTestBase { + protected: + static void SetUpTestSuite() { + avro::RegisterAll(); + parquet::RegisterAll(); + } + + int8_t format_version() const override { return 3; } + + void SetUp() override { + MinimalUpdateTestBase::SetUp(); + ICEBERG_UNWRAP_OR_FAIL(auto properties, table_->NewUpdateProperties()); + properties->Set(TableProperties::kParquetCompression.key(), "uncompressed") + .Set(TableProperties::kDeleteParquetCompression.key(), "uncompressed") + .Set(TableProperties::kCommitNumRetries.key(), "0"); + ASSERT_THAT(properties->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(schema_, table_->schema()); + ICEBERG_UNWRAP_OR_FAIL(spec_, table_->spec()); + auto arrow_io = std::dynamic_pointer_cast(file_io_); + ASSERT_NE(arrow_io, nullptr); + ASSERT_TRUE(arrow_io->fs()->CreateDir(table_location_ + "/data/x=10").ok()); + ASSERT_TRUE(arrow_io->fs()->CreateDir(table_location_ + "/data/x=20").ok()); + } + + Result> WriteDataFile(std::string_view name, + std::string_view json, + std::optional partition = 10) { + const auto path = std::format("{}/data/{}", table_location_, name); + ICEBERG_ASSIGN_OR_RAISE( + auto writer, + DataWriter::Make({ + .path = path, + .schema = schema_, + .spec = spec_, + .partition = PartitionValues( + {partition ? Literal::Long(*partition) : Literal::Null(int64())}), + .format = FileFormatType::kParquet, + .io = file_io_, + .properties = {{"write.parquet.compression-codec", "uncompressed"}}, + })); + + ArrowSchema c_schema{}; + ICEBERG_RETURN_UNEXPECTED(ToArrowSchema(*schema_, &c_schema)); + auto arrow_type = ::arrow::ImportType(&c_schema); + if (!arrow_type.ok()) { + return UnknownError(arrow_type.status().ToString()); + } + auto array = ::arrow::json::ArrayFromJSONString( + ::arrow::struct_(arrow_type.ValueOrDie()->fields()), std::string(json)); + if (!array.ok()) { + return UnknownError(array.status().ToString()); + } + + ArrowArray c_array{}; + auto export_status = ::arrow::ExportArray(*array.ValueOrDie(), &c_array); + if (!export_status.ok()) { + return UnknownError(export_status.ToString()); + } + ICEBERG_RETURN_UNEXPECTED(writer->Write(&c_array)); + ICEBERG_RETURN_UNEXPECTED(writer->Close()); + ICEBERG_ASSIGN_OR_RAISE(auto metadata, writer->Metadata()); + ICEBERG_CHECK(metadata.data_files.size() == 1, "Expected one data file, found {}", + metadata.data_files.size()); + return metadata.data_files.front(); + } + + Status Append(const std::vector>& files) { + ICEBERG_ASSIGN_OR_RAISE(auto append, table_->NewFastAppend()); + for (const auto& file : files) { + append->AppendFile(file); + } + ICEBERG_RETURN_UNEXPECTED(append->Commit()); + return table_->Refresh(); + } + + Result>> Tasks( + const std::shared_ptr
& table) { + ICEBERG_ASSIGN_OR_RAISE(auto builder, table->NewScan()); + ICEBERG_ASSIGN_OR_RAISE(auto scan, builder->Build()); + return scan->PlanFiles(); + } + + Result Plan(const std::vector>& tasks) { + ICEBERG_ASSIGN_OR_RAISE(auto snapshot, table_->current_snapshot()); + ICEBERG_ASSIGN_OR_RAISE( + auto plan, CompactionPlanner::Plan(snapshot->snapshot_id, tasks, + CompactionPlannerConfig{ + .target_file_size_bytes = 1024 * 1024, + .min_file_size_ratio = 1, + .min_input_files = 1, + })); + return plan; + } + + std::shared_ptr LineageProjection() { + std::vector fields(schema_->fields().begin(), schema_->fields().end()); + fields.push_back(MetadataColumns::kRowId); + fields.push_back(MetadataColumns::kLastUpdatedSequenceNumber); + return std::make_shared(std::move(fields), schema_->schema_id()); + } + + Result> ReadRows(const FileScanTask& task) { + ICEBERG_ASSIGN_OR_RAISE(auto reader, FileScanTaskReader::Make({ + .io = file_io_, + .table_schema = schema_, + .schemas = table_->metadata()->schemas, + .projected_schema = LineageProjection(), + })); + ICEBERG_ASSIGN_OR_RAISE(auto stream, reader->Open(task)); + auto batch_reader = ::arrow::ImportRecordBatchReader(&stream); + if (!batch_reader.ok()) { + return UnknownError(batch_reader.status().ToString()); + } + + std::vector rows; + while (true) { + auto batch_result = batch_reader.ValueOrDie()->Next(); + if (!batch_result.ok()) { + return UnknownError(batch_result.status().ToString()); + } + auto batch = batch_result.ValueOrDie(); + if (batch == nullptr) { + break; + } + ICEBERG_CHECK(batch->num_columns() == 5, "Expected five projected columns"); + std::array, 5> columns; + for (size_t i = 0; i < columns.size(); ++i) { + columns[i] = std::static_pointer_cast<::arrow::Int64Array>(batch->column(i)); + } + for (int64_t row = 0; row < batch->num_rows(); ++row) { + Row values; + for (size_t column = 0; column < columns.size(); ++column) { + ICEBERG_CHECK(!columns[column]->IsNull(row), + "Compaction row lineage column is null"); + values[column] = columns[column]->Value(row); + } + rows.push_back(values); + } + } + return rows; + } + + Result> LiveDeleteEntries() { + ICEBERG_ASSIGN_OR_RAISE(auto snapshot, table_->current_snapshot()); + SnapshotReader cache(snapshot.get()); + ICEBERG_ASSIGN_OR_RAISE(auto manifests, cache.DeleteManifests(file_io_)); + std::vector result; + for (const auto& manifest : manifests) { + ICEBERG_ASSIGN_OR_RAISE( + auto spec, table_->metadata()->PartitionSpecById(manifest.partition_spec_id)); + ICEBERG_ASSIGN_OR_RAISE( + auto reader, + ManifestReader::Make(manifest, file_io_, schema_, std::move(spec))); + ICEBERG_ASSIGN_OR_RAISE(auto entries, reader->LiveEntries()); + result.insert(result.end(), std::make_move_iterator(entries.begin()), + std::make_move_iterator(entries.end())); + } + return result; + } + + Result> FailingCommitTable( + std::unexpected failure, std::shared_ptr io = nullptr) { + auto mock = std::make_shared<::testing::NiceMock>(); + ON_CALL(*mock, LoadTable(::testing::_)) + .WillByDefault([catalog = catalog_](const TableIdentifier& name) { + return catalog->LoadTable(name); + }); + EXPECT_CALL(*mock, UpdateTable(::testing::_, ::testing::_, ::testing::_)) + .Times(1) + .WillOnce(::testing::Return(std::move(failure))); + return Table::Make(table_->name(), table_->metadata(), + std::string(table_->metadata_file_location()), + io ? std::move(io) : table_->io(), std::move(mock)); + } + + std::vector CompactedFiles() { + auto arrow_io = std::dynamic_pointer_cast(file_io_); + EXPECT_NE(arrow_io, nullptr); + ::arrow::fs::FileSelector selector; + selector.base_dir = table_location_ + "/data"; + selector.recursive = true; + auto infos = arrow_io->fs()->GetFileInfo(selector); + EXPECT_TRUE(infos.ok()) << infos.status().ToString(); + std::vector result; + if (!infos.ok()) { + return result; + } + for (const auto& info : *infos) { + if (info.path().find("compacted-") != std::string::npos) { + result.push_back(info.path()); + } + } + return result; + } + + std::shared_ptr schema_; + std::shared_ptr spec_; +}; + +class CompactionExecutorV2UpgradeTest : public CompactionExecutorTest { + protected: + int8_t format_version() const override { return 2; } +}; + +TEST_F(CompactionExecutorTest, AppliesDeletesAndPreservesRowLineage) { + ICEBERG_UNWRAP_OR_FAIL( + auto data_file, + WriteDataFile("input.parquet", + R"([[10, 100, 1000], [10, 200, 2000], [10, 300, 3000]])")); + ASSERT_THAT(Append({data_file}), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto deletes, PositionDeleteUpdate::Make(table_)); + deletes->Delete(data_file->file_path, 1); + ASSERT_THAT(deletes->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ASSERT_EQ(tasks.size(), 1); + ICEBERG_UNWRAP_OR_FAIL(auto expected_rows, ReadRows(*tasks.front())); + ICEBERG_UNWRAP_OR_FAIL(auto base_snapshot, table_->current_snapshot()); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(table_)); + ASSERT_THAT(executor->Execute(plan), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto rewritten_tasks, Tasks(table_)); + ASSERT_EQ(rewritten_tasks.size(), 1); + const auto& rewritten = rewritten_tasks.front(); + EXPECT_NE(rewritten->data_file()->file_path, data_file->file_path); + EXPECT_EQ(rewritten->data_file()->record_count, 2); + EXPECT_EQ(rewritten->data_file()->data_sequence_number, base_snapshot->sequence_number); + EXPECT_TRUE(rewritten->delete_files().empty()); + ICEBERG_UNWRAP_OR_FAIL(auto actual_rows, ReadRows(*rewritten)); + EXPECT_EQ(actual_rows, expected_rows); + ICEBERG_UNWRAP_OR_FAIL(auto live_deletes, LiveDeleteEntries()); + EXPECT_TRUE(live_deletes.empty()); +} + +TEST_F(CompactionExecutorTest, CompactsNullPartition) { + auto arrow_io = std::dynamic_pointer_cast(file_io_); + ASSERT_NE(arrow_io, nullptr); + ASSERT_TRUE(arrow_io->fs()->CreateDir(table_location_ + "/data/x=null").ok()); + ICEBERG_UNWRAP_OR_FAIL(auto schema_update, table_->NewUpdateSchema()); + schema_update->MakeColumnOptional("x"); + ASSERT_THAT(schema_update->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(schema_, table_->schema()); + ICEBERG_UNWRAP_OR_FAIL( + auto file, + WriteDataFile("null.parquet", R"([[null, 100, 1000], [null, 200, 2000]])", + std::nullopt)); + ASSERT_THAT(Append({file}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto deletes, PositionDeleteUpdate::Make(table_)); + deletes->Delete(file->file_path, 0); + ASSERT_THAT(deletes->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(table_)); + ASSERT_THAT(executor->Execute(plan), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto rewritten, Tasks(table_)); + ASSERT_EQ(rewritten.size(), 1); + EXPECT_EQ(rewritten.front()->data_file()->record_count, 1); + EXPECT_TRUE(rewritten.front()->data_file()->partition.values().front().IsNull()); + EXPECT_TRUE(rewritten.front()->delete_files().empty()); + + ICEBERG_UNWRAP_OR_FAIL(auto reader, FileScanTaskReader::Make({ + .io = file_io_, + .table_schema = schema_, + .schemas = table_->metadata()->schemas, + .projected_schema = schema_, + })); + ICEBERG_UNWRAP_OR_FAIL(auto stream, reader->Open(*rewritten.front())); + auto batches = ::arrow::ImportRecordBatchReader(&stream).ValueOrDie(); + auto batch = batches->Next().ValueOrDie(); + ASSERT_NE(batch, nullptr); + ASSERT_EQ(batch->num_rows(), 1); + EXPECT_TRUE(batch->column(0)->IsNull(0)); + EXPECT_EQ(std::static_pointer_cast<::arrow::Int64Array>(batch->column(1))->Value(0), + 200); +} + +TEST_F(CompactionExecutorTest, EmptyPlannerResultDoesNotCommit) { + ICEBERG_UNWRAP_OR_FAIL(auto file, + WriteDataFile("input.parquet", R"([[10, 100, 1000]])")); + ASSERT_THAT(Append({file}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto snapshot, table_->current_snapshot()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ICEBERG_UNWRAP_OR_FAIL( + auto plan, + CompactionPlanner::Plan(snapshot->snapshot_id, tasks, + {.target_file_size_bytes = file->file_size_in_bytes})); + ASSERT_TRUE(plan.groups.empty()); + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(table_)); + ASSERT_THAT(executor->Execute(plan), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto after, table_->current_snapshot()); + EXPECT_EQ(after->snapshot_id, snapshot->snapshot_id); + EXPECT_TRUE(CompactedFiles().empty()); +} + +TEST_F(CompactionExecutorTest, CombinesFilesAndRemovesFullyDeletedGroup) { + ICEBERG_UNWRAP_OR_FAIL( + auto first, + WriteDataFile("first.parquet", R"([[10, 100, 1000], [10, 200, 2000]])")); + ICEBERG_UNWRAP_OR_FAIL( + auto second, + WriteDataFile("second.parquet", R"([[10, 300, 3000], [10, 400, 4000]])")); + ICEBERG_UNWRAP_OR_FAIL(auto third, + WriteDataFile("third.parquet", R"([[20, 500, 5000]])", 20)); + ASSERT_THAT(Append({first, second, third}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto deletes, PositionDeleteUpdate::Make(table_)); + deletes->Delete(first->file_path, 0).Delete(third->file_path, 0); + ASSERT_THAT(deletes->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + std::vector expected; + for (const auto& task : tasks) { + ICEBERG_UNWRAP_OR_FAIL(auto rows, ReadRows(*task)); + expected.insert(expected.end(), rows.begin(), rows.end()); + } + std::ranges::sort(expected); + ASSERT_EQ(expected.size(), 3); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + ASSERT_EQ(plan.groups.size(), 2); + ASSERT_EQ(plan.groups.front().files.size(), 2); + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(table_)); + ASSERT_THAT(executor->Execute(plan), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto rewritten, Tasks(table_)); + ASSERT_EQ(rewritten.size(), 1); + EXPECT_TRUE(rewritten.front()->delete_files().empty()); + ICEBERG_UNWRAP_OR_FAIL(auto actual, ReadRows(*rewritten.front())); + std::ranges::sort(actual); + EXPECT_EQ(actual, expected); + ICEBERG_UNWRAP_OR_FAIL(auto live_deletes, LiveDeleteEntries()); + EXPECT_TRUE(live_deletes.empty()); +} + +TEST_F(CompactionExecutorTest, KeepsSharedPuffinForUncompactedDataFile) { + ICEBERG_UNWRAP_OR_FAIL( + auto first, + WriteDataFile("first.parquet", R"([[10, 100, 1000], [10, 200, 2000]])")); + ICEBERG_UNWRAP_OR_FAIL( + auto second, + WriteDataFile("second.parquet", R"([[10, 300, 3000], [10, 400, 4000]])")); + ASSERT_THAT(Append({first, second}), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto deletes, PositionDeleteUpdate::Make(table_)); + deletes->Delete(first->file_path, 0).Delete(second->file_path, 1); + ASSERT_THAT(deletes->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + auto first_task = std::ranges::find_if(tasks, [&](const auto& task) { + return task->data_file()->file_path == first->file_path; + }); + auto second_task = std::ranges::find_if(tasks, [&](const auto& task) { + return task->data_file()->file_path == second->file_path; + }); + ASSERT_NE(first_task, tasks.end()); + ASSERT_NE(second_task, tasks.end()); + ASSERT_EQ((*first_task)->delete_files().size(), 1); + ASSERT_EQ((*second_task)->delete_files().size(), 1); + const auto shared_puffin = (*first_task)->delete_files().front()->file_path; + ASSERT_EQ(shared_puffin, (*second_task)->delete_files().front()->file_path); + + ICEBERG_UNWRAP_OR_FAIL(auto plan, + Plan(std::vector>{*first_task})); + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(table_)); + ASSERT_THAT(executor->Execute(plan), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto rewritten_tasks, Tasks(table_)); + auto remaining = std::ranges::find_if(rewritten_tasks, [&](const auto& task) { + return task->data_file()->file_path == second->file_path; + }); + ASSERT_NE(remaining, rewritten_tasks.end()); + ASSERT_EQ((*remaining)->delete_files().size(), 1); + EXPECT_EQ((*remaining)->delete_files().front()->file_path, shared_puffin); + auto arrow_io = std::dynamic_pointer_cast(file_io_); + ASSERT_NE(arrow_io, nullptr); + auto info = arrow_io->fs()->GetFileInfo(shared_puffin); + ASSERT_TRUE(info.ok()) << info.status().ToString(); + EXPECT_EQ(info->type(), ::arrow::fs::FileType::File); + + ICEBERG_UNWRAP_OR_FAIL(auto live_deletes, LiveDeleteEntries()); + ASSERT_EQ(live_deletes.size(), 1); + EXPECT_EQ(live_deletes.front().data_file->referenced_data_file, second->file_path); +} + +TEST_F(CompactionExecutorTest, CommitStateUnknownRelinquishesOutputOwnership) { + ICEBERG_UNWRAP_OR_FAIL( + auto data_file, + WriteDataFile("input.parquet", R"([[10, 100, 1000], [10, 200, 2000]])")); + ASSERT_THAT(Append({data_file}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + + ICEBERG_UNWRAP_OR_FAIL( + auto mock_table, FailingCommitTable(CommitStateUnknown("injected unknown state"))); + + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(mock_table)); + EXPECT_THAT(executor->Execute(plan), IsError(ErrorKind::kCommitStateUnknown)); + ASSERT_EQ(CompactedFiles().size(), 1); + + EXPECT_THAT(executor->Execute(plan), IsError(ErrorKind::kInvalidArgument)); + EXPECT_EQ(CompactedFiles().size(), 1); +} + +TEST_F(CompactionExecutorTest, RejectsDuplicateSourceFilesBeforeWriting) { + ICEBERG_UNWRAP_OR_FAIL( + auto data_file, + WriteDataFile("input.parquet", R"([[10, 100, 1000], [10, 200, 2000]])")); + ASSERT_THAT(Append({data_file}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + ASSERT_EQ(plan.groups.size(), 1); + ASSERT_EQ(plan.groups.front().files.size(), 1); + + auto within_group = plan; + within_group.groups.front().files.push_back(within_group.groups.front().files.front()); + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(table_)); + EXPECT_THAT(executor->Execute(within_group), IsError(ErrorKind::kInvalidArgument)); + EXPECT_TRUE(CompactedFiles().empty()); + + auto across_groups = plan; + across_groups.groups.push_back(across_groups.groups.front()); + EXPECT_THAT(executor->Execute(across_groups), IsError(ErrorKind::kInvalidArgument)); + EXPECT_TRUE(CompactedFiles().empty()); +} + +TEST_F(CompactionExecutorTest, RejectsPlanAfterNewDeleteSnapshot) { + ICEBERG_UNWRAP_OR_FAIL( + auto data_file, + WriteDataFile("input.parquet", + R"([[10, 100, 1000], [10, 200, 2000], [10, 300, 3000]])")); + ASSERT_THAT(Append({data_file}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + + ICEBERG_UNWRAP_OR_FAIL(auto deletes, PositionDeleteUpdate::Make(table_)); + deletes->Delete(data_file->file_path, 1); + ASSERT_THAT(deletes->Commit(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(table_)); + EXPECT_THAT(executor->Execute(plan), IsError(ErrorKind::kValidationFailed)); + EXPECT_TRUE(CompactedFiles().empty()); + + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto current_tasks, Tasks(table_)); + ASSERT_EQ(current_tasks.size(), 1); + ASSERT_EQ(current_tasks.front()->delete_files().size(), 1); + ICEBERG_UNWRAP_OR_FAIL(auto rows, ReadRows(*current_tasks.front())); + EXPECT_EQ(rows.size(), 2); +} + +TEST_F(CompactionExecutorTest, MultiFileMultiGroupCommitIsAtomicOnFailure) { + ICEBERG_UNWRAP_OR_FAIL( + auto first, + WriteDataFile("first.parquet", R"([[10, 100, 1000], [10, 200, 2000]])")); + ICEBERG_UNWRAP_OR_FAIL( + auto second, + WriteDataFile("second.parquet", R"([[10, 300, 3000], [10, 400, 4000]])")); + ICEBERG_UNWRAP_OR_FAIL( + auto third, + WriteDataFile("third.parquet", R"([[20, 500, 5000], [20, 600, 6000]])", 20)); + ASSERT_THAT(Append({first, second, third}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + ASSERT_EQ(plan.groups.size(), 2); + ASSERT_EQ(plan.groups.front().files.size(), 2); + + ICEBERG_UNWRAP_OR_FAIL(auto mock_table, + FailingCommitTable(CommitFailed("injected failure"))); + + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(mock_table)); + EXPECT_THAT(executor->Execute(plan), IsError(ErrorKind::kCommitFailed)); + EXPECT_TRUE(CompactedFiles().empty()); + + ICEBERG_UNWRAP_OR_FAIL(auto current_tasks, Tasks(table_)); + EXPECT_EQ(current_tasks.size(), 3); +} + +TEST_F(CompactionExecutorTest, CleanupFailureRetainsOutputForExplicitRetry) { + ICEBERG_UNWRAP_OR_FAIL( + auto data_file, + WriteDataFile("input.parquet", R"([[10, 100, 1000], [10, 200, 2000]])")); + ASSERT_THAT(Append({data_file}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + + auto failing_io = std::make_shared(table_->io(), FailurePoint::kDelete); + ICEBERG_UNWRAP_OR_FAIL( + auto mock_table, + FailingCommitTable(CommitFailed("injected commit failure"), failing_io)); + + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(mock_table)); + auto status = executor->Execute(plan); + EXPECT_THAT(status, IsError(ErrorKind::kCommitFailed)); + EXPECT_THAT(status, HasErrorMessage("injected commit failure")); + EXPECT_THAT(status, HasErrorMessage("injected cleanup failure")); + ASSERT_EQ(CompactedFiles().size(), 1); + + failing_io->DisableFailure(); + EXPECT_THAT(executor->Cleanup(), IsOk()); + EXPECT_TRUE(CompactedFiles().empty()); +} + +TEST_F(CompactionExecutorTest, WriterFailureBeforeFileCreationLeavesExecutorReusable) { + ICEBERG_UNWRAP_OR_FAIL( + auto data_file, + WriteDataFile("input.parquet", R"([[10, 100, 1000], [10, 200, 2000]])")); + ASSERT_THAT(Append({data_file}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + + ICEBERG_UNWRAP_OR_FAIL(auto invalid_properties, table_->NewUpdateProperties()); + invalid_properties->Set(WriterProperties::kParquetMaxRowGroupRows.key(), "0"); + ASSERT_THAT(invalid_properties->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(table_)); + auto status = executor->Execute(plan); + EXPECT_THAT(status, IsError(ErrorKind::kInvalidArgument)); + EXPECT_THAT(status, HasErrorMessage("Parquet max row group rows")); + EXPECT_TRUE(CompactedFiles().empty()); + EXPECT_THAT(executor->Cleanup(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto valid_properties, table_->NewUpdateProperties()); + valid_properties->Set(WriterProperties::kParquetMaxRowGroupRows.key(), "100"); + ASSERT_THAT(valid_properties->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ASSERT_THAT(executor->Execute(plan), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto rewritten_tasks, Tasks(table_)); + ASSERT_EQ(rewritten_tasks.size(), 1); + EXPECT_NE(rewritten_tasks.front()->data_file()->file_path, data_file->file_path); +} + +class CompactionWriterFailureTest : public CompactionExecutorTest, + public ::testing::WithParamInterface {}; + +TEST_P(CompactionWriterFailureTest, CleansOutputAndAllowsRetry) { + ICEBERG_UNWRAP_OR_FAIL( + auto file, WriteDataFile("input.parquet", R"([[10, 100, 1000], [10, 200, 2000]])")); + ASSERT_THAT(Append({file}), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + auto io = std::make_shared(table_->io(), GetParam()); + ICEBERG_UNWRAP_OR_FAIL( + auto writer_table, + Table::Make(table_->name(), table_->metadata(), + std::string(table_->metadata_file_location()), io, catalog_)); + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(writer_table)); + EXPECT_THAT(executor->Execute(plan), IsError(ErrorKind::kIOError)); + EXPECT_TRUE(CompactedFiles().empty()); + EXPECT_THAT(executor->Cleanup(), IsOk()); + ASSERT_THAT(executor->Execute(plan), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto rewritten, Tasks(table_)); + ASSERT_EQ(rewritten.size(), 1); + EXPECT_NE(rewritten.front()->data_file()->file_path, file->file_path); +} + +INSTANTIATE_TEST_SUITE_P(OutputCreation, CompactionWriterFailureTest, + ::testing::Values(FailurePoint::kBeforeCreate, + FailurePoint::kAfterCreate)); + +TEST_F(CompactionExecutorV2UpgradeTest, + RemovesParquetDeleteAndAssignsUpgradedRowLineage) { + ICEBERG_UNWRAP_OR_FAIL( + auto data_file, + WriteDataFile("input.parquet", + R"([[10, 100, 1000], [10, 200, 2000], [10, 300, 3000]])")); + ASSERT_THAT(Append({data_file}), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto deletes, PositionDeleteUpdate::Make(table_)); + deletes->Delete(data_file->file_path, 1); + ASSERT_THAT(deletes->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto v2_tasks, Tasks(table_)); + ASSERT_EQ(v2_tasks.size(), 1); + ASSERT_EQ(v2_tasks.front()->delete_files().size(), 1); + EXPECT_EQ(v2_tasks.front()->delete_files().front()->file_format, + FileFormatType::kParquet); + + ICEBERG_UNWRAP_OR_FAIL(auto properties, table_->NewUpdateProperties()); + properties->Set(TableProperties::kFormatVersion.key(), "3"); + ASSERT_THAT(properties->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto tasks, Tasks(table_)); + ASSERT_EQ(tasks.size(), 1); + ASSERT_FALSE(tasks.front()->data_file()->first_row_id.has_value()); + ASSERT_TRUE(tasks.front()->data_file()->data_sequence_number.has_value()); + const auto original_sequence = *tasks.front()->data_file()->data_sequence_number; + ICEBERG_UNWRAP_OR_FAIL(auto plan, Plan(tasks)); + + ICEBERG_UNWRAP_OR_FAIL(auto executor, CompactionExecutor::Make(table_)); + ASSERT_THAT(executor->Execute(plan), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto rewritten_tasks, Tasks(table_)); + ASSERT_EQ(rewritten_tasks.size(), 1); + ASSERT_TRUE(rewritten_tasks.front()->data_file()->first_row_id.has_value()); + const auto first_row_id = *rewritten_tasks.front()->data_file()->first_row_id; + ICEBERG_UNWRAP_OR_FAIL(auto rows, ReadRows(*rewritten_tasks.front())); + ASSERT_EQ(rows.size(), 2); + EXPECT_EQ(rows[0][3], first_row_id); + EXPECT_EQ(rows[1][3], first_row_id + 1); + EXPECT_EQ(rows[0][4], original_sequence); + EXPECT_EQ(rows[1][4], original_sequence); + EXPECT_TRUE(rewritten_tasks.front()->delete_files().empty()); + ICEBERG_UNWRAP_OR_FAIL(auto live_deletes, LiveDeleteEntries()); + EXPECT_TRUE(live_deletes.empty()); +} + +} // namespace +} // namespace iceberg diff --git a/src/iceberg/test/compaction_planner_test.cc b/src/iceberg/test/compaction_planner_test.cc new file mode 100644 index 000000000..600dc562e --- /dev/null +++ b/src/iceberg/test/compaction_planner_test.cc @@ -0,0 +1,378 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#include "iceberg/compaction_planner.h" + +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include + +#include "iceberg/expression/literal.h" +#include "iceberg/file_format.h" +#include "iceberg/manifest/manifest_entry.h" +#include "iceberg/table_scan.h" +#include "iceberg/test/matchers.h" +#include "iceberg/type.h" + +namespace iceberg { +namespace { + +using ::testing::ElementsAre; +using ::testing::IsEmpty; +using ::testing::SizeIs; + +class CompactionPlannerTest : public testing::Test { + protected: + static constexpr int64_t kSnapshotId = 1234; + + static std::shared_ptr Data(std::string path, PartitionValues partition, + int64_t size, std::optional spec_id = 1, + int64_t records = 100) { + return std::make_shared(DataFile{ + .file_path = std::move(path), + .partition = std::move(partition), + .record_count = records, + .file_size_in_bytes = size, + .partition_spec_id = spec_id, + }); + } + + static std::shared_ptr Data(std::string path, int32_t partition, int64_t size, + std::optional spec_id = 1, + int64_t records = 100) { + return Data(std::move(path), PartitionValues(Literal::Int(partition)), size, spec_id, + records); + } + + static std::shared_ptr PositionDelete(std::string path, + const std::string& referenced_file, + int64_t records, + bool file_scoped = true) { + return std::make_shared(DataFile{ + .content = DataFile::Content::kPositionDeletes, + .file_path = std::move(path), + .record_count = records, + .referenced_data_file = + file_scoped ? std::make_optional(referenced_file) : std::nullopt, + }); + } + + static std::shared_ptr DeletionVector(std::string path, + const std::string& referenced_file, + int64_t records, int64_t offset = 0, + int64_t size = 10) { + return std::make_shared(DataFile{ + .content = DataFile::Content::kPositionDeletes, + .file_path = std::move(path), + .file_format = FileFormatType::kPuffin, + .record_count = records, + .referenced_data_file = referenced_file, + .content_offset = offset, + .content_size_in_bytes = size, + }); + } + + static std::shared_ptr EqualityDelete(std::string path, int64_t records) { + return std::make_shared(DataFile{ + .content = DataFile::Content::kEqualityDeletes, + .file_path = std::move(path), + .record_count = records, + }); + } + + static std::shared_ptr Task( + std::shared_ptr data_file, + std::vector> deletes = {}) { + return std::make_shared(std::move(data_file), std::move(deletes)); + } + + static std::vector Paths(const CompactionGroup& group) { + std::vector paths; + for (const auto& file : group.files) { + paths.push_back(file.scan_task->data_file()->file_path); + } + return paths; + } + + static CompactionPlannerConfig Config() { + return CompactionPlannerConfig{ + .target_file_size_bytes = 100, + .min_file_size_ratio = 0.5, + .min_input_files = 2, + .delete_file_threshold = 2, + .delete_ratio_threshold = 0.2, + }; + } +}; + +TEST_F(CompactionPlannerTest, ValidatesConfig) { + std::vector invalid; + + auto config = Config(); + config.target_file_size_bytes = 0; + invalid.push_back(config); + config = Config(); + config.min_file_size_ratio = -0.1; + invalid.push_back(config); + config = Config(); + config.min_file_size_ratio = 1.1; + invalid.push_back(config); + config = Config(); + config.min_file_size_ratio = std::numeric_limits::quiet_NaN(); + invalid.push_back(config); + config = Config(); + config.min_input_files = 0; + invalid.push_back(config); + config = Config(); + config.delete_file_threshold = -1; + invalid.push_back(config); + config = Config(); + config.delete_ratio_threshold = -0.1; + invalid.push_back(config); + config = Config(); + config.delete_ratio_threshold = 1.1; + invalid.push_back(config); + config = Config(); + config.delete_ratio_threshold = std::numeric_limits::quiet_NaN(); + invalid.push_back(config); + + for (const auto& invalid_config : invalid) { + EXPECT_THAT(CompactionPlanner::Plan(kSnapshotId, {}, invalid_config), + IsError(ErrorKind::kInvalidArgument)); + } +} + +TEST_F(CompactionPlannerTest, ZeroDeleteThresholdsDisableDeleteSelection) { + auto config = Config(); + config.delete_file_threshold = 0; + config.delete_ratio_threshold = 0; + std::vector> tasks{ + Task(Data("data", 1, 100), {PositionDelete("delete-a", "data", 100), + PositionDelete("delete-b", "data", 100)})}; + + ICEBERG_UNWRAP_OR_FAIL(auto plan, CompactionPlanner::Plan(kSnapshotId, tasks, config)); + EXPECT_THAT(plan.groups, IsEmpty()); +} + +TEST_F(CompactionPlannerTest, BindsPlanToSourceSnapshot) { + ICEBERG_UNWRAP_OR_FAIL(auto plan, CompactionPlanner::Plan(kSnapshotId, {}, Config())); + + EXPECT_EQ(plan.source_snapshot_id, kSnapshotId); + EXPECT_THAT(CompactionPlanner::Plan(-1, {}, Config()), + IsError(ErrorKind::kInvalidArgument)); +} + +TEST_F(CompactionPlannerTest, SelectsThresholdBoundariesAndIgnoresOtherDeletes) { + std::vector> tasks{ + Task(Data("small-a", 1, 49)), + Task(Data("small-b", 1, 49)), + Task(Data("at-min-size", 1, 50)), + Task(Data("below-ratio", 2, 100), + {DeletionVector("below-ratio.dv", "below-ratio", 19)}), + Task(Data("at-ratio", 2, 100), {DeletionVector("at-ratio.dv", "at-ratio", 20)}), + Task(Data("at-count", 3, 100), {PositionDelete("count-a", "at-count", 0), + PositionDelete("count-b", "at-count", 0)}), + Task(Data("ignored", 4, 100), + {EqualityDelete("equality", 100), + PositionDelete("partition-scoped", "ignored", 100, false)})}; + + ICEBERG_UNWRAP_OR_FAIL(auto plan, + CompactionPlanner::Plan(kSnapshotId, tasks, Config())); + ASSERT_THAT(plan.groups, SizeIs(3)); + EXPECT_THAT(Paths(plan.groups[0]), ElementsAre("small-a", "small-b")); + EXPECT_THAT(Paths(plan.groups[1]), ElementsAre("at-ratio")); + EXPECT_EQ(plan.groups[1].file_scoped_delete_record_count, 20); + EXPECT_THAT(Paths(plan.groups[2]), ElementsAre("at-count")); +} + +TEST_F(CompactionPlannerTest, GroupsNullPartitionsTogether) { + auto null_partition = [] { return PartitionValues(Literal::Null(int32())); }; + std::vector> tasks{ + Task(Data("null-b", null_partition(), 10)), Task(Data("value-a", 1, 10)), + Task(Data("null-a", null_partition(), 10)), Task(Data("value-b", 1, 10))}; + + ICEBERG_UNWRAP_OR_FAIL(auto plan, + CompactionPlanner::Plan(kSnapshotId, tasks, Config())); + ASSERT_THAT(plan.groups, SizeIs(2)); + EXPECT_THAT(Paths(plan.groups[0]), ElementsAre("null-a", "null-b")); + EXPECT_THAT(Paths(plan.groups[1]), ElementsAre("value-a", "value-b")); +} + +TEST_F(CompactionPlannerTest, GroupsCanonicalNaNPartitionsTogether) { + auto quiet_nan = + PartitionValues(Literal::Double(std::numeric_limits::quiet_NaN())); + auto signaling_nan = + PartitionValues(Literal::Double(std::numeric_limits::signaling_NaN())); + std::vector> tasks{ + Task(Data("nan-b", std::move(signaling_nan), 10)), + Task(Data("nan-a", std::move(quiet_nan), 10))}; + + ICEBERG_UNWRAP_OR_FAIL(auto plan, + CompactionPlanner::Plan(kSnapshotId, tasks, Config())); + ASSERT_THAT(plan.groups, SizeIs(1)); + EXPECT_THAT(Paths(plan.groups[0]), ElementsAre("nan-a", "nan-b")); +} + +TEST_F(CompactionPlannerTest, RejectsMissingPartitionSpecId) { + std::vector> tasks{ + Task(Data("missing-spec", 1, 10, std::nullopt))}; + + EXPECT_THAT(CompactionPlanner::Plan(kSnapshotId, tasks, Config()), + IsError(ErrorKind::kInvalidArgument)); +} + +TEST_F(CompactionPlannerTest, RejectsDuplicateDataFileTasks) { + std::vector> tasks{Task(Data("duplicate", 1, 10)), + Task(Data("duplicate", 1, 10))}; + + EXPECT_THAT(CompactionPlanner::Plan(kSnapshotId, tasks, Config()), + IsError(ErrorKind::kInvalidArgument)); +} + +TEST_F(CompactionPlannerTest, ProducesCanonicalPartitionAndFileOrder) { + std::vector> tasks{ + Task(Data("s2-z", 1, 10, 2)), Task(Data("p2-z", 2, 10)), Task(Data("p1-z", 1, 10)), + Task(Data("s2-a", 1, 10, 2)), Task(Data("p2-a", 2, 10)), Task(Data("p1-a", 1, 10))}; + + ICEBERG_UNWRAP_OR_FAIL(auto first, + CompactionPlanner::Plan(kSnapshotId, tasks, Config())); + std::ranges::reverse(tasks); + ICEBERG_UNWRAP_OR_FAIL(auto second, + CompactionPlanner::Plan(kSnapshotId, tasks, Config())); + + ASSERT_THAT(first.groups, SizeIs(3)); + ASSERT_THAT(second.groups, SizeIs(3)); + for (size_t i = 0; i < first.groups.size(); ++i) { + EXPECT_EQ(first.groups[i].partition_spec_id, second.groups[i].partition_spec_id); + EXPECT_EQ(Paths(first.groups[i]), Paths(second.groups[i])); + } + EXPECT_EQ(first.groups[0].partition_spec_id, 1); + EXPECT_THAT(Paths(first.groups[0]), ElementsAre("p1-a", "p1-z")); + EXPECT_THAT(Paths(first.groups[1]), ElementsAre("p2-a", "p2-z")); + EXPECT_EQ(first.groups[2].partition_spec_id, 2); + EXPECT_THAT(Paths(first.groups[2]), ElementsAre("s2-a", "s2-z")); +} + +TEST_F(CompactionPlannerTest, DeduplicatesDeleteReferencesAndDistinguishesDvRanges) { + std::vector> tasks{ + Task(Data("data", 1, 100), + {PositionDelete("deletes", "data", 30), PositionDelete("deletes", "data", 30), + DeletionVector("deletes", "data", 20, 0, 10), + DeletionVector("deletes", "data", 20, 0, 10), + DeletionVector("deletes", "data", 5, 10, 10)})}; + auto config = Config(); + config.delete_file_threshold = 3; + config.delete_ratio_threshold = 0; + + ICEBERG_UNWRAP_OR_FAIL(auto plan, CompactionPlanner::Plan(kSnapshotId, tasks, config)); + ASSERT_THAT(plan.groups, SizeIs(1)); + ASSERT_THAT(plan.groups[0].files, SizeIs(1)); + EXPECT_EQ(plan.groups[0].files[0].file_scoped_delete_count, 3); + EXPECT_EQ(plan.groups[0].files[0].file_scoped_delete_record_count, 55); +} + +TEST_F(CompactionPlannerTest, CapsPositionDeleteCardinalityAtDataRows) { + std::vector> tasks{Task( + Data("data", 1, 100), + {PositionDelete("delete-a", "data", 80), PositionDelete("delete-b", "data", 40)})}; + + ICEBERG_UNWRAP_OR_FAIL(auto plan, + CompactionPlanner::Plan(kSnapshotId, tasks, Config())); + ASSERT_THAT(plan.groups, SizeIs(1)); + EXPECT_EQ(plan.groups[0].files[0].file_scoped_delete_record_count, 100); + EXPECT_EQ(plan.groups[0].file_scoped_delete_record_count, 100); +} + +TEST_F(CompactionPlannerTest, RejectsDvCardinalityAboveDataRows) { + std::vector> tasks{ + Task(Data("data", 1, 100), {DeletionVector("dv", "data", 101)})}; + + EXPECT_THAT(CompactionPlanner::Plan(kSnapshotId, tasks, Config()), + IsError(ErrorKind::kInvalidArgument)); +} + +TEST_F(CompactionPlannerTest, ReturnsErrorsForAggregateOverflow) { + constexpr int64_t kMax = std::numeric_limits::max(); + std::vector> file_overflow{ + Task(Data("data", 1, 100, 1, kMax), {PositionDelete("delete-a", "data", kMax), + PositionDelete("delete-b", "data", kMax)})}; + EXPECT_THAT(CompactionPlanner::Plan(kSnapshotId, file_overflow, Config()), + IsError(ErrorKind::kInvalidArgument)); + + std::vector> group_overflow{ + Task(Data("a", 1, 0, 1, kMax), {PositionDelete("delete-a", "a", kMax)}), + Task(Data("b", 1, 0, 1, kMax), {PositionDelete("delete-b", "b", kMax)})}; + EXPECT_THAT(CompactionPlanner::Plan(kSnapshotId, group_overflow, Config()), + IsError(ErrorKind::kInvalidArgument)); +} + +TEST_F(CompactionPlannerTest, RequiresDeletePressureForOversizedFiles) { + std::vector> tasks{ + Task(Data("no-deletes", 1, 201)), + Task(Data("below-ratio", 1, 201), {DeletionVector("below.dv", "below-ratio", 19)}), + Task(Data("at-ratio", 1, 201), {DeletionVector("at.dv", "at-ratio", 20)})}; + + ICEBERG_UNWRAP_OR_FAIL(auto plan, + CompactionPlanner::Plan(kSnapshotId, tasks, Config())); + ASSERT_THAT(plan.groups, SizeIs(1)); + EXPECT_THAT(Paths(plan.groups[0]), ElementsAre("at-ratio")); +} + +TEST_F(CompactionPlannerTest, SplitsPartitionIntoTargetSizedGroups) { + std::vector> tasks; + for (const auto* path : {"f", "e", "d", "c", "b", "a"}) { + tasks.push_back(Task(Data(path, 1, 40))); + } + + ICEBERG_UNWRAP_OR_FAIL(auto plan, + CompactionPlanner::Plan(kSnapshotId, tasks, Config())); + ASSERT_THAT(plan.groups, SizeIs(3)); + EXPECT_THAT(Paths(plan.groups[0]), ElementsAre("a", "b")); + EXPECT_THAT(Paths(plan.groups[1]), ElementsAre("c", "d")); + EXPECT_THAT(Paths(plan.groups[2]), ElementsAre("e", "f")); + for (const auto& group : plan.groups) { + EXPECT_EQ(group.data_file_size_bytes, 80); + } +} + +TEST_F(CompactionPlannerTest, BestFitAvoidsDroppingCompatibleSmallFiles) { + auto config = Config(); + config.min_file_size_ratio = 1.0; + std::vector> tasks{ + Task(Data("d40", 1, 40)), Task(Data("b60", 1, 60)), Task(Data("c40", 1, 40)), + Task(Data("a60", 1, 60))}; + + ICEBERG_UNWRAP_OR_FAIL(auto plan, CompactionPlanner::Plan(kSnapshotId, tasks, config)); + ASSERT_THAT(plan.groups, SizeIs(2)); + EXPECT_THAT(Paths(plan.groups[0]), ElementsAre("a60", "c40")); + EXPECT_THAT(Paths(plan.groups[1]), ElementsAre("b60", "d40")); + EXPECT_EQ(plan.groups[0].data_file_size_bytes, 100); + EXPECT_EQ(plan.groups[1].data_file_size_bytes, 100); +} + +} // namespace +} // namespace iceberg diff --git a/src/iceberg/test/delete_file_index_test.cc b/src/iceberg/test/delete_file_index_test.cc index 75c82bc39..48693a9b4 100644 --- a/src/iceberg/test/delete_file_index_test.cc +++ b/src/iceberg/test/delete_file_index_test.cc @@ -1113,6 +1113,81 @@ TEST_P(DeleteFileIndexTest, TestMultipleDVs) { EXPECT_THAT(index_result, HasErrorMessage(file_a_->file_path)); } +TEST_P(DeleteFileIndexTest, TestDVApplicability) { + auto version = GetParam(); + if (version < 3) { + GTEST_SKIP() << "DVs only supported in V3+"; + } + + const auto null_partition = PartitionValues({Literal::Null(int32())}); + auto null_partition_file = MakeDataFile("/path/to/data-null.parquet", null_partition, + partitioned_spec_->spec_id()); + + struct TestCase { + std::string name; + PartitionValues dv_partition; + std::shared_ptr dv_spec; + std::shared_ptr data_file; + bool applies; + }; + const std::vector cases = { + { + .name = "equal-partition", + .dv_partition = file_a_->partition, + .dv_spec = partitioned_spec_, + .data_file = file_a_, + .applies = true, + }, + { + .name = "different-spec", + .dv_partition = PartitionValues{}, + .dv_spec = unpartitioned_spec_, + .data_file = file_a_, + .applies = false, + }, + { + .name = "different-partition-value", + .dv_partition = file_b_->partition, + .dv_spec = partitioned_spec_, + .data_file = file_a_, + .applies = false, + }, + { + .name = "equal-null-partition", + .dv_partition = null_partition, + .dv_spec = partitioned_spec_, + .data_file = null_partition_file, + .applies = true, + }, + { + .name = "null-partition-mismatch", + .dv_partition = file_a_->partition, + .dv_spec = partitioned_spec_, + .data_file = null_partition_file, + .applies = false, + }, + }; + + for (const auto& test_case : cases) { + SCOPED_TRACE(test_case.name); + auto dv = MakeDV("/path/to/" + test_case.name + ".puffin", test_case.dv_partition, + test_case.dv_spec->spec_id(), test_case.data_file->file_path); + std::vector entries; + entries.push_back(MakeDeleteEntry(/*snapshot_id=*/1000L, /*sequence_number=*/2, dv)); + auto manifest = WriteDeleteManifest(version, /*snapshot_id=*/1000L, + std::move(entries), test_case.dv_spec); + ICEBERG_UNWRAP_OR_FAIL(auto index, BuildIndex({manifest})); + auto result = index->ForDataFile(1, *test_case.data_file); + if (test_case.applies) { + ICEBERG_UNWRAP_OR_FAIL(auto deletes, std::move(result)); + ASSERT_EQ(deletes.size(), 1); + EXPECT_EQ(deletes[0]->file_path, dv->file_path); + } else { + EXPECT_THAT(result, IsError(ErrorKind::kValidationFailed)); + } + } +} + TEST_P(DeleteFileIndexTest, TestInvalidDVSequenceNumber) { auto version = GetParam(); if (version < 3) { diff --git a/src/iceberg/test/position_delete_update_test.cc b/src/iceberg/test/position_delete_update_test.cc new file mode 100644 index 000000000..2be9c459e --- /dev/null +++ b/src/iceberg/test/position_delete_update_test.cc @@ -0,0 +1,636 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#include "iceberg/data/position_delete_update.h" + +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +#include "iceberg/arrow/arrow_io_internal.h" +#include "iceberg/avro/avro_register.h" +#include "iceberg/data/data_writer.h" +#include "iceberg/data/delete_loader.h" +#include "iceberg/data/file_scan_task_reader.h" +#include "iceberg/data/position_delete_writer.h" +#include "iceberg/deletes/position_delete_index.h" +#include "iceberg/file_reader.h" +#include "iceberg/manifest/manifest_reader.h" +#include "iceberg/metadata_columns.h" +#include "iceberg/parquet/parquet_register.h" +#include "iceberg/partition_spec.h" +#include "iceberg/schema.h" +#include "iceberg/schema_field.h" +#include "iceberg/schema_internal.h" +#include "iceberg/snapshot.h" +#include "iceberg/table.h" +#include "iceberg/table_metadata.h" +#include "iceberg/table_properties.h" +#include "iceberg/table_scan.h" +#include "iceberg/test/matchers.h" +#include "iceberg/test/mock_catalog.h" +#include "iceberg/test/update_test_base.h" +#include "iceberg/update/delete_files.h" +#include "iceberg/update/fast_append.h" +#include "iceberg/update/row_delta.h" +#include "iceberg/update/update_partition_spec.h" +#include "iceberg/update/update_properties.h" +#include "iceberg/util/uuid.h" + +namespace iceberg { + +namespace { + +struct RoutingCase { + int8_t format_version; + bool unpartitioned; + FileFormatType expected_format; +}; + +class FailOncePuffinDeleteFileIO : public arrow::ArrowFileSystemFileIO { + public: + explicit FailOncePuffinDeleteFileIO( + std::shared_ptr<::arrow::fs::FileSystem> file_system) + : ArrowFileSystemFileIO(std::move(file_system)) {} + + Status DeleteFile(const std::string& file_location) override { + if (!file_location.ends_with(".puffin")) { + return ArrowFileSystemFileIO::DeleteFile(file_location); + } + + puffin_delete_attempts.push_back(file_location); + if (fail_next_puffin_delete_) { + fail_next_puffin_delete_ = false; + return IOError("injected cleanup failure for {}", file_location); + } + return ArrowFileSystemFileIO::DeleteFile(file_location); + } + + std::vector puffin_delete_attempts; + + private: + bool fail_next_puffin_delete_ = true; +}; + +class PositionDeleteUpdateTest : public MinimalUpdateTestBase, + public ::testing::WithParamInterface { + protected: + static void SetUpTestSuite() { + avro::RegisterAll(); + parquet::RegisterAll(); + } + + int8_t format_version() const override { return GetParam().format_version; } + + void SetUp() override { + MinimalUpdateTestBase::SetUp(); + if (GetParam().unpartitioned) { + RegisterUnpartitionedTable(); + } + if (GetParam().format_version == 2) { + ICEBERG_UNWRAP_OR_FAIL(auto properties, table_->NewUpdateProperties()); + properties->Set(TableProperties::kDeleteParquetCompression.key(), "uncompressed"); + ASSERT_THAT(properties->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + } + ICEBERG_UNWRAP_OR_FAIL(spec_, table_->spec()); + ICEBERG_UNWRAP_OR_FAIL(data_file_, WriteDataFile()); + AppendDataFile(); + } + + void RegisterUnpartitionedTable() { + ICEBERG_UNWRAP_OR_FAIL( + auto metadata, ReadTableMetadataFromResource("TableMetadataV3ValidMinimal.json")); + metadata->location = table_location_; + metadata->partition_specs = {PartitionSpec::Unpartitioned()}; + metadata->default_spec_id = PartitionSpec::kInitialSpecId; + + const auto metadata_location = + std::format("{}/metadata/00001-{}.metadata.json", table_location_, + Uuid::GenerateV7().ToString()); + ASSERT_THAT(TableMetadataUtil::Write(*file_io_, metadata_location, *metadata), + IsOk()); + ASSERT_THAT(catalog_->DropTable(table_ident_, /*purge=*/false), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(table_, + catalog_->RegisterTable(table_ident_, metadata_location)); + } + + Result> WriteDataFile() { + ICEBERG_ASSIGN_OR_RAISE(auto schema, table_->schema()); + ICEBERG_ASSIGN_OR_RAISE( + auto writer, + DataWriter::Make({ + .path = table_location_ + "/data/file.parquet", + .schema = schema, + .spec = spec_, + .partition = GetParam().unpartitioned ? PartitionValues{} + : PartitionValues({Literal::Long(10)}), + .format = FileFormatType::kParquet, + .io = file_io_, + .properties = {{"write.parquet.compression-codec", "uncompressed"}}, + })); + ArrowSchema c_schema{}; + ICEBERG_RETURN_UNEXPECTED(ToArrowSchema(*schema, &c_schema)); + auto type = ::arrow::ImportType(&c_schema).ValueOrDie(); + std::string json = "["; + for (int64_t pos = 0; pos < 10; ++pos) { + if (pos != 0) json += ","; + json += std::format("[10,{},{}]", pos, pos * 10); + } + json += "]"; + auto array = ::arrow::json::ArrayFromJSONString(type, json).ValueOrDie(); + ArrowArray c_array{}; + auto status = ::arrow::ExportArray(*array, &c_array); + if (!status.ok()) return UnknownError(status.ToString()); + ICEBERG_RETURN_UNEXPECTED(writer->Write(&c_array)); + ICEBERG_RETURN_UNEXPECTED(writer->Close()); + ICEBERG_ASSIGN_OR_RAISE(auto metadata, writer->Metadata()); + return metadata.data_files.front(); + } + + Result> ReadSurvivingRows() { + ICEBERG_ASSIGN_OR_RAISE(table_, catalog_->LoadTable(table_ident_)); + ICEBERG_ASSIGN_OR_RAISE(auto schema, table_->schema()); + ICEBERG_ASSIGN_OR_RAISE(auto task, CurrentTask()); + ICEBERG_ASSIGN_OR_RAISE(auto reader, FileScanTaskReader::Make({ + .io = file_io_, + .table_schema = schema, + .schemas = table_->metadata()->schemas, + .projected_schema = schema, + })); + ICEBERG_ASSIGN_OR_RAISE(auto stream, reader->Open(*task)); + auto batches = ::arrow::ImportRecordBatchReader(&stream).ValueOrDie(); + std::vector rows; + while (auto batch = batches->Next().ValueOrDie()) { + auto values = std::static_pointer_cast<::arrow::Int64Array>(batch->column(1)); + for (int64_t i = 0; i < values->length(); ++i) rows.push_back(values->Value(i)); + } + return rows; + } + + void AppendDataFile() { + ICEBERG_UNWRAP_OR_FAIL(auto append, table_->NewFastAppend()); + append->AppendFile(data_file_); + ASSERT_THAT(append->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + } + + Result> CurrentTask() { + ICEBERG_ASSIGN_OR_RAISE(auto builder, table_->NewScan()); + ICEBERG_ASSIGN_OR_RAISE(auto scan, builder->Build()); + ICEBERG_ASSIGN_OR_RAISE(auto tasks, scan->PlanFiles()); + ICEBERG_CHECK(tasks.size() == 1, "Expected one file scan task, found {}", + tasks.size()); + return tasks.front(); + } + + Result LoadPositions(const FileScanTask& task) { + std::vector> deletes; + std::ranges::copy_if(task.delete_files(), std::back_inserter(deletes), + [](const auto& file) { + return file->content == DataFile::Content::kPositionDeletes; + }); + DeleteLoader loader(file_io_); + return loader.LoadPositionDeletes(deletes, task.data_file()->file_path); + } + + std::shared_ptr spec_; + std::shared_ptr data_file_; +}; + +TEST_P(PositionDeleteUpdateTest, RoutesAndCommitsPositionDeletes) { + ICEBERG_UNWRAP_OR_FAIL(auto update, PositionDeleteUpdate::Make(table_)); + update->Delete(data_file_->file_path, 7).Delete(data_file_->file_path, 2); + ASSERT_THAT(update->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto task, CurrentTask()); + ASSERT_EQ(task->delete_files().size(), 1); + const auto& delete_file = task->delete_files().front(); + EXPECT_EQ(delete_file->file_format, GetParam().expected_format); + EXPECT_EQ(delete_file->partition_spec_id, data_file_->partition_spec_id); + EXPECT_EQ(delete_file->partition, data_file_->partition); + + ICEBERG_UNWRAP_OR_FAIL(auto positions, LoadPositions(*task)); + EXPECT_EQ(positions.Cardinality(), 2); + EXPECT_TRUE(positions.IsDeleted(2)); + EXPECT_TRUE(positions.IsDeleted(7)); + if (GetParam().format_version == 2) { + auto delete_schema = std::make_shared(std::vector{ + MetadataColumns::kDeleteFilePath, MetadataColumns::kDeleteFilePos}); + ICEBERG_UNWRAP_OR_FAIL(auto reader, + ReaderFactoryRegistry::Open(FileFormatType::kParquet, + {.path = delete_file->file_path, + .io = file_io_, + .projection = delete_schema})); + ICEBERG_UNWRAP_OR_FAIL(auto batch, reader->Next()); + ASSERT_TRUE(batch.has_value()); + + ArrowSchema arrow_schema; + ASSERT_THAT(ToArrowSchema(*delete_schema, &arrow_schema), IsOk()); + auto arrow_type = ::arrow::ImportType(&arrow_schema).ValueOrDie(); + auto rows = ::arrow::ImportArray(&batch.value(), arrow_type).ValueOrDie(); + auto struct_rows = std::static_pointer_cast<::arrow::StructArray>(rows); + auto positions = std::static_pointer_cast<::arrow::Int64Array>(struct_rows->field(1)); + ASSERT_EQ(positions->length(), 2); + EXPECT_EQ(positions->Value(0), 2); + EXPECT_EQ(positions->Value(1), 7); + } +} + +TEST_P(PositionDeleteUpdateTest, ReopenedTablePreservesBothDeletes) { + ICEBERG_UNWRAP_OR_FAIL(auto first, PositionDeleteUpdate::Make(table_)); + first->Delete(data_file_->file_path, 2).Delete(data_file_->file_path, 7); + ASSERT_THAT(first->Commit(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto rows, ReadSurvivingRows()); + EXPECT_THAT(rows, ::testing::ElementsAre(0, 1, 3, 4, 5, 6, 8, 9)); + + if (GetParam().format_version == 3 && !GetParam().unpartitioned) { + ICEBERG_UNWRAP_OR_FAIL(auto evolution, table_->NewUpdatePartitionSpec()); + evolution->RemoveField("x"); + ASSERT_THAT(evolution->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + } + ICEBERG_UNWRAP_OR_FAIL(auto second, PositionDeleteUpdate::Make(table_)); + second->Delete(data_file_->file_path, 4); + ASSERT_THAT(second->Commit(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(rows, ReadSurvivingRows()); + EXPECT_THAT(rows, ::testing::ElementsAre(0, 1, 3, 5, 6, 8, 9)); +} + +TEST_P(PositionDeleteUpdateTest, ConcurrentRemovalRejectsPreparedDeletes) { + ICEBERG_UNWRAP_OR_FAIL(auto properties, table_->NewUpdateProperties()); + properties->Set(TableProperties::kCommitMinRetryWaitMs.key(), "1") + .Set(TableProperties::kCommitMaxRetryWaitMs.key(), "1"); + ASSERT_THAT(properties->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + auto mock = std::make_shared<::testing::NiceMock>(); + ON_CALL(*mock, LoadTable(::testing::_)) + .WillByDefault( + [this](const TableIdentifier& name) { return catalog_->LoadTable(name); }); + EXPECT_CALL(*mock, UpdateTable(::testing::_, ::testing::_, ::testing::_)) + .Times(1) + .WillOnce([this](const auto&, const auto&, + const auto&) -> Result> { + ICEBERG_ASSIGN_OR_RAISE(auto removal, table_->NewDeleteFiles()); + removal->DeleteFile(data_file_); + ICEBERG_RETURN_UNEXPECTED(removal->Commit()); + return CommitFailed("concurrent removal committed"); + }); + ICEBERG_UNWRAP_OR_FAIL( + auto competing_table, + Table::Make(table_->name(), table_->metadata(), + std::string(table_->metadata_file_location()), file_io_, mock)); + ICEBERG_UNWRAP_OR_FAIL(auto update, PositionDeleteUpdate::Make(competing_table)); + update->Delete(data_file_->file_path, 2); + EXPECT_THAT(update->Commit(), IsError(ErrorKind::kValidationFailed)); + + ICEBERG_UNWRAP_OR_FAIL(auto reloaded, catalog_->LoadTable(table_ident_)); + ICEBERG_UNWRAP_OR_FAIL(auto builder, reloaded->NewScan()); + ICEBERG_UNWRAP_OR_FAIL(auto scan, builder->Build()); + ICEBERG_UNWRAP_OR_FAIL(auto tasks, scan->PlanFiles()); + EXPECT_TRUE(tasks.empty()); +} + +INSTANTIATE_TEST_SUITE_P( + FormatAndPartitioning, PositionDeleteUpdateTest, + ::testing::Values(RoutingCase{.format_version = 2, + .unpartitioned = false, + .expected_format = FileFormatType::kParquet}, + RoutingCase{.format_version = 3, + .unpartitioned = false, + .expected_format = FileFormatType::kPuffin}, + RoutingCase{.format_version = 3, + .unpartitioned = true, + .expected_format = FileFormatType::kPuffin})); + +class PositionDeleteV3Test : public MinimalUpdateTestBase { + protected: + static void SetUpTestSuite() { + avro::RegisterAll(); + parquet::RegisterAll(); + } + + int8_t format_version() const override { return 3; } + + void SetUp() override { + MinimalUpdateTestBase::SetUp(); + ICEBERG_UNWRAP_OR_FAIL(spec_, table_->spec()); + ICEBERG_UNWRAP_OR_FAIL(schema_, table_->schema()); + data_file_ = MakeDataFile(); + AppendDataFile(); + } + + std::shared_ptr MakeDataFile() const { + auto file = std::make_shared(); + file->content = DataFile::Content::kData; + file->file_path = table_location_ + "/data/file.parquet"; + file->file_format = FileFormatType::kParquet; + file->partition = PartitionValues({Literal::Long(10)}); + file->file_size_in_bytes = 1024; + file->record_count = 10; + file->partition_spec_id = spec_->spec_id(); + return file; + } + + void AppendDataFile() { + ICEBERG_UNWRAP_OR_FAIL(auto append, table_->NewFastAppend()); + append->AppendFile(data_file_); + ASSERT_THAT(append->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + } + + Result> CurrentTask(const std::shared_ptr
& table) { + ICEBERG_ASSIGN_OR_RAISE(auto builder, table->NewScan()); + ICEBERG_ASSIGN_OR_RAISE(auto scan, builder->Build()); + ICEBERG_ASSIGN_OR_RAISE(auto tasks, scan->PlanFiles()); + ICEBERG_CHECK(tasks.size() == 1, "Expected one file scan task, found {}", + tasks.size()); + return tasks.front(); + } + + Result LoadPositions(const FileScanTask& task) { + DeleteLoader loader(file_io_); + return loader.LoadPositionDeletes(task.delete_files(), task.data_file()->file_path); + } + + Result> WritePositionDeletes( + std::span positions) { + const auto path = table_location_ + "/data/existing-position-deletes.parquet"; + ICEBERG_ASSIGN_OR_RAISE( + auto writer, + PositionDeleteWriter::Make(PositionDeleteWriterOptions{ + .path = path, + .schema = schema_, + .spec = spec_, + .partition = data_file_->partition, + .format = FileFormatType::kParquet, + .io = file_io_, + .properties = {{"write.parquet.compression-codec", "uncompressed"}}, + })); + for (int64_t pos : positions) { + ICEBERG_RETURN_UNEXPECTED(writer->WriteDelete(data_file_->file_path, pos)); + } + ICEBERG_RETURN_UNEXPECTED(writer->Close()); + ICEBERG_ASSIGN_OR_RAISE(auto result, writer->Metadata()); + ICEBERG_CHECK(result.data_files.size() == 1, + "Expected one position delete file, found {}", + result.data_files.size()); + return result.data_files.front(); + } + + Result> CurrentDeleteEntries() { + ICEBERG_ASSIGN_OR_RAISE(auto snapshot, table_->current_snapshot()); + SnapshotReader cache(snapshot.get()); + ICEBERG_ASSIGN_OR_RAISE(auto manifests, cache.DeleteManifests(file_io_)); + std::vector entries; + for (const auto& manifest : manifests) { + ICEBERG_ASSIGN_OR_RAISE( + auto spec, table_->metadata()->PartitionSpecById(manifest.partition_spec_id)); + ICEBERG_ASSIGN_OR_RAISE( + auto reader, + ManifestReader::Make(manifest, file_io_, schema_, std::move(spec))); + ICEBERG_ASSIGN_OR_RAISE(auto manifest_entries, reader->Entries()); + entries.insert(entries.end(), std::make_move_iterator(manifest_entries.begin()), + std::make_move_iterator(manifest_entries.end())); + } + return entries; + } + + void ConfigureRetries(int32_t retries) { + ICEBERG_UNWRAP_OR_FAIL(auto properties, table_->NewUpdateProperties()); + properties->Set(TableProperties::kCommitNumRetries.key(), std::to_string(retries)) + .Set(TableProperties::kCommitMinRetryWaitMs.key(), "1") + .Set(TableProperties::kCommitMaxRetryWaitMs.key(), "1") + .Set(TableProperties::kCommitTotalRetryTimeMs.key(), "1000"); + ASSERT_THAT(properties->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + } + + void FailCommit(std::unexpected failure) { + auto mock = std::make_shared<::testing::NiceMock>(); + EXPECT_CALL(*mock, UpdateTable(::testing::_, ::testing::_, ::testing::_)) + .Times(1) + .WillOnce(::testing::Return(std::move(failure))); + ICEBERG_UNWRAP_OR_FAIL( + table_, Table::Make(table_->name(), table_->metadata(), + std::string(table_->metadata_file_location()), file_io_, + std::move(mock))); + } + + std::vector PuffinFiles() { + auto arrow_io = std::dynamic_pointer_cast(file_io_); + EXPECT_NE(arrow_io, nullptr); + ::arrow::fs::FileSelector selector; + selector.base_dir = table_location_ + "/data"; + selector.recursive = true; + auto infos = arrow_io->fs()->GetFileInfo(selector); + EXPECT_TRUE(infos.ok()) << infos.status().ToString(); + std::vector paths; + if (!infos.ok()) { + return paths; + } + for (const auto& info : *infos) { + if (info.path().ends_with(".puffin")) { + paths.push_back(info.path()); + } + } + return paths; + } + + std::shared_ptr spec_; + std::shared_ptr schema_; + std::shared_ptr data_file_; +}; + +TEST_F(PositionDeleteV3Test, SecondDeleteMergesAndSupersedesPreviousDV) { + ICEBERG_UNWRAP_OR_FAIL(auto first, PositionDeleteUpdate::Make(table_)); + first->Delete(data_file_->file_path, 1); + ASSERT_THAT(first->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ICEBERG_UNWRAP_OR_FAIL(auto old_task, CurrentTask(table_)); + const auto old_dv = old_task->delete_files().front(); + + ICEBERG_UNWRAP_OR_FAIL(auto second, PositionDeleteUpdate::Make(table_)); + second->Delete(data_file_->file_path, 3); + ASSERT_THAT(second->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto task, CurrentTask(table_)); + ASSERT_EQ(task->delete_files().size(), 1); + EXPECT_NE(task->delete_files().front()->file_path, old_dv->file_path); + ICEBERG_UNWRAP_OR_FAIL(auto positions, LoadPositions(*task)); + EXPECT_EQ(positions.Cardinality(), 2); + EXPECT_TRUE(positions.IsDeleted(1)); + EXPECT_TRUE(positions.IsDeleted(3)); + + ICEBERG_UNWRAP_OR_FAIL(auto entries, CurrentDeleteEntries()); + EXPECT_TRUE(std::ranges::any_of(entries, [&old_dv](const ManifestEntry& entry) { + return entry.status == ManifestStatus::kDeleted && entry.data_file != nullptr && + entry.data_file->file_path == old_dv->file_path && + entry.data_file->content_offset == old_dv->content_offset; + })); +} + +TEST_F(PositionDeleteV3Test, MergesAndSupersedesFileScopedParquetDelete) { + RegisterTableFromResource("TableMetadataV2ValidMinimal.json"); + ICEBERG_UNWRAP_OR_FAIL(spec_, table_->spec()); + ICEBERG_UNWRAP_OR_FAIL(schema_, table_->schema()); + data_file_ = MakeDataFile(); + AppendDataFile(); + + const std::vector existing_positions{1, 3}; + ICEBERG_UNWRAP_OR_FAIL(auto old_delete, WritePositionDeletes(existing_positions)); + ASSERT_EQ(old_delete->referenced_data_file, data_file_->file_path); + ICEBERG_UNWRAP_OR_FAIL(auto row_delta, table_->NewRowDelta()); + row_delta->AddDeletes(old_delete); + ASSERT_THAT(row_delta->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto properties, table_->NewUpdateProperties()); + properties->Set(TableProperties::kFormatVersion.key(), "3"); + ASSERT_THAT(properties->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto update, PositionDeleteUpdate::Make(table_)); + update->Delete(data_file_->file_path, 5); + ASSERT_THAT(update->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto task, CurrentTask(table_)); + ASSERT_EQ(task->delete_files().size(), 1); + EXPECT_EQ(task->delete_files().front()->file_format, FileFormatType::kPuffin); + ICEBERG_UNWRAP_OR_FAIL(auto positions, LoadPositions(*task)); + EXPECT_EQ(positions.Cardinality(), 3); + EXPECT_TRUE(positions.IsDeleted(1)); + EXPECT_TRUE(positions.IsDeleted(3)); + EXPECT_TRUE(positions.IsDeleted(5)); + + ICEBERG_UNWRAP_OR_FAIL(auto entries, CurrentDeleteEntries()); + EXPECT_TRUE(std::ranges::any_of(entries, [&old_delete](const ManifestEntry& entry) { + return entry.status == ManifestStatus::kDeleted && entry.data_file != nullptr && + entry.data_file->file_path == old_delete->file_path; + })); +} + +TEST_F(PositionDeleteV3Test, UsesTargetDataFileSpecAfterPartitionEvolution) { + ICEBERG_UNWRAP_OR_FAIL(auto spec_update, table_->NewUpdatePartitionSpec()); + spec_update->RemoveField("x"); + ASSERT_THAT(spec_update->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + ASSERT_NE(table_->metadata()->default_spec_id, data_file_->partition_spec_id); + + ICEBERG_UNWRAP_OR_FAIL(auto update, PositionDeleteUpdate::Make(table_)); + update->Delete(data_file_->file_path, 4); + ASSERT_THAT(update->Commit(), IsOk()); + ASSERT_THAT(table_->Refresh(), IsOk()); + + ICEBERG_UNWRAP_OR_FAIL(auto task, CurrentTask(table_)); + const auto& dv = task->delete_files().front(); + EXPECT_EQ(dv->partition_spec_id, data_file_->partition_spec_id); + EXPECT_EQ(dv->partition, data_file_->partition); +} + +TEST_F(PositionDeleteV3Test, CommitRetryReusesWrittenDV) { + ConfigureRetries(1); + FailCommits(1); + ICEBERG_UNWRAP_OR_FAIL(auto update, PositionDeleteUpdate::Make(table_)); + update->Delete(data_file_->file_path, 5); + ASSERT_THAT(update->Commit(), IsOk()); + EXPECT_EQ(PuffinFiles().size(), 1); +} + +TEST_F(PositionDeleteV3Test, FailedCommitCleansWrittenDV) { + ConfigureRetries(0); + FailCommit(CommitFailed("injected failure")); + ICEBERG_UNWRAP_OR_FAIL(auto update, PositionDeleteUpdate::Make(table_)); + update->Delete(data_file_->file_path, 6); + EXPECT_THAT(update->Commit(), IsError(ErrorKind::kCommitFailed)); + EXPECT_TRUE(PuffinFiles().empty()); +} + +TEST_F(PositionDeleteV3Test, CleanupFailureIsReportedAndRetried) { + ConfigureRetries(0); + auto mock_catalog = std::make_shared<::testing::NiceMock>(); + int update_calls = 0; + ON_CALL(*mock_catalog, UpdateTable(::testing::_, ::testing::_, ::testing::_)) + .WillByDefault( + [this, &update_calls]( + const TableIdentifier& identifier, + const std::vector>& requirements, + const std::vector>& updates) + -> Result> { + if (++update_calls == 1) { + return CommitFailed("injected commit failure"); + } + return catalog_->UpdateTable(identifier, requirements, updates); + }); + auto arrow_io = std::dynamic_pointer_cast(file_io_); + ASSERT_NE(arrow_io, nullptr); + auto failing_io = std::make_shared(arrow_io->fs()); + ICEBERG_UNWRAP_OR_FAIL(auto mock_table, + Table::Make(table_->name(), table_->metadata(), + std::string(table_->metadata_file_location()), + failing_io, mock_catalog)); + + ICEBERG_UNWRAP_OR_FAIL(auto update, PositionDeleteUpdate::Make(mock_table)); + update->Delete(data_file_->file_path, 6); + auto first_status = update->Commit(); + EXPECT_THAT(first_status, IsError(ErrorKind::kCommitFailed)); + EXPECT_THAT(first_status, HasErrorMessage("injected commit failure")); + EXPECT_THAT(first_status, HasErrorMessage("injected cleanup failure")); + ASSERT_EQ(PuffinFiles().size(), 1); + ASSERT_EQ(failing_io->puffin_delete_attempts.size(), 1); + const auto retained_path = failing_io->puffin_delete_attempts.front(); + + EXPECT_THAT(update->Commit(), IsOk()); + EXPECT_EQ(update_calls, 2); + EXPECT_EQ(PuffinFiles().size(), 1); + ASSERT_EQ(failing_io->puffin_delete_attempts.size(), 2); + EXPECT_EQ(std::ranges::count(failing_io->puffin_delete_attempts, retained_path), 2); +} + +TEST_F(PositionDeleteV3Test, CommitStateUnknownRelinquishesOutputOwnership) { + ConfigureRetries(1); + FailCommit(CommitStateUnknown("injected unknown state")); + ICEBERG_UNWRAP_OR_FAIL(auto update, PositionDeleteUpdate::Make(table_)); + update->Delete(data_file_->file_path, 6); + EXPECT_THAT(update->Commit(), IsError(ErrorKind::kCommitStateUnknown)); + ASSERT_EQ(PuffinFiles().size(), 1); + EXPECT_THAT(update->Commit(), IsError(ErrorKind::kInvalidArgument)); + EXPECT_EQ(PuffinFiles().size(), 1); +} + +} // namespace + +} // namespace iceberg diff --git a/src/iceberg/test/std_io.h b/src/iceberg/test/std_io.h index 725fc7ba5..c7d5b312b 100644 --- a/src/iceberg/test/std_io.h +++ b/src/iceberg/test/std_io.h @@ -319,11 +319,9 @@ class StdFileIO : public FileIO { Status DeleteFile(const std::string& file_location) override { std::error_code ec; - if (!std::filesystem::remove(file_location, ec)) { - if (ec) { - return IOError("Failed to delete file {}: {}", file_location, ec.message()); - } - return IOError("File does not exist: {}", file_location); + std::filesystem::remove(file_location, ec); + if (ec) { + return IOError("Failed to delete file {}: {}", file_location, ec.message()); } return {}; }