Copilot commented on code in PR #801:
URL: https://github.com/apache/iceberg-cpp/pull/801#discussion_r3958165547


##########
src/iceberg/inspect/snapshots_table.cc:
##########
@@ -19,21 +19,140 @@
 
 #include "iceberg/inspect/snapshots_table.h"
 
+#include <chrono>
+#include <cstddef>
 #include <memory>
+#include <optional>
+#include <tuple>
 #include <utility>
 #include <vector>
 
+#include <nanoarrow/nanoarrow.h>
+
+#include "iceberg/arrow/nanoarrow_status_internal.h"
+#include "iceberg/arrow_c_data_util_internal.h"
+#include "iceberg/arrow_row_builder_internal.h"
 #include "iceberg/schema.h"
 #include "iceberg/schema_field.h"
+#include "iceberg/schema_internal.h"
+#include "iceberg/snapshot.h"
 #include "iceberg/table.h"
-#include "iceberg/table_identifier.h"
+#include "iceberg/table_metadata.h"
 #include "iceberg/type.h"
+#include "iceberg/util/macros.h"
 
 namespace iceberg {
 namespace {
 
-std::shared_ptr<Schema> MakeSnapshotsTableSchema() {
-  return std::make_shared<Schema>(std::vector<SchemaField>{
+Status AppendSnapshot(ArrowRowBuilder& builder, const Snapshot& snapshot) {
+  ICEBERG_RETURN_UNEXPECTED(
+      AppendInt(builder.column(0), 
std::chrono::duration_cast<std::chrono::microseconds>(
+                                       
snapshot.timestamp_ms.time_since_epoch())
+                                       .count()));
+  ICEBERG_RETURN_UNEXPECTED(AppendInt(builder.column(1), 
snapshot.snapshot_id));
+
+  if (snapshot.parent_snapshot_id.has_value()) {
+    ICEBERG_RETURN_UNEXPECTED(AppendInt(builder.column(2), 
*snapshot.parent_snapshot_id));
+  } else {
+    ICEBERG_RETURN_UNEXPECTED(AppendNull(builder.column(2)));
+  }
+
+  auto operation = snapshot.Operation();
+  if (operation.has_value()) {
+    ICEBERG_RETURN_UNEXPECTED(AppendString(builder.column(3), *operation));
+  } else {
+    ICEBERG_RETURN_UNEXPECTED(AppendNull(builder.column(3)));
+  }
+
+  ICEBERG_RETURN_UNEXPECTED(AppendString(builder.column(4), 
snapshot.manifest_list));
+
+  const bool has_summary = !snapshot.summary.empty();
+  auto summary = snapshot.summary;
+  summary.erase(SnapshotSummaryFields::kOperation);
+  if (!has_summary) {
+    ICEBERG_RETURN_UNEXPECTED(AppendNull(builder.column(5)));
+  } else {
+    ICEBERG_RETURN_UNEXPECTED(AppendStringMap(builder.column(5), summary));

Review Comment:
   This copies the entire `snapshot.summary` map for every row, which can be 
expensive for large summaries and large snapshot counts. Consider appending map 
entries directly while skipping `SnapshotSummaryFields::kOperation` (or 
building a small filtered structure only when needed) to avoid per-row full-map 
copies; this also makes it straightforward to detect “empty after filtering”.



##########
src/iceberg/inspect/snapshots_table.cc:
##########
@@ -19,21 +19,140 @@
 
 #include "iceberg/inspect/snapshots_table.h"
 
+#include <chrono>
+#include <cstddef>
 #include <memory>
+#include <optional>
+#include <tuple>
 #include <utility>
 #include <vector>
 
+#include <nanoarrow/nanoarrow.h>
+
+#include "iceberg/arrow/nanoarrow_status_internal.h"
+#include "iceberg/arrow_c_data_util_internal.h"
+#include "iceberg/arrow_row_builder_internal.h"
 #include "iceberg/schema.h"
 #include "iceberg/schema_field.h"
+#include "iceberg/schema_internal.h"
+#include "iceberg/snapshot.h"
 #include "iceberg/table.h"
-#include "iceberg/table_identifier.h"
+#include "iceberg/table_metadata.h"
 #include "iceberg/type.h"
+#include "iceberg/util/macros.h"
 
 namespace iceberg {
 namespace {
 
-std::shared_ptr<Schema> MakeSnapshotsTableSchema() {
-  return std::make_shared<Schema>(std::vector<SchemaField>{
+Status AppendSnapshot(ArrowRowBuilder& builder, const Snapshot& snapshot) {
+  ICEBERG_RETURN_UNEXPECTED(
+      AppendInt(builder.column(0), 
std::chrono::duration_cast<std::chrono::microseconds>(
+                                       
snapshot.timestamp_ms.time_since_epoch())
+                                       .count()));
+  ICEBERG_RETURN_UNEXPECTED(AppendInt(builder.column(1), 
snapshot.snapshot_id));
+
+  if (snapshot.parent_snapshot_id.has_value()) {
+    ICEBERG_RETURN_UNEXPECTED(AppendInt(builder.column(2), 
*snapshot.parent_snapshot_id));
+  } else {
+    ICEBERG_RETURN_UNEXPECTED(AppendNull(builder.column(2)));
+  }
+
+  auto operation = snapshot.Operation();
+  if (operation.has_value()) {
+    ICEBERG_RETURN_UNEXPECTED(AppendString(builder.column(3), *operation));
+  } else {
+    ICEBERG_RETURN_UNEXPECTED(AppendNull(builder.column(3)));
+  }
+
+  ICEBERG_RETURN_UNEXPECTED(AppendString(builder.column(4), 
snapshot.manifest_list));
+
+  const bool has_summary = !snapshot.summary.empty();
+  auto summary = snapshot.summary;
+  summary.erase(SnapshotSummaryFields::kOperation);
+  if (!has_summary) {
+    ICEBERG_RETURN_UNEXPECTED(AppendNull(builder.column(5)));
+  } else {
+    ICEBERG_RETURN_UNEXPECTED(AppendStringMap(builder.column(5), summary));

Review Comment:
   This copies the entire `snapshot.summary` map for every row, which can be 
expensive for large summaries and large snapshot counts. Consider appending map 
entries directly while skipping `SnapshotSummaryFields::kOperation` (or 
building a small filtered structure only when needed) to avoid per-row full-map 
copies; this also makes it straightforward to detect “empty after filtering”.



##########
src/iceberg/inspect/snapshots_table.cc:
##########
@@ -19,21 +19,140 @@
 
 #include "iceberg/inspect/snapshots_table.h"
 
+#include <chrono>
+#include <cstddef>
 #include <memory>
+#include <optional>
+#include <tuple>
 #include <utility>
 #include <vector>
 
+#include <nanoarrow/nanoarrow.h>
+
+#include "iceberg/arrow/nanoarrow_status_internal.h"
+#include "iceberg/arrow_c_data_util_internal.h"
+#include "iceberg/arrow_row_builder_internal.h"
 #include "iceberg/schema.h"
 #include "iceberg/schema_field.h"
+#include "iceberg/schema_internal.h"
+#include "iceberg/snapshot.h"
 #include "iceberg/table.h"
-#include "iceberg/table_identifier.h"
+#include "iceberg/table_metadata.h"
 #include "iceberg/type.h"
+#include "iceberg/util/macros.h"
 
 namespace iceberg {
 namespace {
 
-std::shared_ptr<Schema> MakeSnapshotsTableSchema() {
-  return std::make_shared<Schema>(std::vector<SchemaField>{
+Status AppendSnapshot(ArrowRowBuilder& builder, const Snapshot& snapshot) {
+  ICEBERG_RETURN_UNEXPECTED(
+      AppendInt(builder.column(0), 
std::chrono::duration_cast<std::chrono::microseconds>(
+                                       
snapshot.timestamp_ms.time_since_epoch())
+                                       .count()));
+  ICEBERG_RETURN_UNEXPECTED(AppendInt(builder.column(1), 
snapshot.snapshot_id));
+
+  if (snapshot.parent_snapshot_id.has_value()) {
+    ICEBERG_RETURN_UNEXPECTED(AppendInt(builder.column(2), 
*snapshot.parent_snapshot_id));
+  } else {
+    ICEBERG_RETURN_UNEXPECTED(AppendNull(builder.column(2)));
+  }
+
+  auto operation = snapshot.Operation();
+  if (operation.has_value()) {
+    ICEBERG_RETURN_UNEXPECTED(AppendString(builder.column(3), *operation));
+  } else {
+    ICEBERG_RETURN_UNEXPECTED(AppendNull(builder.column(3)));
+  }
+
+  ICEBERG_RETURN_UNEXPECTED(AppendString(builder.column(4), 
snapshot.manifest_list));
+
+  const bool has_summary = !snapshot.summary.empty();
+  auto summary = snapshot.summary;
+  summary.erase(SnapshotSummaryFields::kOperation);
+  if (!has_summary) {
+    ICEBERG_RETURN_UNEXPECTED(AppendNull(builder.column(5)));
+  } else {
+    ICEBERG_RETURN_UNEXPECTED(AppendStringMap(builder.column(5), summary));
+  }

Review Comment:
   The PR description says “emit null when the remaining summary is empty”, but 
the implementation decides null-vs-map based on the pre-filter 
`snapshot.summary` emptiness. For an “operation-only” summary, this currently 
emits an empty map (not null). If the intended behavior is “null when summary 
is empty after removing `operation`”, compute emptiness after `erase()` (or 
equivalent filtering) and append null when the filtered map is empty; update 
`ScanEmptySummary` expectations accordingly.



##########
src/iceberg/test/snapshots_table_test.cc:
##########
@@ -0,0 +1,230 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements.  See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership.  The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License.  You may obtain a copy of the License at
+ *
+ *   http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied.  See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+#include "iceberg/inspect/snapshots_table.h"
+
+#include <cstdint>
+#include <memory>
+#include <string>
+#include <utility>
+#include <vector>
+
+#include <arrow/array.h>
+#include <arrow/c/bridge.h>
+#include <arrow/record_batch.h>
+#include <gmock/gmock.h>
+#include <gtest/gtest.h>
+
+#include "iceberg/constants.h"
+#include "iceberg/inspect/metadata_table.h"
+#include "iceberg/test/matchers.h"
+#include "iceberg/test/metadata_table_test_base.h"
+
+namespace iceberg {
+namespace {
+
+std::vector<std::pair<std::string, std::string>> GetMapEntries(
+    const std::shared_ptr<::arrow::MapArray>& map_array, int64_t row) {
+  auto keys = 
std::static_pointer_cast<::arrow::StringArray>(map_array->keys());
+  auto values = 
std::static_pointer_cast<::arrow::StringArray>(map_array->items());
+  std::vector<std::pair<std::string, std::string>> entries;
+  entries.reserve(map_array->value_length(row));
+  const auto offset = map_array->value_offset(row);
+  for (int64_t index = offset; index < offset + map_array->value_length(row); 
++index) {
+    entries.emplace_back(keys->GetString(index), values->GetString(index));
+  }
+  return entries;
+}
+
+}  // namespace
+
+class SnapshotsTableTest : public MetadataTableTestBase {
+ protected:
+  void SetUp() override {
+    MetadataTableTestBase::SetUp();
+
+    auto [snap1, snap2] = MakeTestSnapshots();
+    ICEBERG_UNWRAP_OR_FAIL(
+        table_, MakeTableWithSnapshots({snap1, snap2}, 
/*current_snapshot_id=*/2));
+
+    ICEBERG_UNWRAP_OR_FAIL(snapshots_table_, 
MetadataTable::Make<SnapshotsTable>(table_));
+  }
+
+  std::unique_ptr<MetadataTable> snapshots_table_;
+};
+
+TEST_F(SnapshotsTableTest, Construct) {
+  EXPECT_EQ(snapshots_table_->kind(), MetadataTable::Kind::kSnapshots);
+  EXPECT_EQ(snapshots_table_->source_table(), table_);
+  EXPECT_NE(snapshots_table_->schema(), nullptr);
+}
+
+TEST_F(SnapshotsTableTest, Scan) {
+  // Scan the snapshots table once and verify all columns of the result.
+  ICEBERG_UNWRAP_OR_FAIL(auto stream, snapshots_table_->Scan());
+  ICEBERG_UNWRAP_OR_FAIL(auto batches, ReadAllBatches(std::move(stream)));
+  ASSERT_EQ(batches.size(), 1);
+  const auto& batch = batches.front();
+
+  // Row and column counts.
+  EXPECT_EQ(batch->num_rows(), 2);
+  EXPECT_EQ(batch->num_columns(), 6);
+
+  // Column 0: committed_at (timestamptz) — microseconds since epoch.
+  auto committed_at = 
std::static_pointer_cast<::arrow::TimestampArray>(batch->column(0));
+  EXPECT_EQ(committed_at->Value(0), 1234567890000LL * 1000);
+  EXPECT_EQ(committed_at->Value(1), 9876543210000LL * 1000);
+
+  // Column 1: snapshot_id (long) — returned in storage order.
+  auto snapshot_ids = 
std::static_pointer_cast<::arrow::Int64Array>(batch->column(1));
+  EXPECT_EQ(snapshot_ids->Value(0), 1);
+  EXPECT_EQ(snapshot_ids->Value(1), 2);
+
+  // Column 2: parent_id (long) — first snapshot has no parent.
+  auto parent_ids = 
std::static_pointer_cast<::arrow::Int64Array>(batch->column(2));
+  EXPECT_TRUE(parent_ids->IsNull(0));
+  EXPECT_FALSE(parent_ids->IsNull(1));
+  EXPECT_EQ(parent_ids->Value(1), 1);
+
+  // Column 3: operation (string).
+  auto operations = 
std::static_pointer_cast<::arrow::StringArray>(batch->column(3));
+  EXPECT_EQ(operations->GetString(0), "append");
+  EXPECT_EQ(operations->GetString(1), "append");
+
+  // Column 4: manifest_list (string).
+  auto manifest_lists = 
std::static_pointer_cast<::arrow::StringArray>(batch->column(4));
+  EXPECT_EQ(manifest_lists->GetString(0), "file:/tmp/manifest1.avro");
+  EXPECT_EQ(manifest_lists->GetString(1), "file:/tmp/manifest2.avro");
+
+  // Column 5: summary (map<string,string>) excludes the separate operation 
field.
+  auto summaries = 
std::static_pointer_cast<::arrow::MapArray>(batch->column(5));
+  EXPECT_FALSE(summaries->IsNull(0));
+  EXPECT_FALSE(summaries->IsNull(1));
+  EXPECT_EQ(summaries->value_length(0), 10);
+  EXPECT_EQ(summaries->value_length(1), 10);
+
+  auto first_summary = GetMapEntries(summaries, 0);
+  EXPECT_THAT(
+      first_summary,
+      ::testing::Not(::testing::Contains(::testing::Pair("operation", 
"append"))));
+  EXPECT_THAT(first_summary, 
::testing::Contains(::testing::Pair("total-records", "1")));
+
+  auto second_summary = GetMapEntries(summaries, 1);
+  EXPECT_THAT(
+      second_summary,
+      ::testing::Not(::testing::Contains(::testing::Pair("operation", 
"append"))));
+  EXPECT_THAT(second_summary, 
::testing::Contains(::testing::Pair("total-records", "2")));
+}
+
+TEST_F(SnapshotsTableTest, ScanEmptySnapshotList) {
+  // A table with zero snapshots should return zero rows.
+  ICEBERG_UNWRAP_OR_FAIL(
+      auto empty_table,
+      MakeTableWithSnapshots({}, /*current_snapshot_id=*/kInvalidSnapshotId));
+
+  ICEBERG_UNWRAP_OR_FAIL(snapshots_table_,
+                         MetadataTable::Make<SnapshotsTable>(empty_table));
+
+  ICEBERG_UNWRAP_OR_FAIL(auto stream, snapshots_table_->Scan());
+  ICEBERG_UNWRAP_OR_FAIL(auto batches, ReadAllBatches(std::move(stream)));
+  EXPECT_TRUE(batches.empty());
+}
+
+TEST_F(SnapshotsTableTest, ScanSkipsNullSnapshots) {
+  auto [snap1, snap2] = MakeTestSnapshots();
+  ICEBERG_UNWRAP_OR_FAIL(auto table, MakeTableWithSnapshots({snap1, nullptr, 
snap2},
+                                                            
/*current_snapshot_id=*/2));
+  ICEBERG_UNWRAP_OR_FAIL(auto snapshots_table,
+                         MetadataTable::Make<SnapshotsTable>(table));
+
+  ICEBERG_UNWRAP_OR_FAIL(auto stream, snapshots_table->Scan());
+  ICEBERG_UNWRAP_OR_FAIL(auto batches, ReadAllBatches(std::move(stream)));
+  ASSERT_EQ(batches.size(), 1);
+  EXPECT_EQ(batches.front()->num_rows(), 2);
+}
+
+TEST_F(SnapshotsTableTest, ScanEmptySummary) {
+  auto [missing_summary, operation_only_summary] = MakeTestSnapshots();
+  missing_summary->summary.clear();
+  operation_only_summary->summary = {
+      {SnapshotSummaryFields::kOperation, DataOperation::kAppend}};
+  ICEBERG_UNWRAP_OR_FAIL(auto table,
+                         MakeTableWithSnapshots({missing_summary, 
operation_only_summary},
+                                                /*current_snapshot_id=*/2));
+  ICEBERG_UNWRAP_OR_FAIL(auto snapshots_table,
+                         MetadataTable::Make<SnapshotsTable>(table));
+
+  ICEBERG_UNWRAP_OR_FAIL(auto stream, snapshots_table->Scan());
+  ICEBERG_UNWRAP_OR_FAIL(auto batches, ReadAllBatches(std::move(stream)));
+  ASSERT_EQ(batches.size(), 1);
+  auto summaries =
+      std::static_pointer_cast<::arrow::MapArray>(batches.front()->column(5));
+  EXPECT_TRUE(summaries->IsNull(0));
+  EXPECT_FALSE(summaries->IsNull(1));
+  EXPECT_EQ(summaries->value_length(1), 0);
+}
+
+TEST_F(SnapshotsTableTest, ScanReturnsMultipleBatches) {
+  auto snapshot = MakeTestSnapshots().first;
+  std::vector<std::shared_ptr<Snapshot>> snapshots(1025, snapshot);
+  ICEBERG_UNWRAP_OR_FAIL(auto table, 
MakeTableWithSnapshots(std::move(snapshots),
+                                                            
/*current_snapshot_id=*/1));
+  ICEBERG_UNWRAP_OR_FAIL(auto snapshots_table,
+                         MetadataTable::Make<SnapshotsTable>(table));
+
+  ICEBERG_UNWRAP_OR_FAIL(auto stream, snapshots_table->Scan());
+  ICEBERG_UNWRAP_OR_FAIL(auto batches, ReadAllBatches(std::move(stream)));
+  ASSERT_EQ(batches.size(), 2);
+  EXPECT_EQ(batches[0]->num_rows(), 1024);
+  EXPECT_EQ(batches[1]->num_rows(), 1);
+}

Review Comment:
   This test hard-codes the batch size as `1024` and uses `1025` to force a 
second batch. To keep the test robust if `MetadataTable::kBatchSize` changes, 
use `MetadataTable::kBatchSize` (and `kBatchSize + 1`) for both the input size 
and assertions.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to