arunsarin85 commented on code in PR #11094: URL: https://github.com/apache/ozone/pull/11094#discussion_r3898309947
########## hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/om/snapshot/TestOmSnapshotDefragSpaceSavings.java: ########## @@ -0,0 +1,651 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.hadoop.ozone.om.snapshot; + +import static org.apache.hadoop.ozone.OzoneConfigKeys.OZONE_SNAPSHOT_DELETING_SERVICE_INTERVAL; +import static org.apache.hadoop.ozone.OzoneConsts.OM_KEY_PREFIX; +import static org.apache.hadoop.ozone.OzoneConsts.ROCKSDB_SST_SUFFIX; +import static org.apache.hadoop.ozone.om.OMConfigKeys.OZONE_FILESYSTEM_SNAPSHOT_ENABLED_KEY; +import static org.apache.hadoop.ozone.om.OMConfigKeys.OZONE_SNAPSHOT_DEFRAG_SERVICE_INTERVAL; +import static org.apache.hadoop.ozone.om.OMConfigKeys.SNAPSHOT_DEFRAG_LIMIT_PER_TASK; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertNull; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assumptions.assumeTrue; + +import java.io.File; +import java.io.IOException; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.Arrays; +import java.util.HashSet; +import java.util.List; +import java.util.Set; +import java.util.UUID; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.TimeoutException; +import java.util.stream.Collectors; +import java.util.stream.Stream; +import org.apache.hadoop.hdds.conf.OzoneConfiguration; +import org.apache.hadoop.hdds.utils.IOUtils; +import org.apache.hadoop.hdds.utils.db.DBStore; +import org.apache.hadoop.hdds.utils.db.ManagedRawSSTFileReader; +import org.apache.hadoop.ozone.DataTestUtil; +import org.apache.hadoop.ozone.MiniOzoneCluster; +import org.apache.hadoop.ozone.client.ObjectStore; +import org.apache.hadoop.ozone.client.OzoneBucket; +import org.apache.hadoop.ozone.client.OzoneClient; +import org.apache.hadoop.ozone.om.OMMetadataManager; +import org.apache.hadoop.ozone.om.OmSnapshotInternalMetrics; +import org.apache.hadoop.ozone.om.OmSnapshotManager; +import org.apache.hadoop.ozone.om.OzoneManager; +import org.apache.hadoop.ozone.om.helpers.BucketLayout; +import org.apache.hadoop.ozone.om.helpers.SnapshotInfo; +import org.apache.ozone.test.GenericTestUtils; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +/** + * HDDS-13218: integration tests that snapshot defrag reduces checkpoint disk footprint. + * + * <p>Uses inode-aware sizing (matching {@link OMSnapshotDirectoryMetrics}) so hardlinked SST files + * are not double-counted across snapshot checkpoint directories. Version-0 checkpoints hardlink to + * AOS SST files, so their on-disk byte totals are not comparable to materialized post-defrag + * checkpoints. Savings are validated by cross-snapshot SST reference reduction in the chain. + * + * <p>Covers a three-snapshot chain with AOS compactions and insert/overwrite/delete churn on OBS + * and FSO buckets, full-then-incremental defrag paths, footprint checks after deleting the + * middle snapshot and running a follow-up defrag on the remaining youngest snapshot, isolated + * full defrag on a single snapshot, and idempotent repeated defrag on an already-defragged chain. + */ +public class TestOmSnapshotDefragSpaceSavings { + + private static final byte[] TEST_KEY_CONTENT = new byte[] {0x61, 0x62, 0x63}; + private static final byte[] OVERWRITE_KEY_CONTENT = new byte[] {0x64, 0x65, 0x66}; + private static final int INITIAL_KEY_COUNT = 100; + private static final int OVERWRITE_KEY_COUNT = 50; + private static final int DELETE_KEY_COUNT = 25; + private static final int NEW_KEYS_PER_SNAPSHOT = 10; + private static final int CHECKPOINT_WAIT_MS = 120_000; + private static final int PURGE_WAIT_MS = 180_000; + private static final int DEFRAG_WAIT_MS = 600_000; + private static final int KEY_DELETE_WAIT_MS = 60_000; + private static final long FOOTPRINT_TOLERANCE_BYTES = 8192; + + private MiniOzoneCluster cluster; + private OzoneConfiguration conf; + private OzoneClient client; + private ObjectStore store; + + @BeforeEach + void initCluster() throws Exception { + startCluster(); + } + + private void startCluster() throws Exception { + assumeTrue(ManagedRawSSTFileReader.tryLoadLibrary(), + "Snapshot defrag requires rocks-tools native library"); + + conf = new OzoneConfiguration(); + conf.setBoolean(OZONE_FILESYSTEM_SNAPSHOT_ENABLED_KEY, true); + // Keep background defrag idle during the test; manual triggerSnapshotDefrag() still requires + // the service to be initialized (interval must be > 0). + conf.setTimeDuration(OZONE_SNAPSHOT_DEFRAG_SERVICE_INTERVAL, 2, TimeUnit.HOURS); + conf.setInt(SNAPSHOT_DEFRAG_LIMIT_PER_TASK, 10); + conf.setTimeDuration(OZONE_SNAPSHOT_DELETING_SERVICE_INTERVAL, 1, TimeUnit.SECONDS); + + cluster = MiniOzoneCluster.newBuilder(conf).setNumDatanodes(3).build(); + cluster.waitForClusterToBeReady(); + client = cluster.newClient(); + store = client.getObjectStore(); + resumeBackgroundServices(); + } + + private void restartCluster() throws Exception { + IOUtils.closeQuietly(client, cluster); + startCluster(); + } + + private void resumeBackgroundServices() { + OzoneManager om = cluster.getOzoneManager(); + om.getKeyManager().getDeletingService().resume(); + om.getKeyManager().getDirDeletingService().resume(); + om.getKeyManager().getSnapshotDeletingService().resume(); + } + + @AfterEach + void shutdownCluster() { + IOUtils.closeQuietly(client, cluster); + } + + /** + * Three-snapshot chain with churn on OBS then FSO: defrag should reduce aggregate checkpoint + * footprint on both layouts. OBS pass also verifies one full defrag and two incremental defrags. + */ + @Test + public void testSnapshotDefragReducesCheckpointFootprintWithChurn() throws Exception { + runChurnFootprintScenario(BucketLayout.OBJECT_STORE); Review Comment: Done - @ParameterizedTest(name = "layout={0}") + [%s] in assertDefragReducedChainFootprint -- This is an automated message from the Apache Git Service. To respond to the message, please log on to GitHub and use the URL above to go to the specific comment. To unsubscribe, e-mail: [email protected] For queries about this service, please contact Infrastructure at: [email protected] --------------------------------------------------------------------- To unsubscribe, e-mail: [email protected] For additional commands, e-mail: [email protected]
