smtwilio commented on code in PR #19811:
URL: https://github.com/apache/hudi/pull/19811#discussion_r3928622856


##########
hudi-utilities/src/main/java/org/apache/hudi/utilities/streamer/HoodieMultiTableStreamer.java:
##########
@@ -479,10 +516,143 @@ public void sync() {
         }
       }
     }
+  }
 
-    log.info("Ingestion was successful for topics: {}", successTables);
-    if (!failedTables.isEmpty()) {
-      log.info("Ingestion failed for topics: {}", failedTables);
+  /**
+   * Syncs all tables concurrently, one thread per table. Used for continuous 
mode where each table's sync blocks
+   * indefinitely.
+   *
+   * <p>When {@code --fail-fast-on-continuous} is enabled, the first table 
failure fails the whole job. The sibling
+   * streamers are shut down and a {@link HoodieException} is thrown so the 
caller can exit with a non-zero status.
+   * Otherwise, every table is synced independently and a single failure does 
not affect the others.
+   */
+  private void syncContinuously() {
+    // Streamer instances are registered from worker threads, so a thread-safe 
list is required.
+    final List<HoodieStreamer> streamerInstances = new 
CopyOnWriteArrayList<>();
+    // Set once fail fast trips, so tasks that register their streamer 
afterwards stop before starting the sync.
+    final AtomicBoolean shutdownRequested = new AtomicBoolean(false);
+    final ExecutorService executor = 
Executors.newFixedThreadPool(tableExecutionContexts.size(),
+        new CustomizedThreadFactory("multi-table-streamer", true));
+    boolean terminated = false;
+    try {
+      final CompletableFuture<?>[] tableFutures = 
tableExecutionContexts.stream()
+          .map(context -> CompletableFuture.runAsync(() -> {
+            HoodieStreamer streamer = null;
+            try {
+              streamer = new HoodieStreamer(context.getConfig(), jssc, 
Option.ofNullable(context.getProperties()));
+              streamerInstances.add(streamer);
+              // Register before checking the flag so a concurrent 
shutdownStreamers() always sees this streamer.
+              if (shutdownRequested.get()) {
+                return;
+              }
+              streamer.sync();
+              // A streamer registered just before fail fast tripped can reach 
here without ever ingesting.
+              // shutdown() call will be a no-op because its ingestion service 
hadn't started yet.
+              // Don't count that as a success.
+              if (!shutdownRequested.get()) {
+                successTables.add(Helpers.getTableWithDatabase(context));
+              }
+            } catch (Exception e) {
+              log.error("error while running MultiTableDeltaStreamer for 
table: {}", context.getTableName(), e);
+              failedTables.add(Helpers.getTableWithDatabase(context));
+              if (failFastOnContinuousMode) {
+                throw new CompletionException(e);
+              }
+            } finally {
+              if (streamer != null) {
+                streamer.shutdownGracefully();
+              }
+            }
+          }, executor)).toArray(CompletableFuture[]::new);
+
+      if (failFastOnContinuousMode) {
+        log.info("Fail fast enabled in continuous mode. The whole job fails on 
any single table failure");
+        awaitFailFast(tableFutures, streamerInstances, shutdownRequested);
+      } else {
+        CompletableFuture.allOf(tableFutures).join();
+      }
+      log.info("Successful tables: {}, Failed tables: {}", successTables, 
failedTables);
+    } finally {
+      // Wait for every worker thread to finish (including its finally 
cleanup) before returning, so sync() does not
+      // return while a table is still writing and main() then stops the 
shared Spark context under it.
+      terminated = shutdownExecutor(executor);
+    }
+    // If the workers never terminated, ingestion may still be running. Fail 
loudly instead of returning as if the
+    // cleanup succeeded, so the caller does not silently proceed to Spark 
teardown with live writers.
+    if (!terminated) {
+      throw new HoodieException("Timed out shutting down table ingestion 
workers in continuous mode");
+    }
+  }
+
+  /**
+   * Waits until either every table sync finishes successfully or the first 
one fails. On the first failure, the
+   * remaining streamers are shut down and a {@link HoodieException} is 
thrown. Unlike {@code anyOf(...)}, this only
+   * trips on an <em>exceptional</em> completion, so a table that terminates 
normally (e.g. via a
+   * {@link PostWriteTerminationStrategy}) does not abort its siblings.
+   */
+  private void awaitFailFast(CompletableFuture<?>[] tableFutures, 
List<HoodieStreamer> streamerInstances, AtomicBoolean shutdownRequested) {
+    final CompletableFuture<Void> firstOutcome = new CompletableFuture<>();
+    // Trip as soon as any table fails ...
+    for (CompletableFuture<?> tableFuture : tableFutures) {
+      tableFuture.whenComplete((result, throwable) -> {
+        if (throwable != null) {
+          firstOutcome.completeExceptionally(throwable);
+        }
+      });
+    }
+    // ... or complete normally once every table has finished without failure.
+    CompletableFuture.allOf(tableFutures).whenComplete((result, throwable) -> {

Review Comment:
   Good catch! My earlier test failed when swapped with `anyOf`. I dropped the 
test and now relying on `TestFutureUtils` for the guarantee.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to