m1a2st commented on code in PR #22654: URL: https://github.com/apache/kafka/pull/22654#discussion_r3551862446
########## clients/src/main/java/org/apache/kafka/clients/producer/internals/ChunkedBufferPool.java: ########## @@ -0,0 +1,200 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.kafka.clients.producer.internals; + +import org.apache.kafka.clients.producer.BufferExhaustedException; +import org.apache.kafka.common.KafkaException; +import org.apache.kafka.common.metrics.Metrics; +import org.apache.kafka.common.utils.Time; + +import java.nio.ByteBuffer; +import java.util.ArrayList; +import java.util.List; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.locks.Condition; + +/** + * A {@link BufferPool} dedicated to chunk-sized buffer reuse (chunk size = {@link #poolableSize()}). + * <p> + * Adds {@link #allocateChunks(int, long)} to acquire multiple chunks atomically. + */ +public class ChunkedBufferPool extends BufferPool { + + public ChunkedBufferPool(long memory, int chunkSize, Metrics metrics, Time time, String metricGrpName) { + super(memory, chunkSize, metrics, time, metricGrpName); + } + + /** + * Allocate {@code ceil(totalSize / chunkSize)} chunk-sized buffers atomically, mirroring + * {@link BufferPool#allocate}: satisfied immediately if memory is available, else blocks up to + * {@code maxTimeToBlockMs} for the whole request (FIFO on {@link #waiters}). + * The reservation is tracked as bytes against {@link #nonPooledAvailableMemory} plus chunks polled + * from {@link #free}. Any failure refunds the whole reservation and signals the next waiter + * before the exception propagates, so a failed request leaves nothing reserved. + * + * @param totalSize minimum total bytes of capacity required across the returned chunks + * @param maxTimeToBlockMs maximum time in milliseconds to block waiting for memory + * @return list of {@code ceil(totalSize / chunkSize)} {@code ByteBuffer}s, each of capacity + * {@code chunkSize} + * @throws InterruptedException if interrupted while waiting + * @throws IllegalArgumentException if {@code totalSize <= 0}, or if the request rounded up to + * whole chunks exceeds {@code totalMemory()} + * @throws BufferExhaustedException if the request can't be satisfied within {@code maxTimeToBlockMs} + * @throws KafkaException if the pool is closed during the wait + */ + public List<ByteBuffer> allocateChunks(int totalSize, long maxTimeToBlockMs) throws InterruptedException { + if (totalSize <= 0) + throw new IllegalArgumentException("totalSize must be positive: " + totalSize); + throwIfChunksNeededExceedsPool(totalSize); + + int chunkSize = poolableSize(); + int numChunks = (int) (((long) totalSize + chunkSize - 1L) / chunkSize); + long memoryRequired = (long) numChunks * chunkSize; + + // Chunks taken from the free list. The remaining bytes are reserved against + // nonPooledAvailableMemory and materialized as raw allocations after the lock is released. + List<ByteBuffer> pooled = new ArrayList<>(numChunks); + + lock.lock(); + if (this.closed) { + lock.unlock(); + throw new KafkaException("Producer closed while allocating memory"); + } + try { + long freeListBytes = (long) free.size() * chunkSize; + if (this.nonPooledAvailableMemory + freeListBytes >= memoryRequired) { + // Enough memory available to allocate the chunks needed + while (pooled.size() < numChunks && !free.isEmpty()) + pooled.add(free.pollFirst()); + long remainingBytes = memoryRequired - (long) pooled.size() * chunkSize; + if (remainingBytes > 0) { + // remainingBytes > 0 means the free list was fully drained into `pooled`, so the + // remainder comes entirely from non-pooled memory (sufficient per the check above). + this.nonPooledAvailableMemory -= remainingBytes; + } + } else { + // Not enough memory available to allocate the chunks needed, so we need to wait for memory. + // Same as in BufferPool.allocate, but wait to acquire the memory needed for all the chunks. + // A single Condition is added to the waiter's list to ensure FIFO fairness at the request level. + // + // `accumulated` tracks bytes drawn from nonPooledAvailableMemory only (pool chunks + // already taken live in `pooled`), and is always a whole-chunk multiple. If the wait + // does not complete (timeout / close / interrupt), the finally refunds the whole + // reservation: `accumulated` back to non-pooled memory, `pooled` back to the free chunks list. + long accumulated = 0; + boolean allocationCompleted = false; + Condition moreMemory = lock.newCondition(); + try { + long remainingTimeToBlockNs = TimeUnit.MILLISECONDS.toNanos(maxTimeToBlockMs); + waiters.addLast(moreMemory); + while ((long) pooled.size() * chunkSize + accumulated < memoryRequired) { + long startWaitNs = time.nanoseconds(); + long timeNs; + boolean waitingTimeElapsed; + try { + waitingTimeElapsed = !moreMemory.await(remainingTimeToBlockNs, TimeUnit.NANOSECONDS); + } finally { + long endWaitNs = time.nanoseconds(); + timeNs = Math.max(0L, endWaitNs - startWaitNs); + recordWaitTime(timeNs); + } + + if (this.closed) + throw new KafkaException("Producer closed while allocating memory"); + + if (waitingTimeElapsed) { Review Comment: The `allocateChunks(extensionBytesNeeded, 0L)` path is intentionally a non-blocking, fail-fast probe. A `BufferExhaustedException` thrown here is caught, and the record proceeds down the new-batch path rather than being dropped. Calling `recordBufferExhausted()` here would count records that were never actually dropped, inflating the `buffer-exhausted-records` metrics. It could even double-count a single dropped record: once when the probe fails, and again if the subsequent new-batch allocation times out. -- This is an automated message from the Apache Git Service. To respond to the message, please log on to GitHub and use the URL above to go to the specific comment. To unsubscribe, e-mail: [email protected] For queries about this service, please contact Infrastructure at: [email protected]
