From a0f84394b45dc780bda62adab7ffbefbb00dbead Mon Sep 17 00:00:00 2001 From: fhan Date: Wed, 16 Sep 2026 09:34:26 +0800 Subject: [PATCH 1/2] [client] Back off writes rejected by disk protection When disk protection rejects a write with DISK_WRITE_LOCKED, the Log/KV writers apply a bounded exponential backoff before retrying that bucket, instead of hot-looping and hammering the rejecting server. Write-throttling is unified in a new WriteThrottleController that owns two independent gates per bucket: - KV backpressure (wall-clock, quadratic) for KV upsert/delete - disk-write backoff (monotonic nanos, exponential, never-shorten, ceil-to-ms) for DISK_WRITE_LOCKED rejections Per bucket the effective delay is max(kvRemainingMs, diskRemainingMs); the accumulator wakes on the min across buckets. bucketReady keeps leader-first ordering: a leaderless bucket is still reported to unknownLeaderTables while a gate is pending, so metadata refresh (which sends no data) stays orthogonal to throttling. Adds ConfigOptions for the backoff bounds, documentation, unit coverage in RecordAccumulatorTest/SenderTest, and DiskWriteBackoffITCase. --- .../fluss/client/write/RecordAccumulator.java | 237 +++++------ .../org/apache/fluss/client/write/Sender.java | 29 +- .../client/write/WriteThrottleController.java | 265 ++++++++++++ .../client/table/DiskWriteBackoffITCase.java | 217 ++++++++++ .../client/write/RecordAccumulatorTest.java | 216 ++++++++++ .../apache/fluss/client/write/SenderTest.java | 377 +++++++++++++++++- .../apache/fluss/config/ConfigOptions.java | 22 + website/docs/maintenance/configuration.md | 30 ++ 8 files changed, 1262 insertions(+), 131 deletions(-) create mode 100644 fluss-client/src/main/java/org/apache/fluss/client/write/WriteThrottleController.java create mode 100644 fluss-client/src/test/java/org/apache/fluss/client/table/DiskWriteBackoffITCase.java diff --git a/fluss-client/src/main/java/org/apache/fluss/client/write/RecordAccumulator.java b/fluss-client/src/main/java/org/apache/fluss/client/write/RecordAccumulator.java index 32f4b881e4..7e30f412bc 100644 --- a/fluss-client/src/main/java/org/apache/fluss/client/write/RecordAccumulator.java +++ b/fluss-client/src/main/java/org/apache/fluss/client/write/RecordAccumulator.java @@ -42,6 +42,7 @@ import org.apache.fluss.shaded.arrow.org.apache.arrow.memory.BufferAllocator; import org.apache.fluss.shaded.arrow.org.apache.arrow.memory.ChunkedAllocationManager; import org.apache.fluss.utils.CopyOnWriteMap; +import org.apache.fluss.utils.ExponentialBackoff; import org.apache.fluss.utils.MathUtils; import org.apache.fluss.utils.clock.Clock; @@ -52,6 +53,7 @@ import javax.annotation.concurrent.GuardedBy; import java.io.IOException; +import java.time.Duration; import java.util.ArrayDeque; import java.util.ArrayList; import java.util.Collections; @@ -64,12 +66,14 @@ import java.util.Set; import java.util.concurrent.ConcurrentHashMap; import java.util.concurrent.ConcurrentMap; +import java.util.concurrent.atomic.AtomicBoolean; import java.util.concurrent.atomic.AtomicInteger; import static org.apache.fluss.record.LogRecordBatchFormat.NO_BATCH_SEQUENCE; import static org.apache.fluss.record.LogRecordBatchFormat.NO_WRITER_ID; import static org.apache.fluss.shaded.arrow.org.apache.arrow.memory.BufferAllocatorUtil.createBufferAllocator; import static org.apache.fluss.utils.PartitionUtils.HISTORICAL_PARTITION_VALUE; +import static org.apache.fluss.utils.Preconditions.checkArgument; import static org.apache.fluss.utils.Preconditions.checkNotNull; /* This file is based on source code of Apache Kafka Project (https://kafka.apache.org/), licensed by the Apache @@ -113,7 +117,7 @@ public final class RecordAccumulator { private final Object resourcesLock = new Object(); @GuardedBy("resourcesLock") - private boolean resourcesDestroyed; + private final AtomicBoolean resourcesDestroyed = new AtomicBoolean(false); /** The pool of lazily created arrow {@link ArrowWriter}s for arrow log write batch. */ private final ArrowWriterPool arrowWriterPool; @@ -133,17 +137,10 @@ public final class RecordAccumulator { private final Clock clock; private final DynamicWriteBatchSizeEstimator batchSizeEstimator; - // Per-bucket backpressure throttle expiry timestamp. Accessed strictly by key on - // hot paths (get / put / remove); writes happen on every backpressure signal and - // every eviction, so the container is sized for lock-striped O(1) updates without - // any whole-map snapshot cost. - private final ConcurrentMap throttleExpiryMs = new ConcurrentHashMap<>(); - private final long maxThrottleMs; - - // Latest Cluster snapshot fed to the metadata-driven throttle sweep. Identity - // equality against this reference short-circuits the sweep when metadata hasn't - // changed. - private volatile Cluster lastClusterRef = Cluster.empty(); + // Unifies the per-bucket write-throttling gates (KV backpressure and disk-write backoff): + // both paths that decide sendability, ready() and drain(), consult it for the effective + // remaining delay so no single gate can be dropped. + private final WriteThrottleController throttle; // TODO add retryBackoffMs to retry the produce request upon receiving an error. // TODO add deliveryTimeoutMs to report success or failure on record delivery. @@ -159,6 +156,23 @@ public final class RecordAccumulator { idempotenceManager, writerMetricGroup, clock, + createDiskWriteBackoff(conf), + BucketAssignerFactory.defaultFactory(conf)); + } + + @VisibleForTesting + RecordAccumulator( + Configuration conf, + IdempotenceManager idempotenceManager, + WriterMetricGroup writerMetricGroup, + Clock clock, + ExponentialBackoff diskWriteBackoff) { + this( + conf, + idempotenceManager, + writerMetricGroup, + clock, + diskWriteBackoff, BucketAssignerFactory.defaultFactory(conf)); } @@ -169,6 +183,23 @@ public final class RecordAccumulator { WriterMetricGroup writerMetricGroup, Clock clock, BucketAssignerFactory bucketAssignerFactory) { + this( + conf, + idempotenceManager, + writerMetricGroup, + clock, + createDiskWriteBackoff(conf), + bucketAssignerFactory); + } + + @VisibleForTesting + RecordAccumulator( + Configuration conf, + IdempotenceManager idempotenceManager, + WriterMetricGroup writerMetricGroup, + Clock clock, + ExponentialBackoff diskWriteBackoff, + BucketAssignerFactory bucketAssignerFactory) { this.bucketAssignerFactory = checkNotNull(bucketAssignerFactory); this.closed = false; this.flushesInProgress = new AtomicInteger(0); @@ -193,11 +224,30 @@ public final class RecordAccumulator { (int) conf.get(ConfigOptions.CLIENT_WRITER_BUFFER_PAGE_SIZE).getBytes()); this.idempotenceManager = idempotenceManager; this.clock = clock; - this.maxThrottleMs = - conf.get(ConfigOptions.CLIENT_WRITER_KV_BACKPRESSURE_MAX_THROTTLE).toMillis(); + this.throttle = + new WriteThrottleController( + conf.get(ConfigOptions.CLIENT_WRITER_KV_BACKPRESSURE_MAX_THROTTLE) + .toMillis(), + checkNotNull(diskWriteBackoff), + clock, + resourcesDestroyed); registerMetrics(writerMetricGroup); } + private static ExponentialBackoff createDiskWriteBackoff(Configuration conf) { + Duration initial = conf.get(ConfigOptions.CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF); + Duration maximum = conf.get(ConfigOptions.CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF_MAX); + checkArgument( + initial.compareTo(Duration.ofMillis(1)) >= 0 + && initial.compareTo(maximum) <= 0 + && maximum.compareTo(Duration.ofMillis(Integer.MAX_VALUE)) <= 0, + "%s and %s must satisfy 1ms <= initial <= maximum <= %sms.", + ConfigOptions.CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF.key(), + ConfigOptions.CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF_MAX.key(), + Integer.MAX_VALUE); + return new ExponentialBackoff(initial.toMillis(), 2, maximum.toMillis(), 0.2); + } + private void registerMetrics(WriterMetricGroup writerMetricGroup) { // memory segment pool related metrics. writerMetricGroup.gauge(MetricNames.WRITER_BUFFER_TOTAL_BYTES, writerBufferPool::totalSize); @@ -297,7 +347,7 @@ private BucketAndWriteBatches createBucketAndWriteBatches( */ public ReadyCheckResult ready(Cluster cluster) { Set readyNodes = new HashSet<>(); - long nextReadyCheckDelayMs = batchTimeoutMs; + long nextReadyCheckDelayMs = Long.MAX_VALUE; Set unknownLeaderTables = new HashSet<>(); // Go table by table so that we can get queue sizes for buckets in a table and calculate // cumulative frequency table (used in bucket assigner). @@ -312,7 +362,14 @@ public ReadyCheckResult ready(Cluster cluster) { nextReadyCheckDelayMs); } - // TODO and the earliest time at which any non-send-able bucket will be ready; + // When all queued buckets are backing off, wait for the earliest effective deadline. + // In particular, a zero batch timeout must not turn this wait into a busy loop. + // Keep the normal polling cadence for idle writers and unresolved metadata. + if (!readyNodes.isEmpty() + || !unknownLeaderTables.isEmpty() + || nextReadyCheckDelayMs == Long.MAX_VALUE) { + nextReadyCheckDelayMs = Math.min(nextReadyCheckDelayMs, batchTimeoutMs); + } return new ReadyCheckResult(readyNodes, nextReadyCheckDelayMs, unknownLeaderTables); } @@ -620,7 +677,7 @@ public void awaitFlushCompletion() throws InterruptedException { */ public void deallocate(WriteBatch batch) { synchronized (resourcesLock) { - if (incomplete.removeIfPresent(batch) && !resourcesDestroyed) { + if (incomplete.removeIfPresent(batch) && !resourcesDestroyed.get()) { writerBufferPool.returnAll(batch.pooledMemorySegments()); } } @@ -754,36 +811,35 @@ private long bucketReady( int bucketId = entry.getKey(); TableBucket tableBucket = cluster.getTableBucket(tableId, targetPath, bucketId); - // If this bucket is throttled, don't mark its node as ready. - // Instead, factor the remaining throttle time into the next check delay. - Long throttleExpiry = throttleExpiryMs.get(tableBucket); - if (throttleExpiry != null) { - long now = clock.milliseconds(); - if (now < throttleExpiry) { - nextReadyCheckDelayMs = Math.min(nextReadyCheckDelayMs, throttleExpiry - now); - continue; - } - // Expired — evict here to reclaim entries for buckets whose deque - // has gone empty and won't reach the drain-time throttle check. - throttleExpiryMs.remove(tableBucket); - } - Integer leader = cluster.leaderFor(tableBucket); if (leader == null) { // This is a bucket for which leader is not known, but messages are // available to send. Note that entries are currently not removed from - // batches when deque is empty. + // batches when deque is empty. Reported regardless of any pending throttle gate: + // a metadata refresh is orthogonal to the gate (it sends no data), and refreshing + // now means the leader is ready to receive once the gate clears. unknownLeaderTables.add(targetPath); - } else { - nextReadyCheckDelayMs = - batchReady( - exhausted, - leader, - waitedTimeMs, - full, - readyNodes, - nextReadyCheckDelayMs); + continue; + } + + // A bucket blocked by any throttle gate (KV backpressure and/or disk-write backoff) + // cannot become ready until every gate clears, so wake no earlier than the latest of + // them. remainingDelayMs() already folds the reasons into a single per-bucket deadline + // via max(); across buckets we keep the earliest wake-up via min(). + long gateMs = throttle.remainingDelayMs(tableBucket); + if (gateMs > 0) { + nextReadyCheckDelayMs = Math.min(nextReadyCheckDelayMs, gateMs); + continue; } + + nextReadyCheckDelayMs = + batchReady( + exhausted, + leader, + waitedTimeMs, + full, + readyNodes, + nextReadyCheckDelayMs); } return nextReadyCheckDelayMs; @@ -1248,8 +1304,8 @@ private List drainBatchesForOneNode(Cluster cluster, Integer no } private boolean shouldSkipBucket(WriteBatch first, TableBucket tableBucket) { - // Backpressure throttle check: skip this bucket if still under throttle - if (isThrottled(tableBucket)) { + // Skip this bucket while any write-throttling gate still blocks it. + if (throttle.isGated(tableBucket)) { return true; } if (idempotenceManager.idempotenceEnabled()) { @@ -1289,80 +1345,38 @@ private boolean shouldSkipBucket(WriteBatch first, TableBucket tableBucket) { return false; } - // ---- Backpressure throttle methods ---- + // ---- Write-throttling delegates (see WriteThrottleController) ---- - /** - * Check if a bucket is currently under backpressure throttle. - * - *

Performs lazy eviction: if the throttle has expired, the entry is removed from the map to - * prevent unbounded growth. - * - * @return true if the bucket should be skipped during drain - */ boolean isThrottled(TableBucket tableBucket) { - Long expiry = throttleExpiryMs.get(tableBucket); - if (expiry == null) { - return false; - } - if (clock.milliseconds() < expiry) { - return true; - } - // Expired — evict to prevent map leak - throttleExpiryMs.remove(tableBucket); - return false; + return throttle.isKvThrottled(tableBucket); + } + + /** Installs disk backoff before the batch retry count is increased by re-enqueueing. */ + long backoffAfterDiskWriteLocked(ReadyWriteBatch batch) { + return throttle.backoffAfterDiskWrite(batch.tableBucket(), batch.writeBatch().attempts()); + } + + /** Returns the remaining disk backoff without changing its deadline. */ + long diskWriteBackoffRemainingMs(TableBucket tableBucket) { + return throttle.diskRemainingMs(tableBucket); + } + + /** Reclaims expired entries even when their queues no longer contain any batches. */ + void maybeEvictExpiredDiskWriteBackoffs() { + throttle.maybeEvictExpiredDiskBackoffs(); + } + + @VisibleForTesting + int diskWriteBackoffCount() { + return throttle.diskBackoffCount(); } - /** - * Update the throttle state for a bucket based on the received pressure signal. - * - *

The delay grows quadratically with pressure: {@code delay = maxThrottleMs * p^2}, where - * {@code p ∈ [0, 1)}. This provides meaningful throttling across the full ramp-up window while - * remaining gentle at low pressure. - * - * @param tableBucket the bucket to update - * @param pressure value in {@code [0, 1)} on the wire; {@code 0} means recovered, positive - * values trigger a throttle window. {@code 1.0f} is reserved as the internal hard-rejection - * value (never sent by the server): the Sender passes it when the server rejected the write - * outright, and it installs the full {@link #maxThrottleMs} window directly. - */ void updateThrottle(TableBucket tableBucket, float pressure) { - if (pressure >= 1f) { - // Hard rejection: stall the bucket for the full max throttle window, bypassing the - // quadratic curve to avoid long-to-float rounding. - throttleExpiryMs.put(tableBucket, clock.milliseconds() + maxThrottleMs); - return; - } - if (pressure > 0f) { - long delay = (long) (maxThrottleMs * pressure * pressure); - if (delay > 0) { - throttleExpiryMs.put(tableBucket, clock.milliseconds() + delay); - return; - } - } - // Recovered or below the meaningful resolution: remove throttle. - // Note: in production, recovery relies on the last throttle window expiring naturally - // (server stops sending the pressure field once p reaches 0). This branch exists as - // defensive completeness and is exercised by unit tests. - throttleExpiryMs.remove(tableBucket); + throttle.updateKvPressure(tableBucket, pressure); } - /** - * Evict throttle entries whose buckets no longer exist in the given cluster (leader unknown, - * partition dropped, table dropped). - * - *

Invoked on every Sender loop with the current cluster snapshot. The identity short-circuit - * makes this an O(1) no-op when metadata hasn't changed, so the actual O(N) walk only runs once - * per real metadata refresh. - */ void maybeEvictStaleThrottles(Cluster cluster) { - if (cluster == lastClusterRef) { - return; - } - lastClusterRef = cluster; - if (throttleExpiryMs.isEmpty()) { - return; - } - throttleExpiryMs.keySet().removeIf(tb -> cluster.leaderFor(tb) == null); + throttle.maybeEvictStaleThrottles(cluster); } private int getDrainIndex(int id) { @@ -1577,15 +1591,16 @@ public void close() { @VisibleForTesting public void destroyResources() { synchronized (resourcesLock) { - if (resourcesDestroyed) { + if (resourcesDestroyed.get()) { return; } - resourcesDestroyed = true; + resourcesDestroyed.compareAndSet(false, true); writerBufferPool.close(); arrowWriterPool.close(); bufferAllocator.close(); chunkedFactory.close(); } + throttle.clearDiskBackoffs(); } /** Per table bucket and write batches. */ diff --git a/fluss-client/src/main/java/org/apache/fluss/client/write/Sender.java b/fluss-client/src/main/java/org/apache/fluss/client/write/Sender.java index 9f8b1e8ba2..060c2c8160 100644 --- a/fluss-client/src/main/java/org/apache/fluss/client/write/Sender.java +++ b/fluss-client/src/main/java/org/apache/fluss/client/write/Sender.java @@ -247,6 +247,7 @@ private void sendWriteData() throws Exception { // Refresh per-bucket throttle entries against the current cluster snapshot, // dropping any whose bucket has disappeared from metadata. accumulator.maybeEvictStaleThrottles(clusterSnapshot); + accumulator.maybeEvictExpiredDiskWriteBackoffs(); // get the list of buckets with data ready to send. ReadyCheckResult readyCheckResult = accumulator.ready(clusterSnapshot); @@ -359,6 +360,7 @@ private void reEnqueueBatch(ReadyWriteBatch readyWriteBatch) { boolean reEnqueued = accumulator.reEnqueue(readyWriteBatch); maybeRemoveFromInflightBatches(readyWriteBatch); + wakeup(); if (reEnqueued) { // metrics for retry record count. writerMetricGroup @@ -746,13 +748,8 @@ private Set handleWriteBatchException( } else if (canRetry(readyWriteBatch, error.error())) { // if batch failed because of retrievable exception, we need to retry send all those // batches. - LOG.warn( - "Get error write response on table bucket {}, retrying ({} attempts left). Error: {}", - readyWriteBatch.tableBucket(), - retries - writeBatch.attempts(), - error.formatErrMsg()); - if (!idempotenceManager.idempotenceEnabled()) { + prepareWriteRetry(readyWriteBatch, error); reEnqueueBatch(readyWriteBatch); } else if (idempotenceManager.hasWriterId(writeBatch.writerId())) { // If idempotence is enabled only retry the request if the current writer id is @@ -761,6 +758,7 @@ private Set handleWriteBatchException( "Retrying batch to table-bucket {}, Batch sequence : {}", readyWriteBatch.tableBucket(), writeBatch.batchSequence()); + prepareWriteRetry(readyWriteBatch, error); reEnqueueBatch(readyWriteBatch); } else { Exception exception = @@ -802,6 +800,25 @@ private Set handleWriteBatchException( return invalidMetadataTables; } + private void prepareWriteRetry(ReadyWriteBatch batch, ApiError error) { + if (error.error() == Errors.DISK_WRITE_LOCKED) { + long backoffMs = accumulator.backoffAfterDiskWriteLocked(batch); + LOG.warn( + "Get error write response on table bucket {}, disk backoff {} ms " + + "({} attempts left). Error: {}", + batch.tableBucket(), + backoffMs, + retries - batch.writeBatch().attempts(), + error.formatErrMsg()); + } else { + LOG.warn( + "Get error write response on table bucket {}, retrying ({} attempts left). Error: {}", + batch.tableBucket(), + retries - batch.writeBatch().attempts(), + error.formatErrMsg()); + } + } + /** * Rechecks unknown-leader partitions after a bulk metadata update reports {@link * PartitionNotExistException}, and handles missing partitions for tables with historical diff --git a/fluss-client/src/main/java/org/apache/fluss/client/write/WriteThrottleController.java b/fluss-client/src/main/java/org/apache/fluss/client/write/WriteThrottleController.java new file mode 100644 index 0000000000..1b44887981 --- /dev/null +++ b/fluss-client/src/main/java/org/apache/fluss/client/write/WriteThrottleController.java @@ -0,0 +1,265 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.fluss.client.write; + +import org.apache.fluss.annotation.Internal; +import org.apache.fluss.annotation.VisibleForTesting; +import org.apache.fluss.cluster.Cluster; +import org.apache.fluss.metadata.TableBucket; +import org.apache.fluss.utils.ExponentialBackoff; +import org.apache.fluss.utils.clock.Clock; + +import java.util.concurrent.ConcurrentHashMap; +import java.util.concurrent.ConcurrentMap; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.atomic.AtomicBoolean; + +import static org.apache.fluss.utils.Preconditions.checkNotNull; + +/** + * Unifies the write-throttling gates that can delay a bucket from being sent: KV backpressure and + * disk-write backoff. Both gates are per-{@link TableBucket}. + * + *

The two reasons are stored and installed separately because their semantics genuinely differ: + * + *

    + *
  • KV backpressure uses wall-clock deadlines, is latest-wins (a fresher pressure + * signal may shorten or clear the window), and derives its delay quadratically from the + * pressure value. + *
  • Disk-write backoff uses monotonic deadlines (immune to wall-clock shifts), is + * never-shortened under concurrency, and derives its delay from an exponential backoff + * keyed on the batch retry count. + *
+ * + *

What is unified is the read side: a bucket cannot become ready until every active gate + * has cleared, so {@link #remainingDelayMs(TableBucket)} returns the maximum remaining delay across + * reasons and is the single source consulted by both {@code ready()} and {@code drain()}. Adding a + * future throttling reason should only touch this class, not those two paths. + */ +@Internal +final class WriteThrottleController { + + // KV backpressure: wall-clock expiry, latest-wins. Accessed strictly by key on hot paths + // (get / put / remove); the container is sized for lock-striped O(1) updates without any + // whole-map snapshot cost. + private final ConcurrentMap kvThrottleExpiryMs = new ConcurrentHashMap<>(); + private final long maxThrottleMs; + + // Disk protection is independent of KV pressure, whose responses may shorten or clear a + // throttle. Deadlines use monotonic time and are shared by all queues targeting a bucket. + private final ConcurrentMap diskBackoffDeadlineNanos = + new ConcurrentHashMap<>(); + private final ExponentialBackoff diskBackoff; + // Only the sender thread performs periodic sweeps. + private long lastDiskSweepNanos; + + // Latest Cluster snapshot fed to the metadata-driven throttle sweep. Identity equality against + // this reference short-circuits the sweep when metadata hasn't changed. + private volatile Cluster lastClusterRef = Cluster.empty(); + + private final Clock clock; + // Shared with the owning accumulator: a late RPC callback must not retain state after final + // resource destruction. + private final AtomicBoolean resourcesDestroyed; + + WriteThrottleController( + long maxThrottleMs, + ExponentialBackoff diskBackoff, + Clock clock, + AtomicBoolean resourcesDestroyed) { + this.maxThrottleMs = maxThrottleMs; + this.diskBackoff = checkNotNull(diskBackoff); + this.clock = clock; + this.resourcesDestroyed = resourcesDestroyed; + this.lastDiskSweepNanos = clock.nanoseconds(); + } + + // ------------------------------------------------------------------------ + // Unified read side: the single home for "how long until this bucket may send". + // ------------------------------------------------------------------------ + + /** + * Remaining delay before the bucket may be sent, i.e. the latest deadline across all active + * gates. Both reasons express their remainder in milliseconds from now, so the maximum is well + * defined even though they track different clocks internally. Expired entries are evicted + * lazily as a side effect. + */ + long remainingDelayMs(TableBucket tableBucket) { + return Math.max(kvRemainingMs(tableBucket), diskRemainingMs(tableBucket)); + } + + /** Whether any gate currently blocks the bucket from being sent. */ + boolean isGated(TableBucket tableBucket) { + return remainingDelayMs(tableBucket) > 0; + } + + // ------------------------------------------------------------------------ + // KV backpressure (wall-clock millis, latest-wins, quadratic delay). + // ------------------------------------------------------------------------ + + /** + * Update the throttle state for a bucket based on the received pressure signal. + * + *

The delay grows quadratically with pressure: {@code delay = maxThrottleMs * p^2}, where + * {@code p ∈ [0, 1)}. This provides meaningful throttling across the full ramp-up window while + * remaining gentle at low pressure. + * + * @param tableBucket the bucket to update + * @param pressure value in {@code [0, 1)} on the wire; {@code 0} means recovered, positive + * values trigger a throttle window. {@code 1.0f} is reserved as the internal hard-rejection + * value (never sent by the server): the Sender passes it when the server rejected the write + * outright, and it installs the full {@link #maxThrottleMs} window directly. + */ + void updateKvPressure(TableBucket tableBucket, float pressure) { + if (pressure >= 1f) { + // Hard rejection: stall the bucket for the full max throttle window, bypassing the + // quadratic curve to avoid long-to-float rounding. + kvThrottleExpiryMs.put(tableBucket, clock.milliseconds() + maxThrottleMs); + return; + } + if (pressure > 0f) { + long delay = (long) (maxThrottleMs * pressure * pressure); + if (delay > 0) { + kvThrottleExpiryMs.put(tableBucket, clock.milliseconds() + delay); + return; + } + } + // Recovered or below the meaningful resolution: remove throttle. + // Note: in production, recovery relies on the last throttle window expiring naturally + // (server stops sending the pressure field once p reaches 0). This branch exists as + // defensive completeness and is exercised by unit tests. + kvThrottleExpiryMs.remove(tableBucket); + } + + boolean isKvThrottled(TableBucket tableBucket) { + return kvRemainingMs(tableBucket) > 0; + } + + private long kvRemainingMs(TableBucket tableBucket) { + Long expiry = kvThrottleExpiryMs.get(tableBucket); + if (expiry == null) { + return 0; + } + long remainingMs = expiry - clock.milliseconds(); + if (remainingMs > 0) { + return remainingMs; + } + // Expired — evict to prevent map leak. + kvThrottleExpiryMs.remove(tableBucket, expiry); + return 0; + } + + /** + * Evict throttle entries whose buckets no longer exist in the given cluster (leader unknown, + * partition dropped, table dropped). + * + *

Invoked on every Sender loop with the current cluster snapshot. The identity short-circuit + * makes this an O(1) no-op when metadata hasn't changed, so the actual O(N) walk only runs once + * per real metadata refresh. + */ + void maybeEvictStaleThrottles(Cluster cluster) { + if (cluster == lastClusterRef) { + return; + } + lastClusterRef = cluster; + if (kvThrottleExpiryMs.isEmpty()) { + return; + } + kvThrottleExpiryMs.keySet().removeIf(tb -> cluster.leaderFor(tb) == null); + } + + // ------------------------------------------------------------------------ + // Disk protection (monotonic nanos, never-shorten, exponential backoff). + // ------------------------------------------------------------------------ + + /** + * Installs disk backoff for the bucket before the batch retry count is increased by + * re-enqueueing. Concurrent installs never shorten an existing deadline. + * + * @param tableBucket the bucket that was rejected by disk protection + * @param attempts the batch retry count used to derive the exponential backoff + * @return the effective remaining backoff in milliseconds + */ + long backoffAfterDiskWrite(TableBucket tableBucket, int attempts) { + if (resourcesDestroyed.get()) { + return 0; + } + long now = clock.nanoseconds(); + long delayNanos = + TimeUnit.MILLISECONDS.toNanos(Math.max(1L, diskBackoff.backoff(attempts))); + Long deadline = + diskBackoffDeadlineNanos.compute( + tableBucket, + (bucket, previous) -> + previous != null && previous - now > delayNanos + ? previous + : now + delayNanos); + // A late RPC callback must not retain state after final resource destruction. + if (resourcesDestroyed.get()) { + diskBackoffDeadlineNanos.remove(tableBucket, deadline); + return 0; + } + return nanosToCeilMillis(deadline - now); + } + + /** Returns the remaining disk backoff without changing its deadline. */ + long diskRemainingMs(TableBucket tableBucket) { + Long deadline = diskBackoffDeadlineNanos.get(tableBucket); + if (deadline == null) { + return 0; + } + long remainingNanos = deadline - clock.nanoseconds(); + if (remainingNanos > 0) { + return nanosToCeilMillis(remainingNanos); + } + diskBackoffDeadlineNanos.remove(tableBucket, deadline); + return 0; + } + + /** Reclaims expired entries even when their queues no longer contain any batches. */ + void maybeEvictExpiredDiskBackoffs() { + if (diskBackoffDeadlineNanos.isEmpty()) { + return; + } + long now = clock.nanoseconds(); + if (now - lastDiskSweepNanos < TimeUnit.SECONDS.toNanos(1)) { + return; + } + lastDiskSweepNanos = now; + diskBackoffDeadlineNanos.forEach( + (bucket, deadline) -> { + if (deadline - now <= 0) { + diskBackoffDeadlineNanos.remove(bucket, deadline); + } + }); + } + + /** Drops all disk-backoff state; mirrors the accumulator's resource destruction. */ + void clearDiskBackoffs() { + diskBackoffDeadlineNanos.clear(); + } + + @VisibleForTesting + int diskBackoffCount() { + return diskBackoffDeadlineNanos.size(); + } + + private static long nanosToCeilMillis(long nanos) { + return 1 + (nanos - 1) / TimeUnit.MILLISECONDS.toNanos(1); + } +} diff --git a/fluss-client/src/test/java/org/apache/fluss/client/table/DiskWriteBackoffITCase.java b/fluss-client/src/test/java/org/apache/fluss/client/table/DiskWriteBackoffITCase.java new file mode 100644 index 0000000000..8870878e9b --- /dev/null +++ b/fluss-client/src/test/java/org/apache/fluss/client/table/DiskWriteBackoffITCase.java @@ -0,0 +1,217 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.fluss.client.table; + +import org.apache.fluss.client.Connection; +import org.apache.fluss.client.ConnectionFactory; +import org.apache.fluss.client.FlussConnection; +import org.apache.fluss.client.admin.Admin; +import org.apache.fluss.client.metrics.TestingWriterMetricGroup; +import org.apache.fluss.client.table.scanner.log.LogScanner; +import org.apache.fluss.client.write.WriteFormat; +import org.apache.fluss.client.write.WriteRecord; +import org.apache.fluss.client.write.WriterClient; +import org.apache.fluss.config.ConfigOptions; +import org.apache.fluss.config.Configuration; +import org.apache.fluss.config.MemorySize; +import org.apache.fluss.metadata.DatabaseDescriptor; +import org.apache.fluss.metadata.PhysicalTablePath; +import org.apache.fluss.metadata.TableDescriptor; +import org.apache.fluss.metadata.TableInfo; +import org.apache.fluss.metadata.TablePath; +import org.apache.fluss.row.BinaryRow; +import org.apache.fluss.row.encode.CompactedKeyEncoder; +import org.apache.fluss.server.replica.ReplicaManager; +import org.apache.fluss.server.testutils.FlussClusterExtension; + +import org.junit.jupiter.api.extension.RegisterExtension; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.CsvSource; + +import java.time.Duration; +import java.util.concurrent.CompletableFuture; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.TimeoutException; + +import static org.apache.fluss.record.TestData.DATA1_ROW_TYPE; +import static org.apache.fluss.record.TestData.DATA1_SCHEMA; +import static org.apache.fluss.record.TestData.DATA1_SCHEMA_PK; +import static org.apache.fluss.testutils.DataTestUtils.compactedRow; +import static org.apache.fluss.testutils.DataTestUtils.row; +import static org.apache.fluss.testutils.InternalRowAssert.assertThatRow; +import static org.apache.fluss.testutils.common.CommonTestUtils.retry; +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +/** Real RPC tests for retrying writes rejected by disk protection. */ +class DiskWriteBackoffITCase { + + @RegisterExtension + static final FlussClusterExtension CLUSTER = + FlussClusterExtension.builder() + .setNumOfTabletServers(1) + .setClusterConf(clusterConfig()) + .build(); + + @ParameterizedTest + @CsvSource({"false,false", "true,false", "false,true"}) + void testDiskProtectionBackoffAndRecovery(boolean kv, boolean closeWhileLocked) + throws Exception { + Configuration conf = CLUSTER.getClientConfig(); + conf.set(ConfigOptions.CLIENT_WRITER_BATCH_TIMEOUT, Duration.ZERO); + conf.set(ConfigOptions.CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF, Duration.ofSeconds(1)); + conf.set(ConfigOptions.CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF_MAX, Duration.ofSeconds(1)); + TestingWriterMetricGroup metrics = TestingWriterMetricGroup.newInstance(); + ReplicaManager replicas = CLUSTER.getTabletServers().iterator().next().getReplicaManager(); + TablePath path = + TablePath.of( + "disk_backoff", + closeWhileLocked ? "close_table" : kv ? "kv_table" : "log_table"); + try (Connection connection = ConnectionFactory.createConnection(conf); + Admin admin = connection.getAdmin()) { + admin.createDatabase(path.getDatabaseName(), DatabaseDescriptor.EMPTY, true).get(); + admin.createTable( + path, + TableDescriptor.builder() + .schema(kv ? DATA1_SCHEMA_PK : DATA1_SCHEMA) + .distributedBy(1) + .build(), + false) + .get(); + try (Table table = connection.getTable(path)) { + WriterClient writer = + new WriterClient( + conf, + ((FlussConnection) connection).getMetadataUpdater(), + metrics, + admin); + try { + // Establish a writable leader before simulating disk protection. + send(writer, table.getTableInfo(), kv, 1).get(30, TimeUnit.SECONDS); + long retriesBeforeRejection = metrics.recordsRetryTotal().getCount(); + replicas.getDiskUsageMonitor().updateWriteLimitConfig(0.85, 0.80); + replicas.getDiskUsageMonitor().update(0.99); + assertThat(replicas.isDiskWriteLocked()).isTrue(); + CompletableFuture rejected = send(writer, table.getTableInfo(), kv, 2); + retry( + Duration.ofSeconds(10), + () -> + assertThat(metrics.recordsRetryTotal().getCount()) + .isEqualTo(retriesBeforeRejection + 1)); + long bytesAfterRejection = metrics.bytesSendTotal().getCount(); + // The callback remains pending, and the payload is not sent again in the + // window. + assertThatThrownBy(() -> rejected.get(200, TimeUnit.MILLISECONDS)) + .isInstanceOf(TimeoutException.class); + assertThat(metrics.recordsRetryTotal().getCount()) + .isEqualTo(retriesBeforeRejection + 1); + assertThat(metrics.bytesSendTotal().getCount()).isEqualTo(bytesAfterRejection); + if (closeWhileLocked) { + long start = System.nanoTime(); + writer.close(Duration.ofMillis(50)); + assertThat(Duration.ofNanos(System.nanoTime() - start)) + .isLessThan(Duration.ofSeconds(2)); + assertThat(metrics.recordsRetryTotal().getCount()) + .isEqualTo(retriesBeforeRejection + 1); + return; + } + replicas.getDiskUsageMonitor().update(0.5); + assertThat(replicas.isDiskWriteLocked()).isFalse(); + rejected.get(10, TimeUnit.SECONDS); + writer.flush(); + if (kv) { + assertThatRow( + table.newLookup() + .createLookuper() + .lookup(row(2)) + .get() + .getSingletonRow()) + .withSchema(DATA1_ROW_TYPE) + .isEqualTo(row(2, "payload")); + } else { + try (LogScanner scanner = table.newScan().createLogScanner()) { + scanner.subscribeFromBeginning(0); + CompletableFuture read = new CompletableFuture<>(); + long deadline = System.nanoTime() + TimeUnit.SECONDS.toNanos(10); + while (!read.isDone() && System.nanoTime() < deadline) { + scanner.poll(Duration.ofMillis(100)) + .forEach( + record -> { + if (record.getRow().getInt(0) == 2) { + assertThatRow(record.getRow()) + .withSchema(DATA1_ROW_TYPE) + .isEqualTo(row(2, "payload")); + read.complete(null); + } + }); + } + assertThat(read).isCompleted(); + } + } + } finally { + replicas.getDiskUsageMonitor().update(0.5); + writer.close(Duration.ofSeconds(5)); + } + } + } + } + + private static CompletableFuture send( + WriterClient writer, TableInfo info, boolean kv, int key) { + PhysicalTablePath path = PhysicalTablePath.of(info.getTablePath()); + WriteRecord record; + if (kv) { + BinaryRow value = compactedRow(DATA1_ROW_TYPE, new Object[] {key, "payload"}); + byte[] encodedKey = + new CompactedKeyEncoder(DATA1_ROW_TYPE, DATA1_SCHEMA_PK.getPrimaryKeyIndexes()) + .encodeKey(value); + record = + WriteRecord.forUpsert( + info, + path, + value, + encodedKey, + encodedKey, + WriteFormat.COMPACTED_KV, + null); + } else { + record = WriteRecord.forArrowAppend(info, path, row(key, "payload"), null); + } + CompletableFuture result = new CompletableFuture<>(); + writer.send( + record, + (bucket, offset, error) -> { + if (error == null) { + result.complete(null); + } else { + result.completeExceptionally(error); + } + }); + return result; + } + + private static Configuration clusterConfig() { + Configuration conf = new Configuration(); + conf.set(ConfigOptions.DEFAULT_REPLICATION_FACTOR, 1); + // Keep the real periodic sampler from overwriting the simulated usage during each test. + conf.set(ConfigOptions.SERVER_DATA_DISK_CHECK_INTERVAL, Duration.ofHours(1)); + conf.set(ConfigOptions.CLIENT_WRITER_BUFFER_MEMORY_SIZE, MemorySize.parse("1mb")); + conf.set(ConfigOptions.CLIENT_WRITER_BATCH_SIZE, MemorySize.parse("1kb")); + return conf; + } +} diff --git a/fluss-client/src/test/java/org/apache/fluss/client/write/RecordAccumulatorTest.java b/fluss-client/src/test/java/org/apache/fluss/client/write/RecordAccumulatorTest.java index 4ea64f4efe..7925346dfe 100644 --- a/fluss-client/src/test/java/org/apache/fluss/client/write/RecordAccumulatorTest.java +++ b/fluss-client/src/test/java/org/apache/fluss/client/write/RecordAccumulatorTest.java @@ -49,6 +49,7 @@ import org.apache.fluss.rpc.gateway.TabletServerGateway; import org.apache.fluss.rpc.metrics.TestingClientMetricGroup; import org.apache.fluss.utils.CloseableIterator; +import org.apache.fluss.utils.ExponentialBackoff; import org.apache.fluss.utils.clock.ManualClock; import org.apache.commons.lang3.RandomStringUtils; @@ -947,4 +948,219 @@ void testThrottledBucketSkippedInDrain() throws Exception { .extracting(ReadyWriteBatch::tableBucket) .containsExactly(tb2); } + + @Test + void testDiskBackoffIncreasesAndExpiresAtDeadline() throws Exception { + RecordAccumulator accum = createDiskAccumulator(new ExponentialBackoff(1000, 2, 10000, 0)); + ReadyWriteBatch batch = appendAndDrain(accum, 0); + for (long delay : new long[] {1000, 2000, 4000, 8000, 10000, 10000}) { + assertThat(accum.backoffAfterDiskWriteLocked(batch)).isEqualTo(delay); + accum.reEnqueue(batch); + assertThat(accum.ready(cluster).readyNodes).isEmpty(); + assertThat(accum.ready(cluster).nextReadyCheckDelayMs).isEqualTo(delay); + // Repeated readiness checks must neither resample nor extend the deadline. + assertThat(accum.ready(cluster).nextReadyCheckDelayMs).isEqualTo(delay); + clock.advanceTime(Duration.ofMillis(delay - 1)); + assertThat(accum.ready(cluster).nextReadyCheckDelayMs).isOne(); + assertThat(accum.drain(cluster, Collections.singleton(node1.id()), Integer.MAX_VALUE)) + .isEmpty(); + clock.advanceTime(Duration.ofMillis(1)); + assertThat(accum.ready(cluster).readyNodes).containsExactly(node1.id()); + assertThat( + accum.drain( + cluster, + Collections.singleton(node1.id()), + Integer.MAX_VALUE) + .get(node1.id())) + .extracting(ReadyWriteBatch::writeBatch) + .containsExactly(batch.writeBatch()); + } + accum.abortAllBatches(new RuntimeException("test cleanup")); + accum.destroyResources(); + } + + @Test + void testDiskBackoffDoesNotBlockHealthyBucketsOrUnknownMetadata() throws Exception { + RecordAccumulator accum = createDiskAccumulator(new ExponentialBackoff(1000, 2, 10000, 0)); + ReadyWriteBatch blocked = appendAndDrain(accum, 0); + accum.backoffAfterDiskWriteLocked(blocked); + accum.reEnqueue(blocked); + appendDiskTestRecord(accum, 1); + appendDiskTestRecord(accum, 2); + assertThat(accum.ready(cluster).readyNodes) + .containsExactlyInAnyOrder(node1.id(), node2.id()); + Map> drained = + accum.drain( + cluster, + new HashSet<>(Arrays.asList(node1.id(), node2.id())), + Integer.MAX_VALUE); + assertThat(drained.get(node1.id())) + .extracting(ReadyWriteBatch::tableBucket) + .containsExactly(tb2); + assertThat(drained.get(node2.id())) + .extracting(ReadyWriteBatch::tableBucket) + .containsExactly(tb3); + // Even a caller supplying a ready node cannot bypass the disk gate. + assertThat(accum.drain(cluster, Collections.singleton(node1.id()), Integer.MAX_VALUE)) + .isEmpty(); + Cluster unknownLeaderCluster = updateCluster(Collections.singletonList(bucket2)); + assertThat(accum.ready(unknownLeaderCluster).unknownLeaderTables) + .contains(DATA1_PHYSICAL_TABLE_PATH); + assertThat(accum.ready(unknownLeaderCluster).nextReadyCheckDelayMs).isZero(); + accum.abortAllBatches(new RuntimeException("test cleanup")); + accum.destroyResources(); + } + + @Test + void testDiskAndKvBackoffAreIndependentDuringFlushAndClose() throws Exception { + RecordAccumulator accum = createDiskAccumulator(new ExponentialBackoff(1000, 2, 10000, 0)); + ReadyWriteBatch batch = appendAndDrain(accum, 0); + accum.backoffAfterDiskWriteLocked(batch); + accum.reEnqueue(batch); + accum.updateThrottle(tb1, 1.0f); // KV default is three seconds. + assertThat(accum.ready(cluster).nextReadyCheckDelayMs).isEqualTo(3000); + accum.updateThrottle(tb1, 0.1f); + assertThat(accum.ready(cluster).nextReadyCheckDelayMs).isEqualTo(1000); + accum.updateThrottle(tb1, 0f); + accum.beginFlush(); + accum.close(); + assertThat(accum.ready(cluster).readyNodes).isEmpty(); + assertThat(accum.ready(cluster).nextReadyCheckDelayMs).isEqualTo(1000); + assertThat(accum.drain(cluster, Collections.singleton(node1.id()), Integer.MAX_VALUE)) + .isEmpty(); + clock.advanceTime(Duration.ofSeconds(1)); + assertThat(accum.ready(cluster).readyNodes).containsExactly(node1.id()); + accum.abortAllBatches(new RuntimeException("test cleanup")); + accum.destroyResources(); + assertThat(accum.diskWriteBackoffCount()).isZero(); + assertThat(accum.backoffAfterDiskWriteLocked(batch)).isZero(); + } + + @Test + void testConcurrentDiskBackoffsNeverShortenDeadlineAndSweepEmptyQueues() throws Exception { + RecordAccumulator accum = createDiskAccumulator(new ExponentialBackoff(1000, 2, 10000, 0)); + ReadyWriteBatch shortBatch = appendAndDrain(accum, 0); + ReadyWriteBatch longBatch = appendAndDrain(accum, 0); + for (int i = 0; i < 4; i++) { + longBatch.writeBatch().reEnqueued(); + } + for (int i = 0; i < 50; i++) { + CompletableFuture.allOf( + CompletableFuture.runAsync( + () -> accum.backoffAfterDiskWriteLocked(shortBatch)), + CompletableFuture.runAsync( + () -> accum.backoffAfterDiskWriteLocked(longBatch)), + CompletableFuture.runAsync(accum::maybeEvictExpiredDiskWriteBackoffs)) + .get(); + assertThat(accum.diskWriteBackoffRemainingMs(tb1)).isEqualTo(10000); + } + clock.advanceTime(Duration.ofSeconds(10)); + // Race expiry cleanup with a new rejection. Conditional removal must preserve the new gate. + CompletableFuture.allOf( + CompletableFuture.runAsync(accum::maybeEvictExpiredDiskWriteBackoffs), + CompletableFuture.runAsync(() -> accum.diskWriteBackoffRemainingMs(tb1)), + CompletableFuture.runAsync( + () -> accum.backoffAfterDiskWriteLocked(longBatch))) + .get(); + assertThat(accum.diskWriteBackoffRemainingMs(tb1)).isEqualTo(10000); + assertThat(accum.hasUnDrained()).isFalse(); + clock.advanceTime(Duration.ofSeconds(10)); + accum.maybeEvictExpiredDiskWriteBackoffs(); + assertThat(accum.diskWriteBackoffCount()).isZero(); + accum.abortAllBatches(new RuntimeException("test cleanup")); + accum.destroyResources(); + } + + @Test + void testDiskBackoffHandlesNanosecondWrapAndRoundsUp() throws Exception { + clock.advanceTime(Long.MAX_VALUE - clock.nanoseconds() - 500000, TimeUnit.NANOSECONDS); + RecordAccumulator accum = createDiskAccumulator(new ExponentialBackoff(1, 2, 1, 0)); + ReadyWriteBatch batch = appendAndDrain(accum, 0); + assertThat(accum.backoffAfterDiskWriteLocked(batch)).isOne(); + clock.advanceTime(999999, TimeUnit.NANOSECONDS); + assertThat(accum.diskWriteBackoffRemainingMs(tb1)).isOne(); + clock.advanceTime(1, TimeUnit.NANOSECONDS); + assertThat(accum.diskWriteBackoffRemainingMs(tb1)).isZero(); + accum.abortAllBatches(new RuntimeException("test cleanup")); + accum.destroyResources(); + } + + @Test + void testProductionDiskBackoffJitterAndMinimumDelay() throws Exception { + RecordAccumulator accum = createDiskAccumulator(null); + ReadyWriteBatch batch = appendAndDrain(accum, 0); + for (int attempt = 0; attempt < 8; attempt++) { + long baseDelay = Math.min(1000L << attempt, 10000L); + long delay = accum.backoffAfterDiskWriteLocked(batch); + assertThat(delay) + .isBetween( + (long) (baseDelay * 0.8), Math.min((long) (baseDelay * 1.2), 10000L)); + clock.advanceTime(Duration.ofMillis(delay)); + batch.writeBatch().reEnqueued(); + } + accum.abortAllBatches(new RuntimeException("test cleanup")); + accum.destroyResources(); + + accum = createDiskAccumulator(new ExponentialBackoff(0, 2, 0, 0)); + batch = appendAndDrain(accum, 0); + assertThat(accum.backoffAfterDiskWriteLocked(batch)).isOne(); + accum.abortAllBatches(new RuntimeException("test cleanup")); + accum.destroyResources(); + } + + @Test + void testInvalidDiskBackoffConfiguration() { + Duration[][] invalid = { + {Duration.ZERO, Duration.ofSeconds(10)}, + {Duration.ofNanos(999999), Duration.ofSeconds(10)}, + {Duration.ofMillis(-1), Duration.ofSeconds(10)}, + {Duration.ofSeconds(11), Duration.ofSeconds(10)}, + {Duration.ofMillis(1), Duration.ofMillis((long) Integer.MAX_VALUE + 1)}, + {Duration.ofMillis(1), Duration.ofSeconds(Long.MAX_VALUE)} + }; + for (Duration[] values : invalid) { + conf.set(ConfigOptions.CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF, values[0]); + conf.set(ConfigOptions.CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF_MAX, values[1]); + assertThatThrownBy(() -> createDiskAccumulator(null)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("client.writer.disk-write-locked.backoff"); + } + } + + private RecordAccumulator createDiskAccumulator(ExponentialBackoff backoff) { + conf.set(ConfigOptions.CLIENT_WRITER_BATCH_TIMEOUT, Duration.ZERO); + conf.set(ConfigOptions.CLIENT_WRITER_BUFFER_MEMORY_SIZE, new MemorySize(1024 * 1024)); + conf.set(ConfigOptions.CLIENT_WRITER_BATCH_SIZE, new MemorySize(1024)); + conf.set(ConfigOptions.CLIENT_WRITER_BUFFER_PAGE_SIZE, new MemorySize(256)); + IdempotenceManager manager = new IdempotenceManager(false, 5, null, null); + return backoff == null + ? new RecordAccumulator( + conf, + manager, + TestingWriterMetricGroup.newInstance(), + clock, + (tableInfo, path) -> bucketAssigner) + : new RecordAccumulator( + conf, + manager, + TestingWriterMetricGroup.newInstance(), + clock, + backoff, + (tableInfo, path) -> bucketAssigner); + } + + private void appendDiskTestRecord(RecordAccumulator accum, int bucket) throws Exception { + bucketAssigner.setBucketId(bucket); + accum.append( + createRecord(indexedRow(DATA1_ROW_TYPE, new Object[] {1, "a"})), + (tb, offset, error) -> {}, + cluster); + } + + private ReadyWriteBatch appendAndDrain(RecordAccumulator accum, int bucket) throws Exception { + appendDiskTestRecord(accum, bucket); + return accum.drain(cluster, Collections.singleton(node1.id()), Integer.MAX_VALUE) + .get(node1.id()) + .get(0); + } } diff --git a/fluss-client/src/test/java/org/apache/fluss/client/write/SenderTest.java b/fluss-client/src/test/java/org/apache/fluss/client/write/SenderTest.java index 2270d611df..2ee1815208 100644 --- a/fluss-client/src/test/java/org/apache/fluss/client/write/SenderTest.java +++ b/fluss-client/src/test/java/org/apache/fluss/client/write/SenderTest.java @@ -27,6 +27,7 @@ import org.apache.fluss.config.Configuration; import org.apache.fluss.config.MemorySize; import org.apache.fluss.exception.AuthorizationException; +import org.apache.fluss.exception.DiskWriteLockedException; import org.apache.fluss.exception.FlussRuntimeException; import org.apache.fluss.exception.InvalidBucketRoutingException; import org.apache.fluss.exception.NetworkException; @@ -35,6 +36,7 @@ import org.apache.fluss.exception.TableNotExistException; import org.apache.fluss.exception.TimeoutException; import org.apache.fluss.exception.TooManyPartitionsException; +import org.apache.fluss.exception.UnknownWriterIdException; import org.apache.fluss.metadata.DataLakeFormat; import org.apache.fluss.metadata.PhysicalTablePath; import org.apache.fluss.metadata.Schema; @@ -59,6 +61,9 @@ import org.apache.fluss.server.entity.PutKvDataForBucket; import org.apache.fluss.server.tablet.TestTabletServerGateway; import org.apache.fluss.types.DataTypes; +import org.apache.fluss.utils.ExponentialBackoff; +import org.apache.fluss.utils.clock.Clock; +import org.apache.fluss.utils.clock.ManualClock; import org.apache.fluss.utils.clock.SystemClock; import org.junit.jupiter.api.AfterEach; @@ -81,6 +86,7 @@ import java.util.Optional; import java.util.Set; import java.util.concurrent.CompletableFuture; +import java.util.concurrent.CompletionException; import java.util.concurrent.ExecutorService; import java.util.concurrent.Executors; import java.util.concurrent.Future; @@ -129,6 +135,7 @@ final class SenderTest { private TestingBucketAssigner bucketAssigner; private Sender sender = null; private TestingWriterMetricGroup writerMetricGroup; + private Clock clock = SystemClock.getInstance(); // TODO add more tests as kafka SenderTest. @@ -334,8 +341,12 @@ void testAbortsOnlyMissingPartitionWhenRerouteIsUnsafe() throws Exception { assertThat(activeFuture.get()).isNull(); } - @Test - void testNormalAndHistoricalPutRequests() throws Exception { + @ParameterizedTest + @CsvSource({"false,false", "true,false", "true,true"}) + void testNormalAndHistoricalPutRequests(boolean diskRejected, boolean rpcFailure) + throws Exception { + ManualClock manualClock = new ManualClock(); + clock = manualClock; sender.destroyResources(); TableInfo tableInfo = createHistoricalTableInfo(); PhysicalTablePath activePath = PhysicalTablePath.of(tableInfo.getTablePath(), "20990101"); @@ -394,6 +405,41 @@ void testNormalAndHistoricalPutRequests() throws Exception { secondOriginalPath.getPartitionName()); gateway.response(0, createPutKvResponse(activeBucket, 1L)); + if (diskRejected) { + if (rpcFailure) { + gateway.failRequest(0, new DiskWriteLockedException("disk full")); + } else { + gateway.response( + 0, + makePutKvResponse( + Arrays.asList( + PutKvResultForBucket.historicalFailure( + historicalBucket, + Errors.DISK_WRITE_LOCKED.toApiError(), + firstOriginalPath.getPartitionName()), + PutKvResultForBucket.historicalFailure( + historicalBucket, + Errors.DISK_WRITE_LOCKED.toApiError(), + secondOriginalPath.getPartitionName())))); + } + assertThat(accumulator.diskWriteBackoffRemainingMs(historicalBucket)).isEqualTo(1000); + assertThat(accumulator.diskWriteBackoffRemainingMs(activeBucket)).isZero(); + sender.wakeup(); + sender.runOnce(); + assertThat(gateway.pendingRequestSize()).isZero(); + manualClock.advanceTime(Duration.ofSeconds(1)); + sender.runOnce(); + assertThat(gateway.pendingRequestSize()).isOne(); + PutKvRequest retried = (PutKvRequest) gateway.getRequest(0); + assertThat(retried.getBucketsReqsCount()).isEqualTo(2); + Set retriedOriginals = new HashSet<>(); + for (int i = 0; i < retried.getBucketsReqsCount(); i++) { + retriedOriginals.add(retried.getBucketsReqAt(i).getOriginalPartitionName()); + assertThat(retried.getBucketsReqAt(i).getPartitionId()) + .isEqualTo(historicalBucket.getPartitionId()); + } + assertThat(retriedOriginals).isEqualTo(originalPartitionNames); + } gateway.response( 0, makePutKvResponse( @@ -1626,8 +1672,13 @@ void testPutKvStorageExceptionResponseRetriesInsteadOfFailing() throws Exception assertThat(future.get()).isNull(); } - @Test - void testPutKvRpcFailuresRetryOnlyOwnedTableBatches() throws Exception { + @ParameterizedTest + @ValueSource(booleans = {false, true}) + void testPutKvRpcFailuresRetryOnlyOwnedTableBatches(boolean diskRejected) throws Exception { + ManualClock manualClock = new ManualClock(); + clock = manualClock; + sender.destroyResources(); + sender = setupWithIdempotenceState(); TablePath secondTablePath = TablePath.of("test_db_2", "test_pk_table_2"); TableInfo secondTableInfo = TableInfo.of( @@ -1662,7 +1713,12 @@ void testPutKvRpcFailuresRetryOnlyOwnedTableBatches() throws Exception { failRequest( tb1, findRequestIndex(tb1, DATA1_TABLE_ID_PK), - new NetworkException("first table request failed")); + diskRejected + ? new DiskWriteLockedException("disk full") + : new NetworkException("first table request failed")); + assertThat(accumulator.diskWriteBackoffRemainingMs(firstTableBucket)) + .isEqualTo(diskRejected ? 1000L : 0L); + assertThat(accumulator.diskWriteBackoffRemainingMs(secondTableBucket)).isZero(); assertThat(writerMetricGroup.recordsRetryTotal().getCount()).isEqualTo(1L); assertThat(sender.numOfInFlightBatches(firstTableBucket)).isEqualTo(0); assertThat(sender.numOfInFlightBatches(secondTableBucket)).isEqualTo(1); @@ -1677,6 +1733,11 @@ void testPutKvRpcFailuresRetryOnlyOwnedTableBatches() throws Exception { assertThat(sender.numOfInFlightBatches(secondTableBucket)).isEqualTo(0); metadataUpdater.updateCluster(clusterBeforeFailures); + if (diskRejected) { + sender.runOnce(); + assertThat(pendingWriteRequestTableIds(tb1)).containsExactly(DATA2_TABLE_ID); + manualClock.advanceTime(Duration.ofSeconds(1)); + } sender.runOnce(); assertThat(pendingWriteRequestTableIds(tb1)) .containsExactlyInAnyOrder(DATA1_TABLE_ID_PK, DATA2_TABLE_ID); @@ -1693,10 +1754,16 @@ void testPutKvRpcFailuresRetryOnlyOwnedTableBatches() throws Exception { assertThat(secondFuture.get()).isNull(); } - @Test - void testSendWhenTableIdChanges() throws Exception { + @ParameterizedTest + @ValueSource(booleans = {false, true}) + void testSendWhenTableIdChanges(boolean diskRejected) throws Exception { CompletableFuture future1 = new CompletableFuture<>(); appendToAccumulator(tb1, row(1, "a"), (tb, leo, e) -> future1.complete(e)); + if (diskRejected) { + sender.runOnce(); + failRequest(tb1, 0, new DiskWriteLockedException("disk full")); + assertThat(accumulator.diskWriteBackoffRemainingMs(tb1)).isPositive(); + } TableInfo newTableInfo = TableInfo.of( DATA1_TABLE_PATH, @@ -1707,6 +1774,7 @@ void testSendWhenTableIdChanges() throws Exception { System.currentTimeMillis(), System.currentTimeMillis()); TableBucket newTableBucket = new TableBucket(newTableInfo.getTableId(), tb1.getBucket()); + assertThat(accumulator.diskWriteBackoffRemainingMs(newTableBucket)).isZero(); metadataUpdater.updateTableInfos(Collections.singletonMap(DATA1_TABLE_PATH, newTableInfo)); sender.runOnce(); @@ -1800,14 +1868,32 @@ private static TableInfo createHistoricalTableInfo() { return createPartitionedTableInfo(AutoPartitionTimeUnit.DAY, 7, true); } + private static TableInfo createHistoricalTableInfo( + AutoPartitionTimeUnit timeUnit, int numToRetain) { + return createHistoricalTableInfo(timeUnit, numToRetain, true); + } + + private static TableInfo createHistoricalTableInfo( + AutoPartitionTimeUnit timeUnit, int numToRetain, boolean primaryKey) { + return createPartitionedTableInfo(timeUnit, numToRetain, true, primaryKey); + } + private static TableInfo createPartitionedTableInfo( AutoPartitionTimeUnit timeUnit, int numToRetain, boolean historicalPartitionEnabled) { - Schema schema = - Schema.newBuilder() - .column("id", DataTypes.INT()) - .column("dt", DataTypes.STRING()) - .primaryKey("id", "dt") - .build(); + return createPartitionedTableInfo(timeUnit, numToRetain, historicalPartitionEnabled, true); + } + + private static TableInfo createPartitionedTableInfo( + AutoPartitionTimeUnit timeUnit, + int numToRetain, + boolean historicalPartitionEnabled, + boolean primaryKey) { + Schema.Builder schemaBuilder = + Schema.newBuilder().column("id", DataTypes.INT()).column("dt", DataTypes.STRING()); + if (primaryKey) { + schemaBuilder.primaryKey("id", "dt"); + } + Schema schema = schemaBuilder.build(); TableDescriptor descriptor = TableDescriptor.builder() .schema(schema) @@ -2071,6 +2157,268 @@ private PutKvResponse createPutKvResponse(TableBucket tb, Errors error) { Collections.singletonList(new PutKvResultForBucket(tb, error.toApiError()))); } + @ParameterizedTest + @CsvSource({ + "false,false,false", + "false,false,true", + "false,true,false", + "false,true,true", + "true,false,false", + "true,false,true", + "true,true,false", + "true,true,true" + }) + void testDiskWriteLockedRetries(boolean kv, boolean rpcFailure, boolean idempotent) + throws Exception { + sender.destroyResources(); + ManualClock manualClock = new ManualClock(); + clock = manualClock; + IdempotenceManager manager = createIdempotenceManager(idempotent); + manager.setWriterId(42L); + sender = setupWithIdempotenceState(manager, 6, 0); + TableBucket bucket = kv ? new TableBucket(DATA1_TABLE_ID_PK, 0) : tb1; + CompletableFuture result = appendDiskTestRecord(kv, bucket, 1); + sender.runOnce(); + byte[] originalPayload = diskTestPayload(kv, bucket, getRequest(tb1, 0)); + for (long delay : new long[] {1000, 2000, 4000, 8000, 10000, 10000}) { + ApiMessage request = getRequest(tb1, 0); + assertThat(diskTestPayload(kv, bucket, request)).isEqualTo(originalPayload); + if (!kv && idempotent) { + assertThat( + getProduceLogRecords((ProduceLogRequest) request, bucket) + .batchIterator() + .next() + .writerId()) + .isEqualTo(42L); + assertBatchSequenceEquals(bucket, (ProduceLogRequest) request, 0); + } + if (rpcFailure) { + failRequest( + tb1, 0, new CompletionException(new DiskWriteLockedException("disk full"))); + } else { + finishRequest( + tb1, + 0, + kv + ? createPutKvResponse(bucket, Errors.DISK_WRITE_LOCKED) + : createProduceLogResponse(bucket, Errors.DISK_WRITE_LOCKED)); + } + assertThat(result).isNotDone(); + assertThat(accumulator.ready(metadataUpdater.getCluster()).nextReadyCheckDelayMs) + .isEqualTo(delay); + sender.wakeup(); + sender.runOnce(); + assertThat(pendingRequestSize(tb1)).isZero(); + manualClock.advanceTime(Duration.ofMillis(delay - 1)); + sender.wakeup(); + sender.runOnce(); + assertThat(pendingRequestSize(tb1)).isZero(); + manualClock.advanceTime(Duration.ofMillis(1)); + sender.runOnce(); + assertThat(pendingRequestSize(tb1)).isOne(); + } + finishRequest( + tb1, + 0, + kv ? createPutKvResponse(bucket, 1L) : createProduceLogResponse(bucket, 0L, 1L)); + assertThat(result.get()).isNull(); + assertThat(writerMetricGroup.recordsRetryTotal().getCount()).isEqualTo(6L); + } + + @ParameterizedTest + @ValueSource(booleans = {false, true}) + void testDiskWriteLockedFailsWhenRetryIsNotAllowed(boolean writerIdChanged) throws Exception { + sender.destroyResources(); + clock = new ManualClock(); + IdempotenceManager manager = createIdempotenceManager(writerIdChanged); + manager.setWriterId(42L); + sender = setupWithIdempotenceState(manager, writerIdChanged ? 5 : 0, 0); + CompletableFuture result = appendDiskTestRecord(false, tb1, 1); + sender.runOnce(); + if (writerIdChanged) { + manager.setWriterId(43L); + } + failRequest(tb1, 0, new DiskWriteLockedException("disk full")); + assertThat(result.get()) + .isInstanceOf( + writerIdChanged + ? UnknownWriterIdException.class + : DiskWriteLockedException.class); + assertThat(accumulator.diskWriteBackoffCount()).isZero(); + assertThat(writerMetricGroup.recordsRetryTotal().getCount()).isZero(); + assertThat(accumulator.hasUnDrained()).isFalse(); + } + + @ParameterizedTest + @ValueSource(booleans = {false, true}) + void testLateKvSuccessDoesNotClearDiskBackoff(boolean idempotent) throws Exception { + sender.destroyResources(); + ManualClock manualClock = new ManualClock(); + clock = manualClock; + IdempotenceManager manager = createIdempotenceManager(idempotent); + manager.setWriterId(42L); + sender = setupWithIdempotenceState(manager); + TableBucket bucket = new TableBucket(DATA1_TABLE_ID_PK, 0); + CompletableFuture first = appendDiskTestRecord(true, bucket, 1); + sender.runOnce(); + CompletableFuture second = appendDiskTestRecord(true, bucket, 2); + sender.runOnce(); + finishRequest(tb1, 1, createPutKvResponse(bucket, Errors.DISK_WRITE_LOCKED)); + finishRequest(tb1, 0, createPutKvResponse(bucket, 1L, 0f)); + assertThat(first.get()).isNull(); + assertThat(second).isNotDone(); + assertThat(accumulator.diskWriteBackoffRemainingMs(bucket)).isEqualTo(1000); + sender.wakeup(); + sender.runOnce(); + assertThat(pendingRequestSize(tb1)).isZero(); + manualClock.advanceTime(Duration.ofSeconds(1)); + sender.runOnce(); + finishRequest(tb1, 0, createPutKvResponse(bucket, 2L)); + assertThat(second.get()).isNull(); + } + + @ParameterizedTest + @ValueSource(strings = {"append", "retry", "close"}) + void testDiskBackoffWaitCanBeWokenByHealthyWorkOrClose(String cause) throws Exception { + boolean close = cause.equals("close"); + boolean retryResponse = cause.equals("retry"); + sender.destroyResources(); + clock = new ManualClock(); + sender = setupWithIdempotenceState(); + CompletableFuture blocked = appendDiskTestRecord(false, tb1, 1); + sender.runOnce(); + failRequest(tb1, 0, new DiskWriteLockedException("disk full")); + TableBucket healthy = new TableBucket(DATA1_TABLE_ID, 1); + if (retryResponse) { + appendDiskTestRecord(false, healthy, 2); + sender.runOnce(); + } + // Consume the retry wakeup, then enter a real wait in a controlled sender thread. + sender.runOnce(); + CompletableFuture finished = new CompletableFuture<>(); + Thread thread = + new Thread( + () -> { + try { + sender.runOnce(); + finished.complete(null); + } catch (Throwable t) { + finished.completeExceptionally(t); + } + }, + "disk-backoff-wakeup-test"); + thread.start(); + try { + retry( + Duration.ofSeconds(10), + () -> assertThat(thread.getState()).isEqualTo(Thread.State.TIMED_WAITING)); + if (close) { + sender.forceClose(); + } else if (retryResponse) { + failRequest(healthy, 0, new TimeoutException("retry healthy bucket")); + } else { + appendDiskTestRecord(false, healthy, 2); + sender.wakeup(); // WriterClient wakes the sender after appending a new batch. + } + finished.get(10, TimeUnit.SECONDS); + assertThat(blocked).isNotDone(); + assertThat(pendingRequestSize(tb1)).isZero(); + if (!close) { + sender.runOnce(); + assertThat(pendingRequestSize(healthy)).isOne(); + finishRequest(healthy, 0, createProduceLogResponse(healthy, 0L, 1L)); + } + } finally { + sender.wakeup(); + thread.join(10000); + accumulator.abortAllBatches(new RuntimeException("test cleanup")); + } + } + + @ParameterizedTest + @ValueSource(booleans = {false, true}) + void testHistoricalLogDiskRejection(boolean rpcFailure) throws Exception { + sender.destroyResources(); + ManualClock manualClock = new ManualClock(); + clock = manualClock; + TableInfo info = createHistoricalTableInfo(AutoPartitionTimeUnit.DAY, 7, false); + PhysicalTablePath original = PhysicalTablePath.of(info.getTablePath(), "20000101"); + PhysicalTablePath historical = + PhysicalTablePath.of(info.getTablePath(), HISTORICAL_PARTITION_VALUE); + TableBucket bucket = new TableBucket(info.getTableId(), 22L, 0); + metadataUpdater = + new TestingMetadataUpdater(Collections.singletonMap(info.getTablePath(), info)); + metadataUpdater.updateCluster( + partitionedCluster(info, Collections.singletonMap(historical, bucket))); + sender = setupWithIdempotenceState(); + accumulator.checkAndCacheHistoricalPartitionEnabled(info); + accumulator.routeWritesTo(info, original, historical, bucket.getPartitionId()); + CompletableFuture result = new CompletableFuture<>(); + bucketAssigner.setBucketId(bucket.getBucket()); + accumulator.append( + WriteRecord.forArrowAppend(info, original, row(1, "20000101"), null), + (tb, offset, error) -> result.complete(error), + metadataUpdater.getCluster()); + sender.runOnce(); + TestTabletServerGateway gateway = node1Gateway(); + if (rpcFailure) { + gateway.failRequest(0, new DiskWriteLockedException("disk full")); + } else { + gateway.response( + 0, + makeProduceLogResponse( + Collections.singletonList( + ProduceLogResultForBucket.historicalFailure( + bucket, + Errors.DISK_WRITE_LOCKED.toApiError(), + original.getPartitionName())))); + } + sender.wakeup(); + sender.runOnce(); + assertThat(gateway.pendingRequestSize()).isZero(); + manualClock.advanceTime(Duration.ofSeconds(1)); + sender.runOnce(); + ProduceLogRequest request = (ProduceLogRequest) gateway.getRequest(0); + assertThat(request.getBucketsReqAt(0).getPartitionId()).isEqualTo(bucket.getPartitionId()); + assertThat(request.getBucketsReqAt(0).getOriginalPartitionName()) + .isEqualTo(original.getPartitionName()); + gateway.response( + 0, + makeProduceLogResponse( + Collections.singletonList( + ProduceLogResultForBucket.historicalSuccess( + bucket, 0L, 1L, original.getPartitionName())))); + assertThat(result.get()).isNull(); + } + + private CompletableFuture appendDiskTestRecord( + boolean kv, TableBucket bucket, int key) throws Exception { + CompletableFuture result = new CompletableFuture<>(); + if (kv) { + appendKvToAccumulator( + bucket, + compactedRow(DATA1_ROW_TYPE, new Object[] {key, "a"}), + (tb, offset, error) -> result.complete(error)); + } else { + appendToAccumulator( + bucket, row(key, "a"), (tb, offset, error) -> result.complete(error)); + } + return result; + } + + private byte[] diskTestPayload(boolean kv, TableBucket bucket, ApiMessage request) { + java.nio.ByteBuffer buffer = + (kv + ? ((PutKvRequest) request).getBucketsReqAt(0).getRecordsSlice() + : ((ProduceLogRequest) request) + .getBucketsReqAt(0) + .getRecordsSlice()) + .nioBuffer(); + byte[] bytes = new byte[buffer.remaining()]; + buffer.duplicate().get(bytes); + return bytes; + } + private Sender setupWithIdempotenceState() { return setupWithIdempotenceState(createIdempotenceManager(false)); } @@ -2092,7 +2440,8 @@ private Sender setupWithIdempotenceState( conf, idempotenceManager, writerMetricGroup, - SystemClock.getInstance(), + clock, + new ExponentialBackoff(1000, 2, 10000, 0), (tableInfo, path) -> bucketAssigner); return new Sender( accumulator, diff --git a/fluss-common/src/main/java/org/apache/fluss/config/ConfigOptions.java b/fluss-common/src/main/java/org/apache/fluss/config/ConfigOptions.java index d7f0afd796..b06d0efdc3 100644 --- a/fluss-common/src/main/java/org/apache/fluss/config/ConfigOptions.java +++ b/fluss-common/src/main/java/org/apache/fluss/config/ConfigOptions.java @@ -1407,6 +1407,28 @@ public class ConfigOptions { "Setting a value greater than zero will cause the client to resend any record whose " + "send fails with a potentially transient error."); + public static final ConfigOption CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF = + key("client.writer.disk-write-locked.backoff") + .durationType() + .defaultValue(Duration.ofSeconds(1)) + .withDescription( + "The initial retry backoff when disk write protection rejects a write. " + + "Applies to both Log and KV writes, independently of KV backpressure. " + + "The backoff doubles with the batch retry count and uses 20% jitter, " + + "up to client.writer.disk-write-locked.backoff-max. " + + "The value must be at least 1ms and no greater than the maximum backoff."); + + public static final ConfigOption CLIENT_WRITER_DISK_WRITE_LOCKED_BACKOFF_MAX = + key("client.writer.disk-write-locked.backoff-max") + .durationType() + .defaultValue(Duration.ofSeconds(10)) + .withDescription( + "The maximum retry backoff for a bucket whose writes are rejected by disk " + + "write protection. Must be between the initial backoff and 2147483647ms. " + + "Setting this equal to the initial backoff uses a fixed delay without jitter. " + + "Writes retry automatically after the delay, subject to client.writer.retries. " + + "Flush and graceful close also respect this delay."); + public static final ConfigOption CLIENT_WRITER_ENABLE_IDEMPOTENCE = key("client.writer.enable-idempotence") .booleanType() diff --git a/website/docs/maintenance/configuration.md b/website/docs/maintenance/configuration.md index eedb1b821b..bd962a1e87 100644 --- a/website/docs/maintenance/configuration.md +++ b/website/docs/maintenance/configuration.md @@ -215,6 +215,36 @@ The logging-related environment options (`env.log.dir`, `env.log.level`, `env.lo | kv.scanner.max-per-server | Integer | 200 | The maximum total number of concurrent KV scanner sessions allowed across all buckets on a single tablet server. New scan requests that exceed this limit will be rejected with an error. The default value is 200. | | kv.scanner.max-batch-size | MemorySize | 10mb | Server-side cap on the per-batch payload size for KV full-scan responses. The effective batch size is min(client-requested batch_size_bytes, this value). Protects the tablet server from out-of-memory if a client passes an excessively large batch size. The default value is 10mb. | +## Client disk write protection backoff + +| Key | Default | Type | Description | +| :--- | :--- | :--- | :--- | +| `client.writer.disk-write-locked.backoff` | `1 s` | Duration | The initial retry backoff when disk write protection rejects a write. Applies to both Log and KV writes, independently of KV backpressure. The backoff doubles with the batch retry count and uses 20% jitter, up to client.writer.disk-write-locked.backoff-max. The value must be at least 1ms and no greater than the maximum backoff. | +| `client.writer.disk-write-locked.backoff-max` | `10 s` | Duration | The maximum retry backoff for a bucket whose writes are rejected by disk write protection. Must be between the initial backoff and 2147483647ms. Setting this equal to the initial backoff uses a fixed delay without jitter. Writes retry automatically after the delay, subject to client.writer.retries. Flush and graceful close also respect this delay. | + +Java writers back off after a TabletServer rejects a write with `DISK_WRITE_LOCKED`. +The wait applies to both Log and KV writes and to all queued batches targeting the same +physical table bucket, including historical partition writes. Other buckets remain eligible +for normal scheduling. Requests already in flight cannot be recalled. + +Set `client.writer.disk-write-locked.backoff` (default `1s`) and +`client.writer.disk-write-locked.backoff-max` (default `10s`) in the **client** configuration. +The initial value must be at least `1ms`, and the maximum must be between the initial value +and `2147483647ms`. The backoff doubles with the existing batch retry count, with ±20% jitter +and a cap at the maximum. Setting the two values equal selects a fixed delay without jitter. +Actual delays are always at least `1ms`. + +Disk backoff is independent of `client.writer.kv-backpressure.max-throttle`; when both apply, +the writer waits for the longer remaining duration. A successful in-flight write or a KV +pressure value of zero does not clear an active disk wait. Writes retry automatically after +the wait, subject to `client.writer.retries`. Flush and graceful close respect the wait; +the existing close timeout still applies. After disk protection is lifted, the next retry +can take up to the remaining backoff window, plus normal scheduling and RPC time. + +This behavior requires upgrading the Java client or the connector that bundles it. +It uses the existing server error code and requires no server protocol upgrade. +The per-bucket backoff does not impose a cluster-wide bandwidth limit. + ## Metrics More metrics example could be found in [Observability - Metric Reporters](observability/metric-reporters.md). From 99e067b6d93ee4406d9c58190a33e6493de77360 Mon Sep 17 00:00:00 2001 From: fhan Date: Wed, 16 Sep 2026 15:55:06 +0800 Subject: [PATCH 2/2] fix(client): invalidate metadata before re-enqueuing write retries --- .../org/apache/fluss/client/write/Sender.java | 38 ++++++++++--------- .../apache/fluss/client/write/SenderTest.java | 37 ++++++++++++++++++ 2 files changed, 57 insertions(+), 18 deletions(-) diff --git a/fluss-client/src/main/java/org/apache/fluss/client/write/Sender.java b/fluss-client/src/main/java/org/apache/fluss/client/write/Sender.java index 060c2c8160..93aec22a16 100644 --- a/fluss-client/src/main/java/org/apache/fluss/client/write/Sender.java +++ b/fluss-client/src/main/java/org/apache/fluss/client/write/Sender.java @@ -49,6 +49,7 @@ import javax.annotation.concurrent.GuardedBy; import java.util.ArrayList; +import java.util.Collections; import java.util.HashMap; import java.util.HashSet; import java.util.List; @@ -748,6 +749,25 @@ private Set handleWriteBatchException( } else if (canRetry(readyWriteBatch, error.error())) { // if batch failed because of retrievable exception, we need to retry send all those // batches. + if (error.exception() instanceof InvalidMetadataException) { + if (error.exception() instanceof UnknownTableOrBucketException) { + LOG.warn( + "Received unknown table or bucket error in write request on bucket {}. The table-bucket may not exist.", + readyWriteBatch.tableBucket()); + } else { + LOG.warn( + "Received invalid metadata error in write request on bucket {}. " + + "Going to request metadata update.", + readyWriteBatch.tableBucket(), + error.exception()); + } + // Re-enqueuing publishes the retry to the sender and wakes it up. Invalidate the + // actual RPC target first so the retry cannot race ahead using stale metadata. A + // historical batch remains keyed by its original partition path in the + // accumulator, while its RPC is sent to the internal historical partition. + metadataUpdater.invalidPhysicalTableBucketMeta( + Collections.singleton(writeTargetPath)); + } if (!idempotenceManager.idempotenceEnabled()) { prepareWriteRetry(readyWriteBatch, error); reEnqueueBatch(readyWriteBatch); @@ -769,24 +789,6 @@ private Set handleWriteBatchException( writeBatch.writerId(), idempotenceManager.writerId())); failBatch(readyWriteBatch, exception, false); } - - if (error.exception() instanceof InvalidMetadataException) { - if (error.exception() instanceof UnknownTableOrBucketException) { - LOG.warn( - "Received unknown table or bucket error in write request on bucket {}. The table-bucket may not exist.", - readyWriteBatch.tableBucket()); - } else { - LOG.warn( - "Received invalid metadata error in write request on bucket {}. " - + "Going to request metadata update.", - readyWriteBatch.tableBucket(), - error.exception()); - } - // A historical batch remains keyed by its original partition path in the - // accumulator, but its RPC is sent to the internal historical partition. Invalidate - // the actual RPC target so the retry refreshes the historical bucket metadata. - invalidMetadataTables.add(writeTargetPath); - } } else { LOG.warn( "Get error write response on table bucket {}, fail. Error: {}", diff --git a/fluss-client/src/test/java/org/apache/fluss/client/write/SenderTest.java b/fluss-client/src/test/java/org/apache/fluss/client/write/SenderTest.java index 2ee1815208..4e3a667442 100644 --- a/fluss-client/src/test/java/org/apache/fluss/client/write/SenderTest.java +++ b/fluss-client/src/test/java/org/apache/fluss/client/write/SenderTest.java @@ -243,6 +243,43 @@ void testReroutesWriteAfterExplicitMissingPartitionResponse(boolean historicalCo assertThat(future.get()).isNull(); } + @ParameterizedTest + @ValueSource(booleans = {false, true}) + void testInvalidatesMetadataBeforePublishingRetry(boolean kv) throws Exception { + sender.destroyResources(); + Map tableInfos = new HashMap<>(); + tableInfos.put(DATA1_TABLE_PATH, DATA1_TABLE_INFO); + tableInfos.put(DATA1_TABLE_PATH_PK, DATA1_TABLE_INFO_PK); + AtomicReference retryQueuedAtInvalidation = new AtomicReference<>(); + metadataUpdater = + new TestingMetadataUpdater(tableInfos) { + @Override + public void invalidPhysicalTableBucketMeta( + Set physicalTablesToInvalid) { + if (!physicalTablesToInvalid.isEmpty()) { + retryQueuedAtInvalidation.set(accumulator.hasUnDrained()); + } + super.invalidPhysicalTableBucketMeta(physicalTablesToInvalid); + } + }; + sender = setupWithIdempotenceState(); + TableBucket bucket = kv ? new TableBucket(DATA1_TABLE_ID_PK, 0) : tb1; + CompletableFuture result = appendDiskTestRecord(kv, bucket, 1); + sender.runOnce(); + + finishRequest( + tb1, + 0, + kv + ? createPutKvResponse(bucket, Errors.NOT_LEADER_OR_FOLLOWER) + : createProduceLogResponse(bucket, Errors.NOT_LEADER_OR_FOLLOWER)); + + assertThat(retryQueuedAtInvalidation.get()).isFalse(); + assertThat(result).isNotDone(); + assertThat(accumulator.hasUnDrained()).isTrue(); + accumulator.abortAllBatches(new RuntimeException("test cleanup")); + } + @ParameterizedTest @CsvSource({"2, 4", "4, 2"}) void testServerValidatesResolvedBucketCount(int initialCount, int updatedCount)