From eed504b6dfa97b32307f3428cabcf87c7bf33cc2 Mon Sep 17 00:00:00 2001 From: Bee Klimt Date: Sat, 26 Sep 2026 16:27:15 -0700 Subject: [PATCH 1/5] feat: Retry data sources after unexpected HTTP errors instead of stopping --- .../com/launchdarkly/sdktest/TestService.java | 4 +- .../sdk/android/ConnectivityManager.java | 5 +- .../sdk/android/FDv2DataSource.java | 48 ++- .../sdk/android/FDv2DataSourceConditions.java | 43 +- .../android/FDv2StreamingSynchronizer.java | 282 ++++++++----- .../sdk/android/HttpFeatureFlagFetcher.java | 3 +- .../android/LDInvalidResponseCodeFailure.java | 11 +- .../com/launchdarkly/sdk/android/LDUtil.java | 24 +- .../sdk/android/PollingDataSource.java | 108 ++++- .../launchdarkly/sdk/android/RetryState.java | 297 ++++++++++++++ .../sdk/android/SourceManager.java | 152 ++++++- .../sdk/android/StreamingDataSource.java | 277 ++++++++----- .../android/SynchronizerFactoryWithState.java | 83 +++- .../android/FDv2DataSourceConditionsTest.java | 23 ++ .../sdk/android/FDv2DataSourceTest.java | 148 ++++--- .../FDv2StreamingSynchronizerTest.java | 123 +++++- .../launchdarkly/sdk/android/LDUtilTest.java | 32 ++ .../sdk/android/PollingDataSourceTest.java | 189 ++++++++- .../sdk/android/RetryStateTest.java | 376 ++++++++++++++++++ .../sdk/android/SourceManagerTest.java | 177 +++++++++ .../sdk/android/StreamingDataSourceTest.java | 198 ++++++--- .../android/FakeScheduledExecutorService.java | 195 +++++++++ .../sdk/android/FakeTaskExecutor.java | 133 +++++++ 23 files changed, 2568 insertions(+), 363 deletions(-) create mode 100644 launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java create mode 100644 launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java create mode 100644 launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java create mode 100644 shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java create mode 100644 shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java diff --git a/contract-tests/src/main/java/com/launchdarkly/sdktest/TestService.java b/contract-tests/src/main/java/com/launchdarkly/sdktest/TestService.java index b6618d9e0..50acf5e51 100644 --- a/contract-tests/src/main/java/com/launchdarkly/sdktest/TestService.java +++ b/contract-tests/src/main/java/com/launchdarkly/sdktest/TestService.java @@ -43,7 +43,9 @@ public class TestService extends NanoHTTPD { "evaluation-hooks", "track-hooks", "client-per-context-summaries", - "client-event-source-http-errors" + "client-event-source-http-errors", + "retry-conformance-fdv1-streaming", + "retry-conformance-fdv1-polling" }; private static final String MIME_JSON = "application/json"; static final Gson gson = new GsonBuilder() diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/ConnectivityManager.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/ConnectivityManager.java index cbeaf9fb2..d307c9966 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/ConnectivityManager.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/ConnectivityManager.java @@ -136,8 +136,9 @@ public void setStatus(@NonNull DataSourceState state, Throwable failure) { @Override public void shutDown() { - // The DataSource will call this method if it receives an error such as HTTP 401 that - // indicates the mobile key is invalid. + // A custom DataSource may call this to stop the SDK permanently. The SDK's own data + // sources no longer do: they retry every failure, including HTTP 401, with a backoff + // instead of stopping. ConnectivityManager.this.shutDown(); setStatus(ConnectionInformation.ConnectionMode.SHUTDOWN, null); } diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java index 9b8d40562..090dfc99b 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java @@ -149,7 +149,7 @@ public interface DataSourceFactory { } // note that the source manager only uses the initializers after the cache initializers and not the cache initializers - this.sourceManager = new SourceManager(allSynchronizers, generalInitializers); + this.sourceManager = new SourceManager(allSynchronizers, generalInitializers, sharedExecutor); this.fallbackTimeoutSeconds = fallbackTimeoutSeconds; this.recoveryTimeoutSeconds = recoveryTimeoutSeconds; this.sharedExecutor = sharedExecutor; @@ -475,11 +475,36 @@ private List getConditions(int synchronizerC List list = new ArrayList<>(); list.add(new FDv2DataSourceConditions.FallbackCondition(sharedExecutor, fallbackTimeoutSeconds)); if (!isPrime) { - list.add(new FDv2DataSourceConditions.RecoveryCondition(sharedExecutor, recoveryTimeoutSeconds)); + // Recovery only goes ahead if a higher-priority synchronizer is actually available; + // one that is still backing off after an unexpected error keeps the timer running. + list.add(new FDv2DataSourceConditions.RecoveryCondition(sharedExecutor, recoveryTimeoutSeconds, + sourceManager::hasAvailableSynchronizerBeforeCurrent)); } return list; } + /** + * Returns the next available synchronizer. If none is available but at least one is waiting + * out a backoff, waits for the first backoff to end and tries again, so that unexpected errors + * from every synchronizer never stop the data source. Returns null once the data source is + * stopped or there is truly nothing left to try. + */ + @Nullable + private Synchronizer nextSynchronizerOrWaitForBackoff() throws InterruptedException { + while (true) { + Synchronizer synchronizer = sourceManager.getNextAvailableSynchronizerAndSetActive(); + if (synchronizer != null || !sourceManager.hasBackingOffSynchronizers()) { + return synchronizer; + } + logger.info("All synchronizers are waiting out a backoff after unexpected errors; the first to become available will be tried next."); + try { + sourceManager.awaitAvailabilityChange().get(); + } catch (ExecutionException e) { + return null; + } + } + } + private static String detailForThrowable(@Nullable Throwable error) { if (error == null) { return "unknown error"; @@ -515,12 +540,12 @@ private void runSynchronizers( @NonNull DataSourceUpdateSinkV2 sink ) { try { - Synchronizer synchronizer = sourceManager.getNextAvailableSynchronizerAndSetActive(); + Synchronizer synchronizer = nextSynchronizerOrWaitForBackoff(); while (synchronizer != null) { String synchronizerName = synchronizer.name(); logger.info("Synchronizer '{}' is starting.", synchronizerName); resetSynchronizerStatusDedupe(); - int synchronizerCount = sourceManager.getAvailableSynchronizerCount(); + int synchronizerCount = sourceManager.getUsableSynchronizerCount(); boolean isPrime = sourceManager.isPrimeSynchronizer(); try { boolean running = true; @@ -570,6 +595,7 @@ private void runSynchronizers( if (changeSet != null) { sink.apply(context, changeSet); sink.setStatus(DataSourceState.VALID, null); + sourceManager.recordCurrentSynchronizerHealthy(System.currentTimeMillis()); tryCompleteStart(true, null); } break; @@ -595,14 +621,20 @@ private void runSynchronizers( running = false; break; case TERMINAL_ERROR: + // The synchronizer hit an error that is not expected to + // resolve soon, such as HTTP 401. Move on to the next one + // now, and put this one aside for a while rather than + // for good, so that a transient cause still recovers. maybeLogSynchronizerStatusChange( synchronizer.name(), status.getState() ); - sourceManager.blockCurrentSynchronizer(); + long backoffMillis = sourceManager.backOffCurrentSynchronizer( + synchronizer.name(), System.currentTimeMillis()); logger.warn( - "Synchronizer '{}' permanently failed and will not be used again until application restart.", - synchronizer.name() + "Synchronizer '{}' reported an unexpected error and will not be tried again for {} seconds.", + synchronizer.name(), + backoffMillis / 1000 ); running = false; sink.setStatus(DataSourceState.INTERRUPTED, status.getError()); @@ -651,7 +683,7 @@ private void runSynchronizers( Thread.currentThread().interrupt(); return; } - synchronizer = sourceManager.getNextAvailableSynchronizerAndSetActive(); + synchronizer = nextSynchronizerOrWaitForBackoff(); } if (!stopCalled.get()) { logger.warn("No more synchronizers available."); diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSourceConditions.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSourceConditions.java index fd9541d24..43983646a 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSourceConditions.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSourceConditions.java @@ -1,6 +1,7 @@ package com.launchdarkly.sdk.android; import androidx.annotation.NonNull; +import androidx.annotation.Nullable; import com.launchdarkly.sdk.android.subsystems.FDv2SourceResult; import com.launchdarkly.sdk.fdv2.SourceResultType; @@ -99,16 +100,48 @@ public ConditionType getType() { } /** - * Recovery: timer starts when built. Future completes with RECOVERY when timer fires. + * Decides, when a recovery timer fires, whether there is anything to recover to. + */ + interface RecoveryGate { + boolean canRecover(); + } + + /** + * Recovery: timer starts when built. Future completes with RECOVERY when timer fires, unless + * the gate says there is nothing to recover to yet, in which case the timer is re-armed for + * another timeout. */ static final class RecoveryCondition extends TimedCondition { + @Nullable + private final RecoveryGate gate; + private volatile boolean closed = false; RecoveryCondition(@NonNull ScheduledExecutorService executor, long timeoutSeconds) { + this(executor, timeoutSeconds, null); + } + + RecoveryCondition(@NonNull ScheduledExecutorService executor, long timeoutSeconds, @Nullable RecoveryGate gate) { super(executor, timeoutSeconds); - this.timerFuture = executor.schedule( - () -> resultFuture.set(ConditionType.RECOVERY), - timeoutSeconds, - TimeUnit.SECONDS); + this.gate = gate; + arm(); + } + + private void arm() { + this.timerFuture = sharedExecutor.schedule(this::onTimer, timeoutSeconds, TimeUnit.SECONDS); + } + + private void onTimer() { + if (gate == null || gate.canRecover()) { + resultFuture.set(ConditionType.RECOVERY); + } else if (!closed) { + arm(); + } + } + + @Override + public void close() { + closed = true; + super.close(); } @Override diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java index ef8e0d5f8..04460bbbd 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java @@ -5,12 +5,9 @@ import androidx.annotation.VisibleForTesting; import com.launchdarkly.eventsource.ConnectStrategy; -import com.launchdarkly.eventsource.ErrorStrategy; import com.launchdarkly.eventsource.EventSource; -import com.launchdarkly.eventsource.FaultEvent; import com.launchdarkly.eventsource.HttpConnectStrategy; import com.launchdarkly.eventsource.MessageEvent; -import com.launchdarkly.eventsource.RetryDelayStrategy; import com.launchdarkly.eventsource.StreamClosedByCallerException; import com.launchdarkly.eventsource.StreamEvent; import com.launchdarkly.eventsource.ResponseHeaders; @@ -34,8 +31,9 @@ import java.io.IOException; import java.net.URI; import java.util.Map; -import java.util.concurrent.Executor; import java.util.concurrent.Future; +import java.util.concurrent.ScheduledExecutorService; +import java.util.concurrent.ScheduledFuture; import java.util.concurrent.TimeUnit; import java.util.concurrent.atomic.AtomicBoolean; @@ -51,12 +49,17 @@ * If an optional {@link FDv2Requestor} is supplied, {@code ping} SSE events are handled by * issuing a poll request. If no requestor is supplied, {@code ping} events are ignored. *

+ * Reconnection is managed here rather than by the EventSource library. Every connection attempt + * uses a fresh {@link EventSource} that surfaces any failure as an exception and never retries + * on its own. A transport failure, a recoverable HTTP status, or a bad payload is reported as + * INTERRUPTED and followed by a reconnect after the delay computed by {@link RetryState}. An + * HTTP status that is not expected to resolve soon, such as 401, is reported as TERMINAL_ERROR + * and ends this synchronizer; the data source that owns it decides when to try it again. */ final class FDv2StreamingSynchronizer implements Synchronizer { private static final String METHOD_REPORT = "REPORT"; private static final String PING = "ping"; private static final long READ_TIMEOUT_MS = 300_000; // 5 minutes - private static final long MAX_RECONNECT_TIME_MS = 300_000; // 5 minutes private final HttpProperties httpProperties; private final URI streamBaseUri; @@ -67,21 +70,30 @@ final class FDv2StreamingSynchronizer implements Synchronizer { @Nullable private final FDv2Requestor requestor; private final boolean evaluationReasons; - private final int initialReconnectDelayMillis; @Nullable private final DiagnosticStore diagnosticStore; private final LDLogger logger; - private final Executor executor; + private final ScheduledExecutorService executor; private final LDAsyncQueue resultQueue = new LDAsyncQueue<>(); private final LDAwaitFuture shutdownFuture = new LDAwaitFuture<>(); private final AtomicBoolean started = new AtomicBoolean(false); + + // The following are only touched on the streaming thread. Only one connection attempt runs + // at a time, and the next is scheduled by the previous one, so successive attempts are + // ordered even if they run on different threads. private final FDv2ProtocolHandler protocolHandler = new FDv2ProtocolHandler(); + private final RetryState retryState; + // Set while handling a message when the current connection must be dropped and a new one + // made, along with the wait before doing so. + private boolean restartRequested = false; + private long restartDelayMillis = 0; - // closeLock guards: closed and the eventSource assignment in startStream. + // closeLock guards closed, eventSource, and pendingAttempt. private final Object closeLock = new Object(); private boolean closed = false; - private volatile EventSource eventSource; + private EventSource eventSource; + private ScheduledFuture pendingAttempt; private volatile long streamStarted = 0; /** @@ -90,12 +102,14 @@ final class FDv2StreamingSynchronizer implements Synchronizer { * @param streamBaseUri base URI for the stream endpoint * @param streamRequestPath path appended to the base URI for the stream request * @param requestor optional requestor for handling ping events via poll; may be null - * @param initialReconnectDelayMillis delay before reconnecting after an error, in milliseconds + * @param initialReconnectDelayMillis base delay before reconnecting after a failure, in + * milliseconds; later failures back off from this value * @param evaluationReasons true to request evaluation reasons in the stream * @param useReport true to use HTTP REPORT for the request body * @param httpProperties HTTP configuration for the stream request - * @param executor executor used to run the streaming loop on a background - * thread; should use background-priority threads + * @param executor executor used to run each connection attempt on a + * background thread and to schedule the next attempt after + * a failure; should use background-priority threads * @param logger logger * @param diagnosticStore optional store for stream diagnostics; may be null */ @@ -109,22 +123,48 @@ final class FDv2StreamingSynchronizer implements Synchronizer { boolean evaluationReasons, boolean useReport, @NonNull HttpProperties httpProperties, - @NonNull Executor executor, + @NonNull ScheduledExecutorService executor, @NonNull LDLogger logger, @Nullable DiagnosticStore diagnosticStore + ) { + this(evaluationContext, selectorSource, streamBaseUri, streamRequestPath, requestor, + initialReconnectDelayMillis, evaluationReasons, useReport, httpProperties, executor, + logger, diagnosticStore, RetryState.forStreaming(initialReconnectDelayMillis)); + } + + /** + * This constructor allows tests to supply a {@link RetryState} with short delays. See the + * other constructor for the remaining parameters. + * + * @param retryState the retry state that decides the wait before each reconnection + */ + FDv2StreamingSynchronizer( + @NonNull LDContext evaluationContext, + @NonNull SelectorSource selectorSource, + @NonNull URI streamBaseUri, + @NonNull String streamRequestPath, + @Nullable FDv2Requestor requestor, + int initialReconnectDelayMillis, + boolean evaluationReasons, + boolean useReport, + @NonNull HttpProperties httpProperties, + @NonNull ScheduledExecutorService executor, + @NonNull LDLogger logger, + @Nullable DiagnosticStore diagnosticStore, + @NonNull RetryState retryState ) { this.evaluationContext = evaluationContext; this.selectorSource = selectorSource; this.streamBaseUri = streamBaseUri; this.streamRequestPath = streamRequestPath; this.requestor = requestor; - this.initialReconnectDelayMillis = initialReconnectDelayMillis; this.evaluationReasons = evaluationReasons; this.useReport = useReport; this.httpProperties = httpProperties; this.executor = executor; this.logger = logger; this.diagnosticStore = diagnosticStore; + this.retryState = retryState; } @Override @@ -134,7 +174,7 @@ public Future next() { shouldStart = !closed && !started.getAndSet(true); } if (shouldStart) { - startStream(); + executor.execute(this::runConnectionAttempt); } return LDFutures.anyOf(shutdownFuture, resultQueue.take()); } @@ -149,6 +189,11 @@ public void close() { closed = true; esToClose = eventSource; eventSource = null; + // A backoff wait ends with the synchronizer: nothing reconnects after close(). + if (pendingAttempt != null) { + pendingAttempt.cancel(false); + pendingAttempt = null; + } } if (esToClose != null) { esToClose.close(); @@ -162,7 +207,51 @@ public void close() { shutdownFuture.set(FDv2SourceResult.status(FDv2SourceResult.Status.shutdown(), false)); } - private void startStream() { + private boolean isClosed() { + synchronized (closeLock) { + return closed; + } + } + + /** + * One connection attempt: connect, read until the connection ends, then schedule the next + * attempt after the backoff delay. Only {@link #close()} stops the sequence. + */ + private void runConnectionAttempt() { + EventSource es = buildEventSource(); + synchronized (closeLock) { + if (closed) { + es.close(); + return; + } + eventSource = es; + } + streamStarted = System.currentTimeMillis(); + + long delay; + try { + delay = readUntilDisconnected(es); + } finally { + es.close(); + synchronized (closeLock) { + if (eventSource == es) { + eventSource = null; + } + } + } + + if (delay < 0) { + return; + } + synchronized (closeLock) { + if (closed) { + return; + } + pendingAttempt = executor.schedule(this::runConnectionAttempt, delay, TimeUnit.MILLISECONDS); + } + } + + private EventSource buildEventSource() { HttpConnectStrategy connectStrategy = ConnectStrategy.http(getStreamUri()) .clientBuilderActions(clientBuilder -> { httpProperties.applyToHttpClientBuilder(clientBuilder); @@ -194,52 +283,58 @@ private void startStream() { RequestBody.create(JsonSerialization.serialize(evaluationContext), JSON)); } - EventSource es = new EventSource.Builder(connectStrategy) - .retryDelay(initialReconnectDelayMillis, TimeUnit.MILLISECONDS) - .retryDelayStrategy(RetryDelayStrategy.defaultStrategy() - .maxDelay(MAX_RECONNECT_TIME_MS, TimeUnit.MILLISECONDS)) - .errorStrategy(ErrorStrategy.alwaysContinue()) - .build(); - - synchronized (closeLock) { - if (closed) { - es.close(); - return; - } - eventSource = es; - } + // With the default ErrorStrategy, every connection or read failure is thrown from + // readAnyEvent() and the EventSource neither waits nor reconnects on its own. This class + // owns both, using RetryState, and builds a new EventSource for each attempt. + return new EventSource.Builder(connectStrategy).build(); + } - executor.execute(() -> { - streamStarted = System.currentTimeMillis(); - try { - for (StreamEvent event : es.anyEvents()) { - if (closed) { - break; - } - if (event instanceof MessageEvent) { - handleMessage((MessageEvent) event); - } else if (event instanceof FaultEvent) { - handleError((FaultEvent) event); - } - // CommentEvent (SSE comment/heartbeat line) — no action needed + /** + * Reads events from one connection until it ends. + * + * @return the wait in milliseconds before the next connection attempt, or -1 if the + * synchronizer has been closed and must not reconnect + */ + private long readUntilDisconnected(EventSource es) { + try { + while (true) { + StreamEvent event = es.readAnyEvent(); + if (isClosed()) { + return -1; } - } catch (Exception e) { - synchronized (closeLock) { - if (closed) { - return; + if (event instanceof MessageEvent) { + handleMessage((MessageEvent) event); + if (restartRequested) { + restartRequested = false; + return restartDelayMillis; } } - LDUtil.logExceptionAtErrorLevel(logger, e, "Stream thread ended with unexpected exception"); - recordStreamInit(true); - resultQueue.put(FDv2SourceResult.status( - FDv2SourceResult.Status.interrupted( - new LDFailure("Stream thread ended unexpectedly", e, - LDFailure.FailureType.UNKNOWN_ERROR)), - false)); - } finally { - es.close(); + // StartedEvent and CommentEvent (SSE comment/heartbeat line): no action needed } - }); + } catch (StreamException e) { + if (isClosed() || e instanceof StreamClosedByCallerException) { + return -1; + } + return handleError(e); + } catch (RuntimeException e) { + if (isClosed()) { + return -1; + } + LDUtil.logExceptionAtErrorLevel(logger, e, "Unexpected exception while reading stream"); + recordStreamInit(true); + protocolHandler.reset(); + resultQueue.put(FDv2SourceResult.status( + FDv2SourceResult.Status.interrupted( + new LDFailure("Unexpected exception while reading stream", e, + LDFailure.FailureType.UNKNOWN_ERROR)), + false)); + return recordFailureAndGetDelay(false); + } + } + + private long recordFailureAndGetDelay(boolean unexpected) { + retryState.recordFailure(unexpected, System.currentTimeMillis()); + return retryState.nextDelayMillis(); } private URI getStreamUri() { @@ -274,6 +369,10 @@ void handleMessage(MessageEvent event) { String eventData = event.getData(); logger.debug("onMessage: {}: {}", eventName, eventData); + // A message on the stream is healthy operation; enough of it in a row resets the + // backoff. + retryState.recordSuccess(System.currentTimeMillis()); + if (PING.equalsIgnoreCase(eventName)) { handlePing(); return; @@ -387,37 +486,35 @@ private void handlePing() { resultQueue.put(result); } - private void handleError(FaultEvent event) { - StreamException t = event.getCause(); - if (t instanceof StreamClosedByCallerException) { - return; - } - + /** + * Classifies a connection failure and reports it. A recoverable failure is reported as + * INTERRUPTED and followed by a reconnect; an unexpected HTTP status is reported as + * TERMINAL_ERROR and ends this synchronizer. + * + * @return the wait in milliseconds before the next attempt, or -1 if this synchronizer has + * ended and must not reconnect + */ + private long handleError(StreamException t) { recordStreamInit(true); protocolHandler.reset(); - boolean fdv1Fallback = isFdv1Fallback(event.getHeaders()); + boolean fdv1Fallback = false; + LDFailure failure; if (t instanceof StreamHttpErrorException) { StreamHttpErrorException httpError = (StreamHttpErrorException) t; - fdv1Fallback = fdv1Fallback || isFdv1Fallback(httpError.getHeaders()); + fdv1Fallback = isFdv1Fallback(httpError.getHeaders()); int code = httpError.getCode(); boolean recoverable = LDUtil.isHttpErrorRecoverable(code); - LDFailure failure = new LDInvalidResponseCodeFailure( + failure = new LDInvalidResponseCodeFailure( "Unexpected response code from stream", t, code, recoverable); if (!recoverable) { logger.error("Encountered non-retriable error: {}. Aborting connection to stream. Verify correct Mobile Key and Stream URI", code); shutdownFuture.set(FDv2SourceResult.status( FDv2SourceResult.Status.terminalError(failure), fdv1Fallback)); - EventSource es; synchronized (closeLock) { closed = true; - es = eventSource; - eventSource = null; - } - if (es != null) { - es.close(); } if (requestor != null) { try { @@ -425,45 +522,34 @@ private void handleError(FaultEvent event) { } catch (IOException ignored) { } } - } else { - logger.warn("Stream received HTTP error {}; will retry", code); - streamStarted = System.currentTimeMillis(); - resultQueue.put(FDv2SourceResult.status(FDv2SourceResult.Status.interrupted(failure), fdv1Fallback)); + return -1; } + logger.warn("Stream received HTTP error {}; will retry", code); } else { + // Every transport-level failure, including the server closing the connection, is + // recoverable. LDUtil.logExceptionAtWarnLevel(logger, t, "Stream network error"); - streamStarted = System.currentTimeMillis(); - resultQueue.put(FDv2SourceResult.status( - FDv2SourceResult.Status.interrupted( - new LDFailure("Stream network error", t, - LDFailure.FailureType.NETWORK_FAILURE)), - fdv1Fallback)); + failure = new LDFailure("Stream network error", t, LDFailure.FailureType.NETWORK_FAILURE); } + + long delay = recordFailureAndGetDelay(false); + logger.info("Will reconnect to stream in {} ms", delay); + resultQueue.put(FDv2SourceResult.status(FDv2SourceResult.Status.interrupted(failure), fdv1Fallback)); + return delay; } /** - * Interrupts the current connection so the EventSource reconnects immediately on the - * streaming thread, and resets the diagnostic timer for the new connection attempt. - * {@link EventSource#interrupt()} is safe to call from the streaming thread itself. + * Asks the connection attempt to drop the current connection once the current message has + * been handled, and to reconnect after a normal-regime backoff. A malformed payload is a + * normal failure, and so is a server that announces it is about to close the connection. * * @param failed true if the restart is due to an error (for diagnostic recording) */ private void restartStream(boolean failed) { recordStreamInit(failed); - streamStarted = System.currentTimeMillis(); - - EventSource es; - synchronized (closeLock) { - if (closed) { - return; - } - es = eventSource; - } - - if (es != null) { - es.interrupt(); - } protocolHandler.reset(); + restartRequested = true; + restartDelayMillis = recordFailureAndGetDelay(false); } @Override diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/HttpFeatureFlagFetcher.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/HttpFeatureFlagFetcher.java index 782807dc1..429344775 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/HttpFeatureFlagFetcher.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/HttpFeatureFlagFetcher.java @@ -122,7 +122,8 @@ public void onResponse(@NonNull Call call, @NonNull final Response response) { logger.error("Received 400 response when fetching flag values. Please check recommended ProGuard settings"); } callback.onError(new LDInvalidResponseCodeFailure("Unexpected response when retrieving Feature Flags: " + response + " using url: " - + request.url() + " with body: " + body, response.code(), true)); + + request.url() + " with body: " + body, response.code(), + LDUtil.isHttpErrorRecoverable(response.code()))); return; } logger.debug(body); diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDInvalidResponseCodeFailure.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDInvalidResponseCodeFailure.java index 6c5b6f383..18b2f632e 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDInvalidResponseCodeFailure.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDInvalidResponseCodeFailure.java @@ -14,14 +14,16 @@ public class LDInvalidResponseCodeFailure extends LDFailure { private final int responseCode; /** - * Whether or not the failure may be fixed by retrying + * Whether the failure is one that may resolve on its own if retried soon. This is a + * classification of the response code, not a statement about what the SDK will do: the SDK + * retries every failure, and simply waits longer between attempts when this is false. */ private final boolean retryable; /** * @param message the message * @param responseCode the response code - * @param retryable whether or not retrying may resolve the issue + * @param retryable whether the failure may resolve on its own if retried soon */ public LDInvalidResponseCodeFailure(String message, int responseCode, boolean retryable) { super(message, FailureType.UNEXPECTED_RESPONSE_CODE); @@ -33,7 +35,7 @@ public LDInvalidResponseCodeFailure(String message, int responseCode, boolean re * @param message the message * @param cause the cause of the failure * @param responseCode the response code - * @param retryable whether or not retrying may resolve the issue + * @param retryable whether the failure may resolve on its own if retried soon */ public LDInvalidResponseCodeFailure(String message, Throwable cause, int responseCode, boolean retryable) { super(message, cause, FailureType.UNEXPECTED_RESPONSE_CODE); @@ -42,7 +44,8 @@ public LDInvalidResponseCodeFailure(String message, Throwable cause, int respons } /** - * @return true if retrying may resolve the issue + * @return true if the failure may resolve on its own if retried soon; false if it is unlikely + * to, as with an authentication failure */ public boolean isRetryable() { return retryable; diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java index ae8ed1f5b..c1fa7b30b 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java @@ -212,9 +212,14 @@ public void updateHeaders(Map headers) { } /** - * Tests whether an HTTP error status represents a condition that might resolve on its own if we retry. + * Classifies an HTTP error status as either a {@code normal} failure, which may resolve on its + * own if retried soon, or an {@code unexpected} failure, which is not expected to. + *

+ * 400, 408, 429, and all 5xx statuses are {@code normal}; every other 4xx status is + * {@code unexpected}; any other status is {@code normal}. + * * @param statusCode the HTTP status - * @return true if retrying makes sense; false if it should be considered a permanent failure + * @return true if the failure is {@code normal}; false if it is {@code unexpected} */ static boolean isHttpErrorRecoverable(int statusCode) { if (statusCode >= 400 && statusCode < 500) { @@ -230,6 +235,21 @@ static boolean isHttpErrorRecoverable(int statusCode) { return true; } + /** + * Classifies a data source failure as {@code unexpected} or {@code normal}. + *

+ * Only an HTTP response failure whose status {@link #isHttpErrorRecoverable(int)} classifies as + * {@code unexpected} is {@code unexpected}. Every other failure, including transport errors, + * malformed response bodies, and unknown errors, is {@code normal}. + * + * @param failure the failure reported by a data source; may be null + * @return true if the failure is {@code unexpected}; false if it is {@code normal} + */ + static boolean isUnexpectedFailure(Throwable failure) { + return failure instanceof LDInvalidResponseCodeFailure && + !isHttpErrorRecoverable(((LDInvalidResponseCodeFailure) failure).getResponseCode()); + } + static void logExceptionAtErrorLevel(LDLogger logger, Throwable ex, String msgFormat, Object... msgArgs) { logException(logger, ex, true, msgFormat, msgArgs); } diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java index a68697d27..2342e3822 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java @@ -16,6 +16,11 @@ * streaming with Components.pollingDataSource(), or 2. streaming is enabled, but the application is * in the background so we do polling instead. The logic for this is in * ComponentsImpl.PollingDataSourceBuilderImpl and ComponentsImpl.StreamingDataSourceBuilderImpl. + *

+ * Each poll is scheduled individually after the previous one completes, so that the wait can + * be chosen per attempt: after a successful poll the next one happens at the poll + * interval, and after a failed poll the wait comes from {@link RetryState}. No failure stops the + * data source from polling again. */ final class PollingDataSource implements DataSource { private final LDContext context; @@ -29,6 +34,10 @@ final class PollingDataSource implements DataSource { private final LDLogger logger; final AtomicReference> currentPollTask = new AtomicReference<>(); // visible for testing + // Guarded by the lock on this instance. + private final RetryState retryState; + private boolean running = false; + /** * @param context that this data source will fetch data for * @param dataSourceUpdateSink to send data to @@ -52,6 +61,28 @@ final class PollingDataSource implements DataSource { PlatformState platformState, TaskExecutor taskExecutor, LDLogger logger + ) { + this(context, dataSourceUpdateSink, initialDelayMillis, pollIntervalMillis, maxNumberOfPolls, + fetcher, platformState, taskExecutor, RetryState.forPolling(pollIntervalMillis), logger); + } + + /** + * This constructor allows tests to supply a {@link RetryState} with short delays. See the + * other constructor for the remaining parameters. + * + * @param retryState the retry state that decides the wait after a failed poll + */ + PollingDataSource( + LDContext context, + DataSourceUpdateSink dataSourceUpdateSink, + long initialDelayMillis, + long pollIntervalMillis, + long maxNumberOfPolls, + FeatureFetcher fetcher, + PlatformState platformState, + TaskExecutor taskExecutor, + RetryState retryState, + LDLogger logger ) { this.context = context; this.dataSourceUpdateSink = dataSourceUpdateSink; @@ -61,6 +92,7 @@ final class PollingDataSource implements DataSource { this.fetcher = fetcher; this.platformState = platformState; this.taskExecutor = taskExecutor; + this.retryState = retryState; this.logger = logger; } @@ -73,16 +105,20 @@ public void start(final Callback resultCallback) { return; } - Runnable pollRunnable = () -> poll(resultCallback); + synchronized (this) { + running = true; + } logger.debug("Scheduling polling task with interval of {}ms, starting after {}ms, with number of polls {}", pollIntervalMillis, initialDelayMillis, numberOfPollsRemaining); - ScheduledFuture task = taskExecutor.startRepeatingTask(pollRunnable, - initialDelayMillis, pollIntervalMillis); - currentPollTask.set(task); + schedulePoll(initialDelayMillis, resultCallback); } @Override public void stop(Callback completionCallback) { + synchronized (this) { + running = false; + } + // A pending wait, whether the poll interval or a backoff, ends immediately. ScheduledFuture task = currentPollTask.getAndSet(null); if (task != null) { task.cancel(true); @@ -90,18 +126,62 @@ public void stop(Callback completionCallback) { completionCallback.onSuccess(null); } - private void poll(Callback resultCallback) { - // poll if there are polls remaining - if (numberOfPollsRemaining > 0) { + /** + * Schedules the next poll, unless the data source has been stopped or has used up its polls. + */ + private synchronized void schedulePoll(long delayMillis, Callback resultCallback) { + if (!running || numberOfPollsRemaining <= 0) { + return; + } + currentPollTask.set(taskExecutor.scheduleTask(() -> poll(resultCallback), delayMillis)); + } + + private void poll(final Callback resultCallback) { + synchronized (this) { + if (!running || numberOfPollsRemaining <= 0) { + return; + } numberOfPollsRemaining--; - ConnectivityManager.fetchAndSetData(fetcher, context, dataSourceUpdateSink, - resultCallback, logger); - } else { - // terminate if we have no polls remaining - ScheduledFuture task = currentPollTask.getAndSet(null); - if (task != null) { - task.cancel(true); + } + + Callback pollCallback = new Callback() { + @Override + public void onSuccess(Boolean result) { + long delay; + synchronized (PollingDataSource.this) { + retryState.recordSuccess(System.currentTimeMillis()); + delay = retryState.nextDelayMillis(); + } + resultCallback.onSuccess(result); + schedulePoll(delay, resultCallback); } + + @Override + public void onError(Throwable error) { + boolean unexpected = LDUtil.isUnexpectedFailure(error); + long delay; + synchronized (PollingDataSource.this) { + retryState.recordFailure(unexpected, System.currentTimeMillis()); + delay = retryState.nextDelayMillis(); + } + if (unexpected) { + logger.error("Received HTTP error {} from polling request. This is not expected to resolve on its own; verify correct Mobile Key and Polling URI. Will retry in {} ms.", + ((LDInvalidResponseCodeFailure) error).getResponseCode(), delay); + } else { + logger.warn("Polling request failed. Will retry in {} ms.", delay); + } + resultCallback.onError(error); + schedulePoll(delay, resultCallback); + } + }; + + try { + ConnectivityManager.fetchAndSetData(fetcher, context, dataSourceUpdateSink, pollCallback, logger); + } catch (RuntimeException e) { + // A fetcher must report its outcome through the callback, but if one throws instead we + // still owe the caller a result and the next poll. + LDUtil.logExceptionAtErrorLevel(logger, e, "Unexpected exception while polling for flags"); + pollCallback.onError(new LDFailure("Exception while fetching flags", e, LDFailure.FailureType.UNKNOWN_ERROR)); } } } diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java new file mode 100644 index 000000000..acae6889a --- /dev/null +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java @@ -0,0 +1,297 @@ +package com.launchdarkly.sdk.android; + +import androidx.annotation.NonNull; + +import java.util.Random; + +/** + * Retry state for a long-running component such as a streaming or polling data source: + * exponential backoff with jitter, in two regimes. + *

+ * Every failure is classified as either {@code normal} or {@code unexpected}. A {@code normal} + * failure advances the backoff within the current regime. An {@code unexpected} failure (for + * example an HTTP 401 or 403, which is unlikely to resolve on its own quickly) moves the + * component to the extended regime, whose delays run from minutes up to an hour, and the + * component stays there until the reset threshold is met. No failure ever causes the component + * to stop retrying. + *

+ * The reset threshold differs for the two kinds of component: + *

    + *
  • Streaming: continuous healthy operation for {@link #STREAMING_RESET_THRESHOLD_MILLIS}, + * where healthy operation starts with the first payload received on a connection.
  • + *
  • Polling: {@link #POLLING_RESET_THRESHOLD_SUCCESSES} consecutive successful polls.
  • + *
+ *

+ * Timestamps are passed in explicitly so that the state machine is deterministic in tests. Any + * monotonic millisecond clock may be used, as long as the same clock is used for every call. + *

+ * This class is not thread-safe. The owning component must serialize access to it. + */ +final class RetryState { + /** + * The normal-regime ceiling for the wait between attempts. + */ + static final long NORMAL_MAX_DELAY_MILLIS = 30_000L; + /** + * The base delay for the first attempt after an {@code unexpected} failure. + */ + static final long EXTENDED_INITIAL_DELAY_MILLIS = 5L * 60_000L; + /** + * The extended-regime ceiling for the wait between attempts. + */ + static final long EXTENDED_MAX_DELAY_MILLIS = 60L * 60_000L; + /** + * A streaming component resets its retry state after this much continuous healthy operation. + */ + static final long STREAMING_RESET_THRESHOLD_MILLIS = 60_000L; + /** + * A polling component resets its retry state after this many consecutive successful polls. + */ + static final int POLLING_RESET_THRESHOLD_SUCCESSES = 2; + + // 2^30 is far more doubling than any real delay needs, and keeps initialDelay * 2^exponent + // from overflowing a long for any plausible configured delay. + private static final int MAX_EXPONENT = 30; + + private final long normalInitialDelayMillis; + private final long normalMaxDelayMillis; + private final long extendedInitialDelayMillis; + private final long extendedMaxDelayMillis; + // The interval at which the component operates when healthy. The wait after a failure is + // never shorter than this, and it is the wait after a success. Zero for a component with no + // such interval, such as streaming. + private final long operatingCadenceMillis; + // Duration-based reset threshold; zero if this component does not use one. + private final long healthyResetThresholdMillis; + // Count-based reset threshold; zero if this component does not use one. + private final int successResetThreshold; + private final Random random; + + private int attempts = 0; + private boolean extended = false; + private boolean lastOperationFailed = false; + // Start of the current stretch of healthy operation, or zero if not currently healthy. + private long healthySinceMillis = 0; + private int consecutiveSuccesses = 0; + + /** + * Creates the retry state for a streaming data source or synchronizer. + *

+ * The normal regime backs off from the configured initial reconnect delay up to + * {@link #NORMAL_MAX_DELAY_MILLIS}. The extended regime backs off from + * {@link #EXTENDED_INITIAL_DELAY_MILLIS} up to {@link #EXTENDED_MAX_DELAY_MILLIS}. The state + * resets after {@link #STREAMING_RESET_THRESHOLD_MILLIS} of healthy operation. + * + * @param initialReconnectDelayMillis the configured initial reconnect delay; may be zero + * @return the retry state + */ + @NonNull + static RetryState forStreaming(long initialReconnectDelayMillis) { + long initial = Math.max(0, initialReconnectDelayMillis); + return new RetryState( + initial, + Math.max(NORMAL_MAX_DELAY_MILLIS, initial), + Math.max(EXTENDED_INITIAL_DELAY_MILLIS, initial), + Math.max(EXTENDED_MAX_DELAY_MILLIS, initial), + 0, + STREAMING_RESET_THRESHOLD_MILLIS, + 0, + new Random()); + } + + /** + * Creates the retry state for a polling data source or synchronizer. + *

+ * A polling component's operating cadence is its poll interval. In the normal regime it keeps + * polling at that interval after a failure. In the extended regime it backs off from + * {@link #EXTENDED_INITIAL_DELAY_MILLIS} up to {@link #EXTENDED_MAX_DELAY_MILLIS}, but never + * more often than the poll interval. The state resets after + * {@link #POLLING_RESET_THRESHOLD_SUCCESSES} consecutive successful polls. + * + * @param pollIntervalMillis the configured poll interval; must be positive + * @return the retry state + */ + @NonNull + static RetryState forPolling(long pollIntervalMillis) { + long interval = Math.max(1, pollIntervalMillis); + return new RetryState( + interval, + interval, + Math.max(EXTENDED_INITIAL_DELAY_MILLIS, interval), + Math.max(EXTENDED_MAX_DELAY_MILLIS, interval), + interval, + 0, + POLLING_RESET_THRESHOLD_SUCCESSES, + new Random()); + } + + /** + * Creates the retry state that the FDv2 data source keeps for one synchronizer slot. + *

+ * Only unexpected failures are recorded against a slot, so both regimes are bound to the + * extended values: the slot waits {@link #EXTENDED_INITIAL_DELAY_MILLIS} after the first + * unexpected error, doubling up to {@link #EXTENDED_MAX_DELAY_MILLIS}. The state resets after + * {@link #STREAMING_RESET_THRESHOLD_MILLIS} of healthy operation by the slot's synchronizer. + * + * @return the retry state + */ + @NonNull + static RetryState forSynchronizerSlot() { + return new RetryState( + EXTENDED_INITIAL_DELAY_MILLIS, + EXTENDED_MAX_DELAY_MILLIS, + EXTENDED_INITIAL_DELAY_MILLIS, + EXTENDED_MAX_DELAY_MILLIS, + 0, + STREAMING_RESET_THRESHOLD_MILLIS, + 0, + new Random()); + } + + /** + * Creates a retry state with explicit parameters. Production code should use one of the + * static factories; this constructor exists so that tests can use short delays and a + * deterministic random source. + * + * @param normalInitialDelayMillis base delay for the first retry in the normal regime + * @param normalMaxDelayMillis ceiling on the wait in the normal regime + * @param extendedInitialDelayMillis base delay for the first retry in the extended regime; + * raised to the normal initial delay if smaller + * @param extendedMaxDelayMillis ceiling on the wait in the extended regime + * @param operatingCadenceMillis the operating cadence, or zero if the component has none + * @param healthyResetThresholdMillis how long healthy operation must last before the state + * resets, or zero if this component does not reset on + * duration + * @param successResetThreshold how many consecutive successes reset the state, or zero + * if this component does not reset on a count + * @param random source of jitter + */ + RetryState( + long normalInitialDelayMillis, + long normalMaxDelayMillis, + long extendedInitialDelayMillis, + long extendedMaxDelayMillis, + long operatingCadenceMillis, + long healthyResetThresholdMillis, + int successResetThreshold, + @NonNull Random random + ) { + this.normalInitialDelayMillis = Math.max(0, normalInitialDelayMillis); + this.normalMaxDelayMillis = Math.max(this.normalInitialDelayMillis, normalMaxDelayMillis); + this.extendedInitialDelayMillis = Math.max(this.normalInitialDelayMillis, extendedInitialDelayMillis); + this.extendedMaxDelayMillis = Math.max(this.extendedInitialDelayMillis, extendedMaxDelayMillis); + this.operatingCadenceMillis = Math.max(0, operatingCadenceMillis); + this.healthyResetThresholdMillis = Math.max(0, healthyResetThresholdMillis); + this.successResetThreshold = Math.max(0, successResetThreshold); + this.random = random; + } + + /** + * Records healthy operation. + *

+ * For a streaming component this is a payload received on the current connection; the first + * such call after a connection is established starts the clock toward the duration-based + * reset threshold, and later calls on the same connection do not move it. For a polling + * component this is a successful poll, which counts toward the count-based reset threshold. + *

+ * After a success the next wait is the operating cadence, even if the retry state has not + * reset. + * + * @param nowMillis the current time + */ + void recordSuccess(long nowMillis) { + lastOperationFailed = false; + if (healthyResetThresholdMillis > 0 && healthySinceMillis == 0) { + healthySinceMillis = nowMillis; + } + if (successResetThreshold > 0) { + consecutiveSuccesses++; + if (consecutiveSuccesses >= successResetThreshold) { + reset(); + } + } + } + + /** + * Records a failure and updates the retry state. + *

+ * If this component resets on a duration and it had been healthy for at least that long when + * the failure happened, the state is reset first so that the failure is counted against a + * clean slate. Either way, measurement toward the reset threshold restarts. + *

+ * Call this before {@link #nextDelayMillis()}. + * + * @param unexpected true if the failure is classified as {@code unexpected}, false if + * {@code normal} + * @param nowMillis the current time + */ + void recordFailure(boolean unexpected, long nowMillis) { + if (healthyResetThresholdMillis > 0 && healthySinceMillis != 0 + && nowMillis - healthySinceMillis >= healthyResetThresholdMillis) { + reset(); + } + healthySinceMillis = 0; + consecutiveSuccesses = 0; + lastOperationFailed = true; + + if (unexpected && !extended) { + // Move to the extended regime and start its attempt count over, so the first + // extended wait is the extended initial delay. + extended = true; + attempts = 1; + } else { + // A repeated unexpected failure keeps counting in the extended regime. + attempts++; + } + } + + /** + * Clears the retry state back to its initial values: no attempts, and the normal regime. + */ + void reset() { + attempts = 0; + extended = false; + consecutiveSuccesses = 0; + } + + /** + * Computes how long to wait before the next attempt. + *

+ * If the most recent operation succeeded (or no operation has failed yet), this is the + * operating cadence. Otherwise it is {@code T - J}, where {@code T} is the regime's initial + * delay doubled once per attempt after the first and clamped to the regime's ceiling, and + * {@code J} is a uniformly random jitter of up to half of {@code T}. The result is never less + * than the operating cadence. + * + * @return the wait in milliseconds + */ + long nextDelayMillis() { + if (!lastOperationFailed) { + return operatingCadenceMillis; + } + long initial = extended ? extendedInitialDelayMillis : normalInitialDelayMillis; + long max = extended ? extendedMaxDelayMillis : normalMaxDelayMillis; + int exponent = Math.min(Math.max(0, attempts - 1), MAX_EXPONENT); + long base = initial * (1L << exponent); + if (base < 0 || base > max) { // negative means the multiplication overflowed + base = max; + } + long jitter = base > 1 ? (long) (random.nextDouble() * (base / 2.0)) : 0; + return Math.max(operatingCadenceMillis, base - jitter); + } + + /** + * @return the number of failures counted since the last reset + */ + int getAttempts() { + return attempts; + } + + /** + * @return true if an {@code unexpected} failure has moved this component to the extended + * regime and no reset has happened since + */ + boolean isInExtendedRegime() { + return extended; + } +} diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java index bbbd8adb6..66a978dda 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java @@ -1,6 +1,7 @@ package com.launchdarkly.sdk.android; import androidx.annotation.NonNull; +import androidx.annotation.Nullable; import com.launchdarkly.sdk.android.subsystems.Initializer; import com.launchdarkly.sdk.android.subsystems.Synchronizer; @@ -8,18 +9,27 @@ import java.io.Closeable; import java.io.IOException; import java.util.List; +import java.util.concurrent.Future; +import java.util.concurrent.ScheduledExecutorService; +import java.util.concurrent.ScheduledFuture; +import java.util.concurrent.TimeUnit; /** * Manages the state of synchronizers and initializers: tracks which is active, - * advances through the lists (with optional block state for synchronizers), + * advances through the lists (skipping synchronizers that are blocked or backing off), * and closes the previous source when switching. *

+ * A synchronizer that reports an unexpected error is put into a backoff rather than removed: + * its slot is skipped until the backoff ends, and the wait doubles on each repeat, up to an + * hour. No synchronizer is ever permanently removed. + *

* Package-private for internal use by FDv2DataSource. */ final class SourceManager implements Closeable { private final List synchronizerFactories; private final List> initializers; + private final ScheduledExecutorService executor; private final Object activeSourceLock = new Object(); private Closeable activeSource; @@ -31,12 +41,18 @@ final class SourceManager implements Closeable { private SynchronizerFactoryWithState currentSynchronizerFactory; + // Completed, and replaced, whenever a slot's backoff ends or this manager closes, so that a + // caller with no available synchronizer can wait for one. + private LDAwaitFuture availabilityChanged = new LDAwaitFuture<>(); + SourceManager( @NonNull List synchronizerFactories, - @NonNull List> initializers + @NonNull List> initializers, + @NonNull ScheduledExecutorService executor ) { this.synchronizerFactories = synchronizerFactories; this.initializers = initializers; + this.executor = executor; } /** @@ -49,7 +65,7 @@ void resetSourceIndex() { } } - /** True if any synchronizer is marked as FDv1 fallback (Android: not used yet). */ + /** True if any synchronizer is marked as FDv1 fallback. */ boolean hasFDv1Fallback() { for (SynchronizerFactoryWithState s : synchronizerFactories) { if (s.isFDv1Fallback()) { @@ -70,6 +86,7 @@ void fdv1Fallback() { if (s.isFDv1Fallback()) { s.unblock(); } else { + s.cancelPendingUnblock(); s.block(); } } @@ -151,12 +168,105 @@ private FDv2DataSource.DataSourceFactory getNextInitializer() { return initializers.get(initializerIndex); } - /** Block the current synchronizer so it will not be returned again (e.g. after TERMINAL_ERROR). */ - void blockCurrentSynchronizer() { + /** + * Puts the current synchronizer's slot into backoff after it reported an unexpected error. + * The slot is skipped by {@link #getNextAvailableSynchronizerAndSetActive()} until the backoff + * ends, which is scheduled on the executor. + * + * @param synchronizerName the name of the synchronizer that failed, for logging + * @param nowMillis the current time + * @return how long the slot stays in backoff, in milliseconds, or -1 if there is no current + * synchronizer + */ + long backOffCurrentSynchronizer(@NonNull String synchronizerName, long nowMillis) { + synchronized (activeSourceLock) { + final SynchronizerFactoryWithState slot = currentSynchronizerFactory; + if (slot == null || isShutdown) { + return -1; + } + long delayMillis = slot.startBackoff(synchronizerName, nowMillis); + ScheduledFuture unblock = executor.schedule(new Runnable() { + @Override + public void run() { + endBackoff(slot); + } + }, delayMillis, TimeUnit.MILLISECONDS); + slot.setPendingUnblock(unblock); + return delayMillis; + } + } + + /** + * Records healthy operation by the current synchronizer, which counts toward resetting its + * slot's backoff. + */ + void recordCurrentSynchronizerHealthy(long nowMillis) { synchronized (activeSourceLock) { if (currentSynchronizerFactory != null) { - currentSynchronizerFactory.block(); + currentSynchronizerFactory.recordHealthy(nowMillis); + } + } + } + + /** + * Ends a slot's backoff and wakes any caller waiting in {@link #awaitAvailabilityChange()}. + * + * @return the name of the synchronizer whose error started the backoff, if the slot became + * available; null if nothing changed + */ + @Nullable + String endBackoff(@NonNull SynchronizerFactoryWithState slot) { + LDAwaitFuture toComplete; + String name; + synchronized (activeSourceLock) { + if (isShutdown || !slot.endBackoff()) { + return null; + } + name = slot.getLastSynchronizerName(); + toComplete = availabilityChanged; + availabilityChanged = new LDAwaitFuture<>(); + } + toComplete.set(null); + return name == null ? "" : name; + } + + /** True if any synchronizer is waiting out a backoff. Always false once closed. */ + boolean hasBackingOffSynchronizers() { + synchronized (activeSourceLock) { + if (isShutdown) { + return false; + } + for (SynchronizerFactoryWithState s : synchronizerFactories) { + if (s.getState() == SynchronizerFactoryWithState.State.BackingOff) { + return true; + } } + return false; + } + } + + /** + * @return a future that completes the next time a slot's backoff ends, or when this manager + * closes + */ + Future awaitAvailabilityChange() { + synchronized (activeSourceLock) { + return availabilityChanged; + } + } + + /** + * @return true if a synchronizer earlier in the list than the current one is available, so + * that recovering to it would move to a higher-priority synchronizer + */ + boolean hasAvailableSynchronizerBeforeCurrent() { + synchronized (activeSourceLock) { + for (int i = 0; i < synchronizerIndex && i < synchronizerFactories.size(); i++) { + if (synchronizerFactories.get(i).getState() == SynchronizerFactoryWithState.State.Available) { + return true; + } + } + return false; } } @@ -193,11 +303,15 @@ Initializer getNextInitializerAndSetActive() { } } - /** True if the current synchronizer is the first available one (prime). */ + /** + * True if the current synchronizer is the prime one: no synchronizer before it in the list + * is available or merely backing off. A slot that is backing off still outranks the current + * one, so a recovery condition runs while it is unavailable and can return to it later. + */ boolean isPrimeSynchronizer() { synchronized (activeSourceLock) { for (int i = 0; i < synchronizerFactories.size(); i++) { - if (synchronizerFactories.get(i).getState() == SynchronizerFactoryWithState.State.Available) { + if (synchronizerFactories.get(i).getState() != SynchronizerFactoryWithState.State.Blocked) { return synchronizerIndex == i; } } @@ -217,15 +331,37 @@ int getAvailableSynchronizerCount() { } } + /** + * @return the number of synchronizers that are available or backing off, which is the number + * that could run at some point + */ + int getUsableSynchronizerCount() { + synchronized (activeSourceLock) { + int count = 0; + for (SynchronizerFactoryWithState s : synchronizerFactories) { + if (s.getState() != SynchronizerFactoryWithState.State.Blocked) { + count++; + } + } + return count; + } + } + @Override public void close() { + LDAwaitFuture toComplete; synchronized (activeSourceLock) { isShutdown = true; if (activeSource != null) { safeClose(activeSource); activeSource = null; } + for (SynchronizerFactoryWithState s : synchronizerFactories) { + s.cancelPendingUnblock(); + } + toComplete = availabilityChanged; } + toComplete.set(null); } private static void safeClose(Closeable closeable) { diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java index 366254b28..3c472f305 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java @@ -8,7 +8,7 @@ import com.launchdarkly.eventsource.EventSource; import com.launchdarkly.eventsource.HttpConnectStrategy; import com.launchdarkly.eventsource.MessageEvent; -import com.launchdarkly.eventsource.RetryDelayStrategy; +import com.launchdarkly.eventsource.StreamException; import com.launchdarkly.eventsource.StreamHttpErrorException; import com.launchdarkly.eventsource.background.BackgroundEventHandler; import com.launchdarkly.eventsource.background.BackgroundEventSource; @@ -27,6 +27,7 @@ import com.launchdarkly.sdk.json.SerializationException; import java.net.URI; +import java.util.concurrent.ScheduledFuture; import java.util.concurrent.TimeUnit; import okhttp3.RequestBody; @@ -39,6 +40,11 @@ *

* The SDK uses this implementation if streaming is enabled (as it is by default) and the * application is the foreground. The logic for this is in ComponentsImpl.StreamingDataSourceBuilderImpl. + *

+ * Reconnection is managed by this class rather than by the EventSource library, so that the + * backoff is controlled here: every stream failure, including HTTP statuses such as + * 401 and 403 that used to stop the stream permanently, is retried. Failures that are unlikely to + * resolve on their own are retried with a much longer backoff (see {@link RetryState}). */ final class StreamingDataSource implements DataSource { private static final String METHOD_REPORT = "REPORT"; @@ -48,13 +54,10 @@ final class StreamingDataSource implements DataSource { private static final String PATCH = "patch"; private static final String DELETE = "delete"; - private static final long MAX_RECONNECT_TIME_MS = 300_000; // 5 minutes - private static final long READ_TIMEOUT_MS = 300_000; // 5 minutes is the standard read timeout used for all LaunchDarkly stream connections, based on // an expectation that the server will send heartbeats at a shorter interval than that - private BackgroundEventSource es; private final LDContext context; private final HttpProperties httpProperties; private final boolean evaluationReasons; @@ -64,14 +67,21 @@ final class StreamingDataSource implements DataSource { private final DataSourceUpdateSink dataSourceUpdateSink; private final FeatureFetcher fetcher; private final boolean streamEvenInBackground; - private volatile boolean running = false; - // volatile because it is written from the EventSource background thread (onError) - // and read from start(), which may be invoked on a different thread. - private volatile boolean connection401Error = false; private final DiagnosticStore diagnosticStore; - private long eventSourceStarted; + private final TaskExecutor taskExecutor; private final LDLogger logger; + // The following fields are guarded by the lock on this instance. retryState is only ever + // touched while holding that lock. + private final RetryState retryState; + private BackgroundEventSource es; + private BackgroundEventHandler handler; + private ScheduledFuture pendingReconnect; + private boolean running = false; + + // Written when a connection attempt begins, read by the EventSource callbacks for diagnostics. + private volatile long eventSourceStarted; + StreamingDataSource( @NonNull ClientContext clientContext, @NonNull LDContext context, @@ -79,6 +89,22 @@ final class StreamingDataSource implements DataSource { @NonNull FeatureFetcher fetcher, int initialReconnectDelayMillis, boolean streamEvenInBackground + ) { + this(clientContext, context, dataSourceUpdateSink, fetcher, initialReconnectDelayMillis, + streamEvenInBackground, RetryState.forStreaming(initialReconnectDelayMillis)); + } + + /** + * This constructor allows tests to supply a {@link RetryState} with short delays. + */ + StreamingDataSource( + @NonNull ClientContext clientContext, + @NonNull LDContext context, + @NonNull DataSourceUpdateSink dataSourceUpdateSink, + @NonNull FeatureFetcher fetcher, + int initialReconnectDelayMillis, + boolean streamEvenInBackground, + @NonNull RetryState retryState ) { this.context = context; this.dataSourceUpdateSink = dataSourceUpdateSink; @@ -90,111 +116,153 @@ final class StreamingDataSource implements DataSource { this.initialReconnectDelayMillis = initialReconnectDelayMillis; this.streamEvenInBackground = streamEvenInBackground; this.diagnosticStore = ClientContextImpl.get(clientContext).getDiagnosticStore(); + this.taskExecutor = ClientContextImpl.get(clientContext).getTaskExecutor(); + this.retryState = retryState; this.logger = clientContext.getBaseLogger(); } public void start(@NonNull Callback resultCallback) { - if (!running && !connection401Error) { - logger.debug("Starting."); + synchronized (this) { + if (running) { + return; + } + running = true; + handler = makeHandler(resultCallback); + } + logger.debug("Starting."); + connect(); + } - BackgroundEventHandler handler = new BackgroundEventHandler() { - @Override - public void onOpen() { - logger.info("Started LaunchDarkly EventStream"); - if (diagnosticStore != null) { - diagnosticStore.recordStreamInit(eventSourceStarted, (int) (System.currentTimeMillis() - eventSourceStarted), false); - } + private BackgroundEventHandler makeHandler(@NonNull Callback resultCallback) { + return new BackgroundEventHandler() { + @Override + public void onOpen() { + logger.info("Started LaunchDarkly EventStream"); + if (diagnosticStore != null) { + diagnosticStore.recordStreamInit(eventSourceStarted, (int) (System.currentTimeMillis() - eventSourceStarted), false); } + } - @Override - public void onClosed() { - logger.info("Closed LaunchDarkly EventStream"); - } + @Override + public void onClosed() { + logger.info("Closed LaunchDarkly EventStream"); + } - @Override - public void onMessage(final String name, MessageEvent event) { - final String eventData = event.getData(); - logger.debug("onMessage: {}: {}", name, eventData); - handle(name, eventData, resultCallback); + @Override + public void onMessage(final String name, MessageEvent event) { + final String eventData = event.getData(); + logger.debug("onMessage: {}: {}", name, eventData); + // A payload on the stream is healthy operation; enough of it in a row resets + // the backoff. + synchronized (StreamingDataSource.this) { + retryState.recordSuccess(System.currentTimeMillis()); } + handle(name, eventData, resultCallback); + } + + @Override + public void onComment(String comment) { + // intentionally empty + } - @Override - public void onComment(String comment) { - // intentionally empty + @Override + public void onError(Throwable t) { + LDUtil.logExceptionAtErrorLevel(logger, t, + "Encountered EventStream error connecting to URI: {}", + getUri(context)); + + LDFailure failure; + boolean unexpected = false; + int code = 0; + if (t instanceof StreamHttpErrorException) { + if (diagnosticStore != null) { + diagnosticStore.recordStreamInit(eventSourceStarted, (int) (System.currentTimeMillis() - eventSourceStarted), true); + } + code = ((StreamHttpErrorException) t).getCode(); + boolean recoverable = LDUtil.isHttpErrorRecoverable(code); + unexpected = !recoverable; + failure = new LDInvalidResponseCodeFailure("Unexpected Response Code From Stream Connection", t, code, recoverable); + } else { + failure = new LDFailure("Network error in stream connection", t, LDFailure.FailureType.NETWORK_FAILURE); } - @Override - public void onError(Throwable t) { - LDUtil.logExceptionAtErrorLevel(logger, t, - "Encountered EventStream error connecting to URI: {}", - getUri(context)); - if (t instanceof StreamHttpErrorException) { - if (diagnosticStore != null) { - diagnosticStore.recordStreamInit(eventSourceStarted, (int) (System.currentTimeMillis() - eventSourceStarted), true); - } - int code = ((StreamHttpErrorException) t).getCode(); - if (!LDUtil.isHttpErrorRecoverable(code)) { - logger.error("Encountered non-retriable error: {}. Aborting connection to stream. Verify correct Mobile Key and Stream URI", code); - running = false; - // Set the connection401Error guard before notifying the callback. A consumer - // may react to the error by synchronously calling start() again, and the - // guard must already be set so that retry is a no-op. - if (code == 401) { - connection401Error = true; - } - resultCallback.onError(new LDInvalidResponseCodeFailure("Unexpected Response Code From Stream Connection", t, code, false)); - if (code == 401) { - dataSourceUpdateSink.shutDown(); - } - stop(null); + // A StreamException means the connection has ended; every such transport failure + // is a normal failure. Anything else was thrown by this handler while + // processing an event, and the stream is still open, so there is nothing to retry. + if (t instanceof StreamException) { + long delay = scheduleReconnectAfterFailure(unexpected); + if (delay >= 0) { + if (unexpected) { + logger.error("Encountered HTTP error {} from stream. This is not expected to resolve on its own; verify correct Mobile Key and Stream URI. Will retry in {} ms.", code, delay); } else { - eventSourceStarted = System.currentTimeMillis(); - resultCallback.onError(new LDInvalidResponseCodeFailure("Unexpected Response Code From Stream Connection", t, code, true)); + logger.warn("Will retry stream connection in {} ms.", delay); } - } else { - resultCallback.onError(new LDFailure("Network error in stream connection", t, LDFailure.FailureType.NETWORK_FAILURE)); } } - }; - - HttpConnectStrategy connectStrategy = ConnectStrategy.http(getUri(context)) - .clientBuilderActions(clientBuilder -> { - httpProperties.applyToHttpClientBuilder(clientBuilder); - clientBuilder.readTimeout(READ_TIMEOUT_MS, TimeUnit.MILLISECONDS); - }) - .requestTransformer(input -> - input.newBuilder() - .headers( - input.headers().newBuilder().addAll(httpProperties.toHeadersBuilder().build()).build() - ).build()); - - if (useReport) { - connectStrategy = connectStrategy.methodAndBody(METHOD_REPORT, getRequestBody(context)); + resultCallback.onError(failure); } + }; + } - EventSource.Builder esBuilder = new EventSource.Builder(connectStrategy) - .retryDelay(initialReconnectDelayMillis, TimeUnit.MILLISECONDS) - .retryDelayStrategy(RetryDelayStrategy.defaultStrategy() - .maxDelay(MAX_RECONNECT_TIME_MS, TimeUnit.MILLISECONDS)); - - eventSourceStarted = System.currentTimeMillis(); - es = new BackgroundEventSource.Builder(handler, esBuilder) - // The stream thread asks this handler, before it reconnects, whether an - // error ends the stream. The onError callback above runs on a different - // thread, so a stop() from there can arrive after a fast reconnect has - // already opened a new connection. A decision made here cannot lose that race. - .connectionErrorHandler(t -> { - if (t instanceof StreamHttpErrorException && - !LDUtil.isHttpErrorRecoverable(((StreamHttpErrorException) t).getCode())) { - return ConnectionErrorHandler.Action.SHUTDOWN; - } - return ConnectionErrorHandler.Action.PROCEED; - }) - .build(); - es.start(); + /** + * Opens a new stream connection if this data source is still running. Called from + * {@link #start} and from the reconnect task scheduled by {@link #scheduleReconnectAfterFailure}. + */ + private synchronized void connect() { + if (!running) { + return; + } + pendingReconnect = null; + + HttpConnectStrategy connectStrategy = ConnectStrategy.http(getUri(context)) + .clientBuilderActions(clientBuilder -> { + httpProperties.applyToHttpClientBuilder(clientBuilder); + clientBuilder.readTimeout(READ_TIMEOUT_MS, TimeUnit.MILLISECONDS); + }) + .requestTransformer(input -> + input.newBuilder() + .headers( + input.headers().newBuilder().addAll(httpProperties.toHeadersBuilder().build()).build() + ).build()); + + if (useReport) { + connectStrategy = connectStrategy.methodAndBody(METHOD_REPORT, getRequestBody(context)); + } - running = true; + EventSource.Builder esBuilder = new EventSource.Builder(connectStrategy); + + eventSourceStarted = System.currentTimeMillis(); + // The previous BackgroundEventSource, if any, has already shut itself down: the connection + // error handler below ends the stream after every failure, and BackgroundEventSource + // closes itself when that happens. + es = new BackgroundEventSource.Builder(handler, esBuilder) + // The stream thread asks this handler, before it would reconnect, whether an + // error ends the stream. It always does: this class schedules its own reconnect + // from onError with a delay from RetryState. Deciding here, on the stream + // thread, means the library can never race ahead with a reconnect of its own. + .connectionErrorHandler(t -> ConnectionErrorHandler.Action.SHUTDOWN) + .build(); + es.start(); + } + + /** + * Records a stream failure and schedules the next connection attempt. + * + * @param unexpected whether the failure is classified as {@code unexpected} + * @return the delay before the next attempt in milliseconds, or -1 if the data source has been + * stopped and no reconnect was scheduled + */ + private synchronized long scheduleReconnectAfterFailure(boolean unexpected) { + if (!running) { + return -1; + } + retryState.recordFailure(unexpected, System.currentTimeMillis()); + long delay = retryState.nextDelayMillis(); + if (pendingReconnect != null) { + pendingReconnect.cancel(false); } + pendingReconnect = taskExecutor.scheduleTask(this::connect, delay); + return delay; } @NonNull @@ -283,12 +351,23 @@ public boolean needsRefresh(boolean newInBackground, LDContext newEvaluationCont (newInBackground && !streamEvenInBackground); } - private synchronized void stopSync() { - if (es != null) { - es.close(); + private void stopSync() { + BackgroundEventSource esToClose; + synchronized (this) { + running = false; + // A pending backoff wait is interrupted immediately by shutdown. + if (pendingReconnect != null) { + pendingReconnect.cancel(false); + pendingReconnect = null; + } + esToClose = es; + es = null; + } + // Closing waits for the EventSource's threads to finish, so it is done outside the lock + // in case one of those threads is in a callback that needs the lock. + if (esToClose != null) { + esToClose.close(); } - running = false; - es = null; logger.debug("Stopped."); } diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java index 8afcb6589..9ad028891 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java @@ -1,27 +1,45 @@ package com.launchdarkly.sdk.android; import androidx.annotation.NonNull; +import androidx.annotation.Nullable; import com.launchdarkly.sdk.android.subsystems.Synchronizer; +import java.util.concurrent.ScheduledFuture; + /** - * Wraps a synchronizer factory with availability state (available/blocked). - * Used by {@link SourceManager} to skip synchronizers that have been blocked (e.g. after TERMINAL_ERROR). + * Wraps a synchronizer factory with availability state. + * Used by {@link SourceManager} to skip synchronizers that are not currently usable: either + * because they are waiting out a backoff after an unexpected error, or because they are blocked + * (for example the FDv1 fallback synchronizer before the server has directed the SDK to it). *

- * Package-private for internal use by FDv2DataSource. + * Package-private for internal use by FDv2DataSource. Callers synchronize on the + * {@link SourceManager}'s lock. */ final class SynchronizerFactoryWithState { enum State { /** This synchronizer is available to use. */ Available, - /** This synchronizer is no longer available (e.g. after TERMINAL_ERROR). */ + /** + * This synchronizer reported an unexpected error and is waiting out a backoff before it + * may be used again. + */ + BackingOff, + /** This synchronizer is not available until something unblocks it. */ Blocked } private final FDv2DataSource.DataSourceFactory factory; private State state = State.Available; private final boolean isFDv1Fallback; + // Backoff after unexpected errors from this slot's synchronizers. Every failure recorded here + // is unexpected, so only the extended regime ever applies. + private final RetryState retryState = RetryState.forSynchronizerSlot(); + @Nullable + private ScheduledFuture pendingUnblock; + @Nullable + private String lastSynchronizerName; SynchronizerFactoryWithState(@NonNull FDv2DataSource.DataSourceFactory factory) { this(factory, false); @@ -51,4 +69,61 @@ Synchronizer build() { boolean isFDv1Fallback() { return isFDv1Fallback; } + + /** + * Records healthy operation by this slot's synchronizer. + * + * @param nowMillis the current time + */ + void recordHealthy(long nowMillis) { + retryState.recordSuccess(nowMillis); + } + + /** + * Records an unexpected error from this slot's synchronizer and puts the slot into backoff. + * + * @param synchronizerName the name of the synchronizer that failed, for logging + * @param nowMillis the current time + * @return how long the slot stays in backoff, in milliseconds + */ + long startBackoff(@NonNull String synchronizerName, long nowMillis) { + retryState.recordFailure(true, nowMillis); + state = State.BackingOff; + lastSynchronizerName = synchronizerName; + return retryState.nextDelayMillis(); + } + + /** + * Ends this slot's backoff, if it is in one. + * + * @return true if the slot became available + */ + boolean endBackoff() { + pendingUnblock = null; + if (state != State.BackingOff) { + return false; + } + state = State.Available; + return true; + } + + /** + * @return the name of the synchronizer whose unexpected error started the current backoff, + * or null if there has been none + */ + @Nullable + String getLastSynchronizerName() { + return lastSynchronizerName; + } + + void setPendingUnblock(@Nullable ScheduledFuture pendingUnblock) { + this.pendingUnblock = pendingUnblock; + } + + void cancelPendingUnblock() { + if (pendingUnblock != null) { + pendingUnblock.cancel(false); + pendingUnblock = null; + } + } } diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceConditionsTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceConditionsTest.java index 7acade435..854a49a9f 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceConditionsTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceConditionsTest.java @@ -22,6 +22,7 @@ import java.util.concurrent.Future; import java.util.concurrent.ScheduledExecutorService; import java.util.concurrent.TimeUnit; +import java.util.concurrent.atomic.AtomicBoolean; import java.util.concurrent.TimeoutException; import static org.junit.Assert.assertEquals; @@ -196,6 +197,28 @@ public void recovery_informDoesNothing() throws Exception { assertEquals(ConditionType.RECOVERY, type); } + @Test + public void recovery_gateClosed_rearmsTimerUntilGateOpens() throws Exception { + FakeScheduledExecutorService fakeExecutor = new FakeScheduledExecutorService(); + AtomicBoolean canRecover = new AtomicBoolean(false); + try { + RecoveryCondition condition = new RecoveryCondition(fakeExecutor, 1, canRecover::get); + assertEquals(1000, fakeExecutor.awaitScheduledDelayMillis(1000)); + + // With the gate closed, the timer firing re-arms it instead of completing the future. + fakeExecutor.advanceTime(1000); + assertEquals(1000, fakeExecutor.awaitScheduledDelayMillis(1000)); + assertFalse(condition.getFuture().isDone()); + + // With the gate open, the next firing completes the future. + canRecover.set(true); + fakeExecutor.advanceTime(1000); + assertEquals(ConditionType.RECOVERY, condition.getFuture().get(1, TimeUnit.SECONDS)); + } finally { + fakeExecutor.shutdownNow(); + } + } + // ==== Conditions (wrapper) ==== @Test diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java index 33ecfc61a..1981479d2 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java @@ -4,6 +4,7 @@ import static org.junit.Assert.assertFalse; import static org.junit.Assert.assertNotNull; import static org.junit.Assert.assertNull; +import static org.junit.Assert.assertSame; import static org.junit.Assert.assertTrue; import static org.junit.Assert.fail; @@ -59,6 +60,7 @@ public class FDv2DataSourceTest { private static final long ORCHESTRATION_LOG_AWAIT_TIMEOUT_MS = AWAIT_TIMEOUT_SECONDS * 1000L; private ScheduledExecutorService executor; + private final FakeScheduledExecutorService fakeExecutor = new FakeScheduledExecutorService(); @Before public void setUp() { @@ -70,6 +72,7 @@ public void tearDown() { if (executor != null && !executor.isShutdown()) { executor.shutdownNow(); } + fakeExecutor.shutdownNow(); } private FDv2DataSource buildDataSource( @@ -104,6 +107,25 @@ private FDv2DataSource buildDataSource( recoveryTimeoutSeconds); } + /** + * Builds a data source whose timers run on the given executor; tests of backoff behavior + * pass the fake executor so they can move the clock instead of waiting. + */ + private FDv2DataSource buildDataSource( + MockComponents.MockDataSourceUpdateSink sink, + List> initializers, + List> synchronizers, + ScheduledExecutorService executor) { + return new FDv2DataSource( + CONTEXT, + initializers, + synchronizers, + null, + sink, + executor, + logging.logger); + } + /** Starts the data source and returns a callback that will receive the start result. */ private AwaitableCallback startDataSource(FDv2DataSource dataSource) { AwaitableCallback cb = new AwaitableCallback<>(); @@ -607,26 +629,41 @@ public void terminalErrorBlocksSynchronizer() throws Exception { } @Test - public void allThreeSynchronizersFailReportsExhaustion() throws Exception { + public void allSynchronizersFailingWithUnexpectedErrorsAreRetriedAfterBackoff() throws Exception { MockComponents.MockDataSourceUpdateSink sink = new MockComponents.MockDataSourceUpdateSink(); + AtomicInteger firstBuilds = new AtomicInteger(0); + AtomicInteger secondBuilds = new AtomicInteger(0); + // Both synchronizers fail with an unexpected error the first time they are built, and + // deliver data the second time. FDv2DataSource dataSource = buildDataSource(sink, Collections.emptyList(), Arrays.asList( - () -> new MockQueuedSynchronizer(terminalError()), - () -> new MockQueuedSynchronizer(terminalError()), - () -> new MockQueuedSynchronizer(terminalError()))); - - AwaitableCallback startCallback = startDataSource(dataSource); - awaitExpectingError(startCallback); - - List statuses = sink.awaitStatuses(4, AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS); - assertEquals(4, statuses.size()); - assertEquals(DataSourceState.INTERRUPTED, statuses.get(0)); - assertEquals(DataSourceState.INTERRUPTED, statuses.get(1)); - assertEquals(DataSourceState.INTERRUPTED, statuses.get(2)); - assertEquals(DataSourceState.OFF, statuses.get(3)); - assertNotNull(sink.getLastError()); + () -> firstBuilds.incrementAndGet() == 1 + ? new MockQueuedSynchronizer(terminalError()) + : new MockQueuedSynchronizer(FDv2SourceResult.changeSet(makeChangeSet(false), false)), + () -> secondBuilds.incrementAndGet() == 1 + ? new MockQueuedSynchronizer(terminalError()) + : new MockQueuedSynchronizer(FDv2SourceResult.changeSet(makeChangeSet(false), false))), + fakeExecutor); + AwaitableCallback startCallback = startDataSource(dataSource); + + // Each failure is reported as an interruption, and the data source waits for a backoff + // to end rather than reporting OFF. Nothing is rebuilt before the shortest possible + // backoff has elapsed. + assertEquals(Arrays.asList(DataSourceState.INTERRUPTED, DataSourceState.INTERRUPTED), + sink.awaitStatuses(2, AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS)); + fakeExecutor.advanceTime(RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 - 1); + assertEquals(1, firstBuilds.get()); + assertEquals(1, secondBuilds.get()); + + // Once the backoffs end, whichever synchronizer returns first is tried again and + // initialization completes. The two backoffs have independent jitter, so either may be + // the one that is rebuilt. + fakeExecutor.advanceTime(RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 + 1); + assertTrue(startCallback.await(AWAIT_TIMEOUT_SECONDS * 1000)); + assertEquals(3, firstBuilds.get() + secondBuilds.get()); + stopDataSource(dataSource); } @Test @@ -650,20 +687,6 @@ public void blockedSynchronizerSkippedInRotation() throws Exception { stopDataSource(dataSource); } - @Test - public void allSynchronizersBlockedReturnsNullAndExits() throws Exception { - MockComponents.MockDataSourceUpdateSink sink = new MockComponents.MockDataSourceUpdateSink(); - - FDv2DataSource dataSource = buildDataSource(sink, - Collections.emptyList(), - Arrays.asList( - () -> new MockQueuedSynchronizer(terminalError()), - () -> new MockQueuedSynchronizer(terminalError()))); - - AwaitableCallback startCallback = startDataSource(dataSource); - awaitExpectingError(startCallback); - } - @Test public void recoveryResetsToFirstAvailableSynchronizer() throws Exception { MockComponents.MockDataSourceUpdateSink sink = new MockComponents.MockDataSourceUpdateSink(); @@ -1464,19 +1487,18 @@ public void statusIncludesErrorInfoOnFailure() throws Exception { FDv2DataSource dataSource = buildDataSource(sink, Collections.emptyList(), Collections.singletonList(() -> new MockQueuedSynchronizer( - FDv2SourceResult.status(FDv2SourceResult.Status.terminalError(terminalErr), false)))); + FDv2SourceResult.status(FDv2SourceResult.Status.terminalError(terminalErr), false))), + fakeExecutor); - AwaitableCallback startCallback = startDataSource(dataSource); - awaitExpectingError(startCallback); + startDataSource(dataSource); + // The unexpected error is reported as an interruption that carries the error, and the + // data source then waits out the synchronizer's backoff rather than going OFF. DataSourceState first = sink.awaitStatus(AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS); assertEquals(DataSourceState.INTERRUPTED, first); - - DataSourceState second = sink.awaitStatus(AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS); - assertEquals(DataSourceState.OFF, second); - - assertEquals(DataSourceState.OFF, sink.getLastState()); - assertNotNull(sink.getLastError()); + assertEquals(DataSourceState.INTERRUPTED, sink.getLastState()); + assertSame(terminalErr, sink.getLastError()); + stopDataSource(dataSource); } @Test @@ -1500,25 +1522,34 @@ public void statusRemainsValidDuringSynchronizerOperation() throws Exception { } @Test - public void statusTransitionsFromValidToOffWhenAllSynchronizersFail() throws Exception { + public void statusStaysInterruptedWhileTheOnlySynchronizerBacksOffThenReturnsToValid() throws Exception { MockComponents.MockDataSourceUpdateSink sink = new MockComponents.MockDataSourceUpdateSink(); RuntimeException err = new RuntimeException("server error"); + AtomicInteger builds = new AtomicInteger(0); + // The synchronizer delivers data and then fails with an unexpected error; when it is + // built again it delivers data. FDv2DataSource dataSource = buildDataSource(sink, Collections.emptyList(), - Collections.singletonList(() -> new MockQueuedSynchronizer( - FDv2SourceResult.changeSet(makeChangeSet(false), false), - FDv2SourceResult.status(FDv2SourceResult.Status.terminalError(err), false)))); + Collections.singletonList(() -> builds.incrementAndGet() == 1 + ? new MockQueuedSynchronizer( + FDv2SourceResult.changeSet(makeChangeSet(false), false), + FDv2SourceResult.status(FDv2SourceResult.Status.terminalError(err), false)) + : new MockQueuedSynchronizer(FDv2SourceResult.changeSet(makeChangeSet(false), false))), + fakeExecutor); AwaitableCallback startCallback = startDataSource(dataSource); - assertTrue(startCallback.await(AWAIT_TIMEOUT_SECONDS * 1000)); // changeset arrives first, so start succeeds - + assertTrue(startCallback.await(AWAIT_TIMEOUT_SECONDS * 1000)); assertEquals(DataSourceState.VALID, sink.awaitStatus(AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS)); assertEquals(DataSourceState.INTERRUPTED, sink.awaitStatus(AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS)); - assertEquals(DataSourceState.OFF, sink.awaitStatus(AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS)); - - assertEquals(DataSourceState.OFF, sink.getLastState()); assertNotNull(sink.getLastError()); + + // Once the backoff ends the synchronizer is tried again and the status returns to VALID, + // never having reached OFF. + fakeExecutor.advanceTime(RetryState.EXTENDED_INITIAL_DELAY_MILLIS); + assertEquals(DataSourceState.VALID, sink.awaitStatus(AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS)); + assertEquals(2, builds.get()); + stopDataSource(dataSource); } @Test @@ -2079,29 +2110,30 @@ public void orchestrationLogging_fdv1Fallback_logsInfo() throws Exception { } @Test - public void orchestrationLogging_permanentFailure_logsWarn() throws Exception { + public void orchestrationLogging_unexpectedError_logsWarn() throws Exception { MockComponents.MockDataSourceUpdateSink sink = new MockComponents.MockDataSourceUpdateSink(); FDv2DataSource dataSource = buildDataSource(sink, Collections.emptyList(), - Collections.singletonList(() -> new MockQueuedSynchronizer(terminalError()))); - AwaitableCallback startCallback = startDataSource(dataSource); - awaitExpectingError(startCallback); + Collections.singletonList(() -> new MockQueuedSynchronizer(terminalError())), + fakeExecutor); + startDataSource(dataSource); awaitLogContains(logging, - "Synchronizer 'MockQueuedSynchronizer' permanently failed and will not be used again until application restart."); + "Synchronizer 'MockQueuedSynchronizer' reported an unexpected error and will not be tried again for"); + stopDataSource(dataSource); } @Test - public void orchestrationLogging_allSynchronizersExhausted_logsWarn() throws Exception { + public void orchestrationLogging_allSynchronizersBackingOff_logsInfo() throws Exception { MockComponents.MockDataSourceUpdateSink sink = new MockComponents.MockDataSourceUpdateSink(); FDv2DataSource dataSource = buildDataSource(sink, Collections.emptyList(), Arrays.asList( () -> new MockQueuedSynchronizer(terminalError()), - () -> new MockQueuedSynchronizer(terminalError()), - () -> new MockQueuedSynchronizer(terminalError()))); - AwaitableCallback startCallback = startDataSource(dataSource); - awaitExpectingError(startCallback); - awaitLogContains(logging, "No more synchronizers available."); + () -> new MockQueuedSynchronizer(terminalError())), + fakeExecutor); + startDataSource(dataSource); + awaitLogContains(logging, "All synchronizers are waiting out a backoff after unexpected errors"); + stopDataSource(dataSource); } @Test diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java index 9aa3891d9..055ab9405 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java @@ -21,7 +21,8 @@ import java.io.IOException; import java.net.URI; import java.util.HashMap; -import java.util.concurrent.ExecutorService; +import java.util.Random; +import java.util.concurrent.ScheduledExecutorService; import java.util.concurrent.Executors; import java.util.concurrent.Future; import java.util.concurrent.TimeUnit; @@ -38,11 +39,13 @@ public class FDv2StreamingSynchronizerTest { @Rule public Timeout globalTimeout = Timeout.seconds(10); - private final ExecutorService executor = Executors.newCachedThreadPool(); + private final ScheduledExecutorService executor = Executors.newScheduledThreadPool(4); + private final FakeScheduledExecutorService fakeExecutor = new FakeScheduledExecutorService(); @After public void tearDown() { executor.shutdownNow(); + fakeExecutor.shutdownNow(); } private static final LDContext CONTEXT = LDContext.create("test-context"); @@ -99,6 +102,32 @@ private FDv2StreamingSynchronizer makeSynchronizer( httpProperties(), executor, LOGGER, null); } + // The tests of backoff behavior drive the synchronizer's timers with a + // FakeScheduledExecutorService and a RetryState without jitter, so the delay chosen for each + // reconnect can be asserted exactly instead of waited for. + private static final long NORMAL_DELAY_MILLIS = 1000; + private static final long NORMAL_MAX_DELAY_MILLIS = 4000; + // A healthy-operation threshold no test reaches. + private static final long NEVER_RESET_MILLIS = 60_000; + + private static RetryState retryStateWithoutJitter(long healthyResetThresholdMillis) { + return new RetryState(NORMAL_DELAY_MILLIS, NORMAL_MAX_DELAY_MILLIS, + NORMAL_DELAY_MILLIS, NORMAL_MAX_DELAY_MILLIS, 0, healthyResetThresholdMillis, 0, + new Random() { + @Override + public double nextDouble() { + return 0; + } + }); + } + + private FDv2StreamingSynchronizer makeSynchronizer(URI streamBaseUri, RetryState retryState) { + return new FDv2StreamingSynchronizer( + CONTEXT, EMPTY_SELECTOR_SOURCE, streamBaseUri, STREAM_PATH, + null, 1, false, false, + httpProperties(), fakeExecutor, LOGGER, null, retryState); + } + private static DiagnosticStore basicDiagnosticStore() { return new DiagnosticStore(new DiagnosticStore.SdkDiagnosticParams( "mobile-key", "android-client-sdk", "1.0.0", "Android", null, null, null)); @@ -337,6 +366,96 @@ public void httpNonRecoverableError() throws Exception { } } + // ---- backoff after failures ---- + + @Test + public void recoverableErrorSchedulesReconnectWithGrowingDelay() throws Exception { + String serverIntent = makeEvent("server-intent", "{\"payloads\":[{\"id\":\"payload-1\",\"target\":100,\"intentCode\":\"xfer-full\",\"reason\":\"payload-missing\"}]}"); + String payloadTransferred = makeEvent("payload-transferred", "{\"state\":\"(p:payload-1:100)\",\"version\":100}"); + + try (HttpServer server = HttpServer.start(Handlers.sequential( + Handlers.status(503), + Handlers.status(503), + Handlers.all( + Handlers.SSE.start(), + Handlers.SSE.event(serverIntent), + Handlers.SSE.event(payloadTransferred), + Handlers.SSE.leaveOpen())))) { + + FDv2StreamingSynchronizer sync = makeSynchronizer(server.getUri(), retryStateWithoutJitter(NEVER_RESET_MILLIS)); + + // Each 503 is reported as an interruption, and the reconnect is scheduled with double + // the previous delay. + assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); + assertEquals(NORMAL_DELAY_MILLIS, fakeExecutor.awaitScheduledDelayMillis(5000)); + fakeExecutor.advanceTime(NORMAL_DELAY_MILLIS); + assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); + assertEquals(NORMAL_DELAY_MILLIS * 2, fakeExecutor.awaitScheduledDelayMillis(5000)); + + // Once that delay has passed, the synchronizer reconnects and delivers data. + fakeExecutor.advanceTime(NORMAL_DELAY_MILLIS * 2); + assertEquals(SourceResultType.CHANGE_SET, sync.next().get(5, TimeUnit.SECONDS).getResultType()); + + sync.close(); + } + } + + @Test + public void healthyStreamResetsBackoffToInitialDelay() throws Exception { + String serverIntent = makeEvent("server-intent", "{\"payloads\":[{\"id\":\"payload-1\",\"target\":100,\"intentCode\":\"xfer-full\",\"reason\":\"payload-missing\"}]}"); + String payloadTransferred = makeEvent("payload-transferred", "{\"state\":\"(p:payload-1:100)\",\"version\":100}"); + // The synchronizer measures healthy operation on its own clock, so the server must hold + // the stream open for a moment after the data before ending it. This is the shortest + // margin that reliably exceeds the 1 ms threshold used here. + long healthyMarginMillis = 20; + + try (HttpServer server = HttpServer.start(Handlers.sequential( + Handlers.status(503), + Handlers.all( + Handlers.SSE.start(), + Handlers.SSE.event(serverIntent), + Handlers.SSE.event(payloadTransferred), + Handlers.delay(healthyMarginMillis)), + Handlers.all( + Handlers.SSE.start(), + Handlers.SSE.event(serverIntent), + Handlers.SSE.event(payloadTransferred), + Handlers.SSE.leaveOpen())))) { + + FDv2StreamingSynchronizer sync = makeSynchronizer(server.getUri(), retryStateWithoutJitter(1)); + + // The 503 costs one attempt; the reconnect then delivers data and is ended by the + // server. + assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); + assertEquals(NORMAL_DELAY_MILLIS, fakeExecutor.awaitScheduledDelayMillis(5000)); + fakeExecutor.advanceTime(NORMAL_DELAY_MILLIS); + assertEquals(SourceResultType.CHANGE_SET, sync.next().get(5, TimeUnit.SECONDS).getResultType()); + assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); + + // Having been healthy for longer than the threshold, the next delay starts over + // instead of doubling. + assertEquals(NORMAL_DELAY_MILLIS, fakeExecutor.awaitScheduledDelayMillis(5000)); + fakeExecutor.advanceTime(NORMAL_DELAY_MILLIS); + assertEquals(SourceResultType.CHANGE_SET, sync.next().get(5, TimeUnit.SECONDS).getResultType()); + + sync.close(); + } + } + + @Test + public void closeDuringBackoffCancelsReconnect() throws Exception { + try (HttpServer server = HttpServer.start(Handlers.status(503))) { + FDv2StreamingSynchronizer sync = makeSynchronizer(server.getUri(), retryStateWithoutJitter(NEVER_RESET_MILLIS)); + assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); + assertEquals(NORMAL_DELAY_MILLIS, fakeExecutor.awaitScheduledDelayMillis(5000)); + + // Closing cancels the scheduled attempt and reports shutdown. + sync.close(); + assertTrue(fakeExecutor.pendingDelaysMillis().isEmpty()); + assertEquals(SourceSignal.SHUTDOWN, sync.next().get(1, TimeUnit.SECONDS).getStatus().getState()); + } + } + @Test public void httpRecoverableError() throws Exception { try (HttpServer server = HttpServer.start(Handlers.status(503))) { diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/LDUtilTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/LDUtilTest.java index 01b1d4114..77ff0b583 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/LDUtilTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/LDUtilTest.java @@ -27,4 +27,36 @@ public void testSanitizeSpaces() { Assert.assertEquals("--hello--", LDUtil.sanitizeSpaces(" hello ")); Assert.assertEquals("world", LDUtil.sanitizeSpaces("world")); } + + @Test + public void isHttpErrorRecoverableClassifiesStatusCodes() { + // 400, 408, 429, and 5xx are normal; every other 4xx is unexpected. + Assert.assertTrue(LDUtil.isHttpErrorRecoverable(400)); + Assert.assertTrue(LDUtil.isHttpErrorRecoverable(408)); + Assert.assertTrue(LDUtil.isHttpErrorRecoverable(429)); + Assert.assertTrue(LDUtil.isHttpErrorRecoverable(500)); + Assert.assertTrue(LDUtil.isHttpErrorRecoverable(503)); + Assert.assertTrue(LDUtil.isHttpErrorRecoverable(302)); + + Assert.assertFalse(LDUtil.isHttpErrorRecoverable(401)); + Assert.assertFalse(LDUtil.isHttpErrorRecoverable(403)); + Assert.assertFalse(LDUtil.isHttpErrorRecoverable(404)); + Assert.assertFalse(LDUtil.isHttpErrorRecoverable(405)); + } + + @Test + public void isUnexpectedFailureIsTrueOnlyForUnexpectedHttpStatuses() { + Assert.assertTrue(LDUtil.isUnexpectedFailure(new LDInvalidResponseCodeFailure("x", 401, false))); + Assert.assertTrue(LDUtil.isUnexpectedFailure(new LDInvalidResponseCodeFailure("x", 403, false))); + // Classification is by status code, not by the retryable flag the failure was built with. + Assert.assertTrue(LDUtil.isUnexpectedFailure(new LDInvalidResponseCodeFailure("x", 401, true))); + + Assert.assertFalse(LDUtil.isUnexpectedFailure(new LDInvalidResponseCodeFailure("x", 429, true))); + Assert.assertFalse(LDUtil.isUnexpectedFailure(new LDInvalidResponseCodeFailure("x", 500, true))); + // Malformed bodies and transport errors are normal failures. + Assert.assertFalse(LDUtil.isUnexpectedFailure(new LDFailure("x", LDFailure.FailureType.INVALID_RESPONSE_BODY))); + Assert.assertFalse(LDUtil.isUnexpectedFailure(new LDFailure("x", LDFailure.FailureType.NETWORK_FAILURE))); + Assert.assertFalse(LDUtil.isUnexpectedFailure(new RuntimeException("x"))); + Assert.assertFalse(LDUtil.isUnexpectedFailure(null)); + } } diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java index d16fdbc05..76068bfc8 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java @@ -22,7 +22,11 @@ import org.junit.Test; import java.io.IOException; +import java.util.ArrayList; +import java.util.Collections; +import java.util.List; import java.util.Map; +import java.util.Random; import java.util.concurrent.BlockingQueue; import java.util.concurrent.LinkedBlockingQueue; import java.util.concurrent.ScheduledFuture; @@ -271,8 +275,6 @@ public void terminatesAfterMaxNumberOfPolls() throws Exception { try { ds.start(LDUtil.noOpCallback()); - ScheduledFuture pollTask = ds.currentPollTask.get(); - assertFalse(pollTask.isCancelled()); LDContext context1 = requireValue(fetcher.receivedContexts, 500, TimeUnit.MILLISECONDS); @@ -280,12 +282,193 @@ public void terminatesAfterMaxNumberOfPolls() throws Exception { // if a third request is sent, this will fail here requireNoMoreValues(fetcher.receivedContexts, 200, TimeUnit.MILLISECONDS); - assertTrue(pollTask.isCancelled()); + ScheduledFuture pollTask = ds.currentPollTask.get(); + assertTrue("no further poll should be pending", pollTask == null || pollTask.isDone()); } finally { ds.stop(LDUtil.noOpCallback()); } } + // --- backoff after failures --- + // + // These tests drive the data source's timers with a FakeTaskExecutor and a RetryState + // without jitter. The mock fetcher answers synchronously, so every poll and its outcome + // happen inside advanceTime() and the scheduled delays can be asserted exactly. + + private static final long POLL_INTERVAL_MILLIS = 30_000; + private static final long EXTENDED_DELAY_MILLIS = 300_000; + + private final FakeTaskExecutor fakeTaskExecutor = new FakeTaskExecutor(); + + private static LDInvalidResponseCodeFailure httpFailure(int status) { + return new LDInvalidResponseCodeFailure("test failure", status, LDUtil.isHttpErrorRecoverable(status)); + } + + private static RetryState retryStateWithoutJitter() { + return new RetryState(POLL_INTERVAL_MILLIS, POLL_INTERVAL_MILLIS, + EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 4, POLL_INTERVAL_MILLIS, + 0, RetryState.POLLING_RESET_THRESHOLD_SUCCESSES, + new Random() { + @Override + public double nextDouble() { + return 0; + } + }); + } + + private PollingDataSource makePollingDataSource(long maxNumberOfPolls) { + ClientContextImpl clientContext = makeClientContext(false, null); + return new PollingDataSource( + clientContext.getEvaluationContext(), + clientContext.getDataSourceUpdateSink(), + 0, + POLL_INTERVAL_MILLIS, + maxNumberOfPolls, + clientContext.getFetcher(), + clientContext.getPlatformState(), + fakeTaskExecutor, + retryStateWithoutJitter(), + clientContext.getBaseLogger() + ); + } + + private static class TrackingCallback implements Callback { + final List successes = new ArrayList<>(); + final List errors = new ArrayList<>(); + + @Override + public void onSuccess(Boolean result) { + successes.add(result != null ? result : false); + } + + @Override + public void onError(Throwable error) { + errors.add(error); + } + } + + @Test + public void normalErrorIsPolledAgainAfterPollInterval() { + PollingDataSource ds = makePollingDataSource(Long.MAX_VALUE); + fetcher.setupErrorResponse(httpFailure(500)); + fetcher.setupSuccessResponse("{}"); + TrackingCallback callback = new TrackingCallback(); + + // The first poll fails with a 500. + ds.start(callback); + fakeTaskExecutor.runDueTasks(); + assertEquals(1, callback.errors.size()); + + // The next poll is scheduled for the regular interval and succeeds. + assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + fakeTaskExecutor.advanceTime(POLL_INTERVAL_MILLIS); + assertEquals(1, callback.successes.size()); + assertEquals(2, fetcher.receivedContexts.size()); + } + + @Test + public void unexpectedErrorIsPolledAgainAfterExtendedDelay() { + PollingDataSource ds = makePollingDataSource(Long.MAX_VALUE); + fetcher.setupErrorResponse(httpFailure(401)); + fetcher.setupSuccessResponse("{}"); + TrackingCallback callback = new TrackingCallback(); + + // The first poll fails with a 401, which is reported without shutting the SDK down. + ds.start(callback); + fakeTaskExecutor.runDueTasks(); + assertEquals(401, ((LDInvalidResponseCodeFailure) callback.errors.get(0)).getResponseCode()); + assertFalse(dataSourceUpdateSink.shutDownCalled); + + // The next poll is scheduled for the extended delay rather than the poll interval, and + // succeeds once that delay has passed. + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); + assertEquals(1, callback.successes.size()); + assertEquals(2, fetcher.receivedContexts.size()); + } + + @Test + public void sustainedUnexpectedErrorsKeepPollingWithGrowingDelay() { + PollingDataSource ds = makePollingDataSource(Long.MAX_VALUE); + fetcher.setupErrorResponse(httpFailure(401)); + fetcher.setupErrorResponse(httpFailure(403)); + fetcher.setupErrorResponse(httpFailure(405)); + fetcher.setupSuccessResponse("{}"); + TrackingCallback callback = new TrackingCallback(); + + // Each failure schedules the next poll with double the delay, and the data source never + // gives up. + ds.start(callback); + fakeTaskExecutor.runDueTasks(); + for (long expectedDelay : new long[] {EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 2, EXTENDED_DELAY_MILLIS * 4}) { + assertEquals(Collections.singletonList(expectedDelay), fakeTaskExecutor.pendingDelaysMillis()); + fakeTaskExecutor.advanceTime(expectedDelay); + } + + // The fourth poll succeeded. + assertEquals(3, callback.errors.size()); + assertEquals(1, callback.successes.size()); + } + + @Test + public void twoConsecutiveSuccessfulPollsResetBackoff() { + PollingDataSource ds = makePollingDataSource(Long.MAX_VALUE); + fetcher.setupErrorResponse(httpFailure(401)); + fetcher.setupSuccessResponse("{}"); + fetcher.setupSuccessResponse("{}"); + fetcher.setupErrorResponse(httpFailure(500)); + fetcher.setupSuccessResponse("{}"); + TrackingCallback callback = new TrackingCallback(); + + // The 401 moves the data source to the extended delay. + ds.start(callback); + fakeTaskExecutor.runDueTasks(); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + + // One success returns to the regular interval; a second one clears the backoff, so the + // 500 that follows is retried at the regular interval instead of a doubled extended delay. + fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); + assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + fakeTaskExecutor.advanceTime(POLL_INTERVAL_MILLIS); + assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + fakeTaskExecutor.advanceTime(POLL_INTERVAL_MILLIS); + assertEquals(2, callback.errors.size()); + assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + } + + @Test + public void oneShotPollIsNotRetriedAfterFailure() { + PollingDataSource ds = makePollingDataSource(1); + fetcher.setupErrorResponse(httpFailure(500)); + fetcher.setupSuccessResponse("{}"); + TrackingCallback callback = new TrackingCallback(); + + // The single poll fails. + ds.start(callback); + fakeTaskExecutor.runDueTasks(); + assertEquals(1, callback.errors.size()); + + // A one-shot data source is done after its one poll, so nothing further is scheduled. + assertTrue(fakeTaskExecutor.pendingDelaysMillis().isEmpty()); + } + + @Test + public void stopCancelsPendingPoll() { + PollingDataSource ds = makePollingDataSource(Long.MAX_VALUE); + fetcher.setupErrorResponse(httpFailure(401)); + fetcher.setupSuccessResponse("{}"); + TrackingCallback callback = new TrackingCallback(); + + // The 401 schedules a poll for the extended delay. + ds.start(callback); + fakeTaskExecutor.runDueTasks(); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + + // Stopping cancels it. + ds.stop(LDUtil.noOpCallback()); + assertTrue(fakeTaskExecutor.pendingDelaysMillis().isEmpty()); + } + private class MockFetcher implements FeatureFetcher { BlockingQueue receivedContexts = new LinkedBlockingQueue<>(); BlockingQueue responses = new LinkedBlockingQueue<>(); diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java new file mode 100644 index 000000000..0757c3aca --- /dev/null +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java @@ -0,0 +1,376 @@ +package com.launchdarkly.sdk.android; + +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertFalse; +import static org.junit.Assert.assertTrue; + +import org.junit.Test; + +import java.util.Random; + +/** + * Unit tests for {@link RetryState}: two-regime exponential backoff with jitter. + */ +public class RetryStateTest { + private static final long SECOND = 1_000L; + private static final long MINUTE = 60 * SECOND; + + /** A random source that always yields the same fraction, so jitter is predictable. */ + private static Random fixedRandom(final double fraction) { + return new Random() { + @Override + public double nextDouble() { + return fraction; + } + }; + } + + private static final Random NO_JITTER = fixedRandom(0); + private static final Random MAX_JITTER = fixedRandom(0.999_999); + + private static RetryState streaming(Random random) { + return new RetryState( + SECOND, + RetryState.NORMAL_MAX_DELAY_MILLIS, + RetryState.EXTENDED_INITIAL_DELAY_MILLIS, + RetryState.EXTENDED_MAX_DELAY_MILLIS, + 0, + RetryState.STREAMING_RESET_THRESHOLD_MILLIS, + 0, + random); + } + + private static RetryState polling(long pollIntervalMillis, Random random) { + return new RetryState( + pollIntervalMillis, + pollIntervalMillis, + Math.max(RetryState.EXTENDED_INITIAL_DELAY_MILLIS, pollIntervalMillis), + Math.max(RetryState.EXTENDED_MAX_DELAY_MILLIS, pollIntervalMillis), + pollIntervalMillis, + 0, + RetryState.POLLING_RESET_THRESHOLD_SUCCESSES, + random); + } + + // ---- defaults ---- + + @Test + public void defaultsAreThirtySecondNormalCeilingAndFiveMinuteToOneHourExtendedRegime() { + assertEquals(30 * SECOND, RetryState.NORMAL_MAX_DELAY_MILLIS); + assertEquals(5 * MINUTE, RetryState.EXTENDED_INITIAL_DELAY_MILLIS); + assertEquals(60 * MINUTE, RetryState.EXTENDED_MAX_DELAY_MILLIS); + assertEquals(60 * SECOND, RetryState.STREAMING_RESET_THRESHOLD_MILLIS); + assertEquals(2, RetryState.POLLING_RESET_THRESHOLD_SUCCESSES); + } + + // ---- initial state ---- + + @Test + public void startsWithNoAttemptsInNormalRegime() { + RetryState state = streaming(NO_JITTER); + assertEquals(0, state.getAttempts()); + assertFalse(state.isInExtendedRegime()); + } + + @Test + public void delayBeforeAnyFailureIsTheOperatingCadence() { + assertEquals(0, streaming(NO_JITTER).nextDelayMillis()); + assertEquals(10 * SECOND, polling(10 * SECOND, NO_JITTER).nextDelayMillis()); + } + + // ---- normal regime ---- + + @Test + public void normalFailuresDoubleTheDelayUpToTheNormalCeiling() { + RetryState state = streaming(NO_JITTER); + long[] expected = {SECOND, 2 * SECOND, 4 * SECOND, 8 * SECOND, 16 * SECOND, 30 * SECOND, 30 * SECOND}; + for (int i = 0; i < expected.length; i++) { + state.recordFailure(false, 0); + assertEquals("attempt " + (i + 1), i + 1, state.getAttempts()); + assertEquals("delay after attempt " + (i + 1), expected[i], state.nextDelayMillis()); + } + assertFalse(state.isInExtendedRegime()); + } + + // ---- jitter ---- + + @Test + public void jitterSubtractsUpToHalfOfTheBaseDelay() { + RetryState noJitter = streaming(NO_JITTER); + RetryState maxJitter = streaming(MAX_JITTER); + noJitter.recordFailure(false, 0); + maxJitter.recordFailure(false, 0); + assertEquals(SECOND, noJitter.nextDelayMillis()); + // Jitter is chosen from [0, T/2), so the smallest wait is just above T/2. + long minWait = maxJitter.nextDelayMillis(); + assertTrue("wait " + minWait, minWait > SECOND / 2 && minWait <= SECOND); + } + + // ---- unexpected failures and the extended regime ---- + + @Test + public void unexpectedFailureMovesToExtendedRegimeStartingAtExtendedInitialDelay() { + RetryState state = streaming(NO_JITTER); + state.recordFailure(false, 0); + state.recordFailure(false, 0); + state.recordFailure(true, 0); + + assertTrue(state.isInExtendedRegime()); + assertEquals(1, state.getAttempts()); + assertEquals(5 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void extendedRegimeDoublesUpToTheExtendedCeiling() { + RetryState state = streaming(NO_JITTER); + state.recordFailure(true, 0); + long[] expected = {5 * MINUTE, 10 * MINUTE, 20 * MINUTE, 40 * MINUTE, 60 * MINUTE, 60 * MINUTE}; + assertEquals(expected[0], state.nextDelayMillis()); + for (int i = 1; i < expected.length; i++) { + state.recordFailure(false, 0); + assertEquals("delay after attempt " + (i + 1), expected[i], state.nextDelayMillis()); + } + } + + @Test + public void normalFailureAfterUnexpectedStaysInExtendedRegime() { + // Once raised, the ceiling and base stay raised until a reset. + RetryState state = streaming(NO_JITTER); + state.recordFailure(true, 0); + state.recordFailure(false, 0); + assertTrue(state.isInExtendedRegime()); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void repeatedUnexpectedFailuresKeepCountingInExtendedRegime() { + RetryState state = streaming(NO_JITTER); + state.recordFailure(true, 0); + state.recordFailure(true, 0); + assertEquals(2, state.getAttempts()); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void extendedDelaysAreNeverBelowTheNormalInitialDelay() { + // A component whose normal initial delay exceeds the extended bounds uses the normal + // initial delay in their place. + RetryState state = new RetryState(10 * MINUTE, 10 * MINUTE, 5 * MINUTE, 8 * MINUTE, 0, 0, 0, NO_JITTER); + state.recordFailure(true, 0); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + state.recordFailure(false, 0); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + // ---- reset ---- + + @Test + public void resetReturnsToNormalRegimeWithNoAttempts() { + RetryState state = streaming(NO_JITTER); + state.recordFailure(true, 0); + state.recordFailure(false, 0); + state.reset(); + + assertEquals(0, state.getAttempts()); + assertFalse(state.isInExtendedRegime()); + state.recordFailure(false, 0); + assertEquals(SECOND, state.nextDelayMillis()); + } + + @Test + public void streamingHealthyForThresholdResetsWhenTheNextFailureIsRecorded() { + RetryState state = streaming(NO_JITTER); + state.recordFailure(true, 0); + assertTrue(state.isInExtendedRegime()); + + long connectedAt = 10 * SECOND; + state.recordSuccess(connectedAt); + state.recordFailure(false, connectedAt + RetryState.STREAMING_RESET_THRESHOLD_MILLIS); + + assertFalse(state.isInExtendedRegime()); + assertEquals(1, state.getAttempts()); + assertEquals(SECOND, state.nextDelayMillis()); + } + + @Test + public void streamingHealthyForLessThanThresholdDoesNotReset() { + RetryState state = streaming(NO_JITTER); + state.recordFailure(true, 0); + + long connectedAt = 10 * SECOND; + state.recordSuccess(connectedAt); + state.recordFailure(false, connectedAt + RetryState.STREAMING_RESET_THRESHOLD_MILLIS - 1); + + assertTrue(state.isInExtendedRegime()); + assertEquals(2, state.getAttempts()); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void streamingHealthyOperationIsMeasuredFromTheFirstMessageOnTheConnection() { + RetryState state = streaming(NO_JITTER); + state.recordFailure(true, 0); + + state.recordSuccess(10 * SECOND); + state.recordSuccess(65 * SECOND); // must not move the marker + state.recordFailure(false, 70 * SECOND); // 60 seconds since the first message + + assertFalse(state.isInExtendedRegime()); + } + + @Test + public void streamingFailureRestartsHealthyOperationMeasurement() { + // The marker from a previous connection does not count toward the next one. + RetryState state = streaming(NO_JITTER); + state.recordFailure(true, 0); + + state.recordSuccess(10 * SECOND); + state.recordFailure(false, 20 * SECOND); // healthy for 10 seconds only + state.recordFailure(false, 90 * SECOND); // no message on this connection + + assertTrue(state.isInExtendedRegime()); + assertEquals(3, state.getAttempts()); + } + + @Test + public void streamingSuccessAloneDoesNotReset() { + RetryState state = streaming(NO_JITTER); + state.recordFailure(true, 0); + state.recordSuccess(10 * SECOND); + state.recordSuccess(10 * MINUTE); + // The reset is applied when the next failure is recorded, and only then; until that + // point the regime is unchanged. + assertTrue(state.isInExtendedRegime()); + } + + // ---- polling: operating cadence ---- + + @Test + public void pollingNormalFailureWaitsThePollInterval() { + RetryState state = polling(10 * SECOND, MAX_JITTER); + for (int i = 0; i < 5; i++) { + state.recordFailure(false, 0); + // Jitter cannot bring the wait below the cadence. + assertEquals(10 * SECOND, state.nextDelayMillis()); + } + assertFalse(state.isInExtendedRegime()); + } + + @Test + public void pollingUnexpectedFailureWaitsTheExtendedInitialDelay() { + RetryState state = polling(10 * SECOND, NO_JITTER); + state.recordFailure(true, 0); + assertEquals(5 * MINUTE, state.nextDelayMillis()); + state.recordFailure(false, 0); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void pollingExtendedDelayIsFlooredAtThePollInterval() { + RetryState state = polling(10 * MINUTE, NO_JITTER); + state.recordFailure(true, 0); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + state.recordFailure(false, 0); + assertEquals(20 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void pollingExtendedCeilingIsAtLeastThePollInterval() { + RetryState state = polling(90 * MINUTE, NO_JITTER); + state.recordFailure(true, 0); + state.recordFailure(false, 0); + state.recordFailure(false, 0); + assertEquals(90 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void pollingSuccessReturnsToCadenceWithoutResettingTheRegime() { + // One success restores the cadence but does not clear the retry state. + RetryState state = polling(10 * SECOND, NO_JITTER); + state.recordFailure(true, 0); + state.recordSuccess(0); + + assertEquals(10 * SECOND, state.nextDelayMillis()); + assertTrue(state.isInExtendedRegime()); + + state.recordFailure(false, 0); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void pollingTwoConsecutiveSuccessesReset() { + RetryState state = polling(10 * SECOND, NO_JITTER); + state.recordFailure(true, 0); + state.recordSuccess(0); + state.recordSuccess(0); + + assertFalse(state.isInExtendedRegime()); + assertEquals(0, state.getAttempts()); + state.recordFailure(false, 0); + assertEquals(10 * SECOND, state.nextDelayMillis()); + } + + @Test + public void pollingFailureRestartsTheSuccessCount() { + // Successes must be consecutive. + RetryState state = polling(10 * SECOND, NO_JITTER); + state.recordFailure(true, 0); + state.recordSuccess(0); + state.recordFailure(false, 0); + state.recordSuccess(0); + + assertTrue(state.isInExtendedRegime()); + } + + // ---- factories ---- + + @Test + public void forStreamingUsesConfiguredInitialReconnectDelayAsNormalInitialDelay() { + RetryState state = RetryState.forStreaming(250); + state.recordFailure(false, 0); + long wait = state.nextDelayMillis(); + assertTrue("wait " + wait, wait > 125 && wait <= 250); + assertEquals(0, RetryState.forStreaming(250).nextDelayMillis()); + } + + @Test + public void forStreamingWithZeroInitialDelayReconnectsImmediatelyInNormalRegime() { + RetryState state = RetryState.forStreaming(0); + state.recordFailure(false, 0); + assertEquals(0, state.nextDelayMillis()); + state.recordFailure(true, 0); + long wait = state.nextDelayMillis(); + assertTrue("wait " + wait, wait > RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 + && wait <= RetryState.EXTENDED_INITIAL_DELAY_MILLIS); + } + + @Test + public void forPollingUsesPollIntervalAsCadence() { + RetryState state = RetryState.forPolling(30 * SECOND); + assertEquals(30 * SECOND, state.nextDelayMillis()); + state.recordFailure(false, 0); + assertEquals(30 * SECOND, state.nextDelayMillis()); + state.recordFailure(true, 0); + long wait = state.nextDelayMillis(); + assertTrue("wait " + wait, wait > RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 + && wait <= RetryState.EXTENDED_INITIAL_DELAY_MILLIS); + } + + // ---- robustness ---- + + @Test + public void manyFailuresStayAtTheCeilingWithoutOverflow() { + RetryState state = streaming(NO_JITTER); + for (int i = 0; i < 200; i++) { + state.recordFailure(false, 0); + long wait = state.nextDelayMillis(); + assertTrue("wait " + wait, wait > 0 && wait <= RetryState.NORMAL_MAX_DELAY_MILLIS); + } + state.recordFailure(true, 0); + for (int i = 0; i < 200; i++) { + state.recordFailure(false, 0); + long wait = state.nextDelayMillis(); + assertTrue("wait " + wait, wait > 0 && wait <= RetryState.EXTENDED_MAX_DELAY_MILLIS); + } + } +} diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java new file mode 100644 index 000000000..b029417d6 --- /dev/null +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java @@ -0,0 +1,177 @@ +package com.launchdarkly.sdk.android; + +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertFalse; +import static org.junit.Assert.assertNotNull; +import static org.junit.Assert.assertNull; +import static org.junit.Assert.assertTrue; + +import androidx.annotation.NonNull; + +import com.launchdarkly.sdk.android.subsystems.FDv2SourceResult; +import com.launchdarkly.sdk.android.subsystems.Synchronizer; + +import org.junit.After; +import org.junit.Rule; +import org.junit.Test; +import org.junit.rules.Timeout; + +import java.util.Arrays; +import java.util.Collections; +import java.util.concurrent.Future; +import java.util.concurrent.TimeUnit; + +/** + * Unit tests for {@link SourceManager}'s handling of synchronizers that report unexpected + * errors: they are put aside for a backoff and then become available again. + */ +public class SourceManagerTest { + + @Rule + public Timeout globalTimeout = Timeout.seconds(5); + + private final FakeScheduledExecutorService executor = new FakeScheduledExecutorService(); + + @After + public void tearDown() { + executor.shutdownNow(); + } + + /** A synchronizer that never produces a result; only its name matters here. */ + private static final class NamedSynchronizer implements Synchronizer { + private final String name; + + NamedSynchronizer(String name) { + this.name = name; + } + + @Override + @NonNull + public Future next() { + return new LDAwaitFuture<>(); + } + + @Override + public void close() {} + + @Override + @NonNull + public String name() { + return name; + } + } + + private static SynchronizerFactoryWithState slot(final String name) { + return new SynchronizerFactoryWithState(() -> new NamedSynchronizer(name)); + } + + private SourceManager manager(String... names) { + SynchronizerFactoryWithState[] slots = new SynchronizerFactoryWithState[names.length]; + for (int i = 0; i < names.length; i++) { + slots[i] = slot(names[i]); + } + return new SourceManager(Arrays.asList(slots), Collections.emptyList(), executor); + } + + private static void assertFirstBackoff(long delayMillis) { + assertTrue("delay " + delayMillis, + delayMillis > RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 + && delayMillis <= RetryState.EXTENDED_INITIAL_DELAY_MILLIS); + } + + /** Advances the clock past a backoff and waits for the slot to become available again. */ + private void endBackoff(SourceManager manager, long delayMillis) throws Exception { + Future changed = manager.awaitAvailabilityChange(); + executor.advanceTime(delayMillis); + changed.get(1, TimeUnit.SECONDS); + } + + @Test + public void backingOffASlotSkipsItUntilTheBackoffEnds() throws Exception { + SourceManager manager = manager("a", "b"); + assertEquals("a", manager.getNextAvailableSynchronizerAndSetActive().name()); + + // Backing off the current slot schedules its return and makes selection skip it. + long delay = manager.backOffCurrentSynchronizer("a", 0); + assertFirstBackoff(delay); + assertEquals(delay, executor.awaitScheduledDelayMillis(1000)); + assertTrue(manager.hasBackingOffSynchronizers()); + assertEquals("b", manager.getNextAvailableSynchronizerAndSetActive().name()); + assertEquals("b", manager.getNextAvailableSynchronizerAndSetActive().name()); + + // Once the backoff ends, the slot is selected again. + endBackoff(manager, delay); + assertFalse(manager.hasBackingOffSynchronizers()); + assertEquals("a", manager.getNextAvailableSynchronizerAndSetActive().name()); + } + + @Test + public void allSlotsBackingOffYieldsNoSynchronizerUntilOneReturns() throws Exception { + SourceManager manager = manager("a", "b"); + manager.getNextAvailableSynchronizerAndSetActive(); + long firstDelay = manager.backOffCurrentSynchronizer("a", 0); + manager.getNextAvailableSynchronizerAndSetActive(); + long secondDelay = manager.backOffCurrentSynchronizer("b", 0); + + // Nothing can be selected, but the manager still knows a synchronizer will return. + assertNull(manager.getNextAvailableSynchronizerAndSetActive()); + assertTrue(manager.hasBackingOffSynchronizers()); + + // The first backoff to end makes its slot available. + endBackoff(manager, Math.max(firstDelay, secondDelay)); + assertNotNull(manager.getNextAvailableSynchronizerAndSetActive()); + } + + @Test + public void aBackingOffSlotStillOutranksTheCurrentOneForRecovery() throws Exception { + SourceManager manager = manager("a", "b"); + manager.getNextAvailableSynchronizerAndSetActive(); + long delay = manager.backOffCurrentSynchronizer("a", 0); + assertEquals("b", manager.getNextAvailableSynchronizerAndSetActive().name()); + + // While "a" is backing off, "b" is not prime, but there is nothing to recover to yet. + assertFalse(manager.isPrimeSynchronizer()); + assertFalse(manager.hasAvailableSynchronizerBeforeCurrent()); + + // Once "a" is available again, recovery to it is possible. + endBackoff(manager, delay); + assertTrue(manager.hasAvailableSynchronizerBeforeCurrent()); + } + + @Test + public void repeatedUnexpectedErrorsDoubleTheBackoffUntilHealthyOperationResetsIt() throws Exception { + SourceManager manager = manager("a"); + manager.getNextAvailableSynchronizerAndSetActive(); + + // A second unexpected error without any healthy operation in between doubles the wait. + long firstDelay = manager.backOffCurrentSynchronizer("a", 0); + assertFirstBackoff(firstDelay); + endBackoff(manager, firstDelay); + manager.getNextAvailableSynchronizerAndSetActive(); + long secondDelay = manager.backOffCurrentSynchronizer("a", 0); + assertTrue("delay " + secondDelay, secondDelay > RetryState.EXTENDED_INITIAL_DELAY_MILLIS); + endBackoff(manager, secondDelay); + + // Healthy operation for the reset threshold before the next error starts the backoff over. + manager.getNextAvailableSynchronizerAndSetActive(); + long healthyAt = 10_000; + manager.recordCurrentSynchronizerHealthy(healthyAt); + long thirdDelay = manager.backOffCurrentSynchronizer("a", healthyAt + RetryState.STREAMING_RESET_THRESHOLD_MILLIS); + assertFirstBackoff(thirdDelay); + } + + @Test + public void closeCancelsPendingBackoffsAndWakesWaiters() throws Exception { + SourceManager manager = manager("a"); + manager.getNextAvailableSynchronizerAndSetActive(); + manager.backOffCurrentSynchronizer("a", 0); + Future changed = manager.awaitAvailabilityChange(); + + // Closing wakes anyone waiting for a slot and drops the scheduled return. + manager.close(); + changed.get(1, TimeUnit.SECONDS); + assertFalse(manager.hasBackingOffSynchronizers()); + assertTrue(executor.pendingDelaysMillis().isEmpty()); + assertNull(manager.getNextAvailableSynchronizerAndSetActive()); + } +} diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java index aa1bbef88..ad6ab87c4 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java @@ -27,8 +27,10 @@ import java.io.IOException; import java.net.URI; import java.util.ArrayList; +import java.util.Collections; import java.util.List; import java.util.Map; +import java.util.Random; import java.util.concurrent.BlockingQueue; import java.util.concurrent.LinkedBlockingQueue; import java.util.concurrent.TimeUnit; @@ -165,6 +167,41 @@ private StreamingDataSource makeStreamingDataSource( .build(clientContext); } + // The tests of backoff behavior below drive the data source's timers with a FakeTaskExecutor + // and a RetryState without jitter, so the delay chosen for each reconnect can be asserted + // exactly instead of waited for. + private static final long NORMAL_DELAY_MILLIS = 1; + private static final long EXTENDED_DELAY_MILLIS = 300_000; + // A healthy-operation threshold no test reaches. + private static final long NEVER_RESET_MILLIS = 60_000; + + private final FakeTaskExecutor fakeTaskExecutor = new FakeTaskExecutor(); + + private static RetryState retryStateWithoutJitter(long healthyResetThresholdMillis) { + return new RetryState(NORMAL_DELAY_MILLIS, NORMAL_DELAY_MILLIS, + EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 4, 0, healthyResetThresholdMillis, 0, + new Random() { + @Override + public double nextDouble() { + return 0; + } + }); + } + + private StreamingDataSource makeStreamingDataSource(URI streamBaseUri, RetryState retryState) { + LDConfig config = new LDConfig.Builder(AutoEnvAttributes.Disabled) + .serviceEndpoints(Components.serviceEndpoints().streaming(streamBaseUri)) + .build(); + ClientContext baseClientContext = ClientContextImpl.fromConfig( + config, MOBILE_KEY, "", perEnvironmentData, + makeFeatureFetcher(), CONTEXT, + logging.logger, platformState, environmentReporter, fakeTaskExecutor); + ClientContext clientContext = ClientContextImpl.forDataSource( + baseClientContext, dataSourceUpdateSink, CONTEXT, false, false); + return new StreamingDataSource(clientContext, CONTEXT, dataSourceUpdateSink, + makeFeatureFetcher(), 1, false, retryState); + } + private static String makeSseEvent(String type, String data) { return "event: " + type + "\ndata: " + data; } @@ -636,7 +673,7 @@ public void startSendsRequestWithReportAndReasons() throws Exception { // --- start(): error handling verified via HttpServer --- @Test - public void startWithHttp401ShutsDownSink() throws Exception { + public void startWithHttp401DoesNotShutDownSink() throws Exception { try (HttpServer server = HttpServer.start(Handlers.status(401))) { StreamingDataSource sds = makeStreamingDataSource( server.getUri(), false, false); @@ -649,13 +686,9 @@ public void startWithHttp401ShutsDownSink() throws Exception { LDInvalidResponseCodeFailure failure = (LDInvalidResponseCodeFailure) error; assertEquals(401, failure.getResponseCode()); assertFalse(failure.isRetryable()); - // shutDown() runs on the EventSource background thread, so poll briefly - // to allow that thread to complete before asserting. - long deadline = System.currentTimeMillis() + 1000; - while (!dataSourceUpdateSink.shutDownCalled && System.currentTimeMillis() < deadline) { - Thread.sleep(10); - } - assertTrue(dataSourceUpdateSink.shutDownCalled); + // Reporting the error is the last thing the data source does with it, so the SDK + // has not been shut down and will not be. + assertFalse(dataSourceUpdateSink.shutDownCalled); } } @@ -770,91 +803,148 @@ public void startWithNetworkErrorReportsNetworkFailure() throws Exception { assertFalse(dataSourceUpdateSink.shutDownCalled); } + // --- start(): backoff after failures --- + @Test - public void startWithHttp401PreventsSubsequentStart() throws Exception { - try (HttpServer server = HttpServer.start(Handlers.status(401))) { - StreamingDataSource sds = makeStreamingDataSource( - server.getUri(), false, false); - TrackingCallback callback1 = new TrackingCallback(); - startDataSource(sds, callback1); + public void unexpectedErrorSchedulesReconnectAfterExtendedDelay() throws Exception { + String putEvent = makeSseEvent("put", VALID_PUT_JSON); - assertNotNull(callback1.awaitError()); + try (HttpServer server = HttpServer.start(Handlers.sequential( + Handlers.status(401), + Handlers.all( + Handlers.SSE.start(), + Handlers.SSE.event(putEvent), + Handlers.SSE.leaveOpen())))) { - // Second start should be a no-op due to connection401Error flag - TrackingCallback callback2 = new TrackingCallback(); - startDataSource(sds, callback2); + StreamingDataSource sds = makeStreamingDataSource(server.getUri(), + retryStateWithoutJitter(NEVER_RESET_MILLIS)); + TrackingCallback callback = new TrackingCallback(); + startDataSource(sds, callback); - assertNull("Second start should not produce a callback", - callback2.errors.poll(500, TimeUnit.MILLISECONDS)); - assertNull(callback2.successes.poll(200, TimeUnit.MILLISECONDS)); + // The 401 is reported, and the reconnect is scheduled for the extended delay. + assertNotNull(callback.awaitError()); + server.getRecorder().requireRequest(); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + + // Once that delay has passed, the data source reconnects and receives data. + fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); + assertNotNull(callback.awaitSuccess()); + server.getRecorder().requireRequest(); + assertFalse(dataSourceUpdateSink.shutDownCalled); } } - // --- start(): no reconnect after an unrecoverable HTTP error --- - @Test - public void unrecoverableErrorOnInitialConnectDoesNotReconnect() throws Exception { + public void unexpectedErrorOnReconnectSchedulesExtendedDelay() throws Exception { String putEvent = makeSseEvent("put", VALID_PUT_JSON); - // A second request would get a working stream. The SDK must not make it. try (HttpServer server = HttpServer.start(Handlers.sequential( - Handlers.status(401), + Handlers.all( + Handlers.SSE.start(), + Handlers.SSE.event(putEvent)), + Handlers.status(403), Handlers.all( Handlers.SSE.start(), Handlers.SSE.event(putEvent), Handlers.SSE.leaveOpen())))) { - StreamingDataSource sds = makeStreamingDataSource( - server.getUri(), dataSourceUpdateSink, false, false, 1); + StreamingDataSource sds = makeStreamingDataSource(server.getUri(), + retryStateWithoutJitter(NEVER_RESET_MILLIS)); TrackingCallback callback = new TrackingCallback(); - sds.start(callback); + startDataSource(sds, callback); + assertNotNull(callback.awaitSuccess()); + // The server ending the first stream is a normal failure, so the reconnect is + // scheduled for the normal delay. Throwable error = callback.awaitError(); - assertNotNull(error); - assertFalse(((LDInvalidResponseCodeFailure) error).isRetryable()); + assertEquals(LDFailure.FailureType.NETWORK_FAILURE, ((LDFailure) error).getFailureType()); + assertEquals(Collections.singletonList(NORMAL_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + fakeTaskExecutor.advanceTime(NORMAL_DELAY_MILLIS); - server.getRecorder().requireRequest(); - server.getRecorder().requireNoRequests(500, TimeUnit.MILLISECONDS); - assertNull("no stream data expected after the error", - callback.successes.poll(100, TimeUnit.MILLISECONDS)); + // The 403 on that reconnect moves the data source to the extended delay, after which + // it connects again. + error = callback.awaitError(); + assertEquals(403, ((LDInvalidResponseCodeFailure) error).getResponseCode()); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); + assertNotNull(callback.awaitSuccess()); + } + } + + @Test + public void sustainedUnexpectedErrorsKeepReconnectingWithGrowingDelay() throws Exception { + try (HttpServer server = HttpServer.start(Handlers.status(401))) { + StreamingDataSource sds = makeStreamingDataSource(server.getUri(), + retryStateWithoutJitter(NEVER_RESET_MILLIS)); + TrackingCallback callback = new TrackingCallback(); + startDataSource(sds, callback); + + // Every attempt fails; each one is reported and the next is scheduled with double + // the delay, and the data source never gives up. + for (long expectedDelay : new long[] {EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 2, EXTENDED_DELAY_MILLIS * 4}) { + assertNotNull(callback.awaitError()); + server.getRecorder().requireRequest(); + assertEquals(Collections.singletonList(expectedDelay), fakeTaskExecutor.pendingDelaysMillis()); + fakeTaskExecutor.advanceTime(expectedDelay); + } + assertNotNull(callback.awaitError()); + assertFalse(dataSourceUpdateSink.shutDownCalled); } } @Test - public void unrecoverableErrorOnReconnectDoesNotReconnectAgain() throws Exception { + public void healthyStreamResetsBackoffToNormalDelay() throws Exception { String putEvent = makeSseEvent("put", VALID_PUT_JSON); + // The data source measures healthy operation on its own clock, so the server must hold + // the stream open for a moment after the message before ending it. This is the shortest + // margin that reliably exceeds the 1 ms threshold used here. + long healthyMarginMillis = 20; try (HttpServer server = HttpServer.start(Handlers.sequential( - // The first connection delivers data, then the server ends the stream. + Handlers.status(401), Handlers.all( Handlers.SSE.start(), - Handlers.SSE.event(putEvent)), - // The reconnect gets an unrecoverable status. - Handlers.status(403), - // A third request would get a working stream. The SDK must not make it. + Handlers.SSE.event(putEvent), + Handlers.delay(healthyMarginMillis)), Handlers.all( Handlers.SSE.start(), Handlers.SSE.event(putEvent), Handlers.SSE.leaveOpen())))) { - StreamingDataSource sds = makeStreamingDataSource( - server.getUri(), dataSourceUpdateSink, false, false, 1); + StreamingDataSource sds = makeStreamingDataSource(server.getUri(), retryStateWithoutJitter(1)); TrackingCallback callback = new TrackingCallback(); - sds.start(callback); + startDataSource(sds, callback); + // The 401 puts the data source in the extended regime; the reconnect then delivers + // data and is ended by the server. + assertNotNull(callback.awaitError()); + fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); assertNotNull(callback.awaitSuccess()); + assertNotNull(callback.awaitError()); - // The dropped stream reports a network failure first; the 403 follows. - Throwable error = callback.awaitError(); - while (error != null && !(error instanceof LDInvalidResponseCodeFailure)) { - error = callback.awaitError(); - } - assertNotNull(error); - assertEquals(403, ((LDInvalidResponseCodeFailure) error).getResponseCode()); + // Having been healthy for longer than the threshold, the data source is back to the + // normal delay rather than doubling the extended one. + assertEquals(Collections.singletonList(NORMAL_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + fakeTaskExecutor.advanceTime(NORMAL_DELAY_MILLIS); + assertNotNull(callback.awaitSuccess()); + } + } - server.getRecorder().requireRequest(); - server.getRecorder().requireRequest(); - server.getRecorder().requireNoRequests(500, TimeUnit.MILLISECONDS); + @Test + public void stopCancelsPendingReconnect() throws Exception { + try (HttpServer server = HttpServer.start(Handlers.status(401))) { + StreamingDataSource sds = makeStreamingDataSource(server.getUri(), + retryStateWithoutJitter(NEVER_RESET_MILLIS)); + TrackingCallback callback = new TrackingCallback(); + startDataSource(sds, callback); + assertNotNull(callback.awaitError()); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + + // Stopping cancels the scheduled reconnect. + AwaitableCallback stopped = new AwaitableCallback<>(); + sds.stop(stopped); + stopped.await(STOP_TIMEOUT_MILLIS); + assertTrue(fakeTaskExecutor.pendingDelaysMillis().isEmpty()); } } diff --git a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java new file mode 100644 index 000000000..9779d16f5 --- /dev/null +++ b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java @@ -0,0 +1,195 @@ +package com.launchdarkly.sdk.android; + +import androidx.annotation.NonNull; + +import java.util.ArrayList; +import java.util.Collections; +import java.util.Comparator; +import java.util.Iterator; +import java.util.List; +import java.util.concurrent.AbstractExecutorService; +import java.util.concurrent.BlockingQueue; +import java.util.concurrent.Callable; +import java.util.concurrent.Delayed; +import java.util.concurrent.ExecutorService; +import java.util.concurrent.Executors; +import java.util.concurrent.FutureTask; +import java.util.concurrent.LinkedBlockingQueue; +import java.util.concurrent.ScheduledExecutorService; +import java.util.concurrent.ScheduledFuture; +import java.util.concurrent.TimeUnit; + +/** + * A {@link ScheduledExecutorService} with a manual clock, for unit tests of timer-driven code + * whose tasks block (for example on network I/O) and so must run on real background threads. + *

+ * Tasks submitted for immediate execution, and scheduled tasks with no delay, run right away on + * a background thread. A scheduled task with a delay is held until the test moves the clock past + * that delay with {@link #advanceTime(long)}, and is then released to a background thread. A test + * can wait for the code under test to schedule something with + * {@link #awaitScheduledDelayMillis(long)} and assert on the delay it chose, and can assert that + * a held task has not run because the clock has not moved, without sleeping. + */ +public class FakeScheduledExecutorService extends AbstractExecutorService implements ScheduledExecutorService { + private final ExecutorService delegate = Executors.newCachedThreadPool(); + private final Object lock = new Object(); + private final List held = new ArrayList<>(); + private final BlockingQueue scheduledDelays = new LinkedBlockingQueue<>(); + private long nowMillis = 0; + + @Override + public void execute(@NonNull Runnable command) { + delegate.execute(command); + } + + @NonNull + @Override + public ScheduledFuture schedule(@NonNull Runnable command, long delay, @NonNull TimeUnit unit) { + long delayMillis = Math.max(0, unit.toMillis(delay)); + HeldTask task; + synchronized (lock) { + task = new HeldTask(command, nowMillis + delayMillis); + if (delayMillis == 0) { + delegate.execute(task); + } else { + held.add(task); + } + } + scheduledDelays.add(delayMillis); + return task; + } + + /** + * Waits for the code under test to call {@link #schedule(Runnable, long, TimeUnit)} and + * returns the delay it asked for, in milliseconds. Each call consumes one scheduling event, + * in the order they happened. + * + * @param timeoutMillis how long to wait for a scheduling event + * @return the scheduled delay in milliseconds + * @throws AssertionError if nothing is scheduled within the timeout + */ + public long awaitScheduledDelayMillis(long timeoutMillis) throws InterruptedException { + Long delay = scheduledDelays.poll(timeoutMillis, TimeUnit.MILLISECONDS); + if (delay == null) { + throw new AssertionError("timed out waiting for a task to be scheduled"); + } + return delay; + } + + /** + * @return how long, from the fake clock's current time, until each held task is due, in the + * order the tasks were scheduled; cancelled and already-released tasks are omitted + */ + public List pendingDelaysMillis() { + List delays = new ArrayList<>(); + synchronized (lock) { + for (HeldTask task : held) { + if (!task.isCancelled()) { + delays.add(task.dueMillis - nowMillis); + } + } + } + return delays; + } + + /** + * Moves the clock forward and releases every held task that is due by the new time, in due + * order, to a background thread. + */ + public void advanceTime(long millis) { + List due = new ArrayList<>(); + synchronized (lock) { + nowMillis += millis; + Iterator it = held.iterator(); + while (it.hasNext()) { + HeldTask task = it.next(); + if (task.isCancelled()) { + it.remove(); + } else if (task.dueMillis <= nowMillis) { + it.remove(); + due.add(task); + } + } + } + Collections.sort(due, new Comparator() { + @Override + public int compare(HeldTask a, HeldTask b) { + return Long.compare(a.dueMillis, b.dueMillis); + } + }); + for (HeldTask task : due) { + delegate.execute(task); + } + } + + @NonNull + @Override + public ScheduledFuture schedule(@NonNull Callable callable, long delay, @NonNull TimeUnit unit) { + throw new UnsupportedOperationException("FakeScheduledExecutorService does not support Callable scheduling"); + } + + @NonNull + @Override + public ScheduledFuture scheduleAtFixedRate(@NonNull Runnable command, long initialDelay, long period, @NonNull TimeUnit unit) { + throw new UnsupportedOperationException("FakeScheduledExecutorService does not support repeating tasks"); + } + + @NonNull + @Override + public ScheduledFuture scheduleWithFixedDelay(@NonNull Runnable command, long initialDelay, long delay, @NonNull TimeUnit unit) { + throw new UnsupportedOperationException("FakeScheduledExecutorService does not support repeating tasks"); + } + + @Override + public void shutdown() { + synchronized (lock) { + held.clear(); + } + delegate.shutdown(); + } + + @NonNull + @Override + public List shutdownNow() { + synchronized (lock) { + held.clear(); + } + return delegate.shutdownNow(); + } + + @Override + public boolean isShutdown() { + return delegate.isShutdown(); + } + + @Override + public boolean isTerminated() { + return delegate.isTerminated(); + } + + @Override + public boolean awaitTermination(long timeout, @NonNull TimeUnit unit) throws InterruptedException { + return delegate.awaitTermination(timeout, unit); + } + + private final class HeldTask extends FutureTask implements ScheduledFuture { + final long dueMillis; + + HeldTask(Runnable command, long dueMillis) { + super(command, null); + this.dueMillis = dueMillis; + } + + @Override + public long getDelay(@NonNull TimeUnit unit) { + synchronized (lock) { + return unit.convert(dueMillis - nowMillis, TimeUnit.MILLISECONDS); + } + } + + @Override + public int compareTo(@NonNull Delayed other) { + return Long.compare(getDelay(TimeUnit.MILLISECONDS), other.getDelay(TimeUnit.MILLISECONDS)); + } + } +} diff --git a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java new file mode 100644 index 000000000..39eda499d --- /dev/null +++ b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java @@ -0,0 +1,133 @@ +package com.launchdarkly.sdk.android; + +import java.util.ArrayList; +import java.util.Collections; +import java.util.Comparator; +import java.util.Iterator; +import java.util.List; +import java.util.concurrent.Delayed; +import java.util.concurrent.FutureTask; +import java.util.concurrent.ScheduledFuture; +import java.util.concurrent.TimeUnit; + +/** + * A {@link TaskExecutor} with a manual clock, for unit tests of timer-driven code. + *

+ * A scheduled task does not run until the test moves the clock past its delay with + * {@link #advanceTime(long)}, and then it runs synchronously on the calling thread. A test can + * therefore assert exactly which delays were scheduled, and that nothing ran before the clock + * moved, without sleeping. + */ +public class FakeTaskExecutor implements TaskExecutor { + private final Object lock = new Object(); + private final List tasks = new ArrayList<>(); + private long nowMillis = 0; + + @Override + public void executeOnMainThread(Runnable action) { + action.run(); + } + + @Override + public ScheduledFuture scheduleTask(Runnable action, long delayMillis) { + synchronized (lock) { + ScheduledTask task = new ScheduledTask(action, nowMillis + Math.max(0, delayMillis)); + tasks.add(task); + return task; + } + } + + @Override + public ScheduledFuture startRepeatingTask(Runnable action, long initialDelayMillis, long intervalMillis) { + throw new UnsupportedOperationException("FakeTaskExecutor does not support repeating tasks"); + } + + @Override + public void close() { + synchronized (lock) { + tasks.clear(); + } + } + + /** + * @return how long, from the fake clock's current time, until each pending task is due, in + * the order the tasks were scheduled; cancelled tasks are omitted + */ + public List pendingDelaysMillis() { + List delays = new ArrayList<>(); + synchronized (lock) { + for (ScheduledTask task : tasks) { + if (!task.isCancelled()) { + delays.add(task.dueMillis - nowMillis); + } + } + } + return delays; + } + + /** + * Runs every task whose time has already come, without moving the clock. + */ + public void runDueTasks() { + advanceTime(0); + } + + /** + * Moves the clock forward and runs every task that is due by the new time, in due order, on + * the calling thread. A task that schedules another task during its run is also run here if + * that task is due. + */ + public void advanceTime(long millis) { + synchronized (lock) { + nowMillis += millis; + } + while (true) { + List due = new ArrayList<>(); + synchronized (lock) { + Iterator it = tasks.iterator(); + while (it.hasNext()) { + ScheduledTask task = it.next(); + if (task.isCancelled()) { + it.remove(); + } else if (task.dueMillis <= nowMillis) { + it.remove(); + due.add(task); + } + } + } + if (due.isEmpty()) { + return; + } + Collections.sort(due, new Comparator() { + @Override + public int compare(ScheduledTask a, ScheduledTask b) { + return Long.compare(a.dueMillis, b.dueMillis); + } + }); + for (ScheduledTask task : due) { + task.run(); + } + } + } + + private final class ScheduledTask extends FutureTask implements ScheduledFuture { + final long dueMillis; + + ScheduledTask(Runnable action, long dueMillis) { + super(action, null); + this.dueMillis = dueMillis; + } + + @Override + public long getDelay(TimeUnit unit) { + synchronized (lock) { + return unit.convert(dueMillis - nowMillis, TimeUnit.MILLISECONDS); + } + } + + @Override + public int compareTo(Delayed other) { + return Long.compare(getDelay(TimeUnit.MILLISECONDS), other.getDelay(TimeUnit.MILLISECONDS)); + } + } +} From eeed2d6816d6c2d1c8057d38f6e267e55210bf1d Mon Sep 17 00:00:00 2001 From: Bee Klimt Date: Tue, 29 Sep 2026 14:39:54 -0700 Subject: [PATCH 2/5] docs: Clean up comments in the retry and backoff code --- .../sdk/android/ConnectivityManager.java | 3 - .../sdk/android/FDv2DataSource.java | 17 ++- .../android/FDv2StreamingSynchronizer.java | 44 +++---- .../android/LDInvalidResponseCodeFailure.java | 8 +- .../com/launchdarkly/sdk/android/LDUtil.java | 16 +-- .../sdk/android/PollingDataSource.java | 12 +- .../launchdarkly/sdk/android/RetryState.java | 109 ++++++------------ .../sdk/android/SourceManager.java | 26 ++--- .../sdk/android/StreamingDataSource.java | 32 ++--- .../android/SynchronizerFactoryWithState.java | 9 +- .../sdk/android/FDv2DataSourceTest.java | 5 +- .../FDv2StreamingSynchronizerTest.java | 2 +- .../launchdarkly/sdk/android/LDUtilTest.java | 2 +- .../sdk/android/PollingDataSourceTest.java | 2 +- .../sdk/android/RetryStateTest.java | 4 +- .../sdk/android/SourceManagerTest.java | 4 +- .../sdk/android/StreamingDataSourceTest.java | 4 +- .../android/FakeScheduledExecutorService.java | 14 +-- .../sdk/android/FakeTaskExecutor.java | 11 +- 19 files changed, 118 insertions(+), 206 deletions(-) diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/ConnectivityManager.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/ConnectivityManager.java index d307c9966..aae89b116 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/ConnectivityManager.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/ConnectivityManager.java @@ -136,9 +136,6 @@ public void setStatus(@NonNull DataSourceState state, Throwable failure) { @Override public void shutDown() { - // A custom DataSource may call this to stop the SDK permanently. The SDK's own data - // sources no longer do: they retry every failure, including HTTP 401, with a backoff - // instead of stopping. ConnectivityManager.this.shutDown(); setStatus(ConnectionInformation.ConnectionMode.SHUTDOWN, null); } diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java index 090dfc99b..a139c80d8 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java @@ -475,8 +475,8 @@ private List getConditions(int synchronizerC List list = new ArrayList<>(); list.add(new FDv2DataSourceConditions.FallbackCondition(sharedExecutor, fallbackTimeoutSeconds)); if (!isPrime) { - // Recovery only goes ahead if a higher-priority synchronizer is actually available; - // one that is still backing off after an unexpected error keeps the timer running. + // Recovery only goes ahead once a higher-priority synchronizer is available, not + // merely backing off. list.add(new FDv2DataSourceConditions.RecoveryCondition(sharedExecutor, recoveryTimeoutSeconds, sourceManager::hasAvailableSynchronizerBeforeCurrent)); } @@ -484,10 +484,9 @@ private List getConditions(int synchronizerC } /** - * Returns the next available synchronizer. If none is available but at least one is waiting - * out a backoff, waits for the first backoff to end and tries again, so that unexpected errors - * from every synchronizer never stop the data source. Returns null once the data source is - * stopped or there is truly nothing left to try. + * Returns the next available synchronizer, waiting for a backoff to end if that is the only + * way to get one. Returns null once the data source is stopped or there is nothing left to + * try. */ @Nullable private Synchronizer nextSynchronizerOrWaitForBackoff() throws InterruptedException { @@ -621,10 +620,8 @@ private void runSynchronizers( running = false; break; case TERMINAL_ERROR: - // The synchronizer hit an error that is not expected to - // resolve soon, such as HTTP 401. Move on to the next one - // now, and put this one aside for a while rather than - // for good, so that a transient cause still recovers. + // Move on to the next synchronizer now, and put this one + // into a backoff so that it is tried again later. maybeLogSynchronizerStatusChange( synchronizer.name(), status.getState() diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java index 04460bbbd..952a36521 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java @@ -49,12 +49,10 @@ * If an optional {@link FDv2Requestor} is supplied, {@code ping} SSE events are handled by * issuing a poll request. If no requestor is supplied, {@code ping} events are ignored. *

- * Reconnection is managed here rather than by the EventSource library. Every connection attempt - * uses a fresh {@link EventSource} that surfaces any failure as an exception and never retries - * on its own. A transport failure, a recoverable HTTP status, or a bad payload is reported as - * INTERRUPTED and followed by a reconnect after the delay computed by {@link RetryState}. An - * HTTP status that is not expected to resolve soon, such as 401, is reported as TERMINAL_ERROR - * and ends this synchronizer; the data source that owns it decides when to try it again. + * Reconnection is managed here rather than by the EventSource library. A transport failure, a + * recoverable HTTP status, or a bad payload is reported as INTERRUPTED and followed by a + * reconnect after a backoff. An HTTP status that is not expected to resolve soon, such as 401, is + * reported as TERMINAL_ERROR and ends this synchronizer. */ final class FDv2StreamingSynchronizer implements Synchronizer { private static final String METHOD_REPORT = "REPORT"; @@ -79,9 +77,7 @@ final class FDv2StreamingSynchronizer implements Synchronizer { private final LDAwaitFuture shutdownFuture = new LDAwaitFuture<>(); private final AtomicBoolean started = new AtomicBoolean(false); - // The following are only touched on the streaming thread. Only one connection attempt runs - // at a time, and the next is scheduled by the previous one, so successive attempts are - // ordered even if they run on different threads. + // The following are only touched by the current connection attempt. Attempts never overlap. private final FDv2ProtocolHandler protocolHandler = new FDv2ProtocolHandler(); private final RetryState retryState; // Set while handling a message when the current connection must be dropped and a new one @@ -102,14 +98,13 @@ final class FDv2StreamingSynchronizer implements Synchronizer { * @param streamBaseUri base URI for the stream endpoint * @param streamRequestPath path appended to the base URI for the stream request * @param requestor optional requestor for handling ping events via poll; may be null - * @param initialReconnectDelayMillis base delay before reconnecting after a failure, in - * milliseconds; later failures back off from this value + * @param initialReconnectDelayMillis initial delay before reconnecting after a failure, in + * milliseconds * @param evaluationReasons true to request evaluation reasons in the stream * @param useReport true to use HTTP REPORT for the request body * @param httpProperties HTTP configuration for the stream request - * @param executor executor used to run each connection attempt on a - * background thread and to schedule the next attempt after - * a failure; should use background-priority threads + * @param executor executor for the connection attempts and the waits + * between them. Should use background-priority threads. * @param logger logger * @param diagnosticStore optional store for stream diagnostics; may be null */ @@ -189,7 +184,6 @@ public void close() { closed = true; esToClose = eventSource; eventSource = null; - // A backoff wait ends with the synchronizer: nothing reconnects after close(). if (pendingAttempt != null) { pendingAttempt.cancel(false); pendingAttempt = null; @@ -214,8 +208,8 @@ private boolean isClosed() { } /** - * One connection attempt: connect, read until the connection ends, then schedule the next - * attempt after the backoff delay. Only {@link #close()} stops the sequence. + * Runs one connection attempt and schedules the next one after the backoff delay. Only + * {@link #close()} stops the sequence. */ private void runConnectionAttempt() { EventSource es = buildEventSource(); @@ -283,9 +277,8 @@ private EventSource buildEventSource() { RequestBody.create(JsonSerialization.serialize(evaluationContext), JSON)); } - // With the default ErrorStrategy, every connection or read failure is thrown from - // readAnyEvent() and the EventSource neither waits nor reconnects on its own. This class - // owns both, using RetryState, and builds a new EventSource for each attempt. + // The default error strategy throws every failure from readAnyEvent() instead of + // reconnecting, which leaves the backoff to this class. return new EventSource.Builder(connectStrategy).build(); } @@ -309,7 +302,7 @@ private long readUntilDisconnected(EventSource es) { return restartDelayMillis; } } - // StartedEvent and CommentEvent (SSE comment/heartbeat line): no action needed + // StartedEvent and CommentEvent (SSE comment/heartbeat line) need no action. } } catch (StreamException e) { if (isClosed() || e instanceof StreamClosedByCallerException) { @@ -369,8 +362,6 @@ void handleMessage(MessageEvent event) { String eventData = event.getData(); logger.debug("onMessage: {}: {}", eventName, eventData); - // A message on the stream is healthy operation; enough of it in a row resets the - // backoff. retryState.recordSuccess(System.currentTimeMillis()); if (PING.equalsIgnoreCase(eventName)) { @@ -487,9 +478,7 @@ private void handlePing() { } /** - * Classifies a connection failure and reports it. A recoverable failure is reported as - * INTERRUPTED and followed by a reconnect; an unexpected HTTP status is reported as - * TERMINAL_ERROR and ends this synchronizer. + * Classifies a connection failure and reports it. * * @return the wait in milliseconds before the next attempt, or -1 if this synchronizer has * ended and must not reconnect @@ -540,8 +529,7 @@ private long handleError(StreamException t) { /** * Asks the connection attempt to drop the current connection once the current message has - * been handled, and to reconnect after a normal-regime backoff. A malformed payload is a - * normal failure, and so is a server that announces it is about to close the connection. + * been handled, and to reconnect after a backoff. * * @param failed true if the restart is due to an error (for diagnostic recording) */ diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDInvalidResponseCodeFailure.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDInvalidResponseCodeFailure.java index 18b2f632e..d1168162f 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDInvalidResponseCodeFailure.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDInvalidResponseCodeFailure.java @@ -14,9 +14,7 @@ public class LDInvalidResponseCodeFailure extends LDFailure { private final int responseCode; /** - * Whether the failure is one that may resolve on its own if retried soon. This is a - * classification of the response code, not a statement about what the SDK will do: the SDK - * retries every failure, and simply waits longer between attempts when this is false. + * Whether the failure is one that may resolve on its own if retried soon. */ private final boolean retryable; @@ -44,8 +42,8 @@ public LDInvalidResponseCodeFailure(String message, Throwable cause, int respons } /** - * @return true if the failure may resolve on its own if retried soon; false if it is unlikely - * to, as with an authentication failure + * @return true if the failure may resolve on its own if retried soon. An authentication + * failure, for example, is unlikely to. */ public boolean isRetryable() { return retryable; diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java index c1fa7b30b..521562e40 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java @@ -215,11 +215,11 @@ public void updateHeaders(Map headers) { * Classifies an HTTP error status as either a {@code normal} failure, which may resolve on its * own if retried soon, or an {@code unexpected} failure, which is not expected to. *

- * 400, 408, 429, and all 5xx statuses are {@code normal}; every other 4xx status is - * {@code unexpected}; any other status is {@code normal}. + * 400, 408, 429, and all 5xx statuses are {@code normal}. Every other 4xx status is + * {@code unexpected}. Any other status is {@code normal}. * * @param statusCode the HTTP status - * @return true if the failure is {@code normal}; false if it is {@code unexpected} + * @return true if the status is {@code normal}, or false if it is {@code unexpected} */ static boolean isHttpErrorRecoverable(int statusCode) { if (statusCode >= 400 && statusCode < 500) { @@ -238,12 +238,12 @@ static boolean isHttpErrorRecoverable(int statusCode) { /** * Classifies a data source failure as {@code unexpected} or {@code normal}. *

- * Only an HTTP response failure whose status {@link #isHttpErrorRecoverable(int)} classifies as - * {@code unexpected} is {@code unexpected}. Every other failure, including transport errors, - * malformed response bodies, and unknown errors, is {@code normal}. + * Only an HTTP response failure whose status is {@code unexpected} under + * {@link #isHttpErrorRecoverable(int)} is {@code unexpected}. Every other failure is + * {@code normal}. * - * @param failure the failure reported by a data source; may be null - * @return true if the failure is {@code unexpected}; false if it is {@code normal} + * @param failure the failure reported by a data source, or null + * @return true if the failure is {@code unexpected}, or false if it is {@code normal} */ static boolean isUnexpectedFailure(Throwable failure) { return failure instanceof LDInvalidResponseCodeFailure && diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java index 2342e3822..1ed793316 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java @@ -17,10 +17,9 @@ * in the background so we do polling instead. The logic for this is in * ComponentsImpl.PollingDataSourceBuilderImpl and ComponentsImpl.StreamingDataSourceBuilderImpl. *

- * Each poll is scheduled individually after the previous one completes, so that the wait can - * be chosen per attempt: after a successful poll the next one happens at the poll - * interval, and after a failed poll the wait comes from {@link RetryState}. No failure stops the - * data source from polling again. + * Polls happen at the poll interval, except that a failure that is not expected to resolve on + * its own, such as HTTP 401, is followed by a much longer wait. No failure stops the data source + * from polling again. */ final class PollingDataSource implements DataSource { private final LDContext context; @@ -118,7 +117,6 @@ public void stop(Callback completionCallback) { synchronized (this) { running = false; } - // A pending wait, whether the poll interval or a backoff, ends immediately. ScheduledFuture task = currentPollTask.getAndSet(null); if (task != null) { task.cancel(true); @@ -178,8 +176,8 @@ public void onError(Throwable error) { try { ConnectivityManager.fetchAndSetData(fetcher, context, dataSourceUpdateSink, pollCallback, logger); } catch (RuntimeException e) { - // A fetcher must report its outcome through the callback, but if one throws instead we - // still owe the caller a result and the next poll. + // If the fetcher throws instead of using its callback, the caller is still owed a + // result and the next poll. LDUtil.logExceptionAtErrorLevel(logger, e, "Unexpected exception while polling for flags"); pollCallback.onError(new LDFailure("Exception while fetching flags", e, LDFailure.FailureType.UNKNOWN_ERROR)); } diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java index acae6889a..4616cd237 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java @@ -5,25 +5,15 @@ import java.util.Random; /** - * Retry state for a long-running component such as a streaming or polling data source: - * exponential backoff with jitter, in two regimes. + * Backoff state for a long-running component such as a streaming or polling data source. *

- * Every failure is classified as either {@code normal} or {@code unexpected}. A {@code normal} - * failure advances the backoff within the current regime. An {@code unexpected} failure (for - * example an HTTP 401 or 403, which is unlikely to resolve on its own quickly) moves the - * component to the extended regime, whose delays run from minutes up to an hour, and the - * component stays there until the reset threshold is met. No failure ever causes the component - * to stop retrying. + * The owning component reports each success and failure, then asks how long to wait before its + * next attempt. A failure is either {@code normal} or {@code unexpected}. An {@code unexpected} + * failure, such as an HTTP 401, moves the component to a much longer backoff, which it leaves + * only after enough healthy operation. *

- * The reset threshold differs for the two kinds of component: - *

    - *
  • Streaming: continuous healthy operation for {@link #STREAMING_RESET_THRESHOLD_MILLIS}, - * where healthy operation starts with the first payload received on a connection.
  • - *
  • Polling: {@link #POLLING_RESET_THRESHOLD_SUCCESSES} consecutive successful polls.
  • - *
- *

- * Timestamps are passed in explicitly so that the state machine is deterministic in tests. Any - * monotonic millisecond clock may be used, as long as the same clock is used for every call. + * Timestamps are passed in explicitly. Any monotonic millisecond clock may be used, as long as + * the same clock is used for every call. *

* This class is not thread-safe. The owning component must serialize access to it. */ @@ -49,21 +39,18 @@ final class RetryState { */ static final int POLLING_RESET_THRESHOLD_SUCCESSES = 2; - // 2^30 is far more doubling than any real delay needs, and keeps initialDelay * 2^exponent - // from overflowing a long for any plausible configured delay. + // Caps the doubling so that initialDelay * 2^exponent cannot overflow. private static final int MAX_EXPONENT = 30; private final long normalInitialDelayMillis; private final long normalMaxDelayMillis; private final long extendedInitialDelayMillis; private final long extendedMaxDelayMillis; - // The interval at which the component operates when healthy. The wait after a failure is - // never shorter than this, and it is the wait after a success. Zero for a component with no - // such interval, such as streaming. + // The interval at which the component operates when healthy, or zero if it has none. private final long operatingCadenceMillis; - // Duration-based reset threshold; zero if this component does not use one. + // Duration-based reset threshold, or zero if this component does not use one. private final long healthyResetThresholdMillis; - // Count-based reset threshold; zero if this component does not use one. + // Count-based reset threshold, or zero if this component does not use one. private final int successResetThreshold; private final Random random; @@ -75,14 +62,11 @@ final class RetryState { private int consecutiveSuccesses = 0; /** - * Creates the retry state for a streaming data source or synchronizer. - *

- * The normal regime backs off from the configured initial reconnect delay up to - * {@link #NORMAL_MAX_DELAY_MILLIS}. The extended regime backs off from - * {@link #EXTENDED_INITIAL_DELAY_MILLIS} up to {@link #EXTENDED_MAX_DELAY_MILLIS}. The state - * resets after {@link #STREAMING_RESET_THRESHOLD_MILLIS} of healthy operation. + * Creates the retry state for a streaming data source or synchronizer. Normal failures back off + * from the configured initial reconnect delay. The state resets after + * {@link #STREAMING_RESET_THRESHOLD_MILLIS} of healthy operation. * - * @param initialReconnectDelayMillis the configured initial reconnect delay; may be zero + * @param initialReconnectDelayMillis the configured initial reconnect delay, which may be zero * @return the retry state */ @NonNull @@ -100,15 +84,11 @@ static RetryState forStreaming(long initialReconnectDelayMillis) { } /** - * Creates the retry state for a polling data source or synchronizer. - *

- * A polling component's operating cadence is its poll interval. In the normal regime it keeps - * polling at that interval after a failure. In the extended regime it backs off from - * {@link #EXTENDED_INITIAL_DELAY_MILLIS} up to {@link #EXTENDED_MAX_DELAY_MILLIS}, but never - * more often than the poll interval. The state resets after + * Creates the retry state for a polling data source or synchronizer. Normal failures keep the + * poll interval, and no wait is ever shorter than it. The state resets after * {@link #POLLING_RESET_THRESHOLD_SUCCESSES} consecutive successful polls. * - * @param pollIntervalMillis the configured poll interval; must be positive + * @param pollIntervalMillis the configured poll interval, which must be positive * @return the retry state */ @NonNull @@ -126,12 +106,9 @@ static RetryState forPolling(long pollIntervalMillis) { } /** - * Creates the retry state that the FDv2 data source keeps for one synchronizer slot. - *

- * Only unexpected failures are recorded against a slot, so both regimes are bound to the - * extended values: the slot waits {@link #EXTENDED_INITIAL_DELAY_MILLIS} after the first - * unexpected error, doubling up to {@link #EXTENDED_MAX_DELAY_MILLIS}. The state resets after - * {@link #STREAMING_RESET_THRESHOLD_MILLIS} of healthy operation by the slot's synchronizer. + * Creates the retry state for one synchronizer slot. Every failure, {@code normal} or + * {@code unexpected}, backs off in the extended regime. The state resets after + * {@link #STREAMING_RESET_THRESHOLD_MILLIS} of healthy operation. * * @return the retry state */ @@ -149,13 +126,12 @@ static RetryState forSynchronizerSlot() { } /** - * Creates a retry state with explicit parameters. Production code should use one of the - * static factories; this constructor exists so that tests can use short delays and a - * deterministic random source. + * Creates a retry state with explicit parameters, for tests. Production code should use one + * of the static factories. * * @param normalInitialDelayMillis base delay for the first retry in the normal regime * @param normalMaxDelayMillis ceiling on the wait in the normal regime - * @param extendedInitialDelayMillis base delay for the first retry in the extended regime; + * @param extendedInitialDelayMillis base delay for the first retry in the extended regime, * raised to the normal initial delay if smaller * @param extendedMaxDelayMillis ceiling on the wait in the extended regime * @param operatingCadenceMillis the operating cadence, or zero if the component has none @@ -187,15 +163,9 @@ static RetryState forSynchronizerSlot() { } /** - * Records healthy operation. - *

- * For a streaming component this is a payload received on the current connection; the first - * such call after a connection is established starts the clock toward the duration-based - * reset threshold, and later calls on the same connection do not move it. For a polling - * component this is a successful poll, which counts toward the count-based reset threshold. - *

- * After a success the next wait is the operating cadence, even if the retry state has not - * reset. + * Records healthy operation, such as a payload received on a stream or a successful poll. + * Enough healthy operation resets the state. After a success the next wait is the operating + * cadence, even if the state has not reset. * * @param nowMillis the current time */ @@ -213,13 +183,7 @@ void recordSuccess(long nowMillis) { } /** - * Records a failure and updates the retry state. - *

- * If this component resets on a duration and it had been healthy for at least that long when - * the failure happened, the state is reset first so that the failure is counted against a - * clean slate. Either way, measurement toward the reset threshold restarts. - *

- * Call this before {@link #nextDelayMillis()}. + * Records a failure. Call this before {@link #nextDelayMillis()}. * * @param unexpected true if the failure is classified as {@code unexpected}, false if * {@code normal} @@ -235,18 +199,17 @@ void recordFailure(boolean unexpected, long nowMillis) { lastOperationFailed = true; if (unexpected && !extended) { - // Move to the extended regime and start its attempt count over, so the first - // extended wait is the extended initial delay. + // Start the attempt count over so the first extended wait is the extended initial + // delay. extended = true; attempts = 1; } else { - // A repeated unexpected failure keeps counting in the extended regime. attempts++; } } /** - * Clears the retry state back to its initial values: no attempts, and the normal regime. + * Clears the retry state back to no attempts and the normal regime. */ void reset() { attempts = 0; @@ -255,13 +218,9 @@ void reset() { } /** - * Computes how long to wait before the next attempt. - *

- * If the most recent operation succeeded (or no operation has failed yet), this is the - * operating cadence. Otherwise it is {@code T - J}, where {@code T} is the regime's initial - * delay doubled once per attempt after the first and clamped to the regime's ceiling, and - * {@code J} is a uniformly random jitter of up to half of {@code T}. The result is never less - * than the operating cadence. + * Computes how long to wait before the next attempt. After a success this is the operating + * cadence. After a failure it is an exponential backoff with jitter, never less than the + * operating cadence. * * @return the wait in milliseconds */ diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java index 66a978dda..f63a51e87 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java @@ -19,9 +19,8 @@ * advances through the lists (skipping synchronizers that are blocked or backing off), * and closes the previous source when switching. *

- * A synchronizer that reports an unexpected error is put into a backoff rather than removed: - * its slot is skipped until the backoff ends, and the wait doubles on each repeat, up to an - * hour. No synchronizer is ever permanently removed. + * A synchronizer that reports an unexpected error is put into a backoff, and its slot is skipped + * until the backoff ends. No synchronizer is ever permanently removed. *

* Package-private for internal use by FDv2DataSource. */ @@ -41,8 +40,7 @@ final class SourceManager implements Closeable { private SynchronizerFactoryWithState currentSynchronizerFactory; - // Completed, and replaced, whenever a slot's backoff ends or this manager closes, so that a - // caller with no available synchronizer can wait for one. + // Completed and replaced whenever a slot's backoff ends or this manager closes. private LDAwaitFuture availabilityChanged = new LDAwaitFuture<>(); SourceManager( @@ -169,9 +167,8 @@ private FDv2DataSource.DataSourceFactory getNextInitializer() { } /** - * Puts the current synchronizer's slot into backoff after it reported an unexpected error. - * The slot is skipped by {@link #getNextAvailableSynchronizerAndSetActive()} until the backoff - * ends, which is scheduled on the executor. + * Puts the current synchronizer's slot into backoff. The slot is skipped by + * {@link #getNextAvailableSynchronizerAndSetActive()} until the backoff ends. * * @param synchronizerName the name of the synchronizer that failed, for logging * @param nowMillis the current time @@ -212,7 +209,7 @@ void recordCurrentSynchronizerHealthy(long nowMillis) { * Ends a slot's backoff and wakes any caller waiting in {@link #awaitAvailabilityChange()}. * * @return the name of the synchronizer whose error started the backoff, if the slot became - * available; null if nothing changed + * available, or null if nothing changed */ @Nullable String endBackoff(@NonNull SynchronizerFactoryWithState slot) { @@ -256,8 +253,7 @@ Future awaitAvailabilityChange() { } /** - * @return true if a synchronizer earlier in the list than the current one is available, so - * that recovering to it would move to a higher-priority synchronizer + * @return true if a synchronizer earlier in the list than the current one is available */ boolean hasAvailableSynchronizerBeforeCurrent() { synchronized (activeSourceLock) { @@ -304,9 +300,8 @@ Initializer getNextInitializerAndSetActive() { } /** - * True if the current synchronizer is the prime one: no synchronizer before it in the list - * is available or merely backing off. A slot that is backing off still outranks the current - * one, so a recovery condition runs while it is unavailable and can return to it later. + * True if the current synchronizer is the prime one, meaning that no synchronizer before it + * in the list is available or backing off. */ boolean isPrimeSynchronizer() { synchronized (activeSourceLock) { @@ -332,8 +327,7 @@ int getAvailableSynchronizerCount() { } /** - * @return the number of synchronizers that are available or backing off, which is the number - * that could run at some point + * @return the number of synchronizers that are available or backing off */ int getUsableSynchronizerCount() { synchronized (activeSourceLock) { diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java index 3c472f305..0b9d2a49c 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java @@ -41,10 +41,9 @@ * The SDK uses this implementation if streaming is enabled (as it is by default) and the * application is the foreground. The logic for this is in ComponentsImpl.StreamingDataSourceBuilderImpl. *

- * Reconnection is managed by this class rather than by the EventSource library, so that the - * backoff is controlled here: every stream failure, including HTTP statuses such as - * 401 and 403 that used to stop the stream permanently, is retried. Failures that are unlikely to - * resolve on their own are retried with a much longer backoff (see {@link RetryState}). + * Reconnection is managed by this class rather than by the EventSource library. Every stream + * failure is followed by a reconnect after a backoff. Failures that are unlikely to resolve on + * their own, such as HTTP 401, wait much longer. */ final class StreamingDataSource implements DataSource { private static final String METHOD_REPORT = "REPORT"; @@ -71,8 +70,7 @@ final class StreamingDataSource implements DataSource { private final TaskExecutor taskExecutor; private final LDLogger logger; - // The following fields are guarded by the lock on this instance. retryState is only ever - // touched while holding that lock. + // The following fields are guarded by the lock on this instance. private final RetryState retryState; private BackgroundEventSource es; private BackgroundEventHandler handler; @@ -152,8 +150,6 @@ public void onClosed() { public void onMessage(final String name, MessageEvent event) { final String eventData = event.getData(); logger.debug("onMessage: {}: {}", name, eventData); - // A payload on the stream is healthy operation; enough of it in a row resets - // the backoff. synchronized (StreamingDataSource.this) { retryState.recordSuccess(System.currentTimeMillis()); } @@ -186,9 +182,8 @@ public void onError(Throwable t) { failure = new LDFailure("Network error in stream connection", t, LDFailure.FailureType.NETWORK_FAILURE); } - // A StreamException means the connection has ended; every such transport failure - // is a normal failure. Anything else was thrown by this handler while - // processing an event, and the stream is still open, so there is nothing to retry. + // Only a StreamException means the connection has ended. Anything else was thrown + // while processing an event on a stream that is still open. if (t instanceof StreamException) { long delay = scheduleReconnectAfterFailure(unexpected); if (delay >= 0) { @@ -205,8 +200,7 @@ public void onError(Throwable t) { } /** - * Opens a new stream connection if this data source is still running. Called from - * {@link #start} and from the reconnect task scheduled by {@link #scheduleReconnectAfterFailure}. + * Opens a new stream connection if this data source is still running. */ private synchronized void connect() { if (!running) { @@ -232,14 +226,11 @@ private synchronized void connect() { EventSource.Builder esBuilder = new EventSource.Builder(connectStrategy); eventSourceStarted = System.currentTimeMillis(); - // The previous BackgroundEventSource, if any, has already shut itself down: the connection - // error handler below ends the stream after every failure, and BackgroundEventSource - // closes itself when that happens. + // Reconnects run only after the previous stream has ended, so the previous event source + // is not closed here. es = new BackgroundEventSource.Builder(handler, esBuilder) - // The stream thread asks this handler, before it would reconnect, whether an - // error ends the stream. It always does: this class schedules its own reconnect - // from onError with a delay from RetryState. Deciding here, on the stream - // thread, means the library can never race ahead with a reconnect of its own. + // End the stream on every error. This class schedules its own reconnect from + // onError, so the library must never reconnect on its own. .connectionErrorHandler(t -> ConnectionErrorHandler.Action.SHUTDOWN) .build(); es.start(); @@ -355,7 +346,6 @@ private void stopSync() { BackgroundEventSource esToClose; synchronized (this) { running = false; - // A pending backoff wait is interrupted immediately by shutdown. if (pendingReconnect != null) { pendingReconnect.cancel(false); pendingReconnect = null; diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java index 9ad028891..1da3f5ecb 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java @@ -8,10 +8,8 @@ import java.util.concurrent.ScheduledFuture; /** - * Wraps a synchronizer factory with availability state. - * Used by {@link SourceManager} to skip synchronizers that are not currently usable: either - * because they are waiting out a backoff after an unexpected error, or because they are blocked - * (for example the FDv1 fallback synchronizer before the server has directed the SDK to it). + * Wraps a synchronizer factory with availability state. A synchronizer is not usable while it + * is waiting out a backoff after an unexpected error, or while it is blocked. *

* Package-private for internal use by FDv2DataSource. Callers synchronize on the * {@link SourceManager}'s lock. @@ -33,8 +31,7 @@ enum State { private final FDv2DataSource.DataSourceFactory factory; private State state = State.Available; private final boolean isFDv1Fallback; - // Backoff after unexpected errors from this slot's synchronizers. Every failure recorded here - // is unexpected, so only the extended regime ever applies. + // Backoff after unexpected errors from this slot's synchronizers. private final RetryState retryState = RetryState.forSynchronizerSlot(); @Nullable private ScheduledFuture pendingUnblock; diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java index 1981479d2..cb1d711e2 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java @@ -108,8 +108,7 @@ private FDv2DataSource buildDataSource( } /** - * Builds a data source whose timers run on the given executor; tests of backoff behavior - * pass the fake executor so they can move the clock instead of waiting. + * Builds a data source whose timers run on the given executor. */ private FDv2DataSource buildDataSource( MockComponents.MockDataSourceUpdateSink sink, @@ -1527,7 +1526,7 @@ public void statusStaysInterruptedWhileTheOnlySynchronizerBacksOffThenReturnsToV RuntimeException err = new RuntimeException("server error"); AtomicInteger builds = new AtomicInteger(0); - // The synchronizer delivers data and then fails with an unexpected error; when it is + // The synchronizer delivers data and then fails with an unexpected error. When it is // built again it delivers data. FDv2DataSource dataSource = buildDataSource(sink, Collections.emptyList(), diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java index 055ab9405..315ccc435 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java @@ -424,7 +424,7 @@ public void healthyStreamResetsBackoffToInitialDelay() throws Exception { FDv2StreamingSynchronizer sync = makeSynchronizer(server.getUri(), retryStateWithoutJitter(1)); - // The 503 costs one attempt; the reconnect then delivers data and is ended by the + // The 503 costs one attempt. The reconnect then delivers data and is ended by the // server. assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); assertEquals(NORMAL_DELAY_MILLIS, fakeExecutor.awaitScheduledDelayMillis(5000)); diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/LDUtilTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/LDUtilTest.java index 77ff0b583..1bfce9147 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/LDUtilTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/LDUtilTest.java @@ -30,7 +30,7 @@ public void testSanitizeSpaces() { @Test public void isHttpErrorRecoverableClassifiesStatusCodes() { - // 400, 408, 429, and 5xx are normal; every other 4xx is unexpected. + // 400, 408, 429, and 5xx are normal. Every other 4xx is unexpected. Assert.assertTrue(LDUtil.isHttpErrorRecoverable(400)); Assert.assertTrue(LDUtil.isHttpErrorRecoverable(408)); Assert.assertTrue(LDUtil.isHttpErrorRecoverable(429)); diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java index 76068bfc8..6a1fb7692 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java @@ -425,7 +425,7 @@ public void twoConsecutiveSuccessfulPollsResetBackoff() { fakeTaskExecutor.runDueTasks(); assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); - // One success returns to the regular interval; a second one clears the backoff, so the + // One success returns to the regular interval. A second one clears the backoff, so the // 500 that follows is retried at the regular interval instead of a doubled extended delay. fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java index 0757c3aca..f920374a7 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java @@ -9,7 +9,7 @@ import java.util.Random; /** - * Unit tests for {@link RetryState}: two-regime exponential backoff with jitter. + * Unit tests for {@link RetryState}. */ public class RetryStateTest { private static final long SECOND = 1_000L; @@ -238,7 +238,7 @@ public void streamingSuccessAloneDoesNotReset() { state.recordFailure(true, 0); state.recordSuccess(10 * SECOND); state.recordSuccess(10 * MINUTE); - // The reset is applied when the next failure is recorded, and only then; until that + // The reset is applied when the next failure is recorded, and only then. Until that // point the regime is unchanged. assertTrue(state.isInExtendedRegime()); } diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java index b029417d6..eedcf84d1 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java @@ -23,7 +23,7 @@ /** * Unit tests for {@link SourceManager}'s handling of synchronizers that report unexpected - * errors: they are put aside for a backoff and then become available again. + * errors. Such a synchronizer is put aside for a backoff and then becomes available again. */ public class SourceManagerTest { @@ -37,7 +37,7 @@ public void tearDown() { executor.shutdownNow(); } - /** A synchronizer that never produces a result; only its name matters here. */ + /** A synchronizer that never produces a result. Only its name matters here. */ private static final class NamedSynchronizer implements Synchronizer { private final String name; diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java index ad6ab87c4..be360f8da 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java @@ -879,7 +879,7 @@ public void sustainedUnexpectedErrorsKeepReconnectingWithGrowingDelay() throws E TrackingCallback callback = new TrackingCallback(); startDataSource(sds, callback); - // Every attempt fails; each one is reported and the next is scheduled with double + // Every attempt fails. Each one is reported and the next is scheduled with double // the delay, and the data source never gives up. for (long expectedDelay : new long[] {EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 2, EXTENDED_DELAY_MILLIS * 4}) { assertNotNull(callback.awaitError()); @@ -915,7 +915,7 @@ public void healthyStreamResetsBackoffToNormalDelay() throws Exception { TrackingCallback callback = new TrackingCallback(); startDataSource(sds, callback); - // The 401 puts the data source in the extended regime; the reconnect then delivers + // The 401 puts the data source in the extended regime. The reconnect then delivers // data and is ended by the server. assertNotNull(callback.awaitError()); fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); diff --git a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java index 9779d16f5..f95fefe10 100644 --- a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java +++ b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java @@ -21,14 +21,12 @@ /** * A {@link ScheduledExecutorService} with a manual clock, for unit tests of timer-driven code - * whose tasks block (for example on network I/O) and so must run on real background threads. + * whose tasks must run on real background threads. *

- * Tasks submitted for immediate execution, and scheduled tasks with no delay, run right away on - * a background thread. A scheduled task with a delay is held until the test moves the clock past - * that delay with {@link #advanceTime(long)}, and is then released to a background thread. A test - * can wait for the code under test to schedule something with - * {@link #awaitScheduledDelayMillis(long)} and assert on the delay it chose, and can assert that - * a held task has not run because the clock has not moved, without sleeping. + * Tasks submitted without a delay run right away on a background thread. A task scheduled with a + * delay is held until {@link #advanceTime(long)} moves the clock past that delay. Tests can wait + * for a scheduling event with {@link #awaitScheduledDelayMillis(long)} and inspect held tasks + * with {@link #pendingDelaysMillis()}. */ public class FakeScheduledExecutorService extends AbstractExecutorService implements ScheduledExecutorService { private final ExecutorService delegate = Executors.newCachedThreadPool(); @@ -78,7 +76,7 @@ public long awaitScheduledDelayMillis(long timeoutMillis) throws InterruptedExce /** * @return how long, from the fake clock's current time, until each held task is due, in the - * order the tasks were scheduled; cancelled and already-released tasks are omitted + * order the tasks were scheduled. Cancelled and already-released tasks are omitted. */ public List pendingDelaysMillis() { List delays = new ArrayList<>(); diff --git a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java index 39eda499d..aac5059a5 100644 --- a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java +++ b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java @@ -11,12 +11,9 @@ import java.util.concurrent.TimeUnit; /** - * A {@link TaskExecutor} with a manual clock, for unit tests of timer-driven code. - *

- * A scheduled task does not run until the test moves the clock past its delay with - * {@link #advanceTime(long)}, and then it runs synchronously on the calling thread. A test can - * therefore assert exactly which delays were scheduled, and that nothing ran before the clock - * moved, without sleeping. + * A {@link TaskExecutor} with a manual clock, for unit tests of timer-driven code. A scheduled + * task is held until {@link #advanceTime(long)} moves the clock past its delay, and then runs + * synchronously on the calling thread. */ public class FakeTaskExecutor implements TaskExecutor { private final Object lock = new Object(); @@ -51,7 +48,7 @@ public void close() { /** * @return how long, from the fake clock's current time, until each pending task is due, in - * the order the tasks were scheduled; cancelled tasks are omitted + * the order the tasks were scheduled. Cancelled tasks are omitted. */ public List pendingDelaysMillis() { List delays = new ArrayList<>(); From bf180063afbd0090aff915d0a602caf51d8cf1c2 Mon Sep 17 00:00:00 2001 From: Bee Klimt Date: Tue, 29 Sep 2026 16:12:47 -0700 Subject: [PATCH 3/5] refactor: Split RetryState into RetryRegime, StreamingRetryState, and PollingRetryState --- .../android/FDv2StreamingSynchronizer.java | 8 +- .../com/launchdarkly/sdk/android/LDUtil.java | 10 +- .../sdk/android/PollingDataSource.java | 12 +- .../sdk/android/PollingRetryState.java | 90 +++++ .../launchdarkly/sdk/android/RetryRegime.java | 72 ++++ .../launchdarkly/sdk/android/RetryState.java | 256 ------------ .../sdk/android/SourceManager.java | 6 +- .../sdk/android/StreamingDataSource.java | 8 +- .../sdk/android/StreamingRetryState.java | 108 +++++ .../android/SynchronizerFactoryWithState.java | 13 +- .../sdk/android/FDv2DataSourceTest.java | 6 +- .../FDv2StreamingSynchronizerTest.java | 12 +- .../sdk/android/PollingDataSourceTest.java | 9 +- .../sdk/android/PollingRetryStateTest.java | 127 ++++++ .../sdk/android/RetryRegimeTest.java | 82 ++++ .../sdk/android/RetryStateTest.java | 376 ------------------ .../sdk/android/SourceManagerTest.java | 8 +- .../sdk/android/StreamingDataSourceTest.java | 14 +- .../sdk/android/StreamingRetryStateTest.java | 167 ++++++++ 19 files changed, 702 insertions(+), 682 deletions(-) create mode 100644 launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingRetryState.java create mode 100644 launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryRegime.java delete mode 100644 launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java create mode 100644 launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingRetryState.java create mode 100644 launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingRetryStateTest.java create mode 100644 launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryRegimeTest.java delete mode 100644 launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java create mode 100644 launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingRetryStateTest.java diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java index 952a36521..eaf7cbc43 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java @@ -79,7 +79,7 @@ final class FDv2StreamingSynchronizer implements Synchronizer { // The following are only touched by the current connection attempt. Attempts never overlap. private final FDv2ProtocolHandler protocolHandler = new FDv2ProtocolHandler(); - private final RetryState retryState; + private final StreamingRetryState retryState; // Set while handling a message when the current connection must be dropped and a new one // made, along with the wait before doing so. private boolean restartRequested = false; @@ -124,11 +124,11 @@ final class FDv2StreamingSynchronizer implements Synchronizer { ) { this(evaluationContext, selectorSource, streamBaseUri, streamRequestPath, requestor, initialReconnectDelayMillis, evaluationReasons, useReport, httpProperties, executor, - logger, diagnosticStore, RetryState.forStreaming(initialReconnectDelayMillis)); + logger, diagnosticStore, new StreamingRetryState(initialReconnectDelayMillis)); } /** - * This constructor allows tests to supply a {@link RetryState} with short delays. See the + * This constructor allows tests to supply a {@link StreamingRetryState} with short delays. See the * other constructor for the remaining parameters. * * @param retryState the retry state that decides the wait before each reconnection @@ -146,7 +146,7 @@ final class FDv2StreamingSynchronizer implements Synchronizer { @NonNull ScheduledExecutorService executor, @NonNull LDLogger logger, @Nullable DiagnosticStore diagnosticStore, - @NonNull RetryState retryState + @NonNull StreamingRetryState retryState ) { this.evaluationContext = evaluationContext; this.selectorSource = selectorSource; diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java index 521562e40..71a56df1d 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/LDUtil.java @@ -214,9 +214,6 @@ public void updateHeaders(Map headers) { /** * Classifies an HTTP error status as either a {@code normal} failure, which may resolve on its * own if retried soon, or an {@code unexpected} failure, which is not expected to. - *

- * 400, 408, 429, and all 5xx statuses are {@code normal}. Every other 4xx status is - * {@code unexpected}. Any other status is {@code normal}. * * @param statusCode the HTTP status * @return true if the status is {@code normal}, or false if it is {@code unexpected} @@ -236,11 +233,8 @@ static boolean isHttpErrorRecoverable(int statusCode) { } /** - * Classifies a data source failure as {@code unexpected} or {@code normal}. - *

- * Only an HTTP response failure whose status is {@code unexpected} under - * {@link #isHttpErrorRecoverable(int)} is {@code unexpected}. Every other failure is - * {@code normal}. + * Classifies a data source failure as either a {@code normal} failure, which may resolve on + * its own if retried soon, or an {@code unexpected} failure, which is not expected to. * * @param failure the failure reported by a data source, or null * @return true if the failure is {@code unexpected}, or false if it is {@code normal} diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java index 1ed793316..86ae8be6d 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingDataSource.java @@ -34,7 +34,7 @@ final class PollingDataSource implements DataSource { final AtomicReference> currentPollTask = new AtomicReference<>(); // visible for testing // Guarded by the lock on this instance. - private final RetryState retryState; + private final PollingRetryState retryState; private boolean running = false; /** @@ -62,11 +62,11 @@ final class PollingDataSource implements DataSource { LDLogger logger ) { this(context, dataSourceUpdateSink, initialDelayMillis, pollIntervalMillis, maxNumberOfPolls, - fetcher, platformState, taskExecutor, RetryState.forPolling(pollIntervalMillis), logger); + fetcher, platformState, taskExecutor, new PollingRetryState(pollIntervalMillis), logger); } /** - * This constructor allows tests to supply a {@link RetryState} with short delays. See the + * This constructor allows tests to supply a {@link PollingRetryState} with short delays. See the * other constructor for the remaining parameters. * * @param retryState the retry state that decides the wait after a failed poll @@ -80,7 +80,7 @@ final class PollingDataSource implements DataSource { FeatureFetcher fetcher, PlatformState platformState, TaskExecutor taskExecutor, - RetryState retryState, + PollingRetryState retryState, LDLogger logger ) { this.context = context; @@ -147,7 +147,7 @@ private void poll(final Callback resultCallback) { public void onSuccess(Boolean result) { long delay; synchronized (PollingDataSource.this) { - retryState.recordSuccess(System.currentTimeMillis()); + retryState.recordSuccess(); delay = retryState.nextDelayMillis(); } resultCallback.onSuccess(result); @@ -159,7 +159,7 @@ public void onError(Throwable error) { boolean unexpected = LDUtil.isUnexpectedFailure(error); long delay; synchronized (PollingDataSource.this) { - retryState.recordFailure(unexpected, System.currentTimeMillis()); + retryState.recordFailure(unexpected); delay = retryState.nextDelayMillis(); } if (unexpected) { diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingRetryState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingRetryState.java new file mode 100644 index 000000000..130703034 --- /dev/null +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/PollingRetryState.java @@ -0,0 +1,90 @@ +package com.launchdarkly.sdk.android; + +import androidx.annotation.NonNull; + +import java.util.Random; + +/** + * Computes the wait before a polling component polls again. The backoff clears after enough + * consecutive successful polls. + *

+ * This class is not thread-safe. The owning component must serialize access to it. + */ +final class PollingRetryState { + /** + * The backoff clears after this many consecutive successful polls. + */ + static final int RESET_THRESHOLD_SUCCESSES = 2; + + private final long pollIntervalMillis; + private final RetryRegime extended; + private final Random random; + + private boolean inExtendedRegime = false; + private int attempts = 0; + private int consecutiveSuccesses = 0; + + /** + * @param pollIntervalMillis the configured poll interval, which must be positive + */ + PollingRetryState(long pollIntervalMillis) { + this(pollIntervalMillis, RetryRegime.EXTENDED, new Random()); + } + + /** + * Creates a retry state with an explicit extended regime. + * + * @param pollIntervalMillis the configured poll interval, which must be positive + * @param extended the regime entered by an {@code unexpected} failure, raised to the + * poll interval where smaller + * @param random source of jitter + */ + PollingRetryState( + long pollIntervalMillis, + @NonNull RetryRegime extended, + @NonNull Random random + ) { + this.pollIntervalMillis = Math.max(1, pollIntervalMillis); + this.extended = extended.atLeast(this.pollIntervalMillis); + this.random = random; + } + + /** + * Records a successful poll. The backoff clears after {@link #RESET_THRESHOLD_SUCCESSES} + * consecutive successes. + */ + void recordSuccess() { + consecutiveSuccesses++; + if (consecutiveSuccesses >= RESET_THRESHOLD_SUCCESSES) { + inExtendedRegime = false; + attempts = 0; + } + } + + /** + * Records a failed poll. Call this before {@link #nextDelayMillis()}. + * + * @param unexpected true if the failure is {@code unexpected}, false if {@code normal} + */ + void recordFailure(boolean unexpected) { + consecutiveSuccesses = 0; + if (unexpected && !inExtendedRegime) { + inExtendedRegime = true; + attempts = 1; + return; + } + attempts++; + } + + /** + * @return the wait in milliseconds before the next poll + */ + long nextDelayMillis() { + // A normal failure does not grow the delay, and a success shows the service works. Only + // an unresolved unexpected failure waits longer than the poll interval. + if (!inExtendedRegime || consecutiveSuccesses > 0) { + return pollIntervalMillis; + } + return Math.max(pollIntervalMillis, extended.jitteredDelayMillis(attempts, random)); + } +} diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryRegime.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryRegime.java new file mode 100644 index 000000000..e07f8d2ed --- /dev/null +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryRegime.java @@ -0,0 +1,72 @@ +package com.launchdarkly.sdk.android; + +import androidx.annotation.NonNull; + +import java.util.Random; + +/** + * The delay bounds for one retry regime. + */ +final class RetryRegime { + /** + * The base delay for the first attempt after an {@code unexpected} failure. + */ + static final long EXTENDED_INITIAL_DELAY_MILLIS = 5L * 60_000L; + /** + * The largest delay after an {@code unexpected} failure. + */ + static final long EXTENDED_MAX_DELAY_MILLIS = 60L * 60_000L; + /** + * The regime entered after an {@code unexpected} failure. + */ + static final RetryRegime EXTENDED = + new RetryRegime(EXTENDED_INITIAL_DELAY_MILLIS, EXTENDED_MAX_DELAY_MILLIS); + + // Caps the doubling so that initialDelayMillis * 2^exponent cannot overflow. + private static final int MAX_EXPONENT = 30; + + /** + * The base delay for the first attempt in this regime. + */ + final long initialDelayMillis; + /** + * The largest delay this regime produces. + */ + final long maxDelayMillis; + + /** + * @param initialDelayMillis the base delay for the first attempt, floored at zero + * @param maxDelayMillis the largest delay, raised to the initial delay if smaller + */ + RetryRegime(long initialDelayMillis, long maxDelayMillis) { + this.initialDelayMillis = Math.max(0, initialDelayMillis); + this.maxDelayMillis = Math.max(this.initialDelayMillis, maxDelayMillis); + } + + /** + * @param floorMillis the smallest value allowed for either bound + * @return a regime with the same bounds, except that neither is below the floor + */ + @NonNull + RetryRegime atLeast(long floorMillis) { + return new RetryRegime( + Math.max(initialDelayMillis, floorMillis), Math.max(maxDelayMillis, floorMillis)); + } + + /** + * The exponential backoff for an attempt, less a random amount of up to half of it. + * + * @param attempts the number of failures so far, starting at 1 + * @param random source of jitter + * @return the wait in milliseconds + */ + long jitteredDelayMillis(int attempts, @NonNull Random random) { + int exponent = Math.min(Math.max(0, attempts - 1), MAX_EXPONENT); + long base = initialDelayMillis * (1L << exponent); + if (base < 0 || base > maxDelayMillis) { // negative means the multiplication overflowed + base = maxDelayMillis; + } + long jitter = base > 1 ? (long) (random.nextDouble() * (base / 2.0)) : 0; + return base - jitter; + } +} diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java deleted file mode 100644 index 4616cd237..000000000 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/RetryState.java +++ /dev/null @@ -1,256 +0,0 @@ -package com.launchdarkly.sdk.android; - -import androidx.annotation.NonNull; - -import java.util.Random; - -/** - * Backoff state for a long-running component such as a streaming or polling data source. - *

- * The owning component reports each success and failure, then asks how long to wait before its - * next attempt. A failure is either {@code normal} or {@code unexpected}. An {@code unexpected} - * failure, such as an HTTP 401, moves the component to a much longer backoff, which it leaves - * only after enough healthy operation. - *

- * Timestamps are passed in explicitly. Any monotonic millisecond clock may be used, as long as - * the same clock is used for every call. - *

- * This class is not thread-safe. The owning component must serialize access to it. - */ -final class RetryState { - /** - * The normal-regime ceiling for the wait between attempts. - */ - static final long NORMAL_MAX_DELAY_MILLIS = 30_000L; - /** - * The base delay for the first attempt after an {@code unexpected} failure. - */ - static final long EXTENDED_INITIAL_DELAY_MILLIS = 5L * 60_000L; - /** - * The extended-regime ceiling for the wait between attempts. - */ - static final long EXTENDED_MAX_DELAY_MILLIS = 60L * 60_000L; - /** - * A streaming component resets its retry state after this much continuous healthy operation. - */ - static final long STREAMING_RESET_THRESHOLD_MILLIS = 60_000L; - /** - * A polling component resets its retry state after this many consecutive successful polls. - */ - static final int POLLING_RESET_THRESHOLD_SUCCESSES = 2; - - // Caps the doubling so that initialDelay * 2^exponent cannot overflow. - private static final int MAX_EXPONENT = 30; - - private final long normalInitialDelayMillis; - private final long normalMaxDelayMillis; - private final long extendedInitialDelayMillis; - private final long extendedMaxDelayMillis; - // The interval at which the component operates when healthy, or zero if it has none. - private final long operatingCadenceMillis; - // Duration-based reset threshold, or zero if this component does not use one. - private final long healthyResetThresholdMillis; - // Count-based reset threshold, or zero if this component does not use one. - private final int successResetThreshold; - private final Random random; - - private int attempts = 0; - private boolean extended = false; - private boolean lastOperationFailed = false; - // Start of the current stretch of healthy operation, or zero if not currently healthy. - private long healthySinceMillis = 0; - private int consecutiveSuccesses = 0; - - /** - * Creates the retry state for a streaming data source or synchronizer. Normal failures back off - * from the configured initial reconnect delay. The state resets after - * {@link #STREAMING_RESET_THRESHOLD_MILLIS} of healthy operation. - * - * @param initialReconnectDelayMillis the configured initial reconnect delay, which may be zero - * @return the retry state - */ - @NonNull - static RetryState forStreaming(long initialReconnectDelayMillis) { - long initial = Math.max(0, initialReconnectDelayMillis); - return new RetryState( - initial, - Math.max(NORMAL_MAX_DELAY_MILLIS, initial), - Math.max(EXTENDED_INITIAL_DELAY_MILLIS, initial), - Math.max(EXTENDED_MAX_DELAY_MILLIS, initial), - 0, - STREAMING_RESET_THRESHOLD_MILLIS, - 0, - new Random()); - } - - /** - * Creates the retry state for a polling data source or synchronizer. Normal failures keep the - * poll interval, and no wait is ever shorter than it. The state resets after - * {@link #POLLING_RESET_THRESHOLD_SUCCESSES} consecutive successful polls. - * - * @param pollIntervalMillis the configured poll interval, which must be positive - * @return the retry state - */ - @NonNull - static RetryState forPolling(long pollIntervalMillis) { - long interval = Math.max(1, pollIntervalMillis); - return new RetryState( - interval, - interval, - Math.max(EXTENDED_INITIAL_DELAY_MILLIS, interval), - Math.max(EXTENDED_MAX_DELAY_MILLIS, interval), - interval, - 0, - POLLING_RESET_THRESHOLD_SUCCESSES, - new Random()); - } - - /** - * Creates the retry state for one synchronizer slot. Every failure, {@code normal} or - * {@code unexpected}, backs off in the extended regime. The state resets after - * {@link #STREAMING_RESET_THRESHOLD_MILLIS} of healthy operation. - * - * @return the retry state - */ - @NonNull - static RetryState forSynchronizerSlot() { - return new RetryState( - EXTENDED_INITIAL_DELAY_MILLIS, - EXTENDED_MAX_DELAY_MILLIS, - EXTENDED_INITIAL_DELAY_MILLIS, - EXTENDED_MAX_DELAY_MILLIS, - 0, - STREAMING_RESET_THRESHOLD_MILLIS, - 0, - new Random()); - } - - /** - * Creates a retry state with explicit parameters, for tests. Production code should use one - * of the static factories. - * - * @param normalInitialDelayMillis base delay for the first retry in the normal regime - * @param normalMaxDelayMillis ceiling on the wait in the normal regime - * @param extendedInitialDelayMillis base delay for the first retry in the extended regime, - * raised to the normal initial delay if smaller - * @param extendedMaxDelayMillis ceiling on the wait in the extended regime - * @param operatingCadenceMillis the operating cadence, or zero if the component has none - * @param healthyResetThresholdMillis how long healthy operation must last before the state - * resets, or zero if this component does not reset on - * duration - * @param successResetThreshold how many consecutive successes reset the state, or zero - * if this component does not reset on a count - * @param random source of jitter - */ - RetryState( - long normalInitialDelayMillis, - long normalMaxDelayMillis, - long extendedInitialDelayMillis, - long extendedMaxDelayMillis, - long operatingCadenceMillis, - long healthyResetThresholdMillis, - int successResetThreshold, - @NonNull Random random - ) { - this.normalInitialDelayMillis = Math.max(0, normalInitialDelayMillis); - this.normalMaxDelayMillis = Math.max(this.normalInitialDelayMillis, normalMaxDelayMillis); - this.extendedInitialDelayMillis = Math.max(this.normalInitialDelayMillis, extendedInitialDelayMillis); - this.extendedMaxDelayMillis = Math.max(this.extendedInitialDelayMillis, extendedMaxDelayMillis); - this.operatingCadenceMillis = Math.max(0, operatingCadenceMillis); - this.healthyResetThresholdMillis = Math.max(0, healthyResetThresholdMillis); - this.successResetThreshold = Math.max(0, successResetThreshold); - this.random = random; - } - - /** - * Records healthy operation, such as a payload received on a stream or a successful poll. - * Enough healthy operation resets the state. After a success the next wait is the operating - * cadence, even if the state has not reset. - * - * @param nowMillis the current time - */ - void recordSuccess(long nowMillis) { - lastOperationFailed = false; - if (healthyResetThresholdMillis > 0 && healthySinceMillis == 0) { - healthySinceMillis = nowMillis; - } - if (successResetThreshold > 0) { - consecutiveSuccesses++; - if (consecutiveSuccesses >= successResetThreshold) { - reset(); - } - } - } - - /** - * Records a failure. Call this before {@link #nextDelayMillis()}. - * - * @param unexpected true if the failure is classified as {@code unexpected}, false if - * {@code normal} - * @param nowMillis the current time - */ - void recordFailure(boolean unexpected, long nowMillis) { - if (healthyResetThresholdMillis > 0 && healthySinceMillis != 0 - && nowMillis - healthySinceMillis >= healthyResetThresholdMillis) { - reset(); - } - healthySinceMillis = 0; - consecutiveSuccesses = 0; - lastOperationFailed = true; - - if (unexpected && !extended) { - // Start the attempt count over so the first extended wait is the extended initial - // delay. - extended = true; - attempts = 1; - } else { - attempts++; - } - } - - /** - * Clears the retry state back to no attempts and the normal regime. - */ - void reset() { - attempts = 0; - extended = false; - consecutiveSuccesses = 0; - } - - /** - * Computes how long to wait before the next attempt. After a success this is the operating - * cadence. After a failure it is an exponential backoff with jitter, never less than the - * operating cadence. - * - * @return the wait in milliseconds - */ - long nextDelayMillis() { - if (!lastOperationFailed) { - return operatingCadenceMillis; - } - long initial = extended ? extendedInitialDelayMillis : normalInitialDelayMillis; - long max = extended ? extendedMaxDelayMillis : normalMaxDelayMillis; - int exponent = Math.min(Math.max(0, attempts - 1), MAX_EXPONENT); - long base = initial * (1L << exponent); - if (base < 0 || base > max) { // negative means the multiplication overflowed - base = max; - } - long jitter = base > 1 ? (long) (random.nextDouble() * (base / 2.0)) : 0; - return Math.max(operatingCadenceMillis, base - jitter); - } - - /** - * @return the number of failures counted since the last reset - */ - int getAttempts() { - return attempts; - } - - /** - * @return true if an {@code unexpected} failure has moved this component to the extended - * regime and no reset has happened since - */ - boolean isInExtendedRegime() { - return extended; - } -} diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java index f63a51e87..bf4e16b16 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java @@ -171,7 +171,8 @@ private FDv2DataSource.DataSourceFactory getNextInitializer() { * {@link #getNextAvailableSynchronizerAndSetActive()} until the backoff ends. * * @param synchronizerName the name of the synchronizer that failed, for logging - * @param nowMillis the current time + * @param nowMillis the current time in milliseconds, on the same clock as every other + * call on this manager * @return how long the slot stays in backoff, in milliseconds, or -1 if there is no current * synchronizer */ @@ -196,6 +197,9 @@ public void run() { /** * Records healthy operation by the current synchronizer, which counts toward resetting its * slot's backoff. + * + * @param nowMillis the current time in milliseconds, on the same clock as every other call on + * this manager */ void recordCurrentSynchronizerHealthy(long nowMillis) { synchronized (activeSourceLock) { diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java index 0b9d2a49c..32bbd1249 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingDataSource.java @@ -71,7 +71,7 @@ final class StreamingDataSource implements DataSource { private final LDLogger logger; // The following fields are guarded by the lock on this instance. - private final RetryState retryState; + private final StreamingRetryState retryState; private BackgroundEventSource es; private BackgroundEventHandler handler; private ScheduledFuture pendingReconnect; @@ -89,11 +89,11 @@ final class StreamingDataSource implements DataSource { boolean streamEvenInBackground ) { this(clientContext, context, dataSourceUpdateSink, fetcher, initialReconnectDelayMillis, - streamEvenInBackground, RetryState.forStreaming(initialReconnectDelayMillis)); + streamEvenInBackground, new StreamingRetryState(initialReconnectDelayMillis)); } /** - * This constructor allows tests to supply a {@link RetryState} with short delays. + * This constructor allows tests to supply a {@link StreamingRetryState} with short delays. */ StreamingDataSource( @NonNull ClientContext clientContext, @@ -102,7 +102,7 @@ final class StreamingDataSource implements DataSource { @NonNull FeatureFetcher fetcher, int initialReconnectDelayMillis, boolean streamEvenInBackground, - @NonNull RetryState retryState + @NonNull StreamingRetryState retryState ) { this.context = context; this.dataSourceUpdateSink = dataSourceUpdateSink; diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingRetryState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingRetryState.java new file mode 100644 index 000000000..44a4ce1aa --- /dev/null +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/StreamingRetryState.java @@ -0,0 +1,108 @@ +package com.launchdarkly.sdk.android; + +import androidx.annotation.NonNull; + +import java.util.Random; + +/** + * Computes the wait before a streaming connection is retried. The backoff clears after enough + * continuous healthy operation. + *

+ * This class is not thread-safe. The owning component must serialize access to it. + */ +final class StreamingRetryState { + /** + * The largest delay after a {@code normal} failure. + */ + static final long NORMAL_MAX_DELAY_MILLIS = 30_000L; + /** + * The backoff clears after this much continuous healthy operation. + */ + static final long RESET_THRESHOLD_MILLIS = 60_000L; + + private final RetryRegime normal; + private final RetryRegime extended; + private final long resetThresholdMillis; + private final Random random; + + private boolean inExtendedRegime = false; + private int attempts = 0; + private boolean healthy = false; + private long healthySinceMillis = 0; + + /** + * @param initialReconnectDelayMillis the delay before the first reconnect after a + * {@code normal} failure, which may be zero + */ + StreamingRetryState(long initialReconnectDelayMillis) { + this(new RetryRegime(initialReconnectDelayMillis, NORMAL_MAX_DELAY_MILLIS), + RetryRegime.EXTENDED.atLeast(initialReconnectDelayMillis), + RESET_THRESHOLD_MILLIS, + new Random()); + } + + /** + * Creates a retry state with explicit regimes. + * + * @param normal the regime for {@code normal} failures + * @param extended the regime entered by an {@code unexpected} failure + * @param resetThresholdMillis how long healthy operation must last before the backoff clears + * @param random source of jitter + */ + StreamingRetryState( + @NonNull RetryRegime normal, + @NonNull RetryRegime extended, + long resetThresholdMillis, + @NonNull Random random + ) { + this.normal = normal; + this.extended = extended; + this.resetThresholdMillis = resetThresholdMillis; + this.random = random; + } + + /** + * Records healthy operation, such as a payload received on the stream. Healthy operation is + * measured from the first call after a failure. + * + * @param nowMillis the current time in milliseconds. Only differences between times matter, + * so any clock will do as long as every call on this instance uses the same + * one. + */ + void recordSuccess(long nowMillis) { + if (!healthy) { + healthy = true; + healthySinceMillis = nowMillis; + } + } + + /** + * Records a failure. Call this before {@link #nextDelayMillis()}. + * + * @param unexpected true if the failure is {@code unexpected}, false if {@code normal} + * @param nowMillis the current time in milliseconds. Only differences between times matter, + * so any clock will do as long as every call on this instance uses the same + * one. + */ + void recordFailure(boolean unexpected, long nowMillis) { + if (healthy && nowMillis - healthySinceMillis >= resetThresholdMillis) { + // The connection worked for long enough that this failure starts the backoff over. + inExtendedRegime = false; + attempts = 0; + } + healthy = false; + if (unexpected && !inExtendedRegime) { + inExtendedRegime = true; + attempts = 1; + return; + } + attempts++; + } + + /** + * @return the wait in milliseconds before the next connection attempt + */ + long nextDelayMillis() { + return (inExtendedRegime ? extended : normal).jitteredDelayMillis(attempts, random); + } +} diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java index 1da3f5ecb..d92344be2 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java @@ -5,6 +5,7 @@ import com.launchdarkly.sdk.android.subsystems.Synchronizer; +import java.util.Random; import java.util.concurrent.ScheduledFuture; /** @@ -32,7 +33,11 @@ enum State { private State state = State.Available; private final boolean isFDv1Fallback; // Backoff after unexpected errors from this slot's synchronizers. - private final RetryState retryState = RetryState.forSynchronizerSlot(); + private final StreamingRetryState retryState = new StreamingRetryState( + RetryRegime.EXTENDED, + RetryRegime.EXTENDED, + StreamingRetryState.RESET_THRESHOLD_MILLIS, + new Random()); @Nullable private ScheduledFuture pendingUnblock; @Nullable @@ -70,7 +75,8 @@ boolean isFDv1Fallback() { /** * Records healthy operation by this slot's synchronizer. * - * @param nowMillis the current time + * @param nowMillis the current time in milliseconds, on the same clock as every other call on + * this instance */ void recordHealthy(long nowMillis) { retryState.recordSuccess(nowMillis); @@ -80,7 +86,8 @@ void recordHealthy(long nowMillis) { * Records an unexpected error from this slot's synchronizer and puts the slot into backoff. * * @param synchronizerName the name of the synchronizer that failed, for logging - * @param nowMillis the current time + * @param nowMillis the current time in milliseconds, on the same clock as every other + * call on this instance * @return how long the slot stays in backoff, in milliseconds */ long startBackoff(@NonNull String synchronizerName, long nowMillis) { diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java index cb1d711e2..38e08a8c2 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java @@ -652,14 +652,14 @@ public void allSynchronizersFailingWithUnexpectedErrorsAreRetriedAfterBackoff() // backoff has elapsed. assertEquals(Arrays.asList(DataSourceState.INTERRUPTED, DataSourceState.INTERRUPTED), sink.awaitStatuses(2, AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS)); - fakeExecutor.advanceTime(RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 - 1); + fakeExecutor.advanceTime(RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS / 2 - 1); assertEquals(1, firstBuilds.get()); assertEquals(1, secondBuilds.get()); // Once the backoffs end, whichever synchronizer returns first is tried again and // initialization completes. The two backoffs have independent jitter, so either may be // the one that is rebuilt. - fakeExecutor.advanceTime(RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 + 1); + fakeExecutor.advanceTime(RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS / 2 + 1); assertTrue(startCallback.await(AWAIT_TIMEOUT_SECONDS * 1000)); assertEquals(3, firstBuilds.get() + secondBuilds.get()); stopDataSource(dataSource); @@ -1545,7 +1545,7 @@ public void statusStaysInterruptedWhileTheOnlySynchronizerBacksOffThenReturnsToV // Once the backoff ends the synchronizer is tried again and the status returns to VALID, // never having reached OFF. - fakeExecutor.advanceTime(RetryState.EXTENDED_INITIAL_DELAY_MILLIS); + fakeExecutor.advanceTime(RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); assertEquals(DataSourceState.VALID, sink.awaitStatus(AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS)); assertEquals(2, builds.get()); stopDataSource(dataSource); diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java index 315ccc435..a52da249c 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java @@ -103,16 +103,16 @@ private FDv2StreamingSynchronizer makeSynchronizer( } // The tests of backoff behavior drive the synchronizer's timers with a - // FakeScheduledExecutorService and a RetryState without jitter, so the delay chosen for each - // reconnect can be asserted exactly instead of waited for. + // FakeScheduledExecutorService and a StreamingRetryState without jitter, so the delay chosen + // for each reconnect can be asserted exactly instead of waited for. private static final long NORMAL_DELAY_MILLIS = 1000; private static final long NORMAL_MAX_DELAY_MILLIS = 4000; // A healthy-operation threshold no test reaches. private static final long NEVER_RESET_MILLIS = 60_000; - private static RetryState retryStateWithoutJitter(long healthyResetThresholdMillis) { - return new RetryState(NORMAL_DELAY_MILLIS, NORMAL_MAX_DELAY_MILLIS, - NORMAL_DELAY_MILLIS, NORMAL_MAX_DELAY_MILLIS, 0, healthyResetThresholdMillis, 0, + private static StreamingRetryState retryStateWithoutJitter(long healthyResetThresholdMillis) { + RetryRegime regime = new RetryRegime(NORMAL_DELAY_MILLIS, NORMAL_MAX_DELAY_MILLIS); + return new StreamingRetryState(regime, regime, healthyResetThresholdMillis, new Random() { @Override public double nextDouble() { @@ -121,7 +121,7 @@ public double nextDouble() { }); } - private FDv2StreamingSynchronizer makeSynchronizer(URI streamBaseUri, RetryState retryState) { + private FDv2StreamingSynchronizer makeSynchronizer(URI streamBaseUri, StreamingRetryState retryState) { return new FDv2StreamingSynchronizer( CONTEXT, EMPTY_SELECTOR_SOURCE, streamBaseUri, STREAM_PATH, null, 1, false, false, diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java index 6a1fb7692..9e143459f 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java @@ -291,7 +291,7 @@ public void terminatesAfterMaxNumberOfPolls() throws Exception { // --- backoff after failures --- // - // These tests drive the data source's timers with a FakeTaskExecutor and a RetryState + // These tests drive the data source's timers with a FakeTaskExecutor and a PollingRetryState // without jitter. The mock fetcher answers synchronously, so every poll and its outcome // happen inside advanceTime() and the scheduled delays can be asserted exactly. @@ -304,10 +304,9 @@ private static LDInvalidResponseCodeFailure httpFailure(int status) { return new LDInvalidResponseCodeFailure("test failure", status, LDUtil.isHttpErrorRecoverable(status)); } - private static RetryState retryStateWithoutJitter() { - return new RetryState(POLL_INTERVAL_MILLIS, POLL_INTERVAL_MILLIS, - EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 4, POLL_INTERVAL_MILLIS, - 0, RetryState.POLLING_RESET_THRESHOLD_SUCCESSES, + private static PollingRetryState retryStateWithoutJitter() { + return new PollingRetryState(POLL_INTERVAL_MILLIS, + new RetryRegime(EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 4), new Random() { @Override public double nextDouble() { diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingRetryStateTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingRetryStateTest.java new file mode 100644 index 000000000..be5293b14 --- /dev/null +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingRetryStateTest.java @@ -0,0 +1,127 @@ +package com.launchdarkly.sdk.android; + +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertTrue; + +import org.junit.Test; + +import java.util.Random; + +/** + * Unit tests for {@link PollingRetryState}. + */ +public class PollingRetryStateTest { + private static final long SECOND = 1_000L; + private static final long MINUTE = 60 * SECOND; + + /** A random source that always yields the same fraction, so jitter is predictable. */ + private static Random fixedRandom(final double fraction) { + return new Random() { + @Override + public double nextDouble() { + return fraction; + } + }; + } + + private static final Random NO_JITTER = fixedRandom(0); + private static final Random MAX_JITTER = fixedRandom(0.999_999); + + private static PollingRetryState polling(long pollIntervalMillis, Random random) { + return new PollingRetryState(pollIntervalMillis, RetryRegime.EXTENDED, random); + } + + @Test + public void defaultResetThresholdIsTwoSuccesses() { + assertEquals(2, PollingRetryState.RESET_THRESHOLD_SUCCESSES); + } + + @Test + public void delayBeforeAnyFailureIsThePollInterval() { + assertEquals(10 * SECOND, polling(10 * SECOND, NO_JITTER).nextDelayMillis()); + } + + @Test + public void normalFailuresWaitThePollInterval() { + PollingRetryState state = polling(10 * SECOND, MAX_JITTER); + for (int i = 0; i < 5; i++) { + state.recordFailure(false); + assertEquals(10 * SECOND, state.nextDelayMillis()); + } + } + + @Test + public void unexpectedFailureWaitsTheExtendedInitialDelayThenDoubles() { + PollingRetryState state = polling(10 * SECOND, NO_JITTER); + state.recordFailure(true); + assertEquals(5 * MINUTE, state.nextDelayMillis()); + state.recordFailure(false); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void extendedDelayIsFlooredAtThePollInterval() { + PollingRetryState state = polling(10 * MINUTE, NO_JITTER); + state.recordFailure(true); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + state.recordFailure(false); + assertEquals(20 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void extendedCeilingIsAtLeastThePollInterval() { + PollingRetryState state = polling(90 * MINUTE, NO_JITTER); + state.recordFailure(true); + state.recordFailure(false); + state.recordFailure(false); + assertEquals(90 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void successReturnsToPollIntervalWithoutClearingTheBackoff() { + PollingRetryState state = polling(10 * SECOND, NO_JITTER); + state.recordFailure(true); + state.recordSuccess(); + assertEquals(10 * SECOND, state.nextDelayMillis()); + + state.recordFailure(false); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void twoConsecutiveSuccessesClearTheBackoff() { + PollingRetryState state = polling(10 * SECOND, NO_JITTER); + state.recordFailure(true); + state.recordSuccess(); + state.recordSuccess(); + + state.recordFailure(false); + assertEquals(10 * SECOND, state.nextDelayMillis()); + state.recordFailure(true); + assertEquals(5 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void failureRestartsTheSuccessCount() { + PollingRetryState state = polling(10 * SECOND, NO_JITTER); + state.recordFailure(true); + state.recordSuccess(); + state.recordFailure(false); + state.recordSuccess(); + state.recordFailure(false); + assertEquals(20 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void defaultExtendedRegimeIsFiveMinutesWithJitter() { + PollingRetryState state = new PollingRetryState(30 * SECOND); + assertEquals(30 * SECOND, state.nextDelayMillis()); + state.recordFailure(false); + assertEquals(30 * SECOND, state.nextDelayMillis()); + + state.recordFailure(true); + long wait = state.nextDelayMillis(); + assertTrue("wait " + wait, wait > RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS / 2 + && wait <= RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); + } +} diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryRegimeTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryRegimeTest.java new file mode 100644 index 000000000..9fe2535f2 --- /dev/null +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryRegimeTest.java @@ -0,0 +1,82 @@ +package com.launchdarkly.sdk.android; + +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertTrue; + +import org.junit.Test; + +import java.util.Random; + +/** + * Unit tests for {@link RetryRegime}. + */ +public class RetryRegimeTest { + private static final long SECOND = 1_000L; + private static final long MINUTE = 60 * SECOND; + + /** A random source that always yields the same fraction, so jitter is predictable. */ + private static Random fixedRandom(final double fraction) { + return new Random() { + @Override + public double nextDouble() { + return fraction; + } + }; + } + + private static final Random NO_JITTER = fixedRandom(0); + private static final Random MAX_JITTER = fixedRandom(0.999_999); + + @Test + public void extendedRegimeIsFiveMinutesToOneHour() { + assertEquals(5 * MINUTE, RetryRegime.EXTENDED.initialDelayMillis); + assertEquals(60 * MINUTE, RetryRegime.EXTENDED.maxDelayMillis); + } + + @Test + public void delaysDoubleFromTheInitialDelayUpToTheCeiling() { + RetryRegime regime = new RetryRegime(SECOND, 30 * SECOND); + long[] expected = { + SECOND, 2 * SECOND, 4 * SECOND, 8 * SECOND, 16 * SECOND, 30 * SECOND, 30 * SECOND}; + for (int i = 0; i < expected.length; i++) { + int attempts = i + 1; + long wait = regime.jitteredDelayMillis(attempts, NO_JITTER); + assertEquals("attempt " + attempts, expected[i], wait); + } + } + + @Test + public void jitterSubtractsUpToHalfOfTheDelay() { + RetryRegime regime = new RetryRegime(SECOND, 30 * SECOND); + // Jitter is chosen from [0, T/2), so the smallest wait is just above T/2. + long minWait = regime.jitteredDelayMillis(1, MAX_JITTER); + assertTrue("wait " + minWait, minWait > SECOND / 2 && minWait <= SECOND); + } + + @Test + public void ceilingIsRaisedToTheInitialDelay() { + RetryRegime regime = new RetryRegime(10 * MINUTE, 8 * MINUTE); + assertEquals(10 * MINUTE, regime.maxDelayMillis); + assertEquals(10 * MINUTE, regime.jitteredDelayMillis(2, NO_JITTER)); + } + + @Test + public void atLeastRaisesBothBoundsToTheFloor() { + RetryRegime raised = RetryRegime.EXTENDED.atLeast(90 * MINUTE); + assertEquals(90 * MINUTE, raised.initialDelayMillis); + assertEquals(90 * MINUTE, raised.maxDelayMillis); + + RetryRegime unchanged = RetryRegime.EXTENDED.atLeast(SECOND); + assertEquals(5 * MINUTE, unchanged.initialDelayMillis); + assertEquals(60 * MINUTE, unchanged.maxDelayMillis); + } + + @Test + public void manyAttemptsStayAtTheCeilingWithoutOverflow() { + RetryRegime regime = new RetryRegime(SECOND, 30 * SECOND); + for (int attempts = 1; attempts <= 200; attempts++) { + long wait = regime.jitteredDelayMillis(attempts, NO_JITTER); + assertTrue("wait " + wait, wait > 0 && wait <= 30 * SECOND); + } + } +} diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java deleted file mode 100644 index f920374a7..000000000 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/RetryStateTest.java +++ /dev/null @@ -1,376 +0,0 @@ -package com.launchdarkly.sdk.android; - -import static org.junit.Assert.assertEquals; -import static org.junit.Assert.assertFalse; -import static org.junit.Assert.assertTrue; - -import org.junit.Test; - -import java.util.Random; - -/** - * Unit tests for {@link RetryState}. - */ -public class RetryStateTest { - private static final long SECOND = 1_000L; - private static final long MINUTE = 60 * SECOND; - - /** A random source that always yields the same fraction, so jitter is predictable. */ - private static Random fixedRandom(final double fraction) { - return new Random() { - @Override - public double nextDouble() { - return fraction; - } - }; - } - - private static final Random NO_JITTER = fixedRandom(0); - private static final Random MAX_JITTER = fixedRandom(0.999_999); - - private static RetryState streaming(Random random) { - return new RetryState( - SECOND, - RetryState.NORMAL_MAX_DELAY_MILLIS, - RetryState.EXTENDED_INITIAL_DELAY_MILLIS, - RetryState.EXTENDED_MAX_DELAY_MILLIS, - 0, - RetryState.STREAMING_RESET_THRESHOLD_MILLIS, - 0, - random); - } - - private static RetryState polling(long pollIntervalMillis, Random random) { - return new RetryState( - pollIntervalMillis, - pollIntervalMillis, - Math.max(RetryState.EXTENDED_INITIAL_DELAY_MILLIS, pollIntervalMillis), - Math.max(RetryState.EXTENDED_MAX_DELAY_MILLIS, pollIntervalMillis), - pollIntervalMillis, - 0, - RetryState.POLLING_RESET_THRESHOLD_SUCCESSES, - random); - } - - // ---- defaults ---- - - @Test - public void defaultsAreThirtySecondNormalCeilingAndFiveMinuteToOneHourExtendedRegime() { - assertEquals(30 * SECOND, RetryState.NORMAL_MAX_DELAY_MILLIS); - assertEquals(5 * MINUTE, RetryState.EXTENDED_INITIAL_DELAY_MILLIS); - assertEquals(60 * MINUTE, RetryState.EXTENDED_MAX_DELAY_MILLIS); - assertEquals(60 * SECOND, RetryState.STREAMING_RESET_THRESHOLD_MILLIS); - assertEquals(2, RetryState.POLLING_RESET_THRESHOLD_SUCCESSES); - } - - // ---- initial state ---- - - @Test - public void startsWithNoAttemptsInNormalRegime() { - RetryState state = streaming(NO_JITTER); - assertEquals(0, state.getAttempts()); - assertFalse(state.isInExtendedRegime()); - } - - @Test - public void delayBeforeAnyFailureIsTheOperatingCadence() { - assertEquals(0, streaming(NO_JITTER).nextDelayMillis()); - assertEquals(10 * SECOND, polling(10 * SECOND, NO_JITTER).nextDelayMillis()); - } - - // ---- normal regime ---- - - @Test - public void normalFailuresDoubleTheDelayUpToTheNormalCeiling() { - RetryState state = streaming(NO_JITTER); - long[] expected = {SECOND, 2 * SECOND, 4 * SECOND, 8 * SECOND, 16 * SECOND, 30 * SECOND, 30 * SECOND}; - for (int i = 0; i < expected.length; i++) { - state.recordFailure(false, 0); - assertEquals("attempt " + (i + 1), i + 1, state.getAttempts()); - assertEquals("delay after attempt " + (i + 1), expected[i], state.nextDelayMillis()); - } - assertFalse(state.isInExtendedRegime()); - } - - // ---- jitter ---- - - @Test - public void jitterSubtractsUpToHalfOfTheBaseDelay() { - RetryState noJitter = streaming(NO_JITTER); - RetryState maxJitter = streaming(MAX_JITTER); - noJitter.recordFailure(false, 0); - maxJitter.recordFailure(false, 0); - assertEquals(SECOND, noJitter.nextDelayMillis()); - // Jitter is chosen from [0, T/2), so the smallest wait is just above T/2. - long minWait = maxJitter.nextDelayMillis(); - assertTrue("wait " + minWait, minWait > SECOND / 2 && minWait <= SECOND); - } - - // ---- unexpected failures and the extended regime ---- - - @Test - public void unexpectedFailureMovesToExtendedRegimeStartingAtExtendedInitialDelay() { - RetryState state = streaming(NO_JITTER); - state.recordFailure(false, 0); - state.recordFailure(false, 0); - state.recordFailure(true, 0); - - assertTrue(state.isInExtendedRegime()); - assertEquals(1, state.getAttempts()); - assertEquals(5 * MINUTE, state.nextDelayMillis()); - } - - @Test - public void extendedRegimeDoublesUpToTheExtendedCeiling() { - RetryState state = streaming(NO_JITTER); - state.recordFailure(true, 0); - long[] expected = {5 * MINUTE, 10 * MINUTE, 20 * MINUTE, 40 * MINUTE, 60 * MINUTE, 60 * MINUTE}; - assertEquals(expected[0], state.nextDelayMillis()); - for (int i = 1; i < expected.length; i++) { - state.recordFailure(false, 0); - assertEquals("delay after attempt " + (i + 1), expected[i], state.nextDelayMillis()); - } - } - - @Test - public void normalFailureAfterUnexpectedStaysInExtendedRegime() { - // Once raised, the ceiling and base stay raised until a reset. - RetryState state = streaming(NO_JITTER); - state.recordFailure(true, 0); - state.recordFailure(false, 0); - assertTrue(state.isInExtendedRegime()); - assertEquals(10 * MINUTE, state.nextDelayMillis()); - } - - @Test - public void repeatedUnexpectedFailuresKeepCountingInExtendedRegime() { - RetryState state = streaming(NO_JITTER); - state.recordFailure(true, 0); - state.recordFailure(true, 0); - assertEquals(2, state.getAttempts()); - assertEquals(10 * MINUTE, state.nextDelayMillis()); - } - - @Test - public void extendedDelaysAreNeverBelowTheNormalInitialDelay() { - // A component whose normal initial delay exceeds the extended bounds uses the normal - // initial delay in their place. - RetryState state = new RetryState(10 * MINUTE, 10 * MINUTE, 5 * MINUTE, 8 * MINUTE, 0, 0, 0, NO_JITTER); - state.recordFailure(true, 0); - assertEquals(10 * MINUTE, state.nextDelayMillis()); - state.recordFailure(false, 0); - assertEquals(10 * MINUTE, state.nextDelayMillis()); - } - - // ---- reset ---- - - @Test - public void resetReturnsToNormalRegimeWithNoAttempts() { - RetryState state = streaming(NO_JITTER); - state.recordFailure(true, 0); - state.recordFailure(false, 0); - state.reset(); - - assertEquals(0, state.getAttempts()); - assertFalse(state.isInExtendedRegime()); - state.recordFailure(false, 0); - assertEquals(SECOND, state.nextDelayMillis()); - } - - @Test - public void streamingHealthyForThresholdResetsWhenTheNextFailureIsRecorded() { - RetryState state = streaming(NO_JITTER); - state.recordFailure(true, 0); - assertTrue(state.isInExtendedRegime()); - - long connectedAt = 10 * SECOND; - state.recordSuccess(connectedAt); - state.recordFailure(false, connectedAt + RetryState.STREAMING_RESET_THRESHOLD_MILLIS); - - assertFalse(state.isInExtendedRegime()); - assertEquals(1, state.getAttempts()); - assertEquals(SECOND, state.nextDelayMillis()); - } - - @Test - public void streamingHealthyForLessThanThresholdDoesNotReset() { - RetryState state = streaming(NO_JITTER); - state.recordFailure(true, 0); - - long connectedAt = 10 * SECOND; - state.recordSuccess(connectedAt); - state.recordFailure(false, connectedAt + RetryState.STREAMING_RESET_THRESHOLD_MILLIS - 1); - - assertTrue(state.isInExtendedRegime()); - assertEquals(2, state.getAttempts()); - assertEquals(10 * MINUTE, state.nextDelayMillis()); - } - - @Test - public void streamingHealthyOperationIsMeasuredFromTheFirstMessageOnTheConnection() { - RetryState state = streaming(NO_JITTER); - state.recordFailure(true, 0); - - state.recordSuccess(10 * SECOND); - state.recordSuccess(65 * SECOND); // must not move the marker - state.recordFailure(false, 70 * SECOND); // 60 seconds since the first message - - assertFalse(state.isInExtendedRegime()); - } - - @Test - public void streamingFailureRestartsHealthyOperationMeasurement() { - // The marker from a previous connection does not count toward the next one. - RetryState state = streaming(NO_JITTER); - state.recordFailure(true, 0); - - state.recordSuccess(10 * SECOND); - state.recordFailure(false, 20 * SECOND); // healthy for 10 seconds only - state.recordFailure(false, 90 * SECOND); // no message on this connection - - assertTrue(state.isInExtendedRegime()); - assertEquals(3, state.getAttempts()); - } - - @Test - public void streamingSuccessAloneDoesNotReset() { - RetryState state = streaming(NO_JITTER); - state.recordFailure(true, 0); - state.recordSuccess(10 * SECOND); - state.recordSuccess(10 * MINUTE); - // The reset is applied when the next failure is recorded, and only then. Until that - // point the regime is unchanged. - assertTrue(state.isInExtendedRegime()); - } - - // ---- polling: operating cadence ---- - - @Test - public void pollingNormalFailureWaitsThePollInterval() { - RetryState state = polling(10 * SECOND, MAX_JITTER); - for (int i = 0; i < 5; i++) { - state.recordFailure(false, 0); - // Jitter cannot bring the wait below the cadence. - assertEquals(10 * SECOND, state.nextDelayMillis()); - } - assertFalse(state.isInExtendedRegime()); - } - - @Test - public void pollingUnexpectedFailureWaitsTheExtendedInitialDelay() { - RetryState state = polling(10 * SECOND, NO_JITTER); - state.recordFailure(true, 0); - assertEquals(5 * MINUTE, state.nextDelayMillis()); - state.recordFailure(false, 0); - assertEquals(10 * MINUTE, state.nextDelayMillis()); - } - - @Test - public void pollingExtendedDelayIsFlooredAtThePollInterval() { - RetryState state = polling(10 * MINUTE, NO_JITTER); - state.recordFailure(true, 0); - assertEquals(10 * MINUTE, state.nextDelayMillis()); - state.recordFailure(false, 0); - assertEquals(20 * MINUTE, state.nextDelayMillis()); - } - - @Test - public void pollingExtendedCeilingIsAtLeastThePollInterval() { - RetryState state = polling(90 * MINUTE, NO_JITTER); - state.recordFailure(true, 0); - state.recordFailure(false, 0); - state.recordFailure(false, 0); - assertEquals(90 * MINUTE, state.nextDelayMillis()); - } - - @Test - public void pollingSuccessReturnsToCadenceWithoutResettingTheRegime() { - // One success restores the cadence but does not clear the retry state. - RetryState state = polling(10 * SECOND, NO_JITTER); - state.recordFailure(true, 0); - state.recordSuccess(0); - - assertEquals(10 * SECOND, state.nextDelayMillis()); - assertTrue(state.isInExtendedRegime()); - - state.recordFailure(false, 0); - assertEquals(10 * MINUTE, state.nextDelayMillis()); - } - - @Test - public void pollingTwoConsecutiveSuccessesReset() { - RetryState state = polling(10 * SECOND, NO_JITTER); - state.recordFailure(true, 0); - state.recordSuccess(0); - state.recordSuccess(0); - - assertFalse(state.isInExtendedRegime()); - assertEquals(0, state.getAttempts()); - state.recordFailure(false, 0); - assertEquals(10 * SECOND, state.nextDelayMillis()); - } - - @Test - public void pollingFailureRestartsTheSuccessCount() { - // Successes must be consecutive. - RetryState state = polling(10 * SECOND, NO_JITTER); - state.recordFailure(true, 0); - state.recordSuccess(0); - state.recordFailure(false, 0); - state.recordSuccess(0); - - assertTrue(state.isInExtendedRegime()); - } - - // ---- factories ---- - - @Test - public void forStreamingUsesConfiguredInitialReconnectDelayAsNormalInitialDelay() { - RetryState state = RetryState.forStreaming(250); - state.recordFailure(false, 0); - long wait = state.nextDelayMillis(); - assertTrue("wait " + wait, wait > 125 && wait <= 250); - assertEquals(0, RetryState.forStreaming(250).nextDelayMillis()); - } - - @Test - public void forStreamingWithZeroInitialDelayReconnectsImmediatelyInNormalRegime() { - RetryState state = RetryState.forStreaming(0); - state.recordFailure(false, 0); - assertEquals(0, state.nextDelayMillis()); - state.recordFailure(true, 0); - long wait = state.nextDelayMillis(); - assertTrue("wait " + wait, wait > RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 - && wait <= RetryState.EXTENDED_INITIAL_DELAY_MILLIS); - } - - @Test - public void forPollingUsesPollIntervalAsCadence() { - RetryState state = RetryState.forPolling(30 * SECOND); - assertEquals(30 * SECOND, state.nextDelayMillis()); - state.recordFailure(false, 0); - assertEquals(30 * SECOND, state.nextDelayMillis()); - state.recordFailure(true, 0); - long wait = state.nextDelayMillis(); - assertTrue("wait " + wait, wait > RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 - && wait <= RetryState.EXTENDED_INITIAL_DELAY_MILLIS); - } - - // ---- robustness ---- - - @Test - public void manyFailuresStayAtTheCeilingWithoutOverflow() { - RetryState state = streaming(NO_JITTER); - for (int i = 0; i < 200; i++) { - state.recordFailure(false, 0); - long wait = state.nextDelayMillis(); - assertTrue("wait " + wait, wait > 0 && wait <= RetryState.NORMAL_MAX_DELAY_MILLIS); - } - state.recordFailure(true, 0); - for (int i = 0; i < 200; i++) { - state.recordFailure(false, 0); - long wait = state.nextDelayMillis(); - assertTrue("wait " + wait, wait > 0 && wait <= RetryState.EXTENDED_MAX_DELAY_MILLIS); - } - } -} diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java index eedcf84d1..1caece5c4 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java @@ -75,8 +75,8 @@ private SourceManager manager(String... names) { private static void assertFirstBackoff(long delayMillis) { assertTrue("delay " + delayMillis, - delayMillis > RetryState.EXTENDED_INITIAL_DELAY_MILLIS / 2 - && delayMillis <= RetryState.EXTENDED_INITIAL_DELAY_MILLIS); + delayMillis > RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS / 2 + && delayMillis <= RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); } /** Advances the clock past a backoff and waits for the slot to become available again. */ @@ -149,14 +149,14 @@ public void repeatedUnexpectedErrorsDoubleTheBackoffUntilHealthyOperationResetsI endBackoff(manager, firstDelay); manager.getNextAvailableSynchronizerAndSetActive(); long secondDelay = manager.backOffCurrentSynchronizer("a", 0); - assertTrue("delay " + secondDelay, secondDelay > RetryState.EXTENDED_INITIAL_DELAY_MILLIS); + assertTrue("delay " + secondDelay, secondDelay > RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); endBackoff(manager, secondDelay); // Healthy operation for the reset threshold before the next error starts the backoff over. manager.getNextAvailableSynchronizerAndSetActive(); long healthyAt = 10_000; manager.recordCurrentSynchronizerHealthy(healthyAt); - long thirdDelay = manager.backOffCurrentSynchronizer("a", healthyAt + RetryState.STREAMING_RESET_THRESHOLD_MILLIS); + long thirdDelay = manager.backOffCurrentSynchronizer("a", healthyAt + StreamingRetryState.RESET_THRESHOLD_MILLIS); assertFirstBackoff(thirdDelay); } diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java index be360f8da..d7760db78 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java @@ -168,8 +168,8 @@ private StreamingDataSource makeStreamingDataSource( } // The tests of backoff behavior below drive the data source's timers with a FakeTaskExecutor - // and a RetryState without jitter, so the delay chosen for each reconnect can be asserted - // exactly instead of waited for. + // and a StreamingRetryState without jitter, so the delay chosen for each reconnect can be + // asserted exactly instead of waited for. private static final long NORMAL_DELAY_MILLIS = 1; private static final long EXTENDED_DELAY_MILLIS = 300_000; // A healthy-operation threshold no test reaches. @@ -177,9 +177,11 @@ private StreamingDataSource makeStreamingDataSource( private final FakeTaskExecutor fakeTaskExecutor = new FakeTaskExecutor(); - private static RetryState retryStateWithoutJitter(long healthyResetThresholdMillis) { - return new RetryState(NORMAL_DELAY_MILLIS, NORMAL_DELAY_MILLIS, - EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 4, 0, healthyResetThresholdMillis, 0, + private static StreamingRetryState retryStateWithoutJitter(long healthyResetThresholdMillis) { + return new StreamingRetryState( + new RetryRegime(NORMAL_DELAY_MILLIS, NORMAL_DELAY_MILLIS), + new RetryRegime(EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 4), + healthyResetThresholdMillis, new Random() { @Override public double nextDouble() { @@ -188,7 +190,7 @@ public double nextDouble() { }); } - private StreamingDataSource makeStreamingDataSource(URI streamBaseUri, RetryState retryState) { + private StreamingDataSource makeStreamingDataSource(URI streamBaseUri, StreamingRetryState retryState) { LDConfig config = new LDConfig.Builder(AutoEnvAttributes.Disabled) .serviceEndpoints(Components.serviceEndpoints().streaming(streamBaseUri)) .build(); diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingRetryStateTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingRetryStateTest.java new file mode 100644 index 000000000..9f501ae95 --- /dev/null +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingRetryStateTest.java @@ -0,0 +1,167 @@ +package com.launchdarkly.sdk.android; + +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertTrue; + +import org.junit.Test; + +import java.util.Random; + +/** + * Unit tests for {@link StreamingRetryState}. + */ +public class StreamingRetryStateTest { + private static final long SECOND = 1_000L; + private static final long MINUTE = 60 * SECOND; + private static final long THRESHOLD = StreamingRetryState.RESET_THRESHOLD_MILLIS; + + private static final Random NO_JITTER = new Random() { + @Override + public double nextDouble() { + return 0; + } + }; + + private static StreamingRetryState streaming() { + return new StreamingRetryState( + new RetryRegime(SECOND, StreamingRetryState.NORMAL_MAX_DELAY_MILLIS), + RetryRegime.EXTENDED, + THRESHOLD, + NO_JITTER); + } + + @Test + public void defaultsAreThirtySecondNormalCeilingAndOneMinuteResetThreshold() { + assertEquals(30 * SECOND, StreamingRetryState.NORMAL_MAX_DELAY_MILLIS); + assertEquals(60 * SECOND, StreamingRetryState.RESET_THRESHOLD_MILLIS); + } + + // ---- regimes ---- + + @Test + public void normalFailuresBackOffInTheNormalRegime() { + StreamingRetryState state = streaming(); + state.recordFailure(false, 0); + assertEquals(SECOND, state.nextDelayMillis()); + state.recordFailure(false, 0); + assertEquals(2 * SECOND, state.nextDelayMillis()); + state.recordFailure(false, 0); + assertEquals(4 * SECOND, state.nextDelayMillis()); + } + + @Test + public void unexpectedFailureMovesToExtendedRegimeStartingAtExtendedInitialDelay() { + StreamingRetryState state = streaming(); + state.recordFailure(false, 0); + state.recordFailure(false, 0); + state.recordFailure(true, 0); + assertEquals(5 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void normalFailureAfterUnexpectedStaysInExtendedRegime() { + StreamingRetryState state = streaming(); + state.recordFailure(true, 0); + state.recordFailure(false, 0); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void repeatedUnexpectedFailuresKeepCountingInExtendedRegime() { + StreamingRetryState state = streaming(); + state.recordFailure(true, 0); + state.recordFailure(true, 0); + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + // ---- healthy operation ---- + + @Test + public void healthyForThresholdResetsWhenTheNextFailureIsRecorded() { + StreamingRetryState state = streaming(); + state.recordFailure(true, 0); + + long connectedAt = 10 * SECOND; + state.recordSuccess(connectedAt); + state.recordFailure(false, connectedAt + THRESHOLD); + + assertEquals(SECOND, state.nextDelayMillis()); + } + + @Test + public void healthyForLessThanThresholdDoesNotReset() { + StreamingRetryState state = streaming(); + state.recordFailure(true, 0); + + long connectedAt = 10 * SECOND; + state.recordSuccess(connectedAt); + state.recordFailure(false, connectedAt + THRESHOLD - 1); + + assertEquals(10 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void healthyOperationIsMeasuredFromTheFirstMessageOnTheConnection() { + StreamingRetryState state = streaming(); + state.recordFailure(true, 0); + + state.recordSuccess(10 * SECOND); + state.recordSuccess(65 * SECOND); // must not move the marker + state.recordFailure(false, 70 * SECOND); // 60 seconds since the first message + + assertEquals(SECOND, state.nextDelayMillis()); + } + + @Test + public void failureRestartsHealthyOperationMeasurement() { + // The marker from a previous connection does not count toward the next one. + StreamingRetryState state = streaming(); + state.recordFailure(true, 0); + + state.recordSuccess(10 * SECOND); + state.recordFailure(false, 20 * SECOND); // healthy for 10 seconds only + state.recordFailure(false, 90 * SECOND); // no message on this connection + + assertEquals(20 * MINUTE, state.nextDelayMillis()); + } + + @Test + public void healthyOperationStartingAtTimeZeroCounts() { + StreamingRetryState state = streaming(); + state.recordFailure(true, 0); + state.recordSuccess(0); + state.recordFailure(false, THRESHOLD); + + assertEquals(SECOND, state.nextDelayMillis()); + } + + // ---- configured initial reconnect delay ---- + + @Test + public void firstNormalRetryWaitsTheConfiguredInitialReconnectDelay() { + StreamingRetryState state = new StreamingRetryState(250); + state.recordFailure(false, 0); + long wait = state.nextDelayMillis(); + assertTrue("wait " + wait, wait > 125 && wait <= 250); + } + + @Test + public void zeroInitialReconnectDelayReconnectsImmediatelyAfterNormalFailure() { + StreamingRetryState state = new StreamingRetryState(0); + state.recordFailure(false, 0); + assertEquals(0, state.nextDelayMillis()); + + state.recordFailure(true, 0); + long wait = state.nextDelayMillis(); + assertTrue("wait " + wait, wait > RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS / 2 + && wait <= RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); + } + + @Test + public void extendedRegimeIsNeverBelowTheInitialReconnectDelay() { + StreamingRetryState state = new StreamingRetryState(10 * MINUTE); + state.recordFailure(true, 0); + long wait = state.nextDelayMillis(); + assertTrue("wait " + wait, wait > 5 * MINUTE && wait <= 10 * MINUTE); + } +} From 34a1e09b1bfe4130ac92dd8b9bd458b08c2741bf Mon Sep 17 00:00:00 2001 From: Bee Klimt Date: Tue, 29 Sep 2026 16:43:59 -0700 Subject: [PATCH 4/5] refactor: Move the backoff wait into SourceManager.nextAvailableSynchronizer --- .../sdk/android/FDv2DataSource.java | 24 +--- .../sdk/android/SourceManager.java | 103 +++++++++------ .../android/SynchronizerFactoryWithState.java | 19 +-- .../sdk/android/SourceManagerTest.java | 117 +++++++++--------- 4 files changed, 128 insertions(+), 135 deletions(-) diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java index a139c80d8..ee5b7941c 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java @@ -489,21 +489,6 @@ private List getConditions(int synchronizerC * try. */ @Nullable - private Synchronizer nextSynchronizerOrWaitForBackoff() throws InterruptedException { - while (true) { - Synchronizer synchronizer = sourceManager.getNextAvailableSynchronizerAndSetActive(); - if (synchronizer != null || !sourceManager.hasBackingOffSynchronizers()) { - return synchronizer; - } - logger.info("All synchronizers are waiting out a backoff after unexpected errors; the first to become available will be tried next."); - try { - sourceManager.awaitAvailabilityChange().get(); - } catch (ExecutionException e) { - return null; - } - } - } - private static String detailForThrowable(@Nullable Throwable error) { if (error == null) { return "unknown error"; @@ -539,7 +524,7 @@ private void runSynchronizers( @NonNull DataSourceUpdateSinkV2 sink ) { try { - Synchronizer synchronizer = nextSynchronizerOrWaitForBackoff(); + Synchronizer synchronizer = sourceManager.nextAvailableSynchronizer().get(); while (synchronizer != null) { String synchronizerName = synchronizer.name(); logger.info("Synchronizer '{}' is starting.", synchronizerName); @@ -627,12 +612,15 @@ private void runSynchronizers( status.getState() ); long backoffMillis = sourceManager.backOffCurrentSynchronizer( - synchronizer.name(), System.currentTimeMillis()); + System.currentTimeMillis()); logger.warn( "Synchronizer '{}' reported an unexpected error and will not be tried again for {} seconds.", synchronizer.name(), backoffMillis / 1000 ); + if (sourceManager.getAvailableSynchronizerCount() == 0) { + logger.info("All synchronizers are waiting out a backoff after unexpected errors; the first to become available will be tried next."); + } running = false; sink.setStatus(DataSourceState.INTERRUPTED, status.getError()); break; @@ -680,7 +668,7 @@ private void runSynchronizers( Thread.currentThread().interrupt(); return; } - synchronizer = nextSynchronizerOrWaitForBackoff(); + synchronizer = sourceManager.nextAvailableSynchronizer().get(); } if (!stopCalled.get()) { logger.warn("No more synchronizers available."); diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java index bf4e16b16..67e75f5a6 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SourceManager.java @@ -40,8 +40,10 @@ final class SourceManager implements Closeable { private SynchronizerFactoryWithState currentSynchronizerFactory; - // Completed and replaced whenever a slot's backoff ends or this manager closes. - private LDAwaitFuture availabilityChanged = new LDAwaitFuture<>(); + // Handed out by nextAvailableSynchronizer() while every usable slot is backing off, and + // completed by the first backoff to end or by close(). + @Nullable + private LDAwaitFuture pendingNext; SourceManager( @NonNull List synchronizerFactories, @@ -75,8 +77,8 @@ boolean hasFDv1Fallback() { /** * Block all non-FDv1 synchronizers, unblock the FDv1 fallback, and reset the - * synchronizer index so the next {@link #getNextAvailableSynchronizerAndSetActive()} - * picks the now-unblocked FDv1 slot. + * synchronizer index so the next {@link #nextAvailableSynchronizer()} picks the now-unblocked + * FDv1 slot. */ void fdv1Fallback() { synchronized (activeSourceLock) { @@ -115,7 +117,7 @@ private SynchronizerFactoryWithState getNextAvailableSynchronizer() { * and return it. Returns null if shutdown or no available synchronizers. * Skips synchronizers whose factory returns null from build(). */ - Synchronizer getNextAvailableSynchronizerAndSetActive() { + private Synchronizer getNextAvailableSynchronizerAndSetActive() { synchronized (activeSourceLock) { if (isShutdown) { currentSynchronizerFactory = null; @@ -146,6 +148,34 @@ Synchronizer getNextAvailableSynchronizerAndSetActive() { } } + /** + * Selects the synchronizer to run next, builds it, and makes it the active source in place of + * the previous one. If every usable slot is backing off, the returned future completes when + * the first backoff ends. It completes with null once this manager is closed or no + * synchronizer is left to try. + * + * @return a future for the next synchronizer + */ + @NonNull + Future nextAvailableSynchronizer() { + synchronized (activeSourceLock) { + Synchronizer synchronizer = getNextAvailableSynchronizerAndSetActive(); + if (synchronizer != null || !hasBackingOffSynchronizers()) { + return completed(synchronizer); + } + if (pendingNext == null) { + pendingNext = new LDAwaitFuture<>(); + } + return pendingNext; + } + } + + private static LDAwaitFuture completed(@Nullable T value) { + LDAwaitFuture future = new LDAwaitFuture<>(); + future.set(value); + return future; + } + boolean hasAvailableSources() { return hasInitializers() || getAvailableSynchronizerCount() > 0; } @@ -168,21 +198,20 @@ private FDv2DataSource.DataSourceFactory getNextInitializer() { /** * Puts the current synchronizer's slot into backoff. The slot is skipped by - * {@link #getNextAvailableSynchronizerAndSetActive()} until the backoff ends. + * {@link #nextAvailableSynchronizer()} until the backoff ends. * - * @param synchronizerName the name of the synchronizer that failed, for logging - * @param nowMillis the current time in milliseconds, on the same clock as every other - * call on this manager + * @param nowMillis the current time in milliseconds, on the same clock as every other call on + * this manager * @return how long the slot stays in backoff, in milliseconds, or -1 if there is no current * synchronizer */ - long backOffCurrentSynchronizer(@NonNull String synchronizerName, long nowMillis) { + long backOffCurrentSynchronizer(long nowMillis) { synchronized (activeSourceLock) { final SynchronizerFactoryWithState slot = currentSynchronizerFactory; if (slot == null || isShutdown) { return -1; } - long delayMillis = slot.startBackoff(synchronizerName, nowMillis); + long delayMillis = slot.startBackoff(nowMillis); ScheduledFuture unblock = executor.schedule(new Runnable() { @Override public void run() { @@ -210,29 +239,30 @@ void recordCurrentSynchronizerHealthy(long nowMillis) { } /** - * Ends a slot's backoff and wakes any caller waiting in {@link #awaitAvailabilityChange()}. - * - * @return the name of the synchronizer whose error started the backoff, if the slot became - * available, or null if nothing changed + * Ends a slot's backoff. If a caller is waiting in {@link #nextAvailableSynchronizer()}, the + * next synchronizer is selected and handed to it. */ - @Nullable - String endBackoff(@NonNull SynchronizerFactoryWithState slot) { - LDAwaitFuture toComplete; - String name; + void endBackoff(@NonNull SynchronizerFactoryWithState slot) { + LDAwaitFuture waiting = null; + Synchronizer synchronizer = null; synchronized (activeSourceLock) { if (isShutdown || !slot.endBackoff()) { - return null; + return; } - name = slot.getLastSynchronizerName(); - toComplete = availabilityChanged; - availabilityChanged = new LDAwaitFuture<>(); + if (pendingNext != null) { + synchronizer = getNextAvailableSynchronizerAndSetActive(); + if (synchronizer != null || !hasBackingOffSynchronizers()) { + waiting = pendingNext; + pendingNext = null; + } + } + } + if (waiting != null) { + waiting.set(synchronizer); } - toComplete.set(null); - return name == null ? "" : name; } - /** True if any synchronizer is waiting out a backoff. Always false once closed. */ - boolean hasBackingOffSynchronizers() { + private boolean hasBackingOffSynchronizers() { synchronized (activeSourceLock) { if (isShutdown) { return false; @@ -246,16 +276,6 @@ boolean hasBackingOffSynchronizers() { } } - /** - * @return a future that completes the next time a slot's backoff ends, or when this manager - * closes - */ - Future awaitAvailabilityChange() { - synchronized (activeSourceLock) { - return availabilityChanged; - } - } - /** * @return true if a synchronizer earlier in the list than the current one is available */ @@ -347,7 +367,7 @@ int getUsableSynchronizerCount() { @Override public void close() { - LDAwaitFuture toComplete; + LDAwaitFuture waiting; synchronized (activeSourceLock) { isShutdown = true; if (activeSource != null) { @@ -357,9 +377,12 @@ public void close() { for (SynchronizerFactoryWithState s : synchronizerFactories) { s.cancelPendingUnblock(); } - toComplete = availabilityChanged; + waiting = pendingNext; + pendingNext = null; + } + if (waiting != null) { + waiting.set(null); } - toComplete.set(null); } private static void safeClose(Closeable closeable) { diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java index d92344be2..809424d53 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java @@ -40,8 +40,6 @@ enum State { new Random()); @Nullable private ScheduledFuture pendingUnblock; - @Nullable - private String lastSynchronizerName; SynchronizerFactoryWithState(@NonNull FDv2DataSource.DataSourceFactory factory) { this(factory, false); @@ -85,15 +83,13 @@ void recordHealthy(long nowMillis) { /** * Records an unexpected error from this slot's synchronizer and puts the slot into backoff. * - * @param synchronizerName the name of the synchronizer that failed, for logging - * @param nowMillis the current time in milliseconds, on the same clock as every other - * call on this instance + * @param nowMillis the current time in milliseconds, on the same clock as every other call on + * this instance * @return how long the slot stays in backoff, in milliseconds */ - long startBackoff(@NonNull String synchronizerName, long nowMillis) { + long startBackoff(long nowMillis) { retryState.recordFailure(true, nowMillis); state = State.BackingOff; - lastSynchronizerName = synchronizerName; return retryState.nextDelayMillis(); } @@ -111,15 +107,6 @@ boolean endBackoff() { return true; } - /** - * @return the name of the synchronizer whose unexpected error started the current backoff, - * or null if there has been none - */ - @Nullable - String getLastSynchronizerName() { - return lastSynchronizerName; - } - void setPendingUnblock(@Nullable ScheduledFuture pendingUnblock) { this.pendingUnblock = pendingUnblock; } diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java index 1caece5c4..e0f9c0e9d 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java @@ -2,7 +2,6 @@ import static org.junit.Assert.assertEquals; import static org.junit.Assert.assertFalse; -import static org.junit.Assert.assertNotNull; import static org.junit.Assert.assertNull; import static org.junit.Assert.assertTrue; @@ -16,8 +15,9 @@ import org.junit.Test; import org.junit.rules.Timeout; -import java.util.Arrays; +import java.util.ArrayList; import java.util.Collections; +import java.util.List; import java.util.concurrent.Future; import java.util.concurrent.TimeUnit; @@ -31,6 +31,7 @@ public class SourceManagerTest { public Timeout globalTimeout = Timeout.seconds(5); private final FakeScheduledExecutorService executor = new FakeScheduledExecutorService(); + private final List slots = new ArrayList<>(); @After public void tearDown() { @@ -61,16 +62,11 @@ public String name() { } } - private static SynchronizerFactoryWithState slot(final String name) { - return new SynchronizerFactoryWithState(() -> new NamedSynchronizer(name)); - } - private SourceManager manager(String... names) { - SynchronizerFactoryWithState[] slots = new SynchronizerFactoryWithState[names.length]; - for (int i = 0; i < names.length; i++) { - slots[i] = slot(names[i]); + for (final String name : names) { + slots.add(new SynchronizerFactoryWithState(() -> new NamedSynchronizer(name))); } - return new SourceManager(Arrays.asList(slots), Collections.emptyList(), executor); + return new SourceManager(slots, Collections.emptyList(), executor); } private static void assertFirstBackoff(long delayMillis) { @@ -79,99 +75,98 @@ private static void assertFirstBackoff(long delayMillis) { && delayMillis <= RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); } - /** Advances the clock past a backoff and waits for the slot to become available again. */ - private void endBackoff(SourceManager manager, long delayMillis) throws Exception { - Future changed = manager.awaitAvailabilityChange(); - executor.advanceTime(delayMillis); - changed.get(1, TimeUnit.SECONDS); + /** Selects the next synchronizer, asserting that one is known right away. */ + private static String nextNow(SourceManager manager) throws Exception { + Future next = manager.nextAvailableSynchronizer(); + assertTrue(next.isDone()); + Synchronizer synchronizer = next.get(); + return synchronizer == null ? null : synchronizer.name(); } @Test - public void backingOffASlotSkipsItUntilTheBackoffEnds() throws Exception { + public void backingOffTheCurrentSlotSkipsItInFavorOfTheNext() throws Exception { SourceManager manager = manager("a", "b"); - assertEquals("a", manager.getNextAvailableSynchronizerAndSetActive().name()); + assertEquals("a", nextNow(manager)); - // Backing off the current slot schedules its return and makes selection skip it. - long delay = manager.backOffCurrentSynchronizer("a", 0); + long delay = manager.backOffCurrentSynchronizer(0); assertFirstBackoff(delay); assertEquals(delay, executor.awaitScheduledDelayMillis(1000)); - assertTrue(manager.hasBackingOffSynchronizers()); - assertEquals("b", manager.getNextAvailableSynchronizerAndSetActive().name()); - assertEquals("b", manager.getNextAvailableSynchronizerAndSetActive().name()); - - // Once the backoff ends, the slot is selected again. - endBackoff(manager, delay); - assertFalse(manager.hasBackingOffSynchronizers()); - assertEquals("a", manager.getNextAvailableSynchronizerAndSetActive().name()); + + assertEquals("b", nextNow(manager)); + assertEquals("b", nextNow(manager)); } @Test - public void allSlotsBackingOffYieldsNoSynchronizerUntilOneReturns() throws Exception { - SourceManager manager = manager("a", "b"); - manager.getNextAvailableSynchronizerAndSetActive(); - long firstDelay = manager.backOffCurrentSynchronizer("a", 0); - manager.getNextAvailableSynchronizerAndSetActive(); - long secondDelay = manager.backOffCurrentSynchronizer("b", 0); - - // Nothing can be selected, but the manager still knows a synchronizer will return. - assertNull(manager.getNextAvailableSynchronizerAndSetActive()); - assertTrue(manager.hasBackingOffSynchronizers()); - - // The first backoff to end makes its slot available. - endBackoff(manager, Math.max(firstDelay, secondDelay)); - assertNotNull(manager.getNextAvailableSynchronizerAndSetActive()); + public void slotReturnsWhenItsBackoffEnds() throws Exception { + SourceManager manager = manager("a"); + assertEquals("a", nextNow(manager)); + long delay = manager.backOffCurrentSynchronizer(0); + + // With every slot backing off, the next synchronizer is not known yet. + Future next = manager.nextAvailableSynchronizer(); + assertFalse(next.isDone()); + + // The backoff ending supplies it. + executor.advanceTime(delay); + assertEquals("a", next.get(1, TimeUnit.SECONDS).name()); + assertEquals("a", nextNow(manager)); } @Test public void aBackingOffSlotStillOutranksTheCurrentOneForRecovery() throws Exception { SourceManager manager = manager("a", "b"); - manager.getNextAvailableSynchronizerAndSetActive(); - long delay = manager.backOffCurrentSynchronizer("a", 0); - assertEquals("b", manager.getNextAvailableSynchronizerAndSetActive().name()); + nextNow(manager); + manager.backOffCurrentSynchronizer(0); + assertEquals("b", nextNow(manager)); // While "a" is backing off, "b" is not prime, but there is nothing to recover to yet. assertFalse(manager.isPrimeSynchronizer()); assertFalse(manager.hasAvailableSynchronizerBeforeCurrent()); // Once "a" is available again, recovery to it is possible. - endBackoff(manager, delay); + manager.endBackoff(slots.get(0)); assertTrue(manager.hasAvailableSynchronizerBeforeCurrent()); } @Test public void repeatedUnexpectedErrorsDoubleTheBackoffUntilHealthyOperationResetsIt() throws Exception { SourceManager manager = manager("a"); - manager.getNextAvailableSynchronizerAndSetActive(); + nextNow(manager); // A second unexpected error without any healthy operation in between doubles the wait. - long firstDelay = manager.backOffCurrentSynchronizer("a", 0); + long firstDelay = manager.backOffCurrentSynchronizer(0); assertFirstBackoff(firstDelay); - endBackoff(manager, firstDelay); - manager.getNextAvailableSynchronizerAndSetActive(); - long secondDelay = manager.backOffCurrentSynchronizer("a", 0); + manager.endBackoff(slots.get(0)); + nextNow(manager); + long secondDelay = manager.backOffCurrentSynchronizer(0); assertTrue("delay " + secondDelay, secondDelay > RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); - endBackoff(manager, secondDelay); + manager.endBackoff(slots.get(0)); // Healthy operation for the reset threshold before the next error starts the backoff over. - manager.getNextAvailableSynchronizerAndSetActive(); + nextNow(manager); long healthyAt = 10_000; manager.recordCurrentSynchronizerHealthy(healthyAt); - long thirdDelay = manager.backOffCurrentSynchronizer("a", healthyAt + StreamingRetryState.RESET_THRESHOLD_MILLIS); + long thirdDelay = manager.backOffCurrentSynchronizer( + healthyAt + StreamingRetryState.RESET_THRESHOLD_MILLIS); assertFirstBackoff(thirdDelay); } @Test - public void closeCancelsPendingBackoffsAndWakesWaiters() throws Exception { + public void closeCompletesTheWaitWithNullAndCancelsPendingBackoffs() throws Exception { SourceManager manager = manager("a"); - manager.getNextAvailableSynchronizerAndSetActive(); - manager.backOffCurrentSynchronizer("a", 0); - Future changed = manager.awaitAvailabilityChange(); + nextNow(manager); + manager.backOffCurrentSynchronizer(0); + Future next = manager.nextAvailableSynchronizer(); + assertFalse(next.isDone()); - // Closing wakes anyone waiting for a slot and drops the scheduled return. manager.close(); - changed.get(1, TimeUnit.SECONDS); - assertFalse(manager.hasBackingOffSynchronizers()); + assertNull(next.get(1, TimeUnit.SECONDS)); assertTrue(executor.pendingDelaysMillis().isEmpty()); - assertNull(manager.getNextAvailableSynchronizerAndSetActive()); + assertNull(nextNow(manager)); + } + + @Test + public void noSynchronizersYieldsNullRightAway() throws Exception { + assertNull(nextNow(manager())); } } From edabf43fb49912bee1939f0a7d445de0294db1f5 Mon Sep 17 00:00:00 2001 From: Bee Klimt Date: Tue, 29 Sep 2026 20:04:15 -0700 Subject: [PATCH 5/5] test: Consolidate the manual executors and inject the synchronizer backoff --- .../sdk/android/FDv2DataSource.java | 28 ++- .../android/FDv2StreamingSynchronizer.java | 28 +-- .../android/SynchronizerFactoryWithState.java | 23 ++- .../android/FDv2DataSourceConditionsTest.java | 27 +-- .../sdk/android/FDv2DataSourceTest.java | 35 ++-- .../FDv2StreamingSynchronizerTest.java | 96 ++------- .../sdk/android/PollingDataSourceTest.java | 54 ++--- .../sdk/android/SourceManagerTest.java | 54 ++--- .../sdk/android/StreamingDataSourceTest.java | 36 ++-- .../android/FakeScheduledExecutorService.java | 193 ------------------ .../sdk/android/FakeTaskExecutor.java | 130 ------------ .../sdk/android/ManualTaskExecutor.java | 57 ++++-- 12 files changed, 189 insertions(+), 572 deletions(-) delete mode 100644 shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java delete mode 100644 shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java index ee5b7941c..3f9efac06 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2DataSource.java @@ -125,6 +125,29 @@ public interface DataSourceFactory { @NonNull LDLogger logger, long fallbackTimeoutSeconds, long recoveryTimeoutSeconds + ) { + this(evaluationContext, initializers, synchronizers, fdv1FallbackSynchronizer, + dataSourceUpdateSink, sharedExecutor, logger, fallbackTimeoutSeconds, + recoveryTimeoutSeconds, RetryRegime.EXTENDED); + } + + /** + * This constructor allows tests to shorten the backoff after a synchronizer's unexpected + * error. See the other constructor for the remaining parameters. + * + * @param synchronizerBackoff the delay bounds for that backoff + */ + FDv2DataSource( + @NonNull LDContext evaluationContext, + @NonNull List> initializers, + @NonNull List> synchronizers, + @Nullable DataSourceFactory fdv1FallbackSynchronizer, + @NonNull DataSourceUpdateSinkV2 dataSourceUpdateSink, + @NonNull ScheduledExecutorService sharedExecutor, + @NonNull LDLogger logger, + long fallbackTimeoutSeconds, + long recoveryTimeoutSeconds, + @NonNull RetryRegime synchronizerBackoff ) { this.evaluationContext = evaluationContext; this.dataSourceUpdateSink = dataSourceUpdateSink; @@ -140,10 +163,11 @@ public interface DataSourceFactory { List allSynchronizers = new ArrayList<>(); for (DataSourceFactory factory : synchronizers) { - allSynchronizers.add(new SynchronizerFactoryWithState(factory)); + allSynchronizers.add(new SynchronizerFactoryWithState(factory, false, synchronizerBackoff)); } if (fdv1FallbackSynchronizer != null) { - SynchronizerFactoryWithState fdv1 = new SynchronizerFactoryWithState(fdv1FallbackSynchronizer, true); + SynchronizerFactoryWithState fdv1 = + new SynchronizerFactoryWithState(fdv1FallbackSynchronizer, true, synchronizerBackoff); fdv1.block(); allSynchronizers.add(fdv1); } diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java index eaf7cbc43..12048e7ec 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizer.java @@ -121,32 +121,6 @@ final class FDv2StreamingSynchronizer implements Synchronizer { @NonNull ScheduledExecutorService executor, @NonNull LDLogger logger, @Nullable DiagnosticStore diagnosticStore - ) { - this(evaluationContext, selectorSource, streamBaseUri, streamRequestPath, requestor, - initialReconnectDelayMillis, evaluationReasons, useReport, httpProperties, executor, - logger, diagnosticStore, new StreamingRetryState(initialReconnectDelayMillis)); - } - - /** - * This constructor allows tests to supply a {@link StreamingRetryState} with short delays. See the - * other constructor for the remaining parameters. - * - * @param retryState the retry state that decides the wait before each reconnection - */ - FDv2StreamingSynchronizer( - @NonNull LDContext evaluationContext, - @NonNull SelectorSource selectorSource, - @NonNull URI streamBaseUri, - @NonNull String streamRequestPath, - @Nullable FDv2Requestor requestor, - int initialReconnectDelayMillis, - boolean evaluationReasons, - boolean useReport, - @NonNull HttpProperties httpProperties, - @NonNull ScheduledExecutorService executor, - @NonNull LDLogger logger, - @Nullable DiagnosticStore diagnosticStore, - @NonNull StreamingRetryState retryState ) { this.evaluationContext = evaluationContext; this.selectorSource = selectorSource; @@ -159,7 +133,7 @@ final class FDv2StreamingSynchronizer implements Synchronizer { this.executor = executor; this.logger = logger; this.diagnosticStore = diagnosticStore; - this.retryState = retryState; + this.retryState = new StreamingRetryState(initialReconnectDelayMillis); } @Override diff --git a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java index 809424d53..9248c93a1 100644 --- a/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java +++ b/launchdarkly-android-client-sdk/src/main/java/com/launchdarkly/sdk/android/SynchronizerFactoryWithState.java @@ -33,21 +33,24 @@ enum State { private State state = State.Available; private final boolean isFDv1Fallback; // Backoff after unexpected errors from this slot's synchronizers. - private final StreamingRetryState retryState = new StreamingRetryState( - RetryRegime.EXTENDED, - RetryRegime.EXTENDED, - StreamingRetryState.RESET_THRESHOLD_MILLIS, - new Random()); + private final StreamingRetryState retryState; @Nullable private ScheduledFuture pendingUnblock; - SynchronizerFactoryWithState(@NonNull FDv2DataSource.DataSourceFactory factory) { - this(factory, false); - } - - SynchronizerFactoryWithState(@NonNull FDv2DataSource.DataSourceFactory factory, boolean isFDv1Fallback) { + /** + * @param factory builds this slot's synchronizer + * @param isFDv1Fallback true if this slot holds the FDv1 fallback synchronizer + * @param backoff the delay bounds for the backoff after an unexpected error + */ + SynchronizerFactoryWithState( + @NonNull FDv2DataSource.DataSourceFactory factory, + boolean isFDv1Fallback, + @NonNull RetryRegime backoff + ) { this.factory = factory; this.isFDv1Fallback = isFDv1Fallback; + this.retryState = new StreamingRetryState( + backoff, backoff, StreamingRetryState.RESET_THRESHOLD_MILLIS, new Random()); } State getState() { diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceConditionsTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceConditionsTest.java index 854a49a9f..c560f5a22 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceConditionsTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceConditionsTest.java @@ -22,7 +22,7 @@ import java.util.concurrent.Future; import java.util.concurrent.ScheduledExecutorService; import java.util.concurrent.TimeUnit; -import java.util.concurrent.atomic.AtomicBoolean; +import java.util.concurrent.atomic.AtomicInteger; import java.util.concurrent.TimeoutException; import static org.junit.Assert.assertEquals; @@ -199,24 +199,13 @@ public void recovery_informDoesNothing() throws Exception { @Test public void recovery_gateClosed_rearmsTimerUntilGateOpens() throws Exception { - FakeScheduledExecutorService fakeExecutor = new FakeScheduledExecutorService(); - AtomicBoolean canRecover = new AtomicBoolean(false); - try { - RecoveryCondition condition = new RecoveryCondition(fakeExecutor, 1, canRecover::get); - assertEquals(1000, fakeExecutor.awaitScheduledDelayMillis(1000)); - - // With the gate closed, the timer firing re-arms it instead of completing the future. - fakeExecutor.advanceTime(1000); - assertEquals(1000, fakeExecutor.awaitScheduledDelayMillis(1000)); - assertFalse(condition.getFuture().isDone()); - - // With the gate open, the next firing completes the future. - canRecover.set(true); - fakeExecutor.advanceTime(1000); - assertEquals(ConditionType.RECOVERY, condition.getFuture().get(1, TimeUnit.SECONDS)); - } finally { - fakeExecutor.shutdownNow(); - } + // The gate stays closed for the first two firings and opens on the third, so the future + // completes only after the timer has been re-armed twice. + AtomicInteger firings = new AtomicInteger(0); + RecoveryCondition condition = new RecoveryCondition(executor, 0, () -> firings.incrementAndGet() >= 3); + + assertEquals(ConditionType.RECOVERY, condition.getFuture().get(500, TimeUnit.MILLISECONDS)); + assertEquals(3, firings.get()); } // ==== Conditions (wrapper) ==== diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java index 38e08a8c2..8d0316d19 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2DataSourceTest.java @@ -60,7 +60,6 @@ public class FDv2DataSourceTest { private static final long ORCHESTRATION_LOG_AWAIT_TIMEOUT_MS = AWAIT_TIMEOUT_SECONDS * 1000L; private ScheduledExecutorService executor; - private final FakeScheduledExecutorService fakeExecutor = new FakeScheduledExecutorService(); @Before public void setUp() { @@ -72,7 +71,6 @@ public void tearDown() { if (executor != null && !executor.isShutdown()) { executor.shutdownNow(); } - fakeExecutor.shutdownNow(); } private FDv2DataSource buildDataSource( @@ -107,14 +105,17 @@ private FDv2DataSource buildDataSource( recoveryTimeoutSeconds); } + // A synchronizer backoff short enough to wait out in a test. + private static final RetryRegime SHORT_BACKOFF = new RetryRegime(100, 100); + /** - * Builds a data source whose timers run on the given executor. + * Builds a data source with the given backoff after a synchronizer's unexpected error. */ private FDv2DataSource buildDataSource( MockComponents.MockDataSourceUpdateSink sink, List> initializers, List> synchronizers, - ScheduledExecutorService executor) { + RetryRegime synchronizerBackoff) { return new FDv2DataSource( CONTEXT, initializers, @@ -122,7 +123,10 @@ private FDv2DataSource buildDataSource( null, sink, executor, - logging.logger); + logging.logger, + FDv2DataSourceConditions.DEFAULT_FALLBACK_TIMEOUT_SECONDS, + FDv2DataSourceConditions.DEFAULT_RECOVERY_TIMEOUT_SECONDS, + synchronizerBackoff); } /** Starts the data source and returns a callback that will receive the start result. */ @@ -644,22 +648,17 @@ public void allSynchronizersFailingWithUnexpectedErrorsAreRetriedAfterBackoff() () -> secondBuilds.incrementAndGet() == 1 ? new MockQueuedSynchronizer(terminalError()) : new MockQueuedSynchronizer(FDv2SourceResult.changeSet(makeChangeSet(false), false))), - fakeExecutor); + SHORT_BACKOFF); AwaitableCallback startCallback = startDataSource(dataSource); // Each failure is reported as an interruption, and the data source waits for a backoff - // to end rather than reporting OFF. Nothing is rebuilt before the shortest possible - // backoff has elapsed. + // to end rather than reporting OFF. assertEquals(Arrays.asList(DataSourceState.INTERRUPTED, DataSourceState.INTERRUPTED), sink.awaitStatuses(2, AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS)); - fakeExecutor.advanceTime(RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS / 2 - 1); - assertEquals(1, firstBuilds.get()); - assertEquals(1, secondBuilds.get()); // Once the backoffs end, whichever synchronizer returns first is tried again and // initialization completes. The two backoffs have independent jitter, so either may be // the one that is rebuilt. - fakeExecutor.advanceTime(RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS / 2 + 1); assertTrue(startCallback.await(AWAIT_TIMEOUT_SECONDS * 1000)); assertEquals(3, firstBuilds.get() + secondBuilds.get()); stopDataSource(dataSource); @@ -1486,8 +1485,7 @@ public void statusIncludesErrorInfoOnFailure() throws Exception { FDv2DataSource dataSource = buildDataSource(sink, Collections.emptyList(), Collections.singletonList(() -> new MockQueuedSynchronizer( - FDv2SourceResult.status(FDv2SourceResult.Status.terminalError(terminalErr), false))), - fakeExecutor); + FDv2SourceResult.status(FDv2SourceResult.Status.terminalError(terminalErr), false)))); startDataSource(dataSource); @@ -1535,7 +1533,7 @@ public void statusStaysInterruptedWhileTheOnlySynchronizerBacksOffThenReturnsToV FDv2SourceResult.changeSet(makeChangeSet(false), false), FDv2SourceResult.status(FDv2SourceResult.Status.terminalError(err), false)) : new MockQueuedSynchronizer(FDv2SourceResult.changeSet(makeChangeSet(false), false))), - fakeExecutor); + SHORT_BACKOFF); AwaitableCallback startCallback = startDataSource(dataSource); assertTrue(startCallback.await(AWAIT_TIMEOUT_SECONDS * 1000)); @@ -1545,7 +1543,6 @@ public void statusStaysInterruptedWhileTheOnlySynchronizerBacksOffThenReturnsToV // Once the backoff ends the synchronizer is tried again and the status returns to VALID, // never having reached OFF. - fakeExecutor.advanceTime(RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); assertEquals(DataSourceState.VALID, sink.awaitStatus(AWAIT_TIMEOUT_SECONDS, TimeUnit.SECONDS)); assertEquals(2, builds.get()); stopDataSource(dataSource); @@ -2113,8 +2110,7 @@ public void orchestrationLogging_unexpectedError_logsWarn() throws Exception { MockComponents.MockDataSourceUpdateSink sink = new MockComponents.MockDataSourceUpdateSink(); FDv2DataSource dataSource = buildDataSource(sink, Collections.emptyList(), - Collections.singletonList(() -> new MockQueuedSynchronizer(terminalError())), - fakeExecutor); + Collections.singletonList(() -> new MockQueuedSynchronizer(terminalError()))); startDataSource(dataSource); awaitLogContains(logging, "Synchronizer 'MockQueuedSynchronizer' reported an unexpected error and will not be tried again for"); @@ -2128,8 +2124,7 @@ public void orchestrationLogging_allSynchronizersBackingOff_logsInfo() throws Ex Collections.emptyList(), Arrays.asList( () -> new MockQueuedSynchronizer(terminalError()), - () -> new MockQueuedSynchronizer(terminalError())), - fakeExecutor); + () -> new MockQueuedSynchronizer(terminalError()))); startDataSource(dataSource); awaitLogContains(logging, "All synchronizers are waiting out a backoff after unexpected errors"); stopDataSource(dataSource); diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java index a52da249c..f4218d934 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/FDv2StreamingSynchronizerTest.java @@ -21,10 +21,8 @@ import java.io.IOException; import java.net.URI; import java.util.HashMap; -import java.util.Random; -import java.util.concurrent.ScheduledExecutorService; -import java.util.concurrent.Executors; import java.util.concurrent.Future; +import java.util.concurrent.ScheduledThreadPoolExecutor; import java.util.concurrent.TimeUnit; import java.util.concurrent.atomic.AtomicBoolean; import java.util.concurrent.atomic.AtomicInteger; @@ -39,13 +37,11 @@ public class FDv2StreamingSynchronizerTest { @Rule public Timeout globalTimeout = Timeout.seconds(10); - private final ScheduledExecutorService executor = Executors.newScheduledThreadPool(4); - private final FakeScheduledExecutorService fakeExecutor = new FakeScheduledExecutorService(); + private final ScheduledThreadPoolExecutor executor = new ScheduledThreadPoolExecutor(4); @After public void tearDown() { executor.shutdownNow(); - fakeExecutor.shutdownNow(); } private static final LDContext CONTEXT = LDContext.create("test-context"); @@ -102,30 +98,11 @@ private FDv2StreamingSynchronizer makeSynchronizer( httpProperties(), executor, LOGGER, null); } - // The tests of backoff behavior drive the synchronizer's timers with a - // FakeScheduledExecutorService and a StreamingRetryState without jitter, so the delay chosen - // for each reconnect can be asserted exactly instead of waited for. - private static final long NORMAL_DELAY_MILLIS = 1000; - private static final long NORMAL_MAX_DELAY_MILLIS = 4000; - // A healthy-operation threshold no test reaches. - private static final long NEVER_RESET_MILLIS = 60_000; - - private static StreamingRetryState retryStateWithoutJitter(long healthyResetThresholdMillis) { - RetryRegime regime = new RetryRegime(NORMAL_DELAY_MILLIS, NORMAL_MAX_DELAY_MILLIS); - return new StreamingRetryState(regime, regime, healthyResetThresholdMillis, - new Random() { - @Override - public double nextDouble() { - return 0; - } - }); - } - - private FDv2StreamingSynchronizer makeSynchronizer(URI streamBaseUri, StreamingRetryState retryState) { + private FDv2StreamingSynchronizer makeSynchronizer(URI streamBaseUri, int initialReconnectDelayMillis) { return new FDv2StreamingSynchronizer( CONTEXT, EMPTY_SELECTOR_SOURCE, streamBaseUri, STREAM_PATH, - null, 1, false, false, - httpProperties(), fakeExecutor, LOGGER, null, retryState); + null, initialReconnectDelayMillis, false, false, + httpProperties(), executor, LOGGER, null); } private static DiagnosticStore basicDiagnosticStore() { @@ -369,7 +346,7 @@ public void httpNonRecoverableError() throws Exception { // ---- backoff after failures ---- @Test - public void recoverableErrorSchedulesReconnectWithGrowingDelay() throws Exception { + public void recoverableErrorIsReportedAndTheStreamReconnects() throws Exception { String serverIntent = makeEvent("server-intent", "{\"payloads\":[{\"id\":\"payload-1\",\"target\":100,\"intentCode\":\"xfer-full\",\"reason\":\"payload-missing\"}]}"); String payloadTransferred = makeEvent("payload-transferred", "{\"state\":\"(p:payload-1:100)\",\"version\":100}"); @@ -382,60 +359,12 @@ public void recoverableErrorSchedulesReconnectWithGrowingDelay() throws Exceptio Handlers.SSE.event(payloadTransferred), Handlers.SSE.leaveOpen())))) { - FDv2StreamingSynchronizer sync = makeSynchronizer(server.getUri(), retryStateWithoutJitter(NEVER_RESET_MILLIS)); + FDv2StreamingSynchronizer sync = makeSynchronizer(server.getUri(), 1); - // Each 503 is reported as an interruption, and the reconnect is scheduled with double - // the previous delay. + // Each 503 is reported as an interruption, and the synchronizer reconnects on its own + // until the stream delivers data. assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); - assertEquals(NORMAL_DELAY_MILLIS, fakeExecutor.awaitScheduledDelayMillis(5000)); - fakeExecutor.advanceTime(NORMAL_DELAY_MILLIS); assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); - assertEquals(NORMAL_DELAY_MILLIS * 2, fakeExecutor.awaitScheduledDelayMillis(5000)); - - // Once that delay has passed, the synchronizer reconnects and delivers data. - fakeExecutor.advanceTime(NORMAL_DELAY_MILLIS * 2); - assertEquals(SourceResultType.CHANGE_SET, sync.next().get(5, TimeUnit.SECONDS).getResultType()); - - sync.close(); - } - } - - @Test - public void healthyStreamResetsBackoffToInitialDelay() throws Exception { - String serverIntent = makeEvent("server-intent", "{\"payloads\":[{\"id\":\"payload-1\",\"target\":100,\"intentCode\":\"xfer-full\",\"reason\":\"payload-missing\"}]}"); - String payloadTransferred = makeEvent("payload-transferred", "{\"state\":\"(p:payload-1:100)\",\"version\":100}"); - // The synchronizer measures healthy operation on its own clock, so the server must hold - // the stream open for a moment after the data before ending it. This is the shortest - // margin that reliably exceeds the 1 ms threshold used here. - long healthyMarginMillis = 20; - - try (HttpServer server = HttpServer.start(Handlers.sequential( - Handlers.status(503), - Handlers.all( - Handlers.SSE.start(), - Handlers.SSE.event(serverIntent), - Handlers.SSE.event(payloadTransferred), - Handlers.delay(healthyMarginMillis)), - Handlers.all( - Handlers.SSE.start(), - Handlers.SSE.event(serverIntent), - Handlers.SSE.event(payloadTransferred), - Handlers.SSE.leaveOpen())))) { - - FDv2StreamingSynchronizer sync = makeSynchronizer(server.getUri(), retryStateWithoutJitter(1)); - - // The 503 costs one attempt. The reconnect then delivers data and is ended by the - // server. - assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); - assertEquals(NORMAL_DELAY_MILLIS, fakeExecutor.awaitScheduledDelayMillis(5000)); - fakeExecutor.advanceTime(NORMAL_DELAY_MILLIS); - assertEquals(SourceResultType.CHANGE_SET, sync.next().get(5, TimeUnit.SECONDS).getResultType()); - assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); - - // Having been healthy for longer than the threshold, the next delay starts over - // instead of doubling. - assertEquals(NORMAL_DELAY_MILLIS, fakeExecutor.awaitScheduledDelayMillis(5000)); - fakeExecutor.advanceTime(NORMAL_DELAY_MILLIS); assertEquals(SourceResultType.CHANGE_SET, sync.next().get(5, TimeUnit.SECONDS).getResultType()); sync.close(); @@ -445,13 +374,14 @@ public void healthyStreamResetsBackoffToInitialDelay() throws Exception { @Test public void closeDuringBackoffCancelsReconnect() throws Exception { try (HttpServer server = HttpServer.start(Handlers.status(503))) { - FDv2StreamingSynchronizer sync = makeSynchronizer(server.getUri(), retryStateWithoutJitter(NEVER_RESET_MILLIS)); + // A long reconnect delay keeps the scheduled attempt pending until close() runs. + executor.setRemoveOnCancelPolicy(true); + FDv2StreamingSynchronizer sync = makeSynchronizer(server.getUri(), 60_000); assertEquals(SourceSignal.INTERRUPTED, sync.next().get(5, TimeUnit.SECONDS).getStatus().getState()); - assertEquals(NORMAL_DELAY_MILLIS, fakeExecutor.awaitScheduledDelayMillis(5000)); // Closing cancels the scheduled attempt and reports shutdown. sync.close(); - assertTrue(fakeExecutor.pendingDelaysMillis().isEmpty()); + assertTrue(executor.getQueue().isEmpty()); assertEquals(SourceSignal.SHUTDOWN, sync.next().get(1, TimeUnit.SECONDS).getStatus().getState()); } } diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java index 9e143459f..82af9d669 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/PollingDataSourceTest.java @@ -291,14 +291,14 @@ public void terminatesAfterMaxNumberOfPolls() throws Exception { // --- backoff after failures --- // - // These tests drive the data source's timers with a FakeTaskExecutor and a PollingRetryState - // without jitter. The mock fetcher answers synchronously, so every poll and its outcome - // happen inside advanceTime() and the scheduled delays can be asserted exactly. + // These tests drive the data source's timers with a ManualTaskExecutor and a + // PollingRetryState without jitter. The mock fetcher answers synchronously, so every poll and + // its outcome happen inside runPendingTasks() and the scheduled delays can be asserted exactly. private static final long POLL_INTERVAL_MILLIS = 30_000; private static final long EXTENDED_DELAY_MILLIS = 300_000; - private final FakeTaskExecutor fakeTaskExecutor = new FakeTaskExecutor(); + private final ManualTaskExecutor manualTaskExecutor = new ManualTaskExecutor(); private static LDInvalidResponseCodeFailure httpFailure(int status) { return new LDInvalidResponseCodeFailure("test failure", status, LDUtil.isHttpErrorRecoverable(status)); @@ -325,7 +325,7 @@ private PollingDataSource makePollingDataSource(long maxNumberOfPolls) { maxNumberOfPolls, clientContext.getFetcher(), clientContext.getPlatformState(), - fakeTaskExecutor, + manualTaskExecutor, retryStateWithoutJitter(), clientContext.getBaseLogger() ); @@ -355,12 +355,12 @@ public void normalErrorIsPolledAgainAfterPollInterval() { // The first poll fails with a 500. ds.start(callback); - fakeTaskExecutor.runDueTasks(); + manualTaskExecutor.runPendingTasks(); assertEquals(1, callback.errors.size()); // The next poll is scheduled for the regular interval and succeeds. - assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); - fakeTaskExecutor.advanceTime(POLL_INTERVAL_MILLIS); + assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), manualTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); assertEquals(1, callback.successes.size()); assertEquals(2, fetcher.receivedContexts.size()); } @@ -374,14 +374,14 @@ public void unexpectedErrorIsPolledAgainAfterExtendedDelay() { // The first poll fails with a 401, which is reported without shutting the SDK down. ds.start(callback); - fakeTaskExecutor.runDueTasks(); + manualTaskExecutor.runPendingTasks(); assertEquals(401, ((LDInvalidResponseCodeFailure) callback.errors.get(0)).getResponseCode()); assertFalse(dataSourceUpdateSink.shutDownCalled); // The next poll is scheduled for the extended delay rather than the poll interval, and // succeeds once that delay has passed. - assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); - fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), manualTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); assertEquals(1, callback.successes.size()); assertEquals(2, fetcher.receivedContexts.size()); } @@ -398,10 +398,10 @@ public void sustainedUnexpectedErrorsKeepPollingWithGrowingDelay() { // Each failure schedules the next poll with double the delay, and the data source never // gives up. ds.start(callback); - fakeTaskExecutor.runDueTasks(); + manualTaskExecutor.runPendingTasks(); for (long expectedDelay : new long[] {EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 2, EXTENDED_DELAY_MILLIS * 4}) { - assertEquals(Collections.singletonList(expectedDelay), fakeTaskExecutor.pendingDelaysMillis()); - fakeTaskExecutor.advanceTime(expectedDelay); + assertEquals(Collections.singletonList(expectedDelay), manualTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); } // The fourth poll succeeded. @@ -421,18 +421,18 @@ public void twoConsecutiveSuccessfulPollsResetBackoff() { // The 401 moves the data source to the extended delay. ds.start(callback); - fakeTaskExecutor.runDueTasks(); - assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), manualTaskExecutor.pendingDelaysMillis()); // One success returns to the regular interval. A second one clears the backoff, so the // 500 that follows is retried at the regular interval instead of a doubled extended delay. - fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); - assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); - fakeTaskExecutor.advanceTime(POLL_INTERVAL_MILLIS); - assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); - fakeTaskExecutor.advanceTime(POLL_INTERVAL_MILLIS); + manualTaskExecutor.runPendingTasks(); + assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), manualTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); + assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), manualTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); assertEquals(2, callback.errors.size()); - assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + assertEquals(Collections.singletonList(POLL_INTERVAL_MILLIS), manualTaskExecutor.pendingDelaysMillis()); } @Test @@ -444,11 +444,11 @@ public void oneShotPollIsNotRetriedAfterFailure() { // The single poll fails. ds.start(callback); - fakeTaskExecutor.runDueTasks(); + manualTaskExecutor.runPendingTasks(); assertEquals(1, callback.errors.size()); // A one-shot data source is done after its one poll, so nothing further is scheduled. - assertTrue(fakeTaskExecutor.pendingDelaysMillis().isEmpty()); + assertTrue(manualTaskExecutor.pendingDelaysMillis().isEmpty()); } @Test @@ -460,12 +460,12 @@ public void stopCancelsPendingPoll() { // The 401 schedules a poll for the extended delay. ds.start(callback); - fakeTaskExecutor.runDueTasks(); - assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), manualTaskExecutor.pendingDelaysMillis()); // Stopping cancels it. ds.stop(LDUtil.noOpCallback()); - assertTrue(fakeTaskExecutor.pendingDelaysMillis().isEmpty()); + assertTrue(manualTaskExecutor.pendingDelaysMillis().isEmpty()); } private class MockFetcher implements FeatureFetcher { diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java index e0f9c0e9d..5b7121eae 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/SourceManagerTest.java @@ -19,6 +19,7 @@ import java.util.Collections; import java.util.List; import java.util.concurrent.Future; +import java.util.concurrent.ScheduledThreadPoolExecutor; import java.util.concurrent.TimeUnit; /** @@ -30,7 +31,12 @@ public class SourceManagerTest { @Rule public Timeout globalTimeout = Timeout.seconds(5); - private final FakeScheduledExecutorService executor = new FakeScheduledExecutorService(); + // A backoff long enough that no scheduled return runs during a test, with room to double. + private static final RetryRegime LONG_BACKOFF = new RetryRegime(60_000, 240_000); + // A backoff short enough to wait out. + private static final RetryRegime SHORT_BACKOFF = new RetryRegime(50, 50); + + private final ScheduledThreadPoolExecutor executor = new ScheduledThreadPoolExecutor(1); private final List slots = new ArrayList<>(); @After @@ -62,17 +68,17 @@ public String name() { } } - private SourceManager manager(String... names) { + private SourceManager manager(RetryRegime backoff, String... names) { for (final String name : names) { - slots.add(new SynchronizerFactoryWithState(() -> new NamedSynchronizer(name))); + slots.add(new SynchronizerFactoryWithState(() -> new NamedSynchronizer(name), false, backoff)); } return new SourceManager(slots, Collections.emptyList(), executor); } - private static void assertFirstBackoff(long delayMillis) { + /** Asserts that a delay is the first backoff of the regime: its initial delay, less jitter. */ + private static void assertFirstBackoff(RetryRegime regime, long delayMillis) { assertTrue("delay " + delayMillis, - delayMillis > RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS / 2 - && delayMillis <= RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); + delayMillis > regime.initialDelayMillis / 2 && delayMillis <= regime.initialDelayMillis); } /** Selects the next synchronizer, asserting that one is known right away. */ @@ -85,12 +91,10 @@ private static String nextNow(SourceManager manager) throws Exception { @Test public void backingOffTheCurrentSlotSkipsItInFavorOfTheNext() throws Exception { - SourceManager manager = manager("a", "b"); + SourceManager manager = manager(LONG_BACKOFF, "a", "b"); assertEquals("a", nextNow(manager)); - long delay = manager.backOffCurrentSynchronizer(0); - assertFirstBackoff(delay); - assertEquals(delay, executor.awaitScheduledDelayMillis(1000)); + assertFirstBackoff(LONG_BACKOFF, manager.backOffCurrentSynchronizer(0)); assertEquals("b", nextNow(manager)); assertEquals("b", nextNow(manager)); @@ -98,23 +102,19 @@ public void backingOffTheCurrentSlotSkipsItInFavorOfTheNext() throws Exception { @Test public void slotReturnsWhenItsBackoffEnds() throws Exception { - SourceManager manager = manager("a"); + SourceManager manager = manager(SHORT_BACKOFF, "a"); assertEquals("a", nextNow(manager)); - long delay = manager.backOffCurrentSynchronizer(0); + manager.backOffCurrentSynchronizer(0); - // With every slot backing off, the next synchronizer is not known yet. + // With every slot backing off, the next synchronizer is supplied when the backoff ends. Future next = manager.nextAvailableSynchronizer(); - assertFalse(next.isDone()); - - // The backoff ending supplies it. - executor.advanceTime(delay); - assertEquals("a", next.get(1, TimeUnit.SECONDS).name()); + assertEquals("a", next.get(2, TimeUnit.SECONDS).name()); assertEquals("a", nextNow(manager)); } @Test public void aBackingOffSlotStillOutranksTheCurrentOneForRecovery() throws Exception { - SourceManager manager = manager("a", "b"); + SourceManager manager = manager(LONG_BACKOFF, "a", "b"); nextNow(manager); manager.backOffCurrentSynchronizer(0); assertEquals("b", nextNow(manager)); @@ -130,16 +130,15 @@ public void aBackingOffSlotStillOutranksTheCurrentOneForRecovery() throws Except @Test public void repeatedUnexpectedErrorsDoubleTheBackoffUntilHealthyOperationResetsIt() throws Exception { - SourceManager manager = manager("a"); + SourceManager manager = manager(LONG_BACKOFF, "a"); nextNow(manager); // A second unexpected error without any healthy operation in between doubles the wait. - long firstDelay = manager.backOffCurrentSynchronizer(0); - assertFirstBackoff(firstDelay); + assertFirstBackoff(LONG_BACKOFF, manager.backOffCurrentSynchronizer(0)); manager.endBackoff(slots.get(0)); nextNow(manager); long secondDelay = manager.backOffCurrentSynchronizer(0); - assertTrue("delay " + secondDelay, secondDelay > RetryRegime.EXTENDED_INITIAL_DELAY_MILLIS); + assertTrue("delay " + secondDelay, secondDelay > LONG_BACKOFF.initialDelayMillis); manager.endBackoff(slots.get(0)); // Healthy operation for the reset threshold before the next error starts the backoff over. @@ -148,12 +147,13 @@ public void repeatedUnexpectedErrorsDoubleTheBackoffUntilHealthyOperationResetsI manager.recordCurrentSynchronizerHealthy(healthyAt); long thirdDelay = manager.backOffCurrentSynchronizer( healthyAt + StreamingRetryState.RESET_THRESHOLD_MILLIS); - assertFirstBackoff(thirdDelay); + assertFirstBackoff(LONG_BACKOFF, thirdDelay); } @Test public void closeCompletesTheWaitWithNullAndCancelsPendingBackoffs() throws Exception { - SourceManager manager = manager("a"); + executor.setRemoveOnCancelPolicy(true); + SourceManager manager = manager(LONG_BACKOFF, "a"); nextNow(manager); manager.backOffCurrentSynchronizer(0); Future next = manager.nextAvailableSynchronizer(); @@ -161,12 +161,12 @@ public void closeCompletesTheWaitWithNullAndCancelsPendingBackoffs() throws Exce manager.close(); assertNull(next.get(1, TimeUnit.SECONDS)); - assertTrue(executor.pendingDelaysMillis().isEmpty()); + assertTrue(executor.getQueue().isEmpty()); assertNull(nextNow(manager)); } @Test public void noSynchronizersYieldsNullRightAway() throws Exception { - assertNull(nextNow(manager())); + assertNull(nextNow(manager(LONG_BACKOFF))); } } diff --git a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java index d7760db78..74de79c1a 100644 --- a/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java +++ b/launchdarkly-android-client-sdk/src/test/java/com/launchdarkly/sdk/android/StreamingDataSourceTest.java @@ -167,15 +167,15 @@ private StreamingDataSource makeStreamingDataSource( .build(clientContext); } - // The tests of backoff behavior below drive the data source's timers with a FakeTaskExecutor - // and a StreamingRetryState without jitter, so the delay chosen for each reconnect can be - // asserted exactly instead of waited for. + // The tests of backoff behavior below drive the data source's timers with a + // ManualTaskExecutor and a StreamingRetryState without jitter, so the delay chosen for each + // reconnect can be asserted exactly instead of waited for. private static final long NORMAL_DELAY_MILLIS = 1; private static final long EXTENDED_DELAY_MILLIS = 300_000; // A healthy-operation threshold no test reaches. private static final long NEVER_RESET_MILLIS = 60_000; - private final FakeTaskExecutor fakeTaskExecutor = new FakeTaskExecutor(); + private final ManualTaskExecutor manualTaskExecutor = new ManualTaskExecutor(); private static StreamingRetryState retryStateWithoutJitter(long healthyResetThresholdMillis) { return new StreamingRetryState( @@ -197,7 +197,7 @@ private StreamingDataSource makeStreamingDataSource(URI streamBaseUri, Streaming ClientContext baseClientContext = ClientContextImpl.fromConfig( config, MOBILE_KEY, "", perEnvironmentData, makeFeatureFetcher(), CONTEXT, - logging.logger, platformState, environmentReporter, fakeTaskExecutor); + logging.logger, platformState, environmentReporter, manualTaskExecutor); ClientContext clientContext = ClientContextImpl.forDataSource( baseClientContext, dataSourceUpdateSink, CONTEXT, false, false); return new StreamingDataSource(clientContext, CONTEXT, dataSourceUpdateSink, @@ -826,10 +826,10 @@ public void unexpectedErrorSchedulesReconnectAfterExtendedDelay() throws Excepti // The 401 is reported, and the reconnect is scheduled for the extended delay. assertNotNull(callback.awaitError()); server.getRecorder().requireRequest(); - assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), manualTaskExecutor.pendingDelaysMillis()); // Once that delay has passed, the data source reconnects and receives data. - fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); + manualTaskExecutor.runPendingTasks(); assertNotNull(callback.awaitSuccess()); server.getRecorder().requireRequest(); assertFalse(dataSourceUpdateSink.shutDownCalled); @@ -860,15 +860,15 @@ public void unexpectedErrorOnReconnectSchedulesExtendedDelay() throws Exception // scheduled for the normal delay. Throwable error = callback.awaitError(); assertEquals(LDFailure.FailureType.NETWORK_FAILURE, ((LDFailure) error).getFailureType()); - assertEquals(Collections.singletonList(NORMAL_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); - fakeTaskExecutor.advanceTime(NORMAL_DELAY_MILLIS); + assertEquals(Collections.singletonList(NORMAL_DELAY_MILLIS), manualTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); // The 403 on that reconnect moves the data source to the extended delay, after which // it connects again. error = callback.awaitError(); assertEquals(403, ((LDInvalidResponseCodeFailure) error).getResponseCode()); - assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); - fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), manualTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); assertNotNull(callback.awaitSuccess()); } } @@ -886,8 +886,8 @@ public void sustainedUnexpectedErrorsKeepReconnectingWithGrowingDelay() throws E for (long expectedDelay : new long[] {EXTENDED_DELAY_MILLIS, EXTENDED_DELAY_MILLIS * 2, EXTENDED_DELAY_MILLIS * 4}) { assertNotNull(callback.awaitError()); server.getRecorder().requireRequest(); - assertEquals(Collections.singletonList(expectedDelay), fakeTaskExecutor.pendingDelaysMillis()); - fakeTaskExecutor.advanceTime(expectedDelay); + assertEquals(Collections.singletonList(expectedDelay), manualTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); } assertNotNull(callback.awaitError()); assertFalse(dataSourceUpdateSink.shutDownCalled); @@ -920,14 +920,14 @@ public void healthyStreamResetsBackoffToNormalDelay() throws Exception { // The 401 puts the data source in the extended regime. The reconnect then delivers // data and is ended by the server. assertNotNull(callback.awaitError()); - fakeTaskExecutor.advanceTime(EXTENDED_DELAY_MILLIS); + manualTaskExecutor.runPendingTasks(); assertNotNull(callback.awaitSuccess()); assertNotNull(callback.awaitError()); // Having been healthy for longer than the threshold, the data source is back to the // normal delay rather than doubling the extended one. - assertEquals(Collections.singletonList(NORMAL_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); - fakeTaskExecutor.advanceTime(NORMAL_DELAY_MILLIS); + assertEquals(Collections.singletonList(NORMAL_DELAY_MILLIS), manualTaskExecutor.pendingDelaysMillis()); + manualTaskExecutor.runPendingTasks(); assertNotNull(callback.awaitSuccess()); } } @@ -940,13 +940,13 @@ public void stopCancelsPendingReconnect() throws Exception { TrackingCallback callback = new TrackingCallback(); startDataSource(sds, callback); assertNotNull(callback.awaitError()); - assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), fakeTaskExecutor.pendingDelaysMillis()); + assertEquals(Collections.singletonList(EXTENDED_DELAY_MILLIS), manualTaskExecutor.pendingDelaysMillis()); // Stopping cancels the scheduled reconnect. AwaitableCallback stopped = new AwaitableCallback<>(); sds.stop(stopped); stopped.await(STOP_TIMEOUT_MILLIS); - assertTrue(fakeTaskExecutor.pendingDelaysMillis().isEmpty()); + assertTrue(manualTaskExecutor.pendingDelaysMillis().isEmpty()); } } diff --git a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java deleted file mode 100644 index f95fefe10..000000000 --- a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeScheduledExecutorService.java +++ /dev/null @@ -1,193 +0,0 @@ -package com.launchdarkly.sdk.android; - -import androidx.annotation.NonNull; - -import java.util.ArrayList; -import java.util.Collections; -import java.util.Comparator; -import java.util.Iterator; -import java.util.List; -import java.util.concurrent.AbstractExecutorService; -import java.util.concurrent.BlockingQueue; -import java.util.concurrent.Callable; -import java.util.concurrent.Delayed; -import java.util.concurrent.ExecutorService; -import java.util.concurrent.Executors; -import java.util.concurrent.FutureTask; -import java.util.concurrent.LinkedBlockingQueue; -import java.util.concurrent.ScheduledExecutorService; -import java.util.concurrent.ScheduledFuture; -import java.util.concurrent.TimeUnit; - -/** - * A {@link ScheduledExecutorService} with a manual clock, for unit tests of timer-driven code - * whose tasks must run on real background threads. - *

- * Tasks submitted without a delay run right away on a background thread. A task scheduled with a - * delay is held until {@link #advanceTime(long)} moves the clock past that delay. Tests can wait - * for a scheduling event with {@link #awaitScheduledDelayMillis(long)} and inspect held tasks - * with {@link #pendingDelaysMillis()}. - */ -public class FakeScheduledExecutorService extends AbstractExecutorService implements ScheduledExecutorService { - private final ExecutorService delegate = Executors.newCachedThreadPool(); - private final Object lock = new Object(); - private final List held = new ArrayList<>(); - private final BlockingQueue scheduledDelays = new LinkedBlockingQueue<>(); - private long nowMillis = 0; - - @Override - public void execute(@NonNull Runnable command) { - delegate.execute(command); - } - - @NonNull - @Override - public ScheduledFuture schedule(@NonNull Runnable command, long delay, @NonNull TimeUnit unit) { - long delayMillis = Math.max(0, unit.toMillis(delay)); - HeldTask task; - synchronized (lock) { - task = new HeldTask(command, nowMillis + delayMillis); - if (delayMillis == 0) { - delegate.execute(task); - } else { - held.add(task); - } - } - scheduledDelays.add(delayMillis); - return task; - } - - /** - * Waits for the code under test to call {@link #schedule(Runnable, long, TimeUnit)} and - * returns the delay it asked for, in milliseconds. Each call consumes one scheduling event, - * in the order they happened. - * - * @param timeoutMillis how long to wait for a scheduling event - * @return the scheduled delay in milliseconds - * @throws AssertionError if nothing is scheduled within the timeout - */ - public long awaitScheduledDelayMillis(long timeoutMillis) throws InterruptedException { - Long delay = scheduledDelays.poll(timeoutMillis, TimeUnit.MILLISECONDS); - if (delay == null) { - throw new AssertionError("timed out waiting for a task to be scheduled"); - } - return delay; - } - - /** - * @return how long, from the fake clock's current time, until each held task is due, in the - * order the tasks were scheduled. Cancelled and already-released tasks are omitted. - */ - public List pendingDelaysMillis() { - List delays = new ArrayList<>(); - synchronized (lock) { - for (HeldTask task : held) { - if (!task.isCancelled()) { - delays.add(task.dueMillis - nowMillis); - } - } - } - return delays; - } - - /** - * Moves the clock forward and releases every held task that is due by the new time, in due - * order, to a background thread. - */ - public void advanceTime(long millis) { - List due = new ArrayList<>(); - synchronized (lock) { - nowMillis += millis; - Iterator it = held.iterator(); - while (it.hasNext()) { - HeldTask task = it.next(); - if (task.isCancelled()) { - it.remove(); - } else if (task.dueMillis <= nowMillis) { - it.remove(); - due.add(task); - } - } - } - Collections.sort(due, new Comparator() { - @Override - public int compare(HeldTask a, HeldTask b) { - return Long.compare(a.dueMillis, b.dueMillis); - } - }); - for (HeldTask task : due) { - delegate.execute(task); - } - } - - @NonNull - @Override - public ScheduledFuture schedule(@NonNull Callable callable, long delay, @NonNull TimeUnit unit) { - throw new UnsupportedOperationException("FakeScheduledExecutorService does not support Callable scheduling"); - } - - @NonNull - @Override - public ScheduledFuture scheduleAtFixedRate(@NonNull Runnable command, long initialDelay, long period, @NonNull TimeUnit unit) { - throw new UnsupportedOperationException("FakeScheduledExecutorService does not support repeating tasks"); - } - - @NonNull - @Override - public ScheduledFuture scheduleWithFixedDelay(@NonNull Runnable command, long initialDelay, long delay, @NonNull TimeUnit unit) { - throw new UnsupportedOperationException("FakeScheduledExecutorService does not support repeating tasks"); - } - - @Override - public void shutdown() { - synchronized (lock) { - held.clear(); - } - delegate.shutdown(); - } - - @NonNull - @Override - public List shutdownNow() { - synchronized (lock) { - held.clear(); - } - return delegate.shutdownNow(); - } - - @Override - public boolean isShutdown() { - return delegate.isShutdown(); - } - - @Override - public boolean isTerminated() { - return delegate.isTerminated(); - } - - @Override - public boolean awaitTermination(long timeout, @NonNull TimeUnit unit) throws InterruptedException { - return delegate.awaitTermination(timeout, unit); - } - - private final class HeldTask extends FutureTask implements ScheduledFuture { - final long dueMillis; - - HeldTask(Runnable command, long dueMillis) { - super(command, null); - this.dueMillis = dueMillis; - } - - @Override - public long getDelay(@NonNull TimeUnit unit) { - synchronized (lock) { - return unit.convert(dueMillis - nowMillis, TimeUnit.MILLISECONDS); - } - } - - @Override - public int compareTo(@NonNull Delayed other) { - return Long.compare(getDelay(TimeUnit.MILLISECONDS), other.getDelay(TimeUnit.MILLISECONDS)); - } - } -} diff --git a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java deleted file mode 100644 index aac5059a5..000000000 --- a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/FakeTaskExecutor.java +++ /dev/null @@ -1,130 +0,0 @@ -package com.launchdarkly.sdk.android; - -import java.util.ArrayList; -import java.util.Collections; -import java.util.Comparator; -import java.util.Iterator; -import java.util.List; -import java.util.concurrent.Delayed; -import java.util.concurrent.FutureTask; -import java.util.concurrent.ScheduledFuture; -import java.util.concurrent.TimeUnit; - -/** - * A {@link TaskExecutor} with a manual clock, for unit tests of timer-driven code. A scheduled - * task is held until {@link #advanceTime(long)} moves the clock past its delay, and then runs - * synchronously on the calling thread. - */ -public class FakeTaskExecutor implements TaskExecutor { - private final Object lock = new Object(); - private final List tasks = new ArrayList<>(); - private long nowMillis = 0; - - @Override - public void executeOnMainThread(Runnable action) { - action.run(); - } - - @Override - public ScheduledFuture scheduleTask(Runnable action, long delayMillis) { - synchronized (lock) { - ScheduledTask task = new ScheduledTask(action, nowMillis + Math.max(0, delayMillis)); - tasks.add(task); - return task; - } - } - - @Override - public ScheduledFuture startRepeatingTask(Runnable action, long initialDelayMillis, long intervalMillis) { - throw new UnsupportedOperationException("FakeTaskExecutor does not support repeating tasks"); - } - - @Override - public void close() { - synchronized (lock) { - tasks.clear(); - } - } - - /** - * @return how long, from the fake clock's current time, until each pending task is due, in - * the order the tasks were scheduled. Cancelled tasks are omitted. - */ - public List pendingDelaysMillis() { - List delays = new ArrayList<>(); - synchronized (lock) { - for (ScheduledTask task : tasks) { - if (!task.isCancelled()) { - delays.add(task.dueMillis - nowMillis); - } - } - } - return delays; - } - - /** - * Runs every task whose time has already come, without moving the clock. - */ - public void runDueTasks() { - advanceTime(0); - } - - /** - * Moves the clock forward and runs every task that is due by the new time, in due order, on - * the calling thread. A task that schedules another task during its run is also run here if - * that task is due. - */ - public void advanceTime(long millis) { - synchronized (lock) { - nowMillis += millis; - } - while (true) { - List due = new ArrayList<>(); - synchronized (lock) { - Iterator it = tasks.iterator(); - while (it.hasNext()) { - ScheduledTask task = it.next(); - if (task.isCancelled()) { - it.remove(); - } else if (task.dueMillis <= nowMillis) { - it.remove(); - due.add(task); - } - } - } - if (due.isEmpty()) { - return; - } - Collections.sort(due, new Comparator() { - @Override - public int compare(ScheduledTask a, ScheduledTask b) { - return Long.compare(a.dueMillis, b.dueMillis); - } - }); - for (ScheduledTask task : due) { - task.run(); - } - } - } - - private final class ScheduledTask extends FutureTask implements ScheduledFuture { - final long dueMillis; - - ScheduledTask(Runnable action, long dueMillis) { - super(action, null); - this.dueMillis = dueMillis; - } - - @Override - public long getDelay(TimeUnit unit) { - synchronized (lock) { - return unit.convert(dueMillis - nowMillis, TimeUnit.MILLISECONDS); - } - } - - @Override - public int compareTo(Delayed other) { - return Long.compare(getDelay(TimeUnit.MILLISECONDS), other.getDelay(TimeUnit.MILLISECONDS)); - } - } -} diff --git a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/ManualTaskExecutor.java b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/ManualTaskExecutor.java index 0f48f4a27..2bc543400 100644 --- a/shared-test-code/src/main/java/com/launchdarkly/sdk/android/ManualTaskExecutor.java +++ b/shared-test-code/src/main/java/com/launchdarkly/sdk/android/ManualTaskExecutor.java @@ -12,7 +12,10 @@ *

* This avoids {@code Thread.sleep}-based timing, which is flaky on loaded CI runners. Cancelled * tasks (e.g. when a debounce timer is reset) are never run, and {@link #cancelledCount()} lets - * tests assert how many times a task was cancelled/rescheduled. + * tests assert how many times a task was cancelled/rescheduled. {@link #pendingDelaysMillis()} + * lets tests assert the delay each pending task was scheduled with. + *

+ * Tasks may be scheduled from any thread. */ public final class ManualTaskExecutor implements TaskExecutor { private final List pending = new ArrayList<>(); @@ -21,17 +24,35 @@ public final class ManualTaskExecutor implements TaskExecutor { /** * @return the number of scheduled tasks that have been cancelled */ - public int cancelledCount() { + public synchronized int cancelledCount() { return cancelledCount; } + /** + * @return the delay each pending, non-cancelled task was scheduled with, in milliseconds, in + * the order the tasks were scheduled + */ + public synchronized List pendingDelaysMillis() { + List delays = new ArrayList<>(); + for (ManualScheduledFuture task : pending) { + if (!task.cancelled) { + delays.add(task.delayMillis); + } + } + return delays; + } + /** * Runs every pending, non-cancelled task that has been scheduled via - * {@link #scheduleTask(Runnable, long)} and clears the pending queue. + * {@link #scheduleTask(Runnable, long)} and clears the pending queue. A task scheduled while + * this runs is left pending for the next call. */ public void runPendingTasks() { - List toRun = new ArrayList<>(pending); - pending.clear(); + List toRun; + synchronized (this) { + toRun = new ArrayList<>(pending); + pending.clear(); + } for (ManualScheduledFuture task : toRun) { if (!task.cancelled) { task.action.run(); @@ -45,37 +66,41 @@ public void executeOnMainThread(Runnable action) { } @Override - public ScheduledFuture scheduleTask(Runnable action, long delayMillis) { - ManualScheduledFuture future = new ManualScheduledFuture(action); + public synchronized ScheduledFuture scheduleTask(Runnable action, long delayMillis) { + ManualScheduledFuture future = new ManualScheduledFuture(action, delayMillis); pending.add(future); return future; } @Override - public ScheduledFuture startRepeatingTask(Runnable action, long initialDelayMillis, long intervalMillis) { - ManualScheduledFuture future = new ManualScheduledFuture(action); + public synchronized ScheduledFuture startRepeatingTask(Runnable action, long initialDelayMillis, long intervalMillis) { + ManualScheduledFuture future = new ManualScheduledFuture(action, initialDelayMillis); pending.add(future); return future; } @Override - public void close() { + public synchronized void close() { pending.clear(); } private final class ManualScheduledFuture implements ScheduledFuture { private final Runnable action; - private boolean cancelled = false; + private final long delayMillis; + private volatile boolean cancelled = false; - ManualScheduledFuture(Runnable action) { + ManualScheduledFuture(Runnable action, long delayMillis) { this.action = action; + this.delayMillis = delayMillis; } @Override public boolean cancel(boolean mayInterruptIfRunning) { - if (!cancelled) { - cancelled = true; - cancelledCount++; + synchronized (ManualTaskExecutor.this) { + if (!cancelled) { + cancelled = true; + cancelledCount++; + } } return true; } @@ -102,7 +127,7 @@ public Object get(long timeout, TimeUnit unit) { @Override public long getDelay(TimeUnit unit) { - return 0; + return unit.convert(delayMillis, TimeUnit.MILLISECONDS); } @Override