diff --git a/app/common/src/main/java/stirling/software/common/cluster/RateLimitStore.java b/app/common/src/main/java/stirling/software/common/cluster/RateLimitStore.java
index 82351961a2..6c8d393b96 100644
--- a/app/common/src/main/java/stirling/software/common/cluster/RateLimitStore.java
+++ b/app/common/src/main/java/stirling/software/common/cluster/RateLimitStore.java
@@ -2,7 +2,15 @@ package stirling.software.common.cluster;
import java.time.Duration;
-/** Token-bucket rate limiting backed by the cluster backplane. */
+/**
+ * Token-bucket rate limiting backed by the cluster backplane.
+ *
+ *
In-process implementations enforce a per-JVM limit (identical to today's behaviour).
+ * Distributed implementations enforce a single global limit across every node.
+ *
+ *
Both implementations use a Bucket4j greedy-refill token bucket so semantics match across
+ * single-node and cluster deployments (no fixed-window boundary doubling).
+ */
public interface RateLimitStore {
/**
diff --git a/app/common/src/main/java/stirling/software/common/service/JobExecutorService.java b/app/common/src/main/java/stirling/software/common/service/JobExecutorService.java
index 4c51129328..0c4624ccf5 100644
--- a/app/common/src/main/java/stirling/software/common/service/JobExecutorService.java
+++ b/app/common/src/main/java/stirling/software/common/service/JobExecutorService.java
@@ -112,25 +112,12 @@ public class JobExecutorService {
log.debug("Generated jobId: {} (base: {})", scopedJobKey, baseJobId);
- // Store the scoped job ID in the request for potential use by other components
+ // Store the scoped job ID in the request for potential use by other components.
+ // Ownership lives in the scoped key itself (userId:jobId) plus the cluster-visible
+ // JobStore entry, so we no longer mirror it into the HTTP session - that did not
+ // survive a node hop in cluster mode.
if (request != null) {
request.setAttribute("jobId", scopedJobKey);
-
- // Also track this job ID in the user's session for authorization purposes
- // This ensures users can only cancel their own jobs
- if (request.getSession() != null) {
- @SuppressWarnings("unchecked")
- java.util.Set userJobIds =
- (java.util.Set) request.getSession().getAttribute("userJobIds");
-
- if (userJobIds == null) {
- userJobIds = new java.util.concurrent.ConcurrentSkipListSet<>();
- request.getSession().setAttribute("userJobIds", userJobIds);
- }
-
- userJobIds.add(scopedJobKey);
- log.debug("Added scoped job ID {} to user session", scopedJobKey);
- }
}
String jobId = scopedJobKey;
diff --git a/app/common/src/test/java/stirling/software/common/cluster/inprocess/InProcessDistributedLockTest.java b/app/common/src/test/java/stirling/software/common/cluster/inprocess/InProcessDistributedLockTest.java
index 30b738a3b9..2656c0d09c 100644
--- a/app/common/src/test/java/stirling/software/common/cluster/inprocess/InProcessDistributedLockTest.java
+++ b/app/common/src/test/java/stirling/software/common/cluster/inprocess/InProcessDistributedLockTest.java
@@ -24,7 +24,9 @@ class InProcessDistributedLockTest {
}
@Test
- void reentryFromSameThreadFails() {
+ void reentryFromSameThreadFails_parityWithValkey() {
+ // The Valkey impl refuses reentry (SET NX semantics); the in-process impl must match,
+ // otherwise code working in single-instance silently breaks in cluster mode.
DistributedLock lock = new InProcessDistributedLock();
DistributedLock.LockHandle h1 = lock.tryAcquire("k", Duration.ofSeconds(30)).orElseThrow();
Optional reentry = lock.tryAcquire("k", Duration.ofSeconds(30));
diff --git a/app/common/src/test/java/stirling/software/common/service/TaskManagerJobStoreDelegationTest.java b/app/common/src/test/java/stirling/software/common/service/TaskManagerJobStoreDelegationTest.java
index 1316c3c137..23165a14ae 100644
--- a/app/common/src/test/java/stirling/software/common/service/TaskManagerJobStoreDelegationTest.java
+++ b/app/common/src/test/java/stirling/software/common/service/TaskManagerJobStoreDelegationTest.java
@@ -86,6 +86,8 @@ class TaskManagerJobStoreDelegationTest {
@Override
public boolean shouldRunLocalCleanup() {
+ // Distributed backplanes own job TTL eviction themselves; this mock
+ // mirrors the real ValkeyClusterBackplane override of the default true.
return false;
}
};
diff --git a/app/core/src/main/java/stirling/software/common/controller/JobController.java b/app/core/src/main/java/stirling/software/common/controller/JobController.java
index 7bd81741ed..407d59d97e 100644
--- a/app/core/src/main/java/stirling/software/common/controller/JobController.java
+++ b/app/core/src/main/java/stirling/software/common/controller/JobController.java
@@ -4,6 +4,7 @@ import java.net.URLEncoder;
import java.nio.charset.StandardCharsets;
import java.util.List;
import java.util.Map;
+import java.util.Optional;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.http.MediaType;
@@ -22,6 +23,10 @@ import jakarta.servlet.http.HttpServletRequest;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
+import stirling.software.common.cluster.ClusterBackplane;
+import stirling.software.common.cluster.JobStore;
+import stirling.software.common.cluster.JobStoreEntry;
+import stirling.software.common.cluster.StickyMissRecorder;
import stirling.software.common.model.job.JobResult;
import stirling.software.common.model.job.ResultFile;
import stirling.software.common.service.FileStorage;
@@ -42,10 +47,22 @@ public class JobController {
private final FileStorage fileStorage;
private final JobQueue jobQueue;
private final HttpServletRequest request;
+ private final ClusterBackplane clusterBackplane;
+ private final JobStore jobStore;
+
+ /**
+ * Process-local short-TTL cache fronting {@link JobStore#get(String)} on the sticky-410 path.
+ * Without this every result download / status poll fires a Valkey HGETALL which doubles RTT on
+ * the hot path when the same client re-requests the same job within seconds.
+ */
+ private final JobOwnershipCache ownershipCache = new JobOwnershipCache();
@Autowired(required = false)
private JobOwnershipService jobOwnershipService;
+ @Autowired(required = false)
+ private StickyMissRecorder stickyMissRecorder;
+
/**
* Get the status of a job
*
@@ -55,6 +72,14 @@ public class JobController {
@GetMapping("/job/{jobId}")
@Operation(summary = "Get job status")
public ResponseEntity> getJobStatus(@PathVariable("jobId") String jobId) {
+ // Sticky-410 must precede user-auth (403): a non-owner node has no way to verify
+ // ownership for a job it doesn't own, and a 403 here would leak job existence to
+ // unauthorized callers. Return 410 first so the LB re-routes to the owner.
+ Optional> peerOwned = guardNonOwner(jobId);
+ if (peerOwned.isPresent()) {
+ return peerOwned.get();
+ }
+
// Validate job ownership
if (!validateJobAccess(jobId)) {
log.warn("Unauthorized attempt to access job status: {}", jobId);
@@ -91,6 +116,14 @@ public class JobController {
@GetMapping("/job/{jobId}/result")
@Operation(summary = "Get job result")
public ResponseEntity> getJobResult(@PathVariable("jobId") String jobId) {
+ // Sticky-410 must precede user-auth (403): a non-owner node has no way to verify
+ // ownership for a job it doesn't own, and a 403 here would leak job existence to
+ // unauthorized callers. Return 410 first so the LB re-routes to the owner.
+ Optional> peerOwned = guardNonOwner(jobId);
+ if (peerOwned.isPresent()) {
+ return peerOwned.get();
+ }
+
// Validate job ownership
if (!validateJobAccess(jobId)) {
log.warn("Unauthorized attempt to access job result: {}", jobId);
@@ -125,11 +158,14 @@ public class JobController {
result.getAllResultFiles()));
}
- // Handle single file (download directly)
+ // Handle single file (download directly). Cross-node ownership was already resolved
+ // at the top of this method, so reaching here means we ARE the owner (or single-node)
+ // and the bytes live on our local disk.
if (result.hasFiles() && !result.hasMultipleFiles()) {
try {
List files = result.getAllResultFiles();
ResultFile singleFile = files.get(0);
+
byte[] fileContent = fileStorage.retrieveBytes(singleFile.getFileId());
return ResponseEntity.ok()
.header("Content-Type", singleFile.getContentType())
@@ -163,6 +199,15 @@ public class JobController {
public ResponseEntity> cancelJob(@PathVariable("jobId") String jobId) {
log.debug("Request to cancel job: {}", jobId);
+ // Sticky-410 must precede user-auth (403): a non-owner node has no way to verify
+ // ownership for a job it doesn't own, and a 403 here would leak job existence to
+ // unauthorized callers. Return 410 first so the LB re-routes to the owner who can
+ // actually cancel.
+ Optional> peerOwned = guardNonOwner(jobId);
+ if (peerOwned.isPresent()) {
+ return peerOwned.get();
+ }
+
// Validate job ownership
if (!validateJobAccess(jobId)) {
log.warn("Unauthorized attempt to cancel job: {}", jobId);
@@ -201,7 +246,9 @@ public class JobController {
"queuePosition",
queuePosition >= 0 ? queuePosition : "n/a"));
} else {
- // Job not found or already complete
+ // Job not found or already complete. Cross-node ownership was already resolved at
+ // the top of this method (sticky-410 precedes user-auth), so any peer-owned case
+ // has been returned already; reaching here means we ARE the owner (or single-node).
JobResult result = taskManager.getJobResult(jobId);
if (result == null) {
return ResponseEntity.notFound().build();
@@ -224,6 +271,14 @@ public class JobController {
@GetMapping("/job/{jobId}/result/files")
@Operation(summary = "Get job result files")
public ResponseEntity> getJobFiles(@PathVariable("jobId") String jobId) {
+ // Sticky-410 must precede user-auth (403): a non-owner node has no way to verify
+ // ownership for a job it doesn't own, and a 403 here would leak job existence to
+ // unauthorized callers. Return 410 first so the LB re-routes to the owner.
+ Optional> peerOwned = guardNonOwner(jobId);
+ if (peerOwned.isPresent()) {
+ return peerOwned.get();
+ }
+
// Validate job ownership
if (!validateJobAccess(jobId)) {
log.warn("Unauthorized attempt to access job files: {}", jobId);
@@ -267,6 +322,14 @@ public class JobController {
return ResponseEntity.notFound().build();
}
+ // Sticky-410 must precede user-auth (403): a non-owner node has no way to verify
+ // ownership for a job it doesn't own, and a 403 here would leak file existence to
+ // unauthorized callers. Return 410 first so the LB re-routes to the owner.
+ Optional> notOwner = guardNonOwner(jobKey);
+ if (notOwner.isPresent()) {
+ return notOwner.get();
+ }
+
if (!validateJobAccess(jobKey)) {
log.warn("Unauthorized attempt to access file metadata: {}", fileId);
return ResponseEntity.status(403)
@@ -323,15 +386,21 @@ public class JobController {
return ResponseEntity.notFound().build();
}
+ // Sticky-410 must precede the user-auth (403) check: a non-owner node has no way to
+ // verify ownership for a job it doesn't own, and a 403 here would leak file existence
+ // to unauthorized callers. Return 410 first so the LB re-routes to the owner where
+ // the real auth check can run.
+ Optional> notOwner = guardNonOwner(jobKey);
+ if (notOwner.isPresent()) {
+ return notOwner.get();
+ }
+
if (!validateJobAccess(jobKey)) {
log.warn("Unauthorized attempt to download file: {}", fileId);
return ResponseEntity.status(403)
.body(Map.of("message", "You are not authorized to access this file"));
}
- // Retrieve file content
- byte[] fileContent = fileStorage.retrieveBytes(fileId);
-
// Find the file metadata from any job that contains this file
// This is for getting the original filename and content type
ResultFile resultFile = taskManager.findResultFileByFileId(fileId);
@@ -342,6 +411,9 @@ public class JobController {
? resultFile.getContentType()
: MediaType.APPLICATION_OCTET_STREAM_VALUE;
+ // Retrieve file content from local disk
+ byte[] fileContent = fileStorage.retrieveBytes(fileId);
+
return ResponseEntity.ok()
.header("Content-Type", contentType)
.header("Content-Disposition", createContentDispositionHeader(fileName))
@@ -356,6 +428,76 @@ public class JobController {
return jobOwnershipService != null;
}
+ /**
+ * Returns {@code 410 Gone} with {@code {message, ownedBy, currentNode}} and {@code Retry-After:
+ * 0} when the job is owned by a peer node. Returns {@link Optional#empty()} when we are the
+ * owner, when cluster mode is off / JobStore has no entry, or when {@code owningNodeId} is
+ * blank (caller proceeds with its normal not-found / 200 path).
+ *
+ *
Wraps the {@link JobStore#get(String)} call in a short-TTL local cache and a defensive
+ * try/catch so that Valkey RTT cost is not multiplied by every download retry and so that a
+ * Valkey timeout falls through to the local-disk path instead of surfacing as 500.
+ */
+ private Optional> guardNonOwner(String jobId) {
+ if (clusterBackplane == null || jobStore == null) {
+ return Optional.empty();
+ }
+ Optional entry;
+ Optional> cached = ownershipCache.get(jobId);
+ if (cached.isPresent()) {
+ entry = cached.get();
+ } else {
+ try {
+ entry = jobStore.get(jobId);
+ } catch (RuntimeException ex) {
+ // Valkey unavailable / timeout: treat as "no cluster-visible entry" so the request
+ // can proceed to the local-disk path. Surfacing a 500 here would break every
+ // download attempt during a brief Valkey blip; the worst case if we miss a real
+ // peer-owned entry is one wasted round trip + a 404 from the local node.
+ log.warn(
+ "JobStore lookup failed for jobId={} - treating as not-found and falling"
+ + " through to local path: {}",
+ jobId,
+ ex.getMessage());
+ return Optional.empty();
+ }
+ ownershipCache.put(jobId, entry);
+ }
+ if (entry.isEmpty()) {
+ return Optional.empty();
+ }
+ String owner = entry.get().owningNodeId();
+ if (owner == null || owner.isBlank()) {
+ return Optional.empty();
+ }
+ String localId = clusterBackplane.localNodeId();
+ if (owner.equals(localId)) {
+ return Optional.empty();
+ }
+ log.info(
+ "Sticky-session miss for jobId={} (owner={}, local={}); returning 410 so client"
+ + " retries via LB affinity",
+ jobId,
+ owner,
+ localId);
+ if (stickyMissRecorder != null) {
+ stickyMissRecorder.recordStickyMiss();
+ }
+ return Optional.of(
+ ResponseEntity.status(410)
+ .header("Retry-After", "0")
+ .body(
+ Map.of(
+ "message",
+ "Result lives on another node. Retry to be routed there"
+ + " by the load balancer's sticky-session"
+ + " affinity, or re-run the job.",
+ "ownedBy",
+ owner,
+ "currentNode",
+ localId == null ? "" : localId)));
+ }
+
/**
* Create Content-Disposition header with UTF-8 filename support
*
diff --git a/app/core/src/main/java/stirling/software/common/controller/JobOwnershipCache.java b/app/core/src/main/java/stirling/software/common/controller/JobOwnershipCache.java
new file mode 100644
index 0000000000..d4e6444c21
--- /dev/null
+++ b/app/core/src/main/java/stirling/software/common/controller/JobOwnershipCache.java
@@ -0,0 +1,51 @@
+package stirling.software.common.controller;
+
+import java.util.Optional;
+import java.util.concurrent.ConcurrentHashMap;
+import java.util.concurrent.ConcurrentMap;
+
+import stirling.software.common.cluster.JobStoreEntry;
+
+/**
+ * Process-local TTL cache for {@link JobStoreEntry} lookups to suppress redundant Valkey HGETALL
+ * round-trips on the hot result-download path (sticky-410 ownership check).
+ *
+ *
5 second TTL is short enough that a job's lifecycle transitions (RUNNING -> COMPLETE -> TTL
+ * expiry) propagate to all nodes within the LB's sticky-session window, and short enough that a
+ * mistakenly-cached "not found" recovers quickly when an entry actually shows up. Cap the map at
+ * 2048 entries to bound memory; eviction is best-effort (clear-and-restart) since the cache is
+ * advisory.
+ */
+final class JobOwnershipCache {
+
+ private static final long TTL_NANOS = 5L * 1_000_000_000L; // 5 s
+ private static final int MAX_ENTRIES = 2048;
+
+ private final ConcurrentMap entries = new ConcurrentHashMap<>();
+
+ Optional> get(String jobId) {
+ Entry e = entries.get(jobId);
+ if (e == null) {
+ return Optional.empty();
+ }
+ if (System.nanoTime() - e.storedAtNanos > TTL_NANOS) {
+ entries.remove(jobId, e);
+ return Optional.empty();
+ }
+ return Optional.of(e.value);
+ }
+
+ void put(String jobId, Optional value) {
+ if (entries.size() >= MAX_ENTRIES) {
+ // Best-effort eviction; under burst the cache simply rebuilds.
+ entries.clear();
+ }
+ entries.put(jobId, new Entry(value, System.nanoTime()));
+ }
+
+ void invalidate(String jobId) {
+ entries.remove(jobId);
+ }
+
+ private record Entry(Optional value, long storedAtNanos) {}
+}
diff --git a/app/core/src/test/java/stirling/software/common/controller/JobControllerOwnershipTest.java b/app/core/src/test/java/stirling/software/common/controller/JobControllerOwnershipTest.java
new file mode 100644
index 0000000000..b3acb5ebc6
--- /dev/null
+++ b/app/core/src/test/java/stirling/software/common/controller/JobControllerOwnershipTest.java
@@ -0,0 +1,502 @@
+package stirling.software.common.controller;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertInstanceOf;
+import static org.junit.jupiter.api.Assertions.assertNotNull;
+import static org.junit.jupiter.api.Assertions.assertNull;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+import static org.mockito.Mockito.lenient;
+import static org.mockito.Mockito.mock;
+import static org.mockito.Mockito.never;
+import static org.mockito.Mockito.times;
+import static org.mockito.Mockito.verify;
+import static org.mockito.Mockito.when;
+
+import java.time.Instant;
+import java.util.List;
+import java.util.Map;
+import java.util.Optional;
+import java.util.stream.Stream;
+
+import org.junit.jupiter.api.BeforeEach;
+import org.junit.jupiter.api.DisplayName;
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.params.ParameterizedTest;
+import org.junit.jupiter.params.provider.Arguments;
+import org.junit.jupiter.params.provider.MethodSource;
+import org.springframework.http.HttpStatus;
+import org.springframework.http.ResponseEntity;
+import org.springframework.test.util.ReflectionTestUtils;
+
+import jakarta.servlet.http.HttpServletRequest;
+
+import stirling.software.common.cluster.ClusterBackplane;
+import stirling.software.common.cluster.JobStore;
+import stirling.software.common.cluster.JobStoreEntry;
+import stirling.software.common.cluster.StickyMissRecorder;
+import stirling.software.common.model.job.JobResult;
+import stirling.software.common.service.FileStorage;
+import stirling.software.common.service.JobOwnershipService;
+import stirling.software.common.service.JobQueue;
+import stirling.software.common.service.TaskManager;
+
+/**
+ * Sticky-session ownership behavior for {@link JobController}.
+ *
+ *
Result PDFs live on the local disk of whichever node ran the job. When the load balancer's
+ * cookie/IP affinity *misses* and routes a download to a non-owner node, the controller must return
+ * {@code 410 Gone} with a structured payload that tells the client to retry (the LB will usually
+ * route them to the owner on the second attempt).
+ *
+ *
Contract verified here:
+ *
+ *
+ *
Owner == this node → file is read from local disk (normal 200).
+ *
Owner == another node → {@code 410 Gone} with {@code ownedBy} + {@code currentNode} fields,
+ * and {@code FileStorage} is never touched.
+ *
JobStore has no entry → behave as single-instance (no 410).
+ *
{@code owningNodeId} blank → behave as single-instance (no 410).
+ *
Single-instance install (no JobStore / no ClusterBackplane) → no 410, no NPE.
+ *
+ *
+ *
Manual mock construction (no {@code MockitoExtension}) so each test can wire its own
+ * controller with a different {@code ClusterBackplane} / {@code JobStore} combo without setUp stubs
+ * leaking across cases.
+ */
+class JobControllerOwnershipTest {
+
+ private TaskManager taskManager;
+ private FileStorage fileStorage;
+ private JobQueue jobQueue;
+ private HttpServletRequest request;
+ private JobOwnershipService jobOwnershipService;
+ private ClusterBackplane clusterBackplane;
+ private JobStore jobStore;
+ private StickyMissRecorder stickyMissRecorder;
+
+ private static final String JOB_ID = "job-42";
+ private static final String FILE_ID = "file-abc";
+ private static final String LOCAL_NODE = "node-self";
+ private static final String PEER_NODE = "node-peer";
+
+ @BeforeEach
+ void setUp() {
+ taskManager = mock(TaskManager.class);
+ fileStorage = mock(FileStorage.class);
+ jobQueue = mock(JobQueue.class);
+ request = mock(HttpServletRequest.class);
+ jobOwnershipService = mock(JobOwnershipService.class);
+ clusterBackplane = mock(ClusterBackplane.class);
+ jobStore = mock(JobStore.class);
+ stickyMissRecorder = mock(StickyMissRecorder.class);
+ }
+
+ private JobController makeController(ClusterBackplane backplane, JobStore store) {
+ JobController c =
+ new JobController(taskManager, fileStorage, jobQueue, request, backplane, store);
+ // jobOwnershipService is @Autowired(required=false) - field-injected. When non-null,
+ // validateJobAccess delegates to it. We leave it null by default so the security
+ // check is a no-op (backwards compat path) and the test focuses on sticky-410.
+ // stickyMissRecorder is also field-injected; wire by default so the metric assertions
+ // work without per-test setup.
+ ReflectionTestUtils.setField(c, "stickyMissRecorder", stickyMissRecorder);
+ return c;
+ }
+
+ private JobController makeController() {
+ return makeController(clusterBackplane, jobStore);
+ }
+
+ private JobStoreEntry entryOwnedBy(String ownerNodeId) {
+ return new JobStoreEntry(
+ JOB_ID,
+ JobStoreEntry.JobState.COMPLETE,
+ ownerNodeId,
+ Instant.now(),
+ Instant.now(),
+ null,
+ List.of(FILE_ID),
+ Map.of());
+ }
+
+ private JobResult completedJobWithFile() {
+ JobResult result = new JobResult();
+ result.setJobId(JOB_ID);
+ // completeWithSingleFile populates the resultFiles list, sets complete=true,
+ // and sets completedAt - all required for the getJobResult single-file branch.
+ result.completeWithSingleFile(FILE_ID, "out.pdf", "application/pdf", 7L);
+ return result;
+ }
+
+ /**
+ * Full sticky-410 contract for {@code downloadFile} when the requested job is owned by a peer.
+ * Asserts everything in one place (status, Retry-After header, payload shape, no
+ * implementation-detail leak, metric incremented, storage never touched).
+ *
+ *
Other tests cover the edge cases independently (locally-owned, no JobStore entry, blank
+ * owner, etc.) so a single failure here points at exactly one missing or broken contract
+ * property.
+ */
+ @Test
+ @DisplayName(
+ "downloadFile peer-owned → full sticky-410 contract"
+ + " (status + Retry-After + payload + metric + storage untouched)")
+ void downloadFile_peerOwned_fullStickyContract() throws Exception {
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(PEER_NODE)));
+
+ ResponseEntity> response = makeController().downloadFile(FILE_ID);
+
+ // 1. Status + Retry-After header (immediate-retry hint).
+ assertEquals(HttpStatus.GONE, response.getStatusCode());
+ assertEquals("0", response.getHeaders().getFirst("Retry-After"));
+
+ // 2. Payload shape: exactly { message, ownedBy, currentNode }, with no leaked secrets.
+ assertInstanceOf(Map.class, response.getBody());
+ Map, ?> body = (Map, ?>) response.getBody();
+ assertEquals(3, body.size(), "exactly: message, ownedBy, currentNode");
+ assertEquals(PEER_NODE, body.get("ownedBy"));
+ assertEquals(LOCAL_NODE, body.get("currentNode"));
+ assertNotNull(body.get("message"));
+ assertTrue(((String) body.get("message")).toLowerCase().contains("retry"));
+ assertNull(body.get("internalSecret"));
+ assertNull(body.get("filePath"));
+
+ // 3. Operator-alert metric incremented exactly once.
+ verify(stickyMissRecorder).recordStickyMiss();
+
+ // 4. Storage layer NEVER touched - bytes don't live here, so reading would be wrong.
+ verify(fileStorage, never()).retrieveBytes(FILE_ID);
+ }
+
+ // --------------------------------------------------------------------------------------
+ // Happy-path ownership matrix: any non-peer signal (local owner, no entry, blank owner)
+ // must produce a 200 from FileStorage with NO sticky-miss metric increment.
+ // --------------------------------------------------------------------------------------
+
+ private static Stream downloadHappyPathScenarios() {
+ return Stream.of(
+ Arguments.of("locallyOwned", LOCAL_NODE, true),
+ Arguments.of("noJobStoreEntry", null, false),
+ Arguments.of("blankOwningNodeId", "", true));
+ }
+
+ @ParameterizedTest(name = "downloadFile {0} -> 200, no sticky-miss")
+ @MethodSource("downloadHappyPathScenarios")
+ void downloadFile_happyPath_returnsOkAndNoMetric(
+ String scenario, String ownerNodeId, boolean entryPresent) throws Exception {
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ when(jobStore.get(JOB_ID))
+ .thenReturn(
+ entryPresent ? Optional.of(entryOwnedBy(ownerNodeId)) : Optional.empty());
+ when(fileStorage.retrieveBytes(FILE_ID)).thenReturn("payload".getBytes());
+
+ ResponseEntity> response = makeController().downloadFile(FILE_ID);
+
+ assertEquals(HttpStatus.OK, response.getStatusCode(), scenario);
+ verify(fileStorage).retrieveBytes(FILE_ID);
+ verify(stickyMissRecorder, never()).recordStickyMiss();
+ }
+
+ @Test
+ @DisplayName("getJobResult: locally-owned single-file result → reads from FileStorage, 200 OK")
+ void getJobResult_singleFile_locallyOwned_readsFromStorage() throws Exception {
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(taskManager.getJobResult(JOB_ID)).thenReturn(completedJobWithFile());
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(LOCAL_NODE)));
+ when(fileStorage.retrieveBytes(FILE_ID)).thenReturn("payload".getBytes());
+
+ ResponseEntity> response = makeController().getJobResult(JOB_ID);
+
+ assertEquals(HttpStatus.OK, response.getStatusCode());
+ }
+
+ // --------------------------------------------------------------------------------------
+ // Peer-owned 410 matrix across endpoints. Cross-cutting contract:
+ // - status = 410, Retry-After = 0
+ // - body.ownedBy = peer node, body.currentNode = local node
+ // - sticky-miss metric incremented exactly once per request
+ // The contract test above covers ALL of these properties for downloadFile in detail; this
+ // parameterized matrix asserts the same status + ownedBy + metric signals on every endpoint.
+ // --------------------------------------------------------------------------------------
+
+ private enum Endpoint {
+ DOWNLOAD_FILE,
+ GET_JOB_RESULT,
+ GET_JOB_STATUS,
+ CANCEL_JOB
+ }
+
+ private static Stream peerOwned410Scenarios() {
+ return Stream.of(
+ Arguments.of(Endpoint.DOWNLOAD_FILE),
+ Arguments.of(Endpoint.GET_JOB_RESULT),
+ Arguments.of(Endpoint.GET_JOB_STATUS),
+ Arguments.of(Endpoint.CANCEL_JOB));
+ }
+
+ @ParameterizedTest(name = "{0} peer-owned -> 410, ownedBy=peer, metric++")
+ @MethodSource("peerOwned410Scenarios")
+ void endpoint_peerOwned_returns410(Endpoint endpoint) throws Exception {
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(PEER_NODE)));
+ // Endpoint-specific wiring: the per-endpoint code path needs different mock setup
+ // before it reaches the sticky-410 guard.
+ switch (endpoint) {
+ case DOWNLOAD_FILE -> when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ case GET_JOB_RESULT ->
+ when(taskManager.getJobResult(JOB_ID)).thenReturn(completedJobWithFile());
+ case GET_JOB_STATUS -> when(taskManager.getJobResult(JOB_ID)).thenReturn(null);
+ case CANCEL_JOB -> {
+ when(jobQueue.isJobQueued(JOB_ID)).thenReturn(false);
+ when(taskManager.getJobResult(JOB_ID)).thenReturn(null);
+ }
+ }
+
+ ResponseEntity> response =
+ switch (endpoint) {
+ case DOWNLOAD_FILE -> makeController().downloadFile(FILE_ID);
+ case GET_JOB_RESULT -> makeController().getJobResult(JOB_ID);
+ case GET_JOB_STATUS -> makeController().getJobStatus(JOB_ID);
+ case CANCEL_JOB -> makeController().cancelJob(JOB_ID);
+ };
+
+ assertEquals(HttpStatus.GONE, response.getStatusCode());
+ Map, ?> body = (Map, ?>) response.getBody();
+ assertEquals(PEER_NODE, body.get("ownedBy"));
+ assertEquals(LOCAL_NODE, body.get("currentNode"));
+ verify(stickyMissRecorder).recordStickyMiss();
+ // Storage / mutation must never be touched by a peer-routed request.
+ verify(fileStorage, never()).retrieveBytes(FILE_ID);
+ if (endpoint == Endpoint.CANCEL_JOB) {
+ verify(taskManager, never()).setError(JOB_ID, "Job was cancelled by user");
+ }
+ }
+
+ // --------------------------------------------------------------------------------------
+ // Unknown-job 404 matrix: when neither TaskManager nor JobStore knows the jobId, the
+ // controller must return 404 (not 410) and must NOT count it as a sticky miss.
+ // --------------------------------------------------------------------------------------
+
+ private static Stream unknownJob404Scenarios() {
+ return Stream.of(Arguments.of(Endpoint.GET_JOB_STATUS), Arguments.of(Endpoint.CANCEL_JOB));
+ }
+
+ @ParameterizedTest(name = "{0} unknown jobId -> 404 (not 410), no metric")
+ @MethodSource("unknownJob404Scenarios")
+ void endpoint_unknownJob_returns404(Endpoint endpoint) {
+ when(taskManager.getJobResult(JOB_ID)).thenReturn(null);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.empty());
+ if (endpoint == Endpoint.CANCEL_JOB) {
+ when(jobQueue.isJobQueued(JOB_ID)).thenReturn(false);
+ }
+
+ ResponseEntity> response =
+ switch (endpoint) {
+ case GET_JOB_STATUS -> makeController().getJobStatus(JOB_ID);
+ case CANCEL_JOB -> makeController().cancelJob(JOB_ID);
+ default -> throw new IllegalArgumentException(endpoint.name());
+ };
+
+ assertEquals(HttpStatus.NOT_FOUND, response.getStatusCode());
+ verify(stickyMissRecorder, never()).recordStickyMiss();
+ }
+
+ // --------------------------------------------------------------------------------------
+ // Single-instance / null-bean wiring: no NPE, no 410, no metric. These test SPECIFIC
+ // wiring permutations and so stay as discrete tests rather than rows.
+ // --------------------------------------------------------------------------------------
+
+ @Test
+ @DisplayName("Single-instance install (no ClusterBackplane bean): no 410, no NPE")
+ void singleInstance_noClusterBackplane_noGoneResponse() throws Exception {
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ when(fileStorage.retrieveBytes(FILE_ID)).thenReturn("payload".getBytes());
+
+ ResponseEntity> response = makeController(null, jobStore).downloadFile(FILE_ID);
+
+ assertEquals(HttpStatus.OK, response.getStatusCode());
+ verify(fileStorage).retrieveBytes(FILE_ID);
+ }
+
+ @Test
+ @DisplayName("Single-instance install (no JobStore bean): no 410, no NPE")
+ void singleInstance_noJobStore_noGoneResponse() throws Exception {
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ when(fileStorage.retrieveBytes(FILE_ID)).thenReturn("payload".getBytes());
+
+ ResponseEntity> response = makeController(clusterBackplane, null).downloadFile(FILE_ID);
+
+ assertEquals(HttpStatus.OK, response.getStatusCode());
+ verify(fileStorage).retrieveBytes(FILE_ID);
+ }
+
+ @Test
+ @DisplayName("Single-instance (no StickyMissRecorder bean) → no NPE, still 200 OK")
+ void noStickyMissRecorder_works() throws Exception {
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(LOCAL_NODE)));
+ when(fileStorage.retrieveBytes(FILE_ID)).thenReturn("payload".getBytes());
+
+ JobController c = makeController();
+ ReflectionTestUtils.setField(c, "stickyMissRecorder", null);
+
+ ResponseEntity> response = c.downloadFile(FILE_ID);
+
+ assertEquals(HttpStatus.OK, response.getStatusCode());
+ }
+
+ @Test
+ @DisplayName(
+ "cluster-mode but localNodeId is null → no NPE; 410 because owner is set and"
+ + " differs from blank")
+ void clusterBackplanePresent_butLocalNodeIdNull_falsBackGracefully() throws Exception {
+ when(clusterBackplane.localNodeId()).thenReturn(null);
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(PEER_NODE)));
+
+ // We still 410: owner is "node-peer", local is null → they don't match. Rather than
+ // silently 200-from-wrong-disk (which would serve garbage), we surface the mismatch.
+ ResponseEntity> response = makeController().downloadFile(FILE_ID);
+
+ assertEquals(HttpStatus.GONE, response.getStatusCode());
+ Map, ?> body = (Map, ?>) response.getBody();
+ assertEquals("", body.get("currentNode"), "blank when localNodeId is null");
+ assertEquals(PEER_NODE, body.get("ownedBy"));
+ }
+
+ // --------------------------------------------------------------------------------------
+ // Ownership-service interaction: sticky-410 must take precedence over per-user auth so
+ // we never 403 on a peer-owned resource (which would leak existence + defeat the redirect).
+ // --------------------------------------------------------------------------------------
+
+ @Test
+ @DisplayName("Owner returns 410 even when JobOwnershipService allows access (orthogonal)")
+ void ownershipService_passes_butStickyStillReturns410() throws Exception {
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(PEER_NODE)));
+ lenient().when(jobOwnershipService.validateJobAccess(JOB_ID)).thenReturn(true);
+
+ JobController c = makeController();
+ ReflectionTestUtils.setField(c, "jobOwnershipService", jobOwnershipService);
+ ResponseEntity> response = c.downloadFile(FILE_ID);
+
+ // OwnershipService is about *user* auth; sticky-410 is about *node* topology.
+ // Both must pass for a 200, and node-ownership is checked first so a non-owner
+ // never even runs the user-auth check.
+ assertEquals(HttpStatus.GONE, response.getStatusCode());
+ }
+
+ @Test
+ @DisplayName(
+ "downloadFile: peer-owned + ownership-denied → 410 (NOT 403) so we don't leak"
+ + " file existence")
+ void downloadFile_peerOwned_ownershipDenied_returns410NotForbidden() throws Exception {
+ // Guard ordering: sticky-410 must run before user-auth. If user-auth ran first, a
+ // peer-owned-file request on the wrong node would fail user-auth (this node cannot
+ // verify access to a job it does not own) and return 403, which leaks file existence
+ // AND defeats the sticky-410 design (the frontend can't retry-with-affinity off a 403).
+ // Guard first so the user is redirected to the owner where the real auth check happens.
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(PEER_NODE)));
+ lenient().when(jobOwnershipService.validateJobAccess(JOB_ID)).thenReturn(false);
+
+ JobController c = makeController();
+ ReflectionTestUtils.setField(c, "jobOwnershipService", jobOwnershipService);
+ ResponseEntity> response = c.downloadFile(FILE_ID);
+
+ assertEquals(HttpStatus.GONE, response.getStatusCode());
+ Map, ?> body = (Map, ?>) response.getBody();
+ assertEquals(PEER_NODE, body.get("ownedBy"));
+ // Crucially: never reached fileStorage, never returned 403.
+ verify(fileStorage, never()).retrieveBytes(FILE_ID);
+ }
+
+ @Test
+ @DisplayName(
+ "getJobStatus: peer-owned + ownership-denied → 410 (NOT 403) so we don't leak"
+ + " job existence")
+ void getJobStatus_peerOwned_ownershipDenied_returns410NotForbidden() {
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(PEER_NODE)));
+ lenient().when(jobOwnershipService.validateJobAccess(JOB_ID)).thenReturn(false);
+
+ JobController c = makeController();
+ ReflectionTestUtils.setField(c, "jobOwnershipService", jobOwnershipService);
+ ResponseEntity> response = c.getJobStatus(JOB_ID);
+
+ assertEquals(HttpStatus.GONE, response.getStatusCode());
+ Map, ?> body = (Map, ?>) response.getBody();
+ assertEquals(PEER_NODE, body.get("ownedBy"));
+ }
+
+ @Test
+ @DisplayName(
+ "cancelJob: peer-owned + ownership-denied → 410 (NOT 403) so we don't leak job"
+ + " existence")
+ void cancelJob_peerOwned_ownershipDenied_returns410NotForbidden() {
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(jobQueue.isJobQueued(JOB_ID)).thenReturn(false);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(PEER_NODE)));
+ lenient().when(jobOwnershipService.validateJobAccess(JOB_ID)).thenReturn(false);
+
+ JobController c = makeController();
+ ReflectionTestUtils.setField(c, "jobOwnershipService", jobOwnershipService);
+ ResponseEntity> response = c.cancelJob(JOB_ID);
+
+ assertEquals(HttpStatus.GONE, response.getStatusCode());
+ Map, ?> body = (Map, ?>) response.getBody();
+ assertEquals(PEER_NODE, body.get("ownedBy"));
+ verify(taskManager, never()).setError(JOB_ID, "Job was cancelled by user");
+ }
+
+ // --------------------------------------------------------------------------------------
+ // S2: JobStore lookup hardening (local cache + Valkey-fault graceful-degrade)
+ // --------------------------------------------------------------------------------------
+
+ @Test
+ @DisplayName(
+ "guardNonOwner caches JobStore.get within TTL window: second call same jobId hits"
+ + " cache, not Valkey")
+ void guardNonOwner_cachesJobStoreLookupWithinTtl() throws Exception {
+ when(clusterBackplane.localNodeId()).thenReturn(LOCAL_NODE);
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ when(jobStore.get(JOB_ID)).thenReturn(Optional.of(entryOwnedBy(LOCAL_NODE)));
+ when(fileStorage.retrieveBytes(FILE_ID)).thenReturn("payload".getBytes());
+
+ JobController c = makeController();
+ c.downloadFile(FILE_ID);
+ c.downloadFile(FILE_ID);
+ c.downloadFile(FILE_ID);
+
+ // Three downloads of the same fileId == one HGETALL. Without the cache this would have
+ // been three Valkey round-trips on the hot download path.
+ verify(jobStore, times(1)).get(JOB_ID);
+ }
+
+ @Test
+ @DisplayName(
+ "guardNonOwner: JobStore.get throws (Valkey timeout) → falls through to local-disk"
+ + " path, no 500 leaks to caller")
+ void guardNonOwner_jobStoreException_fallsThroughToLocalPath() throws Exception {
+ when(taskManager.findJobKeyByFileId(FILE_ID)).thenReturn(JOB_ID);
+ // Simulate a Valkey timeout. spring-data-redis surfaces these as runtime exceptions
+ // wrapping Lettuce errors; the controller must not let any RuntimeException leak out.
+ when(jobStore.get(JOB_ID)).thenThrow(new RuntimeException("Valkey command timeout"));
+ when(fileStorage.retrieveBytes(FILE_ID)).thenReturn("payload".getBytes());
+
+ ResponseEntity> response = makeController().downloadFile(FILE_ID);
+
+ // The download must succeed via the local-disk path. A brief Valkey blip cannot break
+ // every download attempt with 500 - we cleanly degrade to single-node behavior.
+ assertEquals(HttpStatus.OK, response.getStatusCode());
+ verify(fileStorage).retrieveBytes(FILE_ID);
+ // The exception did NOT count as a sticky-miss (it wasn't one; we just couldn't see).
+ verify(stickyMissRecorder, never()).recordStickyMiss();
+ }
+}
diff --git a/app/core/src/test/java/stirling/software/common/controller/JobControllerTest.java b/app/core/src/test/java/stirling/software/common/controller/JobControllerTest.java
index 1388abeae6..53212a2e65 100644
--- a/app/core/src/test/java/stirling/software/common/controller/JobControllerTest.java
+++ b/app/core/src/test/java/stirling/software/common/controller/JobControllerTest.java
@@ -19,6 +19,8 @@ import org.springframework.test.util.ReflectionTestUtils;
import jakarta.servlet.http.HttpServletRequest;
+import stirling.software.common.cluster.ClusterBackplane;
+import stirling.software.common.cluster.JobStore;
import stirling.software.common.model.job.JobResult;
import stirling.software.common.service.FileStorage;
import stirling.software.common.service.JobOwnershipService;
@@ -37,6 +39,10 @@ class JobControllerTest {
@Mock private JobOwnershipService jobOwnershipService;
+ @Mock private ClusterBackplane clusterBackplane;
+
+ @Mock private JobStore jobStore;
+
private MockHttpSession session;
@InjectMocks private JobController controller;
diff --git a/app/proprietary/build.gradle b/app/proprietary/build.gradle
index e923ae6229..0b4523d0e3 100644
--- a/app/proprietary/build.gradle
+++ b/app/proprietary/build.gradle
@@ -53,8 +53,13 @@ dependencies {
api 'org.springframework.boot:spring-boot-starter-mail'
api 'org.springframework.boot:spring-boot-starter-cache'
api 'com.github.ben-manes.caffeine:caffeine'
+ implementation 'org.springframework.boot:spring-boot-starter-data-redis'
api 'io.swagger.core.v3:swagger-core-jakarta:2.2.46'
- implementation 'com.bucket4j:bucket4j_jdk17-core:8.18.0'
+ implementation 'com.bucket4j:bucket4j_jdk17-core:8.19.0'
+ // Lettuce-backed Bucket4j ProxyManager used by ValkeyRateLimitStore for cluster-wide
+ // token-bucket rate limiting (parity with in-process Bucket4j semantics; no fixed-window
+ // boundary doubling).
+ implementation 'com.bucket4j:bucket4j_jdk17-lettuce:8.19.0'
// https://mvnrepository.com/artifact/com.bucket4j/bucket4j_jdk17
implementation "org.bouncycastle:bcprov-jdk18on:$bouncycastleVersion"
@@ -71,6 +76,11 @@ dependencies {
implementation('com.coveo:saml-client:5.0.0') {
exclude group: 'org.opensaml', module: 'opensaml-core'
}
+
+ // Testcontainers: spins up a real Valkey for LiveValkeyIntegrationTest in CI without
+ // needing a manually-started instance. Tests skip cleanly when Docker is unavailable.
+ testImplementation 'org.testcontainers:testcontainers:1.21.4'
+ testImplementation 'org.testcontainers:junit-jupiter:1.21.4'
}
tasks.register('prepareKotlinBuildScriptModel') {}
diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/cluster/ClusterLicenseGate.java b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/ClusterLicenseGate.java
new file mode 100644
index 0000000000..db562611a7
--- /dev/null
+++ b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/ClusterLicenseGate.java
@@ -0,0 +1,47 @@
+package stirling.software.proprietary.cluster;
+
+import org.springframework.beans.factory.annotation.Autowired;
+import org.springframework.beans.factory.annotation.Qualifier;
+import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
+import org.springframework.context.annotation.Configuration;
+import org.springframework.core.Ordered;
+import org.springframework.core.annotation.Order;
+
+import jakarta.annotation.PostConstruct;
+
+import lombok.extern.slf4j.Slf4j;
+
+/**
+ * Runtime license gate for cluster mode. Cluster mode requires a SERVER or ENTERPRISE license; the
+ * SaaS flavor bypasses (no {@code runningProOrHigher} bean is published). Fires before any Valkey
+ * bean construction via {@link Ordered#HIGHEST_PRECEDENCE}.
+ *
+ *
There is no testing/development bypass. Live e2e tests that need cluster mode must inject a
+ * valid {@code stirling.premium.key} for a test-tier SERVER/ENTERPRISE license. Unit tests stub the
+ * {@code runningProOrHigher} bean directly.
+ */
+@Configuration
+@ConditionalOnProperty(name = "cluster.enabled", havingValue = "true")
+@Order(Ordered.HIGHEST_PRECEDENCE)
+@Slf4j
+public class ClusterLicenseGate {
+
+ @Autowired(required = false)
+ @Qualifier("runningProOrHigher")
+ private Boolean runningProOrHigher;
+
+ @PostConstruct
+ void verifyLicense() {
+ if (runningProOrHigher == null) {
+ return; // saas flavor - licensed via Stripe elsewhere
+ }
+ if (!runningProOrHigher) {
+ throw new IllegalStateException(
+ "Cluster mode (cluster.enabled=true) requires a SERVER or"
+ + " ENTERPRISE license. Configure stirling.premium.key with a valid"
+ + " license key (contact sales@stirlingpdf.com to obtain one), or set"
+ + " cluster.enabled=false.");
+ }
+ log.info("Cluster license gate: SERVER/ENTERPRISE license verified, cluster mode allowed.");
+ }
+}
diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/cluster/ClusterMetrics.java b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/ClusterMetrics.java
new file mode 100644
index 0000000000..c0f1f974a3
--- /dev/null
+++ b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/ClusterMetrics.java
@@ -0,0 +1,123 @@
+package stirling.software.proprietary.cluster;
+
+import java.util.List;
+import java.util.concurrent.ConcurrentHashMap;
+import java.util.concurrent.atomic.AtomicLong;
+
+import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
+import org.springframework.stereotype.Component;
+
+import io.micrometer.core.instrument.Counter;
+import io.micrometer.core.instrument.Gauge;
+import io.micrometer.core.instrument.MeterRegistry;
+import io.micrometer.core.instrument.Timer;
+
+import stirling.software.common.cluster.StickyMissRecorder;
+import stirling.software.common.model.ApplicationProperties;
+
+/**
+ * Cluster operation metrics exposed via {@code /actuator/prometheus}. Registered only when cluster
+ * mode is on.
+ */
+@Component
+@ConditionalOnProperty(name = "cluster.enabled", havingValue = "true")
+public class ClusterMetrics implements StickyMissRecorder {
+
+ private final MeterRegistry registry;
+ private final ApplicationProperties applicationProperties;
+
+ private final Counter stickyMissTotal;
+ private final Counter rateLimitRejected;
+ private final Timer backplaneLatency;
+ private final Timer jobWaitSeconds;
+
+ // Per-lane queue depth gauges. Lanes are a fixed enum (FAST, SLOW, AI), so we register all
+ // three eagerly so dashboards never have a missing series.
+ private static final List KNOWN_LANES = List.of("FAST", "SLOW", "AI");
+ private final ConcurrentHashMap queueDepth = new ConcurrentHashMap<>();
+
+ // In-flight job count for THIS node.
+ private final AtomicLong jobsInflight = new AtomicLong();
+
+ public ClusterMetrics(MeterRegistry registry, ApplicationProperties applicationProperties) {
+ this.registry = registry;
+ this.applicationProperties = applicationProperties;
+ this.stickyMissTotal =
+ Counter.builder("stirling_cluster_sticky_miss_total")
+ .description(
+ "Sticky-session misses: a download for a job whose result lives on"
+ + " a peer node landed on this node. High sustained value means"
+ + " LB affinity is broken.")
+ .register(registry);
+ this.rateLimitRejected =
+ Counter.builder("stirling_cluster_ratelimit_rejected_total")
+ .description("Cluster-wide rate limit rejections")
+ .register(registry);
+ this.backplaneLatency =
+ Timer.builder("stirling_cluster_backplane_latency_seconds")
+ .description("Backplane round-trip latency")
+ .register(registry);
+ this.jobWaitSeconds =
+ Timer.builder("stirling_cluster_job_wait_seconds")
+ .description("Time jobs spend queued before execution")
+ .register(registry);
+ Gauge.builder("stirling_cluster_jobs_inflight", jobsInflight, AtomicLong::doubleValue)
+ .description("Jobs currently in flight on this node")
+ .tag("node", applicationProperties.getCluster().resolvedNodeId())
+ .register(registry);
+ for (String lane : KNOWN_LANES) {
+ ensureLaneGauge(lane);
+ }
+ }
+
+ /**
+ * Increment when {@code JobController} returns 410 Gone because the requested job's owner is a
+ * peer node. Surfaces to {@code stirling_cluster_sticky_miss_total}.
+ */
+ @Override
+ public void recordStickyMiss() {
+ stickyMissTotal.increment();
+ }
+
+ public void recordRateLimitReject() {
+ rateLimitRejected.increment();
+ }
+
+ public Timer backplaneLatency() {
+ return backplaneLatency;
+ }
+
+ public Timer jobWaitSeconds() {
+ return jobWaitSeconds;
+ }
+
+ public void incrementInflight() {
+ jobsInflight.incrementAndGet();
+ }
+
+ public void decrementInflight() {
+ jobsInflight.decrementAndGet();
+ }
+
+ /**
+ * Publish (or update) the queue depth gauge for {@code lane}. Idempotent - safe to call hot.
+ * Known lanes (FAST, SLOW, AI) are pre-registered at construction; unknown lanes register on
+ * first call.
+ */
+ public void setQueueDepth(String lane, long depth) {
+ ensureLaneGauge(lane).set(depth);
+ }
+
+ private AtomicLong ensureLaneGauge(String lane) {
+ return queueDepth.computeIfAbsent(
+ lane,
+ l -> {
+ AtomicLong holder = new AtomicLong();
+ Gauge.builder("stirling_cluster_queue_depth", holder, AtomicLong::doubleValue)
+ .description("Pending items in a job queue lane")
+ .tag("lane", l)
+ .register(registry);
+ return holder;
+ });
+ }
+}
diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/cluster/ClusterNodeBootstrap.java b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/ClusterNodeBootstrap.java
new file mode 100644
index 0000000000..61a921c8e6
--- /dev/null
+++ b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/ClusterNodeBootstrap.java
@@ -0,0 +1,201 @@
+package stirling.software.proprietary.cluster;
+
+import java.net.InetAddress;
+import java.net.UnknownHostException;
+import java.time.Duration;
+import java.time.Instant;
+import java.util.Locale;
+
+import org.springframework.beans.factory.annotation.Value;
+import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
+import org.springframework.boot.context.event.ApplicationReadyEvent;
+import org.springframework.context.SmartLifecycle;
+import org.springframework.context.event.EventListener;
+import org.springframework.scheduling.annotation.Scheduled;
+import org.springframework.stereotype.Component;
+
+import lombok.extern.slf4j.Slf4j;
+
+import stirling.software.common.cluster.ClusterNode;
+import stirling.software.common.cluster.InstanceRegistry;
+import stirling.software.common.model.ApplicationProperties;
+import stirling.software.common.model.ApplicationProperties.Cluster;
+
+/**
+ * Registers the local node with {@link InstanceRegistry} on startup, refreshes the entry at 1/3 of
+ * the TTL, and deregisters cleanly on shutdown.
+ *
+ *
Implements {@link SmartLifecycle} with {@code getPhase() == Integer.MAX_VALUE} so Spring tears
+ * this bean down before {@code LettuceConnectionFactory} - deregister therefore runs while the
+ * Valkey connection is still alive.
+ */
+@Component
+@Slf4j
+@ConditionalOnProperty(name = "cluster.enabled", havingValue = "true")
+public class ClusterNodeBootstrap implements SmartLifecycle {
+
+ /** TTL of the node entry in the registry. Set to 3x the heartbeat interval. */
+ private final Duration heartbeatTtl;
+
+ private final ApplicationProperties applicationProperties;
+ private final InstanceRegistry instanceRegistry;
+
+ @Value("${server.port:8080}")
+ private int serverPort;
+
+ private volatile String nodeId;
+ private volatile String internalAddress;
+ private volatile boolean running = false;
+
+ public ClusterNodeBootstrap(
+ ApplicationProperties applicationProperties, InstanceRegistry instanceRegistry) {
+ this.applicationProperties = applicationProperties;
+ this.instanceRegistry = instanceRegistry;
+ Cluster cluster = applicationProperties.getCluster();
+ long heartbeatMs =
+ cluster.getNode() == null ? 10_000L : cluster.getNode().getHeartbeatIntervalMs();
+ // TTL = 3x heartbeat: tolerate one missed tick before the node drops out of the registry.
+ this.heartbeatTtl = Duration.ofMillis(heartbeatMs * 3);
+ }
+
+ @EventListener(ApplicationReadyEvent.class)
+ public void registerOnStartup() {
+ nodeId = applicationProperties.getCluster().resolvedNodeId();
+ internalAddress = resolveInternalAddress();
+ registerSelf("register");
+ }
+
+ @Scheduled(fixedDelayString = "${cluster.node.heartbeat-interval-ms:10000}")
+ public void heartbeat() {
+ // Heartbeat-after-stop race: SmartLifecycle.stop() deregisters, but the @Scheduled
+ // tick keeps firing during a slow drain. Without this guard, the next tick re-registers
+ // the dead node and the entry resurfaces in the registry until TTL expiry.
+ if (!running) {
+ return;
+ }
+ if (nodeId == null) {
+ return; // not yet registered (startup race)
+ }
+ // Self-healing: register() is idempotent and re-populates every field, so a wiped
+ // Valkey (FLUSHALL, hash eviction) recovers on the next tick without operator action.
+ registerSelf("heartbeat");
+ }
+
+ private void registerSelf(String reason) {
+ try {
+ instanceRegistry.register(
+ new ClusterNode(nodeId, internalAddress, Instant.now(), role()), heartbeatTtl);
+ if ("register".equals(reason)) {
+ log.info(
+ "Cluster node registered: nodeId={}, internalAddress={}, role={}, ttl={}s",
+ nodeId,
+ internalAddress,
+ role(),
+ heartbeatTtl.toSeconds());
+ }
+ } catch (RuntimeException e) {
+ log.debug("Cluster {} failed for {}", reason, nodeId, e);
+ }
+ }
+
+ // ---------- SmartLifecycle (see class javadoc for ordering rationale) ----------
+
+ @Override
+ public void start() {
+ running = true;
+ }
+
+ @Override
+ public void stop() {
+ running = false;
+ if (nodeId == null) {
+ return;
+ }
+ try {
+ instanceRegistry.deregister(nodeId);
+ log.info("Cluster node deregistered: {}", nodeId);
+ } catch (RuntimeException e) {
+ // Registry entry will TTL-expire within heartbeatTtl anyway.
+ log.warn(
+ "Cluster deregister failed for {} (will TTL-expire within {}s): {}",
+ nodeId,
+ heartbeatTtl.toSeconds(),
+ e.getMessage());
+ }
+ }
+
+ @Override
+ public boolean isRunning() {
+ return running;
+ }
+
+ @Override
+ public int getPhase() {
+ return Integer.MAX_VALUE; // stopped first; LettuceConnectionFactory's phase is 0
+ }
+
+ @Override
+ public boolean isAutoStartup() {
+ return true;
+ }
+
+ /**
+ * Resolve the address peers should hit. Order: explicit config -> {@code POD_IP} env (K8s
+ * downward API) -> JDK hostname -> fail loud (never silently fall back to a loopback).
+ *
+ *
Scheme is taken from {@code cluster.node.scheme} (default {@code http}). Set to {@code
+ * https} when nodes terminate TLS themselves; leave as {@code http} when an upstream LB
+ * terminates TLS and intra-cluster traffic is plain HTTP.
+ */
+ private String resolveInternalAddress() {
+ Cluster cluster = applicationProperties.getCluster();
+ String configured =
+ cluster.getNode() == null ? null : cluster.getNode().getInternalAddress();
+ if (configured != null && !configured.isBlank()) {
+ return ensurePort(configured);
+ }
+ String podIp = System.getenv("POD_IP");
+ if (podIp != null && !podIp.isBlank()) {
+ return scheme() + "://" + podIp + ":" + serverPort;
+ }
+ try {
+ return scheme()
+ + "://"
+ + InetAddress.getLocalHost().getHostAddress()
+ + ":"
+ + serverPort;
+ } catch (UnknownHostException e) {
+ throw new IllegalStateException(
+ "Could not resolve this host's address for cluster registration; set"
+ + " cluster.node.internal-address explicitly (or set POD_IP"
+ + " in the Kubernetes downward API).",
+ e);
+ }
+ }
+
+ private String ensurePort(String addr) {
+ if (addr.startsWith("http://") || addr.startsWith("https://")) {
+ return addr;
+ }
+ if (addr.contains(":")) {
+ return scheme() + "://" + addr;
+ }
+ return scheme() + "://" + addr + ":" + serverPort;
+ }
+
+ private String scheme() {
+ Cluster cluster = applicationProperties.getCluster();
+ if (cluster.getNode() == null
+ || cluster.getNode().getScheme() == null
+ || cluster.getNode().getScheme().isBlank()) {
+ return "http";
+ }
+ String s = cluster.getNode().getScheme().trim().toLowerCase(Locale.ROOT);
+ return "https".equals(s) ? "https" : "http";
+ }
+
+ private String role() {
+ Cluster.NodeRole r = applicationProperties.getCluster().resolvedRole();
+ return r == null ? "BOTH" : r.name();
+ }
+}
diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ConditionalOnValkeyBackplane.java b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ConditionalOnValkeyBackplane.java
new file mode 100644
index 0000000000..8952c580ea
--- /dev/null
+++ b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ConditionalOnValkeyBackplane.java
@@ -0,0 +1,25 @@
+package stirling.software.proprietary.cluster.valkey;
+
+import java.lang.annotation.ElementType;
+import java.lang.annotation.Retention;
+import java.lang.annotation.RetentionPolicy;
+import java.lang.annotation.Target;
+
+import org.springframework.boot.autoconfigure.condition.ConditionalOnExpression;
+
+/**
+ * Composite condition: cluster mode is on AND the configured backplane is Valkey.
+ *
+ *
Either condition alone is insufficient to load a Valkey bean. With {@code enabled=true} but
+ * {@code backplane=inprocess}, loading the Valkey beans would crash at boot because there's no
+ * {@code StringRedisTemplate}; with {@code enabled=false} the whole cluster mode is off. Combining
+ * the two stops both footguns.
+ *
+ *
Spring's {@code @ConditionalOnProperty} cannot be applied twice on the same class, so we use
+ * {@code @ConditionalOnExpression} via this meta-annotation.
+ */
+@Target({ElementType.TYPE, ElementType.METHOD})
+@Retention(RetentionPolicy.RUNTIME)
+@ConditionalOnExpression(
+ "${cluster.enabled:false} and '${cluster.backplane:inprocess}'.equals('valkey')")
+public @interface ConditionalOnValkeyBackplane {}
diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyClusterBackplane.java b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyClusterBackplane.java
new file mode 100644
index 0000000000..7f333537e3
--- /dev/null
+++ b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyClusterBackplane.java
@@ -0,0 +1,62 @@
+package stirling.software.proprietary.cluster.valkey;
+
+import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
+import org.springframework.data.redis.core.RedisCallback;
+import org.springframework.data.redis.core.StringRedisTemplate;
+import org.springframework.stereotype.Component;
+
+import lombok.RequiredArgsConstructor;
+import lombok.extern.slf4j.Slf4j;
+
+import stirling.software.common.cluster.ClusterBackplane;
+import stirling.software.common.model.ApplicationProperties;
+
+@Slf4j
+@Component
+@RequiredArgsConstructor
+@ConditionalOnProperty(name = "cluster.enabled", havingValue = "true")
+@org.springframework.boot.autoconfigure.condition.ConditionalOnProperty(
+ name = "cluster.backplane",
+ havingValue = "valkey")
+public class ValkeyClusterBackplane implements ClusterBackplane {
+
+ private final ApplicationProperties applicationProperties;
+ private final StringRedisTemplate template;
+
+ @Override
+ public boolean isHealthy() {
+ try {
+ // template.execute() borrows from the pool and returns the connection in a finally
+ // block - critical because isHealthy() is hit on every k8s liveness/readiness probe
+ // tick. Calling getConnectionFactory().getConnection() directly leaks the connection
+ // and exhausts the pool under monitoring load.
+ String pong = template.execute((RedisCallback) connection -> connection.ping());
+ return "PONG".equalsIgnoreCase(pong);
+ } catch (RuntimeException ex) {
+ log.warn("Valkey backplane health check failed: {}", ex.getMessage());
+ return false;
+ }
+ }
+
+ @Override
+ public String backplaneType() {
+ return "valkey";
+ }
+
+ @Override
+ public String localNodeId() {
+ return applicationProperties.getCluster().resolvedNodeId();
+ }
+
+ /**
+ * Disable the local {@code TaskManager#cleanupOldJobs} loop on Valkey-backed clusters: {@link
+ * ValkeyJobStore} stores every entry with a TTL pExpire and the reverse-index entries share
+ * that TTL, so Valkey itself evicts expired job state. Running the local cleanup loop on top of
+ * that would only delete per-node in-memory {@code TaskManager} caches that the cluster-visible
+ * {@code JobStore} has already authoritative state for.
+ */
+ @Override
+ public boolean shouldRunLocalCleanup() {
+ return false;
+ }
+}
diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyConnectionConfiguration.java b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyConnectionConfiguration.java
new file mode 100644
index 0000000000..514cba876a
--- /dev/null
+++ b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyConnectionConfiguration.java
@@ -0,0 +1,222 @@
+package stirling.software.proprietary.cluster.valkey;
+
+import java.net.URI;
+
+import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
+import org.springframework.context.annotation.Bean;
+import org.springframework.context.annotation.Configuration;
+import org.springframework.context.annotation.DependsOn;
+import org.springframework.data.redis.connection.RedisPassword;
+import org.springframework.data.redis.connection.RedisStandaloneConfiguration;
+import org.springframework.data.redis.connection.lettuce.LettuceClientConfiguration;
+import org.springframework.data.redis.connection.lettuce.LettuceConnectionFactory;
+import org.springframework.data.redis.core.StringRedisTemplate;
+
+import io.lettuce.core.RedisCommandExecutionException;
+import io.lettuce.core.SslVerifyMode;
+
+import lombok.RequiredArgsConstructor;
+import lombok.extern.slf4j.Slf4j;
+
+import stirling.software.common.model.ApplicationProperties;
+import stirling.software.common.model.ApplicationProperties.Cluster;
+
+/** Wires the LettuceConnectionFactory and StringRedisTemplate for cluster mode. */
+@Slf4j
+@Configuration
+@RequiredArgsConstructor
+@ConditionalOnProperty(name = "cluster.enabled", havingValue = "true")
+@DependsOn("clusterLicenseGate")
+public class ValkeyConnectionConfiguration {
+
+ private final ApplicationProperties applicationProperties;
+
+ @Bean(destroyMethod = "destroy")
+ @ConditionalOnProperty(name = "cluster.backplane", havingValue = "valkey")
+ public LettuceConnectionFactory valkeyConnectionFactory() {
+ Cluster cluster = applicationProperties.getCluster();
+ String url = cluster.getValkey().getUrl();
+ if (url == null || url.isBlank()) {
+ throw new IllegalStateException("cluster.valkey.url must be set when backplane=valkey");
+ }
+ URI uri = URI.create(url);
+ boolean tls = "rediss".equalsIgnoreCase(uri.getScheme());
+ int port = uri.getPort() <= 0 ? 6379 : uri.getPort();
+ RedisStandaloneConfiguration cfg = new RedisStandaloneConfiguration(uri.getHost(), port);
+ if (uri.getUserInfo() != null) {
+ String[] parts = uri.getUserInfo().split(":", 2);
+ if (parts.length == 2) {
+ cfg.setUsername(parts[0]);
+ cfg.setPassword(RedisPassword.of(parts[1]));
+ } else if (parts.length == 1 && !parts[0].isBlank()) {
+ cfg.setPassword(RedisPassword.of(parts[0]));
+ }
+ }
+ boolean skipCertVerification =
+ cluster.getValkey().getTls() != null
+ && cluster.getValkey().getTls().isSkipCertVerification();
+ LettuceClientConfiguration clientConfig =
+ buildClientConfiguration(tls, skipCertVerification);
+ LettuceConnectionFactory factory = new LettuceConnectionFactory(cfg, clientConfig);
+ factory.afterPropertiesSet();
+ // Eager handshake with retry tolerates docker-compose DNS races; fails boot loudly
+ // if Valkey is genuinely unreachable.
+ eagerHandshake(factory, uri.getHost(), port, tls);
+ log.info(
+ "Valkey connection configured: {}:{} tls={} verifyPeer={}",
+ uri.getHost(),
+ port,
+ tls,
+ tls ? clientConfig.getVerifyMode() : "n/a");
+ return factory;
+ }
+
+ /**
+ * Builds the Lettuce client configuration with TLS verification pinned. Package-private so unit
+ * tests can verify the {@code verifyPeer} mode without standing up a real Valkey.
+ *
+ *
{@code verifyPeer(FULL)} is pinned explicitly so a future Spring Data Redis default change
+ * cannot silently weaken our TLS handshake. {@code FULL} = X.509 chain + hostname check (per
+ * Lettuce's {@link SslVerifyMode}). The {@code skipCertVerification} opt-out is for local dev
+ * with self-signed certs only; production deployments MUST leave it false.
+ */
+ static LettuceClientConfiguration buildClientConfiguration(
+ boolean tls, boolean skipCertVerification) {
+ LettuceClientConfiguration.LettuceClientConfigurationBuilder clientBuilder =
+ LettuceClientConfiguration.builder();
+ if (tls) {
+ clientBuilder
+ .useSsl()
+ .verifyPeer(skipCertVerification ? SslVerifyMode.NONE : SslVerifyMode.FULL);
+ if (skipCertVerification) {
+ log.warn(
+ "Valkey TLS hostname/chain verification DISABLED via"
+ + " cluster.valkey.tls.skip-cert-verification=true"
+ + " - insecure, dev-only");
+ }
+ }
+ return clientBuilder.build();
+ }
+
+ /**
+ * 10 x 3s = 30s of retry. Boot-time only.
+ *
+ *
Auth-class failures (WRONGPASS / NOAUTH / NOPERM) are unrecoverable and surfaced
+ * immediately on the first attempt; only transport-level errors (connection refused, timeout,
+ * host unreachable) get the retry loop.
+ *
+ *
Package-private so unit tests can drive it with a mocked connection factory.
+ */
+ static void eagerHandshake(
+ LettuceConnectionFactory factory, String host, int port, boolean tls) {
+ RuntimeException last = null;
+ for (int attempt = 1; attempt <= 10; attempt++) {
+ try {
+ String pong = factory.getConnection().ping();
+ if (!"PONG".equalsIgnoreCase(pong)) {
+ throw new IllegalStateException(
+ "Valkey PING returned '" + pong + "' (expected PONG)");
+ }
+ if (attempt > 1) {
+ log.info("Valkey reachable after {} attempts", attempt);
+ }
+ return;
+ } catch (RuntimeException ex) {
+ if (isAuthFailure(ex)) {
+ factory.destroy();
+ throw new IllegalStateException(
+ "Valkey authentication failed for "
+ + host
+ + ":"
+ + port
+ + " (tls="
+ + tls
+ + "): "
+ + rootAuthMessage(ex)
+ + ". Check cluster.valkey.url credentials"
+ + " (user/password and ACL permissions).",
+ ex);
+ }
+ last = ex;
+ log.warn(
+ "Valkey PING attempt {}/10 failed ({}:{}, tls={}): {}",
+ attempt,
+ host,
+ port,
+ tls,
+ ex.getMessage());
+ try {
+ Thread.sleep(3000);
+ } catch (InterruptedException ie) {
+ Thread.currentThread().interrupt();
+ break;
+ }
+ }
+ }
+ factory.destroy();
+ throw new IllegalStateException(
+ "Valkey unreachable at boot after 10 attempts ("
+ + host
+ + ":"
+ + port
+ + ", tls="
+ + tls
+ + "): "
+ + (last == null ? "no detail" : last.getMessage()),
+ last);
+ }
+
+ /**
+ * Walks the cause chain for a Lettuce {@link RedisCommandExecutionException} whose message
+ * starts with an auth-class server reply (WRONGPASS, NOAUTH, NOPERM). Spring Data Redis wraps
+ * Lettuce errors in a {@code RedisSystemException}, so the auth signal usually lives one level
+ * down from the thrown exception.
+ *
+ *
Checked for a typed alternative: neither Spring Data Redis 4.0.5 nor Lettuce 6.8.2 ships a
+ * {@code RedisAuthenticationException} on the classpath, so we keep the message-prefix match.
+ * Revisit when upgrading Spring Data Redis if a typed exception lands upstream.
+ */
+ static boolean isAuthFailure(Throwable t) {
+ for (Throwable cur = t; cur != null; cur = cur.getCause()) {
+ if (cur instanceof RedisCommandExecutionException && hasAuthPrefix(cur.getMessage())) {
+ return true;
+ }
+ // Defensive: some translations preserve the original message on the wrapper itself.
+ if (hasAuthPrefix(cur.getMessage())) {
+ return true;
+ }
+ if (cur.getCause() == cur) {
+ break;
+ }
+ }
+ return false;
+ }
+
+ private static boolean hasAuthPrefix(String message) {
+ if (message == null) {
+ return false;
+ }
+ String upper = message.toUpperCase(java.util.Locale.ROOT).stripLeading();
+ return upper.startsWith("WRONGPASS")
+ || upper.startsWith("NOAUTH")
+ || upper.startsWith("NOPERM");
+ }
+
+ private static String rootAuthMessage(Throwable t) {
+ for (Throwable cur = t; cur != null; cur = cur.getCause()) {
+ if (cur instanceof RedisCommandExecutionException && cur.getMessage() != null) {
+ return cur.getMessage();
+ }
+ if (cur.getCause() == cur) {
+ break;
+ }
+ }
+ return t.getMessage();
+ }
+
+ @Bean
+ @ConditionalOnProperty(name = "cluster.backplane", havingValue = "valkey")
+ public StringRedisTemplate valkeyTemplate(LettuceConnectionFactory factory) {
+ return new StringRedisTemplate(factory);
+ }
+}
diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyDistributedLock.java b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyDistributedLock.java
new file mode 100644
index 0000000000..44075a0b93
--- /dev/null
+++ b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyDistributedLock.java
@@ -0,0 +1,92 @@
+package stirling.software.proprietary.cluster.valkey;
+
+import java.time.Duration;
+import java.util.Collections;
+import java.util.Optional;
+import java.util.UUID;
+
+import org.springframework.data.redis.core.StringRedisTemplate;
+import org.springframework.data.redis.core.script.DefaultRedisScript;
+import org.springframework.data.redis.core.script.RedisScript;
+import org.springframework.stereotype.Component;
+
+import lombok.RequiredArgsConstructor;
+import lombok.extern.slf4j.Slf4j;
+
+import stirling.software.common.cluster.DistributedLock;
+
+@Component
+@RequiredArgsConstructor
+@ConditionalOnValkeyBackplane
+@Slf4j
+public class ValkeyDistributedLock implements DistributedLock {
+
+ private static final String PREFIX = "stirling:lock:";
+
+ private static final RedisScript RELEASE_SCRIPT =
+ new DefaultRedisScript<>(
+ "if redis.call('get', KEYS[1]) == ARGV[1] then return redis.call('del', KEYS[1]) else return 0 end",
+ Long.class);
+
+ private static final RedisScript RENEW_SCRIPT =
+ new DefaultRedisScript<>(
+ "if redis.call('get', KEYS[1]) == ARGV[1] then return redis.call('pexpire', KEYS[1], ARGV[2]) else return 0 end",
+ Long.class);
+
+ private final StringRedisTemplate template;
+
+ @Override
+ public Optional tryAcquire(String lockKey, Duration leaseTime) {
+ String key = PREFIX + lockKey;
+ String value = UUID.randomUUID().toString();
+ Boolean ok = template.opsForValue().setIfAbsent(key, value, leaseTime);
+ if (Boolean.TRUE.equals(ok)) {
+ return Optional.of(new ValkeyHandle(template, key, value));
+ }
+ return Optional.empty();
+ }
+
+ private static final class ValkeyHandle implements LockHandle {
+ private final StringRedisTemplate template;
+ private final String key;
+ private final String value;
+ private boolean released;
+
+ ValkeyHandle(StringRedisTemplate template, String key, String value) {
+ this.template = template;
+ this.key = key;
+ this.value = value;
+ }
+
+ @Override
+ public synchronized void release() {
+ if (released) {
+ return;
+ }
+ released = true;
+ template.execute(RELEASE_SCRIPT, Collections.singletonList(key), value);
+ }
+
+ @Override
+ public synchronized boolean renew(Duration leaseTime) {
+ if (released) {
+ return false;
+ }
+ try {
+ Long result =
+ template.execute(
+ RENEW_SCRIPT,
+ Collections.singletonList(key),
+ value,
+ Long.toString(leaseTime.toMillis()));
+ return result != null && result == 1L;
+ } catch (RuntimeException ex) {
+ log.warn(
+ "Lock renew failed for {} (treated as lost lease): {}",
+ key,
+ ex.getMessage());
+ return false;
+ }
+ }
+ }
+}
diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyInstanceRegistry.java b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyInstanceRegistry.java
new file mode 100644
index 0000000000..d23932506d
--- /dev/null
+++ b/app/proprietary/src/main/java/stirling/software/proprietary/cluster/valkey/ValkeyInstanceRegistry.java
@@ -0,0 +1,116 @@
+package stirling.software.proprietary.cluster.valkey;
+
+import java.nio.charset.StandardCharsets;
+import java.time.Duration;
+import java.time.Instant;
+import java.util.ArrayList;
+import java.util.Collection;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Optional;
+
+import org.springframework.data.redis.core.Cursor;
+import org.springframework.data.redis.core.RedisCallback;
+import org.springframework.data.redis.core.ScanOptions;
+import org.springframework.data.redis.core.StringRedisTemplate;
+import org.springframework.stereotype.Component;
+
+import lombok.RequiredArgsConstructor;
+
+import stirling.software.common.cluster.ClusterNode;
+import stirling.software.common.cluster.InstanceRegistry;
+
+/**
+ * Valkey-backed {@link InstanceRegistry}. Each node is stored as a hash with a TTL equal to the
+ * configured heartbeat TTL; the heartbeat re-arms the TTL.
+ */
+@Component
+@RequiredArgsConstructor
+@ConditionalOnValkeyBackplane
+public class ValkeyInstanceRegistry implements InstanceRegistry {
+
+ private static final String PREFIX = "stirling:nodes:";
+
+ private final StringRedisTemplate template;
+
+ @Override
+ public void register(ClusterNode node, Duration heartbeatTtl) {
+ String key = PREFIX + node.nodeId();
+ long ttlMs = heartbeatTtl.toMillis();
+ Map fields = new LinkedHashMap<>();
+ fields.put("nodeId", node.nodeId());
+ fields.put("internalAddress", node.internalAddress());
+ fields.put("role", node.role());
+ fields.put("lastHeartbeat", node.lastHeartbeat().toString());
+
+ // MULTI/EXEC so the hash fields and the TTL commit together. Without this, a crash
+ // between HSET and EXPIRE leaves the hash with no TTL: it never expires, masks the
+ // dead node as alive, and only a subsequent successful register() would re-arm it.
+ template.execute(
+ (RedisCallback