Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
25 commits
Select commit Hold shift + click to select a range
473acdb
feat(gitignore): add docs/plans and logs-smoke to ignore list
Jonathan-Eid Aug 26, 2026
9285bf3
feat(catchup): retire workers the queue cannot keep busy
Jonathan-Eid Aug 26, 2026
d9d9eb6
fix(catchup): a Terminating pod is not ready
Jonathan-Eid Aug 26, 2026
233285a
perf(catchup): skip the pod list and exec when nothing can come of them
Jonathan-Eid Aug 26, 2026
6b4f927
fix(catchup): bound how many workers one pass drains
Jonathan-Eid Aug 26, 2026
ed59cf0
fix(catchup): pass --namespace to every helm call
Jonathan-Eid Aug 26, 2026
cfe2a5e
fix(catchup): keep a reserve of workers unmarked
Jonathan-Eid Aug 27, 2026
e29f2f2
fix(gitignore): remove unnecessary entries for docs/plans and logs-smoke
Jonathan-Eid Aug 27, 2026
56c99f5
fix(catchup): act on the exec exit code instead of printing it
Jonathan-Eid Aug 27, 2026
520696c
fix(catchup): update comments for clarity and increase max retired po…
Jonathan-Eid Aug 27, 2026
960cb34
fix(catchup): measure surplus against the whole unmarked fleet
Jonathan-Eid Aug 27, 2026
753f255
fix(catchup): stop treating an empty log archive as a failed collection
Jonathan-Eid Aug 27, 2026
8e9bcf2
fix(catchup): take the worker index from the chart, not the pod name
Jonathan-Eid Aug 27, 2026
e178fc2
fix(catchup): recognise per-worker StatefulSets in the orphan sweep
Jonathan-Eid Aug 27, 2026
95851b0
fix(catchup): read the job-owners hash name from the pod environment
Jonathan-Eid Aug 27, 2026
c489d09
fix(catchup): split release names at the last -stellar-core, record m…
Jonathan-Eid Aug 27, 2026
040e465
Merge branch 'main' into jonathan/catchup-scale-down-v2
Jonathan-Eid Sep 8, 2026
8a186ab
Drop stale collectLogsFromPods header lines
Jonathan-Eid Sep 8, 2026
72a5baa
Move scale-down marking into job monitor; workers as bare pods
Jonathan-Eid Sep 22, 2026
e2ef41f
Restore unrelated lines to match main
Jonathan-Eid Sep 22, 2026
9de0692
Merge branch 'main' into jonathan/catchup-scale-down-v2
Jonathan-Eid Sep 22, 2026
e79e2da
Document ownerless-Pod deletion in the sweep
Jonathan-Eid Sep 22, 2026
8046411
worker.sh: fail closed on the retiring check; plumb RETIRING through …
Jonathan-Eid Oct 5, 2026
775e515
Orphan sweep: list pods after the helm uninstalls
Jonathan-Eid Oct 5, 2026
9ae0847
Merge branch 'main' into jonathan/catchup-scale-down-v2
Jonathan-Eid Oct 8, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion src/App/Program.fs
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@ type PollOptions(kubeConfig: string, namespaceProperty: string option) =
inherit KubeOption(kubeConfig, namespaceProperty)

[<Verb("force-clean-namespace",
HelpText = "Delete every Service/ConfigMap/StatefulSet/Ingress/Job/DaemonSet/Deployment in the namespace and helm-uninstall any parallel-catchup-* releases. Destructive: meant for explicit recovery of a shared namespace, not for routine use.")>]
HelpText = "Delete every Service/ConfigMap/StatefulSet/Ingress/Job/DaemonSet/Deployment and ownerless Pod in the namespace and helm-uninstall any parallel-catchup-* releases. Destructive: meant for explicit recovery of a shared namespace, not for routine use.")>]
type ForceCleanNamespaceOptions(kubeConfig: string, namespaceProperty: string option) =
inherit KubeOption(kubeConfig, namespaceProperty)

Expand Down
29 changes: 13 additions & 16 deletions src/CSLibrary/RemoteCommandRunner.cs
Original file line number Diff line number Diff line change
Expand Up @@ -66,14 +66,15 @@ await kube.MuxedStreamNamespacedPodExecAsync(name: podName, @namespace: ns,
}

// Execute a command and capture stdout to a file (for copying files from pod)
public static void RunRemoteCommandAndCaptureOutput(Kubernetes kube, string ns, string podName,
// Returns the command's exit code
public static int RunRemoteCommandAndCaptureOutput(Kubernetes kube, string ns, string podName,
string containerName, string[] command, string outputFilePath)
{
Task task = RunRemoteCommandAndCaptureOutputAsync(kube, ns, podName, containerName, command, outputFilePath);
task.Wait();
Task<int> task = RunRemoteCommandAndCaptureOutputAsync(kube, ns, podName, containerName, command, outputFilePath);
return task.Result;
}

public static async Task RunRemoteCommandAndCaptureOutputAsync(Kubernetes kube, string ns, string podName,
public static async Task<int> RunRemoteCommandAndCaptureOutputAsync(Kubernetes kube, string ns, string podName,
string containerName, string[] command, string outputFilePath)
{
// The `using` lifetime guard ensure these objects lifetimes last the entire task,
Expand All @@ -89,30 +90,26 @@ await kube.MuxedStreamNamespacedPodExecAsync(
stderr: true,
tty: false).ConfigureAwait(false))
using (System.IO.Stream stdout = mstr.GetStream(ChannelIndex.StdOut, null))
using (System.IO.Stream stderr = mstr.GetStream(ChannelIndex.Error, null))
using (System.IO.StreamReader errorReader = new System.IO.StreamReader(stderr))
using (System.IO.Stream statusChannel = mstr.GetStream(ChannelIndex.Error, null))
using (System.IO.StreamReader statusReader = new System.IO.StreamReader(statusChannel))
using (System.IO.FileStream fileStream = new System.IO.FileStream(outputFilePath, FileMode.Create, FileAccess.Write, FileShare.None,
bufferSize: 128 * 1024, useAsync: true))
{
// Start the MuxStream, this establishes the connection and routes bytes back into separate channels.
// We only care about stdout(1) and stderr(2).
// We only care about stdout(1) and the status channel(3).
mstr.Start();

// Copy stdout → file asynchronously, and drain stderr concurrently.
// Copy stdout → file asynchronously, and drain the status channel concurrently.
var copyTask = stdout.CopyToAsync(fileStream);
var errorTask = errorReader.ReadToEndAsync();
var statusTask = statusReader.ReadToEndAsync();

await Task.WhenAll(copyTask, errorTask).ConfigureAwait(false);
await Task.WhenAll(copyTask, statusTask).ConfigureAwait(false);

// Flush the file stream to ensure all data is written
await fileStream.FlushAsync().ConfigureAwait(false);

// Log any errors to console
string errors = errorTask.Result;
if (!string.IsNullOrEmpty(errors))
{
Console.WriteLine($"Command stderr from pod {podName}: {errors}");
}
string status = statusTask.Result;
return Kubernetes.GetExitCodeOrThrow(SafeJsonConvert.DeserializeObject<V1Status>(status));
}
}
}
Expand Down
69 changes: 53 additions & 16 deletions src/FSLibrary/MissionHistoryPubnetParallelCatchupV2.fs
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,10 @@ let failedJobLogStreamLineCount = 1000

let mutable nonce : String = ""
let mutable helmReleaseName : String = ""
// Pods not yet retired; module scope because cleanup runs from a signal handler.
let mutable livePods : Set<string> = Set.empty
// Log collection is serial and runs inside the poll loop, so bound what one pass can block on.
let maxRetiredPerPass = 64

let jobMonitorHostName (context: MissionContext) =
match context.jobMonitorExternalHost with
Expand Down Expand Up @@ -227,6 +231,8 @@ let installProject (context: MissionContext) =
"install"
helmReleaseName
helmChartPath
"--namespace"
context.namespaceProperty
"--values"
valuesFilePath
"--set"
Expand All @@ -236,22 +242,14 @@ let installProject (context: MissionContext) =
match RunShellCommand [| "helm"
"get"
"values"
helmReleaseName |] with
helmReleaseName
"--namespace"
context.namespaceProperty |] with
| Some valuesOutput -> LogInfo "%s" valuesOutput
| _ -> ()

// Collect log files from all parallel catchup worker pods
// This function:
// 1. Automatically determines worker pod names from context.pubnetParallelCatchupNumWorkers
// 2. For each pod, finds all files matching "stellar-core-*.log" in /data
// 3. Creates a tar.gz archive and copies it to context.destination directory
let collectLogsFromPods (context: MissionContext) =
// Generate pod names based on number of workers
// Pod names follow the pattern: <helmReleaseName>-stellar-core-0, <helmReleaseName>-stellar-core-1, etc.
let podNames =
[ 0 .. context.pubnetParallelCatchupNumWorkers - 1 ]
|> List.map (fun i -> sprintf "%s-stellar-core-%d" helmReleaseName i)

// Collect log files from the given worker pods; a pod whose collection fails is skipped.
let collectLogsFromPods (context: MissionContext) (podNames: string list) : unit =
LogInfo "Collecting logs from %d worker pods to directory: %s" (List.length podNames) context.destination.Path

for podName in podNames do
Expand All @@ -275,6 +273,7 @@ let collectLogsFromPods (context: MissionContext) =
command = command,
outputFilePath = outputFile
)
|> ignore
Comment thread
Jonathan-Eid marked this conversation as resolved.

let fileInfo = FileInfo(outputFile)

Expand Down Expand Up @@ -307,7 +306,9 @@ let cleanup (signalTriggered: bool) (context: MissionContext) =

RunShellCommand [| "helm"
"uninstall"
helmReleaseName |]
helmReleaseName
"--namespace"
context.namespaceProperty |]
|> ignore
else
// Normal / legitimate-failure path: pods are still alive through
Expand All @@ -317,14 +318,16 @@ let cleanup (signalTriggered: bool) (context: MissionContext) =
try
LogInfo "Attempting to collect worker logs before cleanup..."
let stopwatch = Stopwatch.StartNew()
collectLogsFromPods context
collectLogsFromPods context (List.ofSeq livePods)
stopwatch.Stop()
LogInfo "Log collection completed in %.2f seconds" stopwatch.Elapsed.TotalSeconds
with ex -> LogWarn "Failed to collect some or all worker logs: %s" ex.Message

RunShellCommand [| "helm"
"uninstall"
helmReleaseName |]
helmReleaseName
"--namespace"
context.namespaceProperty |]
|> ignore

let mutable cleanupContext : MissionContext option = None
Expand Down Expand Up @@ -399,6 +402,11 @@ let historyPubnetParallelCatchupV2 (context: MissionContext) =
installProject context

let mutable allJobsFinished = false

livePods <-
Set.ofList [ for i in 0 .. context.pubnetParallelCatchupNumWorkers - 1 ->
sprintf "%s-stellar-core-%d" helmReleaseName i ]

let mutable timeoutLeft = jobMonitorStatusCheckTimeOutSecs
let mutable timeBeforeNextMetricsCheck = jobMonitorMetricsCheckIntervalSecs
let mutable stalledForSecs = 0
Expand Down Expand Up @@ -431,6 +439,35 @@ let historyPubnetParallelCatchupV2 (context: MissionContext) =

failwith "Catch up failed, check logs for more info"

// `queue_remain_count`, not `num_remain`, which is 1 as a pre-first-poll sentinel.
let outstanding = status.Value<int>("queue_remain_count") + JobsInProgress.Count

// The job monitor owns the marking decision; we only collect logs and delete.
try
let retirable = status.["retirable"] :?> JArray

// Already-deleted names stay in the monitor's set, so filter by what is still live.
let toRetire =
retirable
|> Seq.map (fun name -> name.ToString())
|> Seq.filter livePods.Contains
|> Seq.truncate maxRetiredPerPass
|> List.ofSeq

if not toRetire.IsEmpty then
// /data is emptyDir, so collect before deleting or the logs are lost.
collectLogsFromPods context toRetire

for pod in toRetire do
try
context.kube.DeleteNamespacedPod(pod, context.namespaceProperty) |> ignore
Comment on lines +461 to +463
with :? k8s.Autorest.HttpOperationException as ex when
ex.Response.StatusCode = Net.HttpStatusCode.NotFound ->
LogInfo "Pod %s already gone" pod

livePods <- Set.difference livePods (Set.ofList toRetire)
LogInfo "Retired %d workers (%d outstanding)" toRetire.Length outstanding
with ex -> LogWarn "Worker scale-down skipped this pass: %s" ex.Message
// Detect if the mission is stuck from two signals: 1. job queue
// has in progress items but no live workers, which we fail the
// mission 2. the job monitor itself gets stuck unable to
Expand Down
52 changes: 33 additions & 19 deletions src/FSLibrary/StellarOrphanSweep.fs
Original file line number Diff line number Diff line change
Expand Up @@ -54,38 +54,39 @@ let private sweepKind

// Delete resources older than `cutoff` in the given namespace. Targets the same
// resource type set the retired `clean` verb did (Service, ConfigMap,
// StatefulSet, Ingress, Job, DaemonSet, Deployment) and also helm-uninstalls
// any `parallel-catchup-*` releases (PCv2) so helm's release secrets get
// tidied along with the workloads.
// StatefulSet, Ingress, Job, DaemonSet, Deployment) plus ownerless Pods, and
// also helm-uninstalls any `parallel-catchup-*` releases (PCv2) so helm's
// release secrets get tidied along with the workloads.
//
// Used by both the automatic on-startup orphan sweep (cutoff = now - 2 days)
// and the explicit `force-clean-namespace` verb (cutoff = DateTime.MaxValue,
// i.e. everything).
let private sweepWithCutoff (cutoff: DateTime) (kube: Kubernetes) (ns: string) (apiRateLimit: int) : unit =
LogInfo "Orphan sweep starting: namespace=%s cutoff=%s" ns (cutoff.ToString("o"))

// 1. helm-uninstall PCv2 releases whose StatefulSet is older than the cutoff.
// 1. helm-uninstall PCv2 releases whose worker pod is older than the cutoff.
// Done before the kubectl-style deletes so helm's release secrets get
// cleaned up properly, rather than left dangling pointing at deleted
// resources.
ApiRateLimit.sleepUntilNextRateLimitedApiCallTime apiRateLimit

let stsItems = kube.ListNamespacedStatefulSet(namespaceParameter = ns).Items

for sts in stsItems do
if isOlderThan cutoff sts.Metadata then
let name = sts.Metadata.Name

if name.StartsWith("parallel-catchup-") && name.EndsWith("-stellar-core") then
let release = name.Substring(0, name.Length - "-stellar-core".Length)
LogInfo "Orphan sweep: helm uninstall %s" release

RunShellCommand [| "helm"
"uninstall"
release
"-n"
ns |]
|> ignore
let podItems = kube.ListNamespacedPod(namespaceParameter = ns).Items

for release in podItems
|> Seq.filter (fun pod -> isOlderThan cutoff pod.Metadata)
|> Seq.map (fun pod -> pod.Metadata.Name)
|> Seq.filter (fun name -> name.StartsWith("parallel-catchup-") && name.Contains("-stellar-core"))
|> Seq.map (fun name -> name.Substring(0, name.LastIndexOf("-stellar-core")))
|> Set.ofSeq do
LogInfo "Orphan sweep: helm uninstall %s" release

RunShellCommand [| "helm"
"uninstall"
release
"-n"
ns |]
|> ignore

// 2. Delete the same resource type set the retired `clean` verb targeted.
// The order matches the old NamespaceContent.Cleanup so dependent
Expand All @@ -111,6 +112,19 @@ let private sweepWithCutoff (cutoff: DateTime) (kube: Kubernetes) (ns: string) (
|> ignore)
cutoff

// PCv2 workers are bare pods, so nothing else in this sweep would reap them.
// Restricted to ownerless pods: everything else goes away with its controller.
// Listed here, after the helm uninstalls above, so pods they already removed are not deleted again.
sweepKind
Comment thread
Jonathan-Eid marked this conversation as resolved.
apiRateLimit
"Pod"
(fun () ->
kube.ListNamespacedPod(namespaceParameter = ns).Items
|> Seq.filter (fun p -> isNull p.Metadata.OwnerReferences || p.Metadata.OwnerReferences.Count = 0)

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This deletes every ownerless pod in the namespace that is older than 2 days, not just PCv2 workers. The sweep runs at every mission start, so any kubectl run or debug pod, or another tool's bare pod, in a shared namespace gets killed, even while running. That includes default when --namespace isn't passed. Suggest applying the same parallel-catchup-* / -stellar-core name filter the helm step uses, or matching on the worker app label.


Generated by Claude Code

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The goal of the sweep is to keep the supercluster namespace completely clean, so far we've had no need to keep recurring resources for a next run, so the extra crossfire would be okay.

Also, these changes would be the first to introduce ownerless pods in this namespace, as no other mission spins up bare pods as spinning up bare pods without a deployment or statefulset, etc is unconventional.

|> Seq.map (fun p -> p.Metadata))
(fun n -> kube.DeleteNamespacedPod(namespaceParameter = ns, name = n) |> ignore)
cutoff

sweepKind
apiRateLimit
"Deployment"
Expand Down
34 changes: 33 additions & 1 deletion src/MissionParallelCatchup/job_monitor.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import os
import redis
import socket
import requests
import json
import sys
Expand All @@ -26,6 +27,8 @@
PROGRESS_QUEUE = os.getenv('PROGRESS_QUEUE', 'in_progress') #LIST
METRICS = os.getenv('METRICS', 'metrics') # SET
JOB_OWNERS = os.getenv('JOB_OWNERS', 'job_owners') # HASH
RETIRING = os.getenv('RETIRING', 'retiring') # SET
MIN_UNMARKED_WORKERS = int(os.getenv('MIN_UNMARKED_WORKERS', 8))
WORKER_PREFIX = os.getenv('WORKER_PREFIX', 'stellar-core')
NAMESPACE = os.getenv('NAMESPACE', 'default')
WORKER_COUNT = int(os.getenv('WORKER_COUNT', 3))
Expand Down Expand Up @@ -77,6 +80,7 @@ def get_logging_level():
'jobs_failed': [],
'jobs_in_progress': [],
'workers': [],
'retirable': [],
'workers_up': 0,
'workers_down': 0,
'workers_refresh_duration': 0,
Expand Down Expand Up @@ -148,6 +152,14 @@ def ping_worker(pod_name, retries=1):
time.sleep(STUCK_JOB_PING_DELAY_SECS)
return False

def pod_exists(pod_name):
# Trailing dot skips the resolv.conf search list, so a miss costs one query.
try:
socket.gethostbyname(f"{pod_name}.{WORKER_PREFIX}.{NAMESPACE}.svc.cluster.local.")
return True
except socket.gaierror:
return False

def update_status_and_metrics():
global status
mission_start_time = time.time()
Expand All @@ -174,8 +186,27 @@ def update_status_and_metrics():
queue_in_progress_count = len(jobs_in_progress)
queue_remain_count = redis_client.llen(JOB_QUEUE)

# --- Phase 1c: Mark surplus idle workers as retiring; worker.sh stops claiming once marked ---
# Names here are pod names ("{WORKER_PREFIX}-{i}") as worker.sh stores them in JOB_OWNERS.
busy = set(job_owners.values())
retiring = redis_client.smembers(RETIRING)
candidates = sorted(w for w in (f"{WORKER_PREFIX}-{i}" for i in range(WORKER_COUNT))
if w not in busy and w not in retiring)
outstanding = queue_remain_count + queue_in_progress_count
keep = max(outstanding, MIN_UNMARKED_WORKERS)

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

If the unmarked reserve (as few as MIN_UNMARKED_WORKERS) is lost to eviction or node failure, the mission hangs instead of failing. Bare pods aren't recreated, Phase 3 requeues their jobs so in_progress empties, and the driver's stall check only fires when in_progress > 0. That leaves queued jobs no pod can claim, and the monitor is still reachable, so no timeout trips. Before this PR, the StatefulSet would have recreated the pods. Suggest failing in the driver when queue_remain_count > 0 and no live unmarked worker remains.


Generated by Claude Code

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yes, this is why the default for MIN_UNMARKED_WORKERS left some headroom to account for some nodes going down, but it's not expected in an ondemand setup for nodes to go down and be reclaimed.

That's what led me to the tradeoff of using n bare pods instead of n statefulsets so less load can be applied on the k8s api server.

to_mark = []
# Resolve only when the name count says something could be marked: at t=0 outstanding >= fleet, so zero lookups.
if len(busy - retiring) + len(candidates) > keep:
unmarked_idle = [w for w in candidates if pod_exists(w)]
to_mark = unmarked_idle[:max(0, len(busy - retiring) + len(unmarked_idle) - keep)]
if to_mark:
redis_client.sadd(RETIRING, *to_mark)
logger.info("Marked %d workers retiring (%d outstanding)", len(to_mark), outstanding)
# Marked on an earlier pass and still idle; names stay here after the driver deletes them.
retirable = sorted(retiring - busy)

# --- Phase 2: Quick single-ping check of workers that own in-progress jobs ---
active_workers = set(job_owners.values())
active_workers = busy
worker_statuses = []
workers_up = 0
workers_down = 0
Expand Down Expand Up @@ -231,6 +262,7 @@ def update_status_and_metrics():
'jobs_failed': jobs_failed,
'jobs_in_progress': jobs_in_progress,
'workers': worker_statuses,
'retirable': retirable,
'workers_up': workers_up,
'workers_down': workers_down,
'workers_refresh_duration': workers_refresh_duration,
Expand Down
13 changes: 11 additions & 2 deletions src/MissionParallelCatchup/parallel_catchup_helm/files/worker.sh
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@ if [ -z "$FAILED_QUEUE" ]; then echo "FAILED_QUEUE not set"; exit 1; fi
if [ -z "$SUCCESS_QUEUE" ]; then echo "SUCCESS_QUEUE not set"; exit 1; fi
if [ -z "$METRICS" ]; then echo "METRICS not set"; exit 1; fi
if [ -z "$JOB_OWNERS" ]; then echo "JOB_OWNERS not set"; exit 1; fi
if [ -z "$RETIRING" ]; then echo "RETIRING not set"; exit 1; fi
if [ -z "$RELEASE_NAME" ]; then echo "RELEASE_NAME not set"; exit 1; fi
if [ -z "$POD_NAME" ]; then echo "POD_NAME not set"; exit 1; fi

Expand All @@ -27,6 +28,15 @@ if job then redis.call("HSET", KEYS[3], job, ARGV[1]) end
return job'

while true; do
# Stop claiming once the job monitor marks us, so the driver can remove us without interrupting a range.
# Fail closed: anything but an explicit 0 (marked, or redis-cli error) means don't claim.
if [ "$(redis-cli -h "$REDIS_HOST" -p "$REDIS_PORT" SISMEMBER "$RETIRING" "$POD_NAME")" != "0" ]; then
echo "$(date) $POD_NAME is retiring or redis unreachable; not claiming."
sleep $SLEEP_INTERVAL
continue
fi


# Claim the next job: atomically move it from the job queue to the progress
# queue and record this pod as its owner. Our ranges are generated in the order
# we want to run them from left to right, so we always pull from the left
Expand Down Expand Up @@ -84,8 +94,7 @@ if [ $CLAIM_EXIT_CODE -eq 0 ] && [ "$CLAIM_VALID" = true ]; then
fi

# Push metrics to redis in a transaction to ensure data consistency. Retry for 5min on failures
# Extract the pod ordinal (last hyphen-separated segment) from pod name like "release-name-stellar-core-0"
core_id=$(echo "$POD_NAME" | awk -F'-' '{print $NF}')
core_id="$WORKER_INDEX"
# Validate core_id was extracted successfully
if [ -z "$core_id" ]; then
echo "Error: Failed to extract core_id from POD_NAME: $POD_NAME"
Expand Down
Loading
Loading