yrenat commented on code in PR #6046:
URL: https://github.com/apache/texera/pull/6046#discussion_r3835504334
##########
computing-unit-managing-service/src/main/scala/org/apache/texera/service/resource/ComputingUnitManagingResource.scala:
##########
@@ -69,6 +77,136 @@ object ComputingUnitManagingResource {
.getInstance()
.createDSLContext()
+ private[resource] final class IdleComputingUnitCleanupConfig(
+ val enabled: Boolean,
+ val idleTimeoutMinutes: Long
+ ) {
+ def copy(
+ enabled: Boolean = this.enabled,
+ idleTimeoutMinutes: Long = this.idleTimeoutMinutes
+ ): IdleComputingUnitCleanupConfig =
+ new IdleComputingUnitCleanupConfig(enabled, idleTimeoutMinutes)
+ }
+
+ private[resource] trait KubernetesPodOperations {
+ val podExists: Int => Boolean
+ val deletePod: Int => Unit
+ }
+
+ private[resource] object DefaultKubernetesPodOperations extends
KubernetesPodOperations {
+ private[resource] var podExistsDelegate: Int => Boolean =
KubernetesClient.podExists
+ private[resource] var deletePodDelegate: Int => Unit =
KubernetesClient.deletePod
+
+ override val podExists: Int => Boolean = cuid => podExistsDelegate(cuid)
+ override val deletePod: Int => Unit = cuid => deletePodDelegate(cuid)
+ }
+
+ private[resource] def lastComputingUnitActivityTime(
+ unit: WorkflowComputingUnit,
+ latestUpdateTime: Option[Timestamp],
+ latestStartTime: Option[Timestamp]
+ ): Timestamp =
+ Seq(
+ latestUpdateTime,
+ latestStartTime,
+ Option(unit.getCreationTime)
+ ).flatten.maxBy(_.getTime)
+
+ private[resource] def shouldTerminateIdleComputingUnit(
+ hasActiveExecution: Boolean,
+ lastExecutionTime: Timestamp,
+ cutoff: Timestamp
+ ): Boolean =
+ !hasActiveExecution && lastExecutionTime.before(cutoff)
+
+ def terminateIdleKubernetesComputingUnits():
List[TerminatedComputingUnitInfo] =
+ terminateIdleKubernetesComputingUnits(
+ new IdleComputingUnitCleanupConfig(
+ KubernetesConfig.kubernetesComputingUnitEnabled,
+ KubernetesConfig.computingUnitIdleTimeoutMinutes
+ ),
+ () => new Timestamp(System.currentTimeMillis()),
+ DefaultKubernetesPodOperations
+ )
+
+ private[resource] def terminateIdleKubernetesComputingUnits(
+ cleanupConfig: IdleComputingUnitCleanupConfig,
+ currentTime: () => Timestamp,
+ podOperations: KubernetesPodOperations
+ ): List[TerminatedComputingUnitInfo] = {
+ if (!cleanupConfig.enabled || cleanupConfig.idleTimeoutMinutes <= 0) {
+ return List.empty
+ }
+
+ val now = currentTime()
+ val cutoff = new Timestamp(now.getTime - cleanupConfig.idleTimeoutMinutes
* 60 * 1000)
+ val activeStatuses = Seq(Short.box(0), Short.box(1), Short.box(2))
+
+ withTransaction(context) { ctx =>
Review Comment:
Done. The cleanup logic now first scans idle CU candidates in
`idleKubernetesComputingUnitCandidates`, then processes each candidate
separately in `terminateIdleKubernetesComputingUnitCandidate`. Pod deletion
happens outside the scan transaction, and the DB update for each CU uses its
own transaction. Per-unit failures are caught and logged, so one failed cleanup
does not roll back or block the rest of the batch.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]