zhengcanbin commented on a change in pull request #11323: [FLINK-16439][k8s]
Make KubernetesResourceManager starts workers using WorkerResourceSpec
requested by SlotManager
URL: https://github.com/apache/flink/pull/11323#discussion_r409991098
##########
File path:
flink-kubernetes/src/main/java/org/apache/flink/kubernetes/KubernetesResourceManager.java
##########
@@ -237,57 +230,73 @@ private void recoverWorkerNodesFromPreviousAttempts()
throws ResourceManagerExce
++currentMaxAttemptId);
}
- private void requestKubernetesPod() {
- numPendingPodRequests++;
+ private void requestKubernetesPod(WorkerResourceSpec
workerResourceSpec) {
+ final KubernetesTaskManagerParameters parameters =
+
createKubernetesTaskManagerParameters(workerResourceSpec);
+
+ final KubernetesPod taskManagerPod =
+
KubernetesTaskManagerFactory.createTaskManagerComponent(parameters);
+ kubeClient.createTaskManagerPod(taskManagerPod);
+
+ podWorkerResources.put(parameters.getPodName(),
workerResourceSpec);
+ final int pendingWorkerNum =
notifyNewWorkerRequested(workerResourceSpec);
log.info("Requesting new TaskManager pod with <{},{}>. Number
pending requests {}.",
- defaultMemoryMB,
- defaultCpus,
- numPendingPodRequests);
+ parameters.getTaskManagerMemoryMB(),
+ parameters.getTaskManagerCPU(),
+ pendingWorkerNum);
+ log.info("TaskManager {} will be started with {}.",
parameters.getPodName(), workerResourceSpec);
+ }
+
+ private KubernetesTaskManagerParameters
createKubernetesTaskManagerParameters(WorkerResourceSpec workerResourceSpec) {
+ final TaskExecutorProcessSpec taskExecutorProcessSpec =
+
TaskExecutorProcessUtils.processSpecFromWorkerResourceSpec(flinkConfig,
workerResourceSpec);
final String podName = String.format(
TASK_MANAGER_POD_FORMAT,
clusterId,
currentMaxAttemptId,
++currentMaxPodId);
+ final ContaineredTaskManagerParameters taskManagerParameters =
+ ContaineredTaskManagerParameters.create(flinkConfig,
taskExecutorProcessSpec);
+
final String dynamicProperties =
BootstrapTools.getDynamicPropertiesAsString(flinkClientConfig, flinkConfig);
- final KubernetesTaskManagerParameters
kubernetesTaskManagerParameters = new KubernetesTaskManagerParameters(
+ return new KubernetesTaskManagerParameters(
flinkConfig,
podName,
dynamicProperties,
taskManagerParameters);
-
- final KubernetesPod taskManagerPod =
-
KubernetesTaskManagerFactory.createTaskManagerComponent(kubernetesTaskManagerParameters);
-
- log.info("TaskManager {} will be started with {}.", podName,
taskExecutorProcessSpec);
- kubeClient.createTaskManagerPod(taskManagerPod);
}
/**
* Request new pod if pending pods cannot satisfy pending slot requests.
*/
- private void requestKubernetesPodIfRequired() {
- final int requiredTaskManagers =
getNumberRequiredTaskManagers();
+ private void requestKubernetesPodIfRequired(WorkerResourceSpec
workerResourceSpec) {
+ final int pendingWorkerNum =
getNumPendingWorkersFor(workerResourceSpec);
+ int requiredTaskManagers =
getRequiredResources().get(workerResourceSpec);
- while (requiredTaskManagers > numPendingPodRequests) {
- requestKubernetesPod();
+ while (requiredTaskManagers-- > pendingWorkerNum) {
+ requestKubernetesPod(workerResourceSpec);
}
}
private void removePodIfTerminated(KubernetesPod pod) {
if (pod.isTerminated()) {
kubeClient.stopPod(pod.getName());
Review comment:
> So this means that `KubernetesResourceManager.onError` will only be called
if `onAdded` has been called before? I guess this is also a question for
@wangyang0918.
There is no guarantee for this, a case in K8s client-go:
`ERROR` could be thrown in
https://github.com/kubernetes/kubernetes/blob/343c1e7636fe5c75cdd378c0b170b26935806de5/staging/src/k8s.io/apimachinery/pkg/watch/streamwatcher.go#L121
It seems that the `ERROR` is an HTTP exception that rarely happens and we
shouldn't rely on it to handle Pod failure. Once a Pod is terminated, whether
or not exited normally, it is expected we would receive a `MODIFIED` instead of
`ERROR` event.
----------------------------------------------------------------
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
For queries about this service, please contact Infrastructure at:
[email protected]
With regards,
Apache Git Services