From e47177bced72402c6ce4352e5f0a9a83f5727b20 Mon Sep 17 00:00:00 2001 From: Xuan Lu Date: Wed, 9 Sep 2026 17:27:14 +0000 Subject: [PATCH] chore(helm): bump inference-operator to v3.6 (chart 2.6.0) - Chart 2.5.0 -> 2.6.0, appVersion 3.5 -> 3.6, image tag v3.5 -> v3.6 - Regenerated IEC/JSM CRDs: add spec/status.inferenceGateway opt-in (CR-295860367) and IEC spec.kubernetes.podTemplateAnnotations (CR-299953131). Additive, backward compatible. --- helm_chart/HyperPodHelmChart/Chart.yaml | 2 +- .../charts/inference-operator/Chart.yaml | 4 +- ...s.amazon.com_inferenceendpointconfigs.yaml | 59 +++++++++++++++++++ ...emaker.aws.amazon.com_jumpstartmodels.yaml | 47 +++++++++++++++ .../charts/inference-operator/values.yaml | 2 +- 5 files changed, 110 insertions(+), 4 deletions(-) diff --git a/helm_chart/HyperPodHelmChart/Chart.yaml b/helm_chart/HyperPodHelmChart/Chart.yaml index b49e71b9..3252831a 100644 --- a/helm_chart/HyperPodHelmChart/Chart.yaml +++ b/helm_chart/HyperPodHelmChart/Chart.yaml @@ -81,7 +81,7 @@ dependencies: repository: "file://charts/team-role-and-bindings" condition: team-role-and-bindings.enabled - name: hyperpod-inference-operator - version: "2.5.0" + version: "2.6.0" repository: "file://charts/inference-operator" condition: inferenceOperators.enabled - name: hyperpod-patching diff --git a/helm_chart/HyperPodHelmChart/charts/inference-operator/Chart.yaml b/helm_chart/HyperPodHelmChart/charts/inference-operator/Chart.yaml index 8c319e01..43da4435 100644 --- a/helm_chart/HyperPodHelmChart/charts/inference-operator/Chart.yaml +++ b/helm_chart/HyperPodHelmChart/charts/inference-operator/Chart.yaml @@ -15,11 +15,11 @@ type: application # This is the chart version. This version number should be incremented each time you make changes # to the chart and its templates, including the app version. # Versions are expected to follow Semantic Versioning (https://semver.org/) -version: 2.5.0 +version: 2.6.0 # This is the version number of the application being deployed. Keep this aligned # with operator image MAJOR.MINOR version. -appVersion: "3.5" +appVersion: "3.6" dependencies: - name: aws-mountpoint-s3-csi-driver diff --git a/helm_chart/HyperPodHelmChart/charts/inference-operator/config/crd/inference.sagemaker.aws.amazon.com_inferenceendpointconfigs.yaml b/helm_chart/HyperPodHelmChart/charts/inference-operator/config/crd/inference.sagemaker.aws.amazon.com_inferenceendpointconfigs.yaml index 3431bc30..a5bcb224 100644 --- a/helm_chart/HyperPodHelmChart/charts/inference-operator/config/crd/inference.sagemaker.aws.amazon.com_inferenceendpointconfigs.yaml +++ b/helm_chart/HyperPodHelmChart/charts/inference-operator/config/crd/inference.sagemaker.aws.amazon.com_inferenceendpointconfigs.yaml @@ -620,6 +620,25 @@ spec: maxLength: 63 pattern: ^$|^[a-zA-Z0-9](-*[a-zA-Z0-9]){0,62}$ type: string + inferenceGateway: + description: |- + InferenceGateway opts this model into the shared inference gateway. When + enabled, the operator registers a scheduler entry for this model on an + InferenceGatewayConfig so it is served behind the shared gateway endpoint. + Mutually exclusive with intelligentRoutingSpec. + properties: + enabled: + default: false + description: Enabled opts this model into the shared inference + gateway. + type: boolean + name: + description: |- + Name of the gateway (InferenceGatewayConfig) this model attaches to. + Models with the same name in the same namespace share one gateway. + When empty, the operator creates a new gateway named inf-igw-. + type: string + type: object instanceType: description: |- Single instance type to deploy the model on. @@ -2493,6 +2512,18 @@ spec: - name type: object type: array + podTemplateAnnotations: + additionalProperties: + type: string + description: |- + Custom annotations to add to the pods that run this inference endpoint. They + are applied to the Deployment's pod template + (spec.template.metadata.annotations), so every pod the endpoint creates carries + them, and they are kept in sync across updates. Keys in the + inference.sagemaker.aws.amazon.com namespace (including its subdomains) and the + kubectl.kubernetes.io/last-applied-configuration key are reserved and will be + rejected. + type: object schedulerName: description: |- Name of the scheduler to use for pod scheduling. @@ -6829,6 +6860,11 @@ spec: - modelSourceConfig - worker type: object + x-kubernetes-validations: + - message: inferenceGateway.enabled and intelligentRoutingSpec.enabled + are mutually exclusive + rule: '!(has(self.inferenceGateway) && self.inferenceGateway.enabled + && has(self.intelligentRoutingSpec) && self.intelligentRoutingSpec.enabled)' status: description: ModelDeploymentStatus defines the observed state of ModelDeployment properties: @@ -7120,6 +7156,29 @@ spec: format: int32 type: integer type: object + inferenceGateway: + description: |- + Status of the shared inference gateway for this model. Present only when + spec.inferenceGateway.enable is true; mirrored from the backing + InferenceGatewayConfig. + properties: + name: + description: |- + Name is the InferenceGatewayConfig this model is wired into. When + spec.inferenceGateway.name is set it mirrors that value; when the name is + left empty the operator generates one on the first reconcile and records + it here so subsequent reconciles reuse the same gateway. + type: string + state: + description: |- + State is Pending while the gateway provisions, Ready when it can serve + traffic, and Failed on a provisioning error. + enum: + - Pending + - Ready + - Failed + type: string + type: object metricsStatus: description: Status of metrics collection properties: diff --git a/helm_chart/HyperPodHelmChart/charts/inference-operator/config/crd/inference.sagemaker.aws.amazon.com_jumpstartmodels.yaml b/helm_chart/HyperPodHelmChart/charts/inference-operator/config/crd/inference.sagemaker.aws.amazon.com_jumpstartmodels.yaml index 003df746..d05ebae5 100644 --- a/helm_chart/HyperPodHelmChart/charts/inference-operator/config/crd/inference.sagemaker.aws.amazon.com_jumpstartmodels.yaml +++ b/helm_chart/HyperPodHelmChart/charts/inference-operator/config/crd/inference.sagemaker.aws.amazon.com_jumpstartmodels.yaml @@ -619,6 +619,25 @@ spec: type: object maxItems: 100 type: array + inferenceGateway: + description: |- + InferenceGateway opts this model into the shared inference gateway. When + enabled, the operator registers a scheduler entry for this model on an + InferenceGatewayConfig so it is served behind the shared gateway endpoint. + Mutually exclusive with intelligentRoutingSpec. + properties: + enabled: + default: false + description: Enabled opts this model into the shared inference + gateway. + type: boolean + name: + description: |- + Name of the gateway (InferenceGatewayConfig) this model attaches to. + Models with the same name in the same namespace share one gateway. + When empty, the operator creates a new gateway named inf-igw-. + type: string + type: object intelligentRoutingSpec: description: |- Configuration for intelligent routing @@ -1520,6 +1539,11 @@ spec: - model - server type: object + x-kubernetes-validations: + - message: inferenceGateway.enabled and intelligentRoutingSpec.enabled + are mutually exclusive + rule: '!(has(self.inferenceGateway) && self.inferenceGateway.enabled + && has(self.intelligentRoutingSpec) && self.intelligentRoutingSpec.enabled)' status: description: ModelDeploymentStatus defines the observed state of ModelDeployment properties: @@ -1811,6 +1835,29 @@ spec: format: int32 type: integer type: object + inferenceGateway: + description: |- + Status of the shared inference gateway for this model. Present only when + spec.inferenceGateway.enable is true; mirrored from the backing + InferenceGatewayConfig. + properties: + name: + description: |- + Name is the InferenceGatewayConfig this model is wired into. When + spec.inferenceGateway.name is set it mirrors that value; when the name is + left empty the operator generates one on the first reconcile and records + it here so subsequent reconciles reuse the same gateway. + type: string + state: + description: |- + State is Pending while the gateway provisions, Ready when it can serve + traffic, and Failed on a provisioning error. + enum: + - Pending + - Ready + - Failed + type: string + type: object metricsStatus: description: Status of metrics collection properties: diff --git a/helm_chart/HyperPodHelmChart/charts/inference-operator/values.yaml b/helm_chart/HyperPodHelmChart/charts/inference-operator/values.yaml index 0c636eac..4d0994cf 100644 --- a/helm_chart/HyperPodHelmChart/charts/inference-operator/values.yaml +++ b/helm_chart/HyperPodHelmChart/charts/inference-operator/values.yaml @@ -25,7 +25,7 @@ image: ap-southeast-3: 158128612970.dkr.ecr.ap-southeast-3.amazonaws.com ap-south-2: 680458885894.dkr.ecr.ap-south-2.amazonaws.com eu-south-2: 025050981094.dkr.ecr.eu-south-2.amazonaws.com - tag: v3.5 + tag: v3.6 pullPolicy: Always repository: initContainer: