Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion helm_chart/HyperPodHelmChart/Chart.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -81,7 +81,7 @@ dependencies:
repository: "file://charts/team-role-and-bindings"
condition: team-role-and-bindings.enabled
- name: hyperpod-inference-operator
version: "2.5.0"
version: "2.6.0"
repository: "file://charts/inference-operator"
condition: inferenceOperators.enabled
- name: hyperpod-patching
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -15,11 +15,11 @@ type: application
# This is the chart version. This version number should be incremented each time you make changes
# to the chart and its templates, including the app version.
# Versions are expected to follow Semantic Versioning (https://semver.org/)
version: 2.5.0
version: 2.6.0

# This is the version number of the application being deployed. Keep this aligned
# with operator image MAJOR.MINOR version.
appVersion: "3.5"
appVersion: "3.6"

dependencies:
- name: aws-mountpoint-s3-csi-driver
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -620,6 +620,25 @@ spec:
maxLength: 63
pattern: ^$|^[a-zA-Z0-9](-*[a-zA-Z0-9]){0,62}$
type: string
inferenceGateway:
description: |-
InferenceGateway opts this model into the shared inference gateway. When
enabled, the operator registers a scheduler entry for this model on an
InferenceGatewayConfig so it is served behind the shared gateway endpoint.
Mutually exclusive with intelligentRoutingSpec.
properties:
enabled:
default: false
description: Enabled opts this model into the shared inference
gateway.
type: boolean
name:
description: |-
Name of the gateway (InferenceGatewayConfig) this model attaches to.
Models with the same name in the same namespace share one gateway.
When empty, the operator creates a new gateway named inf-igw-<uuid>.
type: string
type: object
instanceType:
description: |-
Single instance type to deploy the model on.
Expand Down Expand Up @@ -2493,6 +2512,18 @@ spec:
- name
type: object
type: array
podTemplateAnnotations:
additionalProperties:
type: string
description: |-
Custom annotations to add to the pods that run this inference endpoint. They
are applied to the Deployment's pod template
(spec.template.metadata.annotations), so every pod the endpoint creates carries
them, and they are kept in sync across updates. Keys in the
inference.sagemaker.aws.amazon.com namespace (including its subdomains) and the
kubectl.kubernetes.io/last-applied-configuration key are reserved and will be
rejected.
type: object
schedulerName:
description: |-
Name of the scheduler to use for pod scheduling.
Expand Down Expand Up @@ -6829,6 +6860,11 @@ spec:
- modelSourceConfig
- worker
type: object
x-kubernetes-validations:
- message: inferenceGateway.enabled and intelligentRoutingSpec.enabled
are mutually exclusive
rule: '!(has(self.inferenceGateway) && self.inferenceGateway.enabled
&& has(self.intelligentRoutingSpec) && self.intelligentRoutingSpec.enabled)'
status:
description: ModelDeploymentStatus defines the observed state of ModelDeployment
properties:
Expand Down Expand Up @@ -7120,6 +7156,29 @@ spec:
format: int32
type: integer
type: object
inferenceGateway:
description: |-
Status of the shared inference gateway for this model. Present only when
spec.inferenceGateway.enable is true; mirrored from the backing
InferenceGatewayConfig.
properties:
name:
description: |-
Name is the InferenceGatewayConfig this model is wired into. When
spec.inferenceGateway.name is set it mirrors that value; when the name is
left empty the operator generates one on the first reconcile and records
it here so subsequent reconciles reuse the same gateway.
type: string
state:
description: |-
State is Pending while the gateway provisions, Ready when it can serve
traffic, and Failed on a provisioning error.
enum:
- Pending
- Ready
- Failed
type: string
type: object
metricsStatus:
description: Status of metrics collection
properties:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -619,6 +619,25 @@ spec:
type: object
maxItems: 100
type: array
inferenceGateway:
description: |-
InferenceGateway opts this model into the shared inference gateway. When
enabled, the operator registers a scheduler entry for this model on an
InferenceGatewayConfig so it is served behind the shared gateway endpoint.
Mutually exclusive with intelligentRoutingSpec.
properties:
enabled:
default: false
description: Enabled opts this model into the shared inference
gateway.
type: boolean
name:
description: |-
Name of the gateway (InferenceGatewayConfig) this model attaches to.
Models with the same name in the same namespace share one gateway.
When empty, the operator creates a new gateway named inf-igw-<uuid>.
type: string
type: object
intelligentRoutingSpec:
description: |-
Configuration for intelligent routing
Expand Down Expand Up @@ -1520,6 +1539,11 @@ spec:
- model
- server
type: object
x-kubernetes-validations:
- message: inferenceGateway.enabled and intelligentRoutingSpec.enabled
are mutually exclusive
rule: '!(has(self.inferenceGateway) && self.inferenceGateway.enabled
&& has(self.intelligentRoutingSpec) && self.intelligentRoutingSpec.enabled)'
status:
description: ModelDeploymentStatus defines the observed state of ModelDeployment
properties:
Expand Down Expand Up @@ -1811,6 +1835,29 @@ spec:
format: int32
type: integer
type: object
inferenceGateway:
description: |-
Status of the shared inference gateway for this model. Present only when
spec.inferenceGateway.enable is true; mirrored from the backing
InferenceGatewayConfig.
properties:
name:
description: |-
Name is the InferenceGatewayConfig this model is wired into. When
spec.inferenceGateway.name is set it mirrors that value; when the name is
left empty the operator generates one on the first reconcile and records
it here so subsequent reconciles reuse the same gateway.
type: string
state:
description: |-
State is Pending while the gateway provisions, Ready when it can serve
traffic, and Failed on a provisioning error.
enum:
- Pending
- Ready
- Failed
type: string
type: object
metricsStatus:
description: Status of metrics collection
properties:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,7 @@ image:
ap-southeast-3: 158128612970.dkr.ecr.ap-southeast-3.amazonaws.com
ap-south-2: 680458885894.dkr.ecr.ap-south-2.amazonaws.com
eu-south-2: 025050981094.dkr.ecr.eu-south-2.amazonaws.com
tag: v3.5
tag: v3.6
pullPolicy: Always
repository:
initContainer:
Expand Down
Loading