diff --git a/api/serverless/openapi.yaml b/api/serverless/openapi.yaml index c13df8b..4db8a51 100644 --- a/api/serverless/openapi.yaml +++ b/api/serverless/openapi.yaml @@ -3778,7 +3778,6 @@ components: - maxWorkers - idleTtlSecs - scalingDelaySecs - - concurrency properties: id: type: string @@ -3849,11 +3848,6 @@ components: type: integer format: int32 description: Cooldown between consecutive scaling decisions. - concurrency: - type: integer - format: int32 - default: 1 - description: Max tasks a single worker handles simultaneously. createdAt: type: string format: date-time @@ -3956,11 +3950,6 @@ components: type: integer format: int32 minimum: 1 - concurrency: - type: integer - format: int32 - minimum: 1 - default: 1 WorkerConfigPatch: type: object @@ -4042,10 +4031,6 @@ components: type: integer format: int32 minimum: 1 - concurrency: - type: integer - format: int32 - minimum: 1 Version: type: object diff --git a/docs/runware_serverless_apps_scale.md b/docs/runware_serverless_apps_scale.md index 5752248..9098a0e 100644 --- a/docs/runware_serverless_apps_scale.md +++ b/docs/runware_serverless_apps_scale.md @@ -32,7 +32,6 @@ runware serverless apps scale [flags] ``` --available-workers-pct int32 Idle-worker buffer as a percentage of load (0-100) - --concurrency int32 Max tasks a single worker handles simultaneously --fallback-gpu-type string Secondary GPU type if the preferred type is unavailable --gpu-type string Preferred GPU type ID (see 'serverless gpus') --gpus-per-worker int32 GPUs allocated per worker diff --git a/internal/api/serverless/client_test.go b/internal/api/serverless/client_test.go index 6971237..99a9a97 100644 --- a/internal/api/serverless/client_test.go +++ b/internal/api/serverless/client_test.go @@ -370,7 +370,7 @@ func TestListApps(t *testing.T) { "appId":"my-app", "appName":"My App", "status":"active", - "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"concurrency":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, + "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, "environmentVariables":[], "secrets":[], "createdAt":"2026-07-30T12:00:00Z", @@ -468,7 +468,7 @@ func TestGetApp(t *testing.T) { "appId":"my-app", "appName":"My App", "status":"initializing", - "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"concurrency":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, + "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, "environmentVariables":[], "secrets":[], "createdAt":"2026-07-30T12:00:00Z", @@ -628,7 +628,7 @@ func TestUpdateApp(t *testing.T) { "appId":"my-app", "appName":"My App", "status":"active", - "configuration":{"maxWorkers":2,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"concurrency":1,"computeType":"gpu"}, + "configuration":{"maxWorkers":2,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"computeType":"gpu"}, "environmentVariables":[], "secrets":[], "createdAt":"2026-07-30T12:00:00Z", @@ -704,7 +704,7 @@ func TestUpdateApp_AppSource(t *testing.T) { "appId":"my-app", "appName":"My App", "status":"initializing", - "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"concurrency":1,"computeType":"gpu"}, + "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"computeType":"gpu"}, "environmentVariables":[], "secrets":[], "createdAt":"2026-07-30T12:00:00Z", @@ -908,7 +908,7 @@ func lifecycleAppJSON(status string) string { "appId":"my-app", "appName":"My App", "status":"` + status + `", - "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"concurrency":1,"computeType":"gpu"}, + "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"computeType":"gpu"}, "environmentVariables":[], "secrets":[], "createdAt":"2026-07-30T12:00:00Z", @@ -1345,7 +1345,7 @@ func TestDeployVersion(t *testing.T) { "appName":"My App", "status":"initializing", "activeVersionId":"` + testVersionID + `", - "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"concurrency":1,"computeType":"gpu"}, + "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"computeType":"gpu"}, "environmentVariables":[], "secrets":[], "createdAt":"2026-07-30T12:00:00Z", @@ -1618,7 +1618,7 @@ func testAppJSON(status AppStatus) string { "appId":"my-app", "appName":"My App", "status":"` + string(status) + `", - "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"concurrency":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, + "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, "environmentVariables":[], "secrets":[], "createdAt":"2026-07-30T12:00:00Z", diff --git a/internal/api/serverless/gen/client.gen.go b/internal/api/serverless/gen/client.gen.go index 955be0a..a47bab7 100644 --- a/internal/api/serverless/gen/client.gen.go +++ b/internal/api/serverless/gen/client.gen.go @@ -1775,10 +1775,7 @@ type WorkerConfig struct { // ComputeType Worker compute class. GPU is the only supported value. CPU workloads are not supported. ComputeType ComputeType `json:"computeType"` - - // Concurrency Max tasks a single worker handles simultaneously. - Concurrency int32 `json:"concurrency"` - CreatedAt *time.Time `json:"createdAt,omitempty"` + CreatedAt *time.Time `json:"createdAt,omitempty"` // FallbackGpuType Secondary GPU type used if the preferred type is unavailable. FallbackGpuType *GpuTypeId `json:"fallbackGpuType,omitempty"` @@ -1812,7 +1809,6 @@ type WorkerConfigCreate struct { // ComputeType GPU is the only supported compute type. Omitting the field selects GPU. CPU workloads are not supported; a request that names `cpu` is rejected with 422 before a build or deploy starts. ComputeType *ComputeType `json:"computeType,omitempty"` - Concurrency *int32 `json:"concurrency,omitempty"` // FallbackGpuType Secondary GPU type used if the preferred type is unavailable. Omit, send JSON null, or send an empty string for no fallback — forms bind an unselected dropdown as `""`, which is not a `GpuTypeId`. A non-empty value must be an active catalogue code. Unlike `gpuType` this is an existence check only: it does not require admitted capacity, so the code may not appear in the customer `GET /v1/gpu-types` list. FallbackGpuType *GpuTypeIdOrEmpty `json:"fallbackGpuType,omitempty"` @@ -1835,7 +1831,6 @@ type WorkerConfigCreate struct { type WorkerConfigPatch struct { // AvailableWorkersPct Idle workers held above current demand, as a percentage of that demand, rounded up. Omit to leave unchanged; send 0 to remove the buffer. Null is refused, because omitting a field and clearing it mean different things here. AvailableWorkersPct *int32 `json:"availableWorkersPct,omitempty"` - Concurrency *int32 `json:"concurrency,omitempty"` // FallbackGpuType Secondary GPU type. Omit to leave unchanged. Send an empty string to clear — forms bind an unselected dropdown as `""`. JSON null is rejected: this field is not nullable, so a client that meant to clear must send `""` rather than null. A non-empty value must be an active catalogue code. Unlike `gpuType` this is an existence check only: it does not require admitted capacity, so the code may not appear in the customer `GET /v1/gpu-types` list. FallbackGpuType *GpuTypeIdOrEmpty `json:"fallbackGpuType,omitempty"` diff --git a/internal/cmd/serverless/apps_scale.go b/internal/cmd/serverless/apps_scale.go index 65a415f..d0ae003 100644 --- a/internal/cmd/serverless/apps_scale.go +++ b/internal/cmd/serverless/apps_scale.go @@ -19,7 +19,6 @@ type scaleFlags struct { minWorkers int32 idleTTL int32 scalingDelay int32 - concurrency int32 gpuType string gpusPerWorker int32 fallbackGPUType string @@ -87,7 +86,6 @@ func bindScaleFlags(cmd *cobra.Command, flags *scaleFlags) { f.Int32Var(&flags.minWorkers, "min-workers", 0, "Minimum number of workers (0 = scale to zero)") f.Int32Var(&flags.idleTTL, "idle-ttl", 0, "Idle TTL in seconds before scaling down") f.Int32Var(&flags.scalingDelay, "scaling-delay", 0, "Scaling delay in seconds") - f.Int32Var(&flags.concurrency, "concurrency", 0, "Max tasks a single worker handles simultaneously") f.StringVar(&flags.gpuType, "gpu-type", "", "Preferred GPU type ID (see 'serverless gpus')") f.Int32Var(&flags.gpusPerWorker, "gpus-per-worker", 0, "GPUs allocated per worker") f.StringVar(&flags.fallbackGPUType, "fallback-gpu-type", "", "Secondary GPU type if the preferred type is unavailable") @@ -104,7 +102,6 @@ func workerConfigPatchFromFlags(cmd *cobra.Command, flags scaleFlags) (*serverle MinWorkers: optionalInt32Ptr(cmd, "min-workers", flags.minWorkers), IdleTtlSecs: optionalInt32Ptr(cmd, "idle-ttl", flags.idleTTL), ScalingDelaySecs: optionalInt32Ptr(cmd, "scaling-delay", flags.scalingDelay), - Concurrency: optionalInt32Ptr(cmd, "concurrency", flags.concurrency), GpuType: optionalFlagStringPtr(cmd, "gpu-type", flags.gpuType), GpusPerWorker: optionalInt32Ptr(cmd, "gpus-per-worker", flags.gpusPerWorker), FallbackGpuType: optionalFlagStringPtr(cmd, "fallback-gpu-type", flags.fallbackGPUType), diff --git a/internal/cmd/serverless/apps_scale_test.go b/internal/cmd/serverless/apps_scale_test.go index 4ed754b..b6ec537 100644 --- a/internal/cmd/serverless/apps_scale_test.go +++ b/internal/cmd/serverless/apps_scale_test.go @@ -20,7 +20,6 @@ func TestWorkerConfigPatchFromFlags_EachFlag(t *testing.T) { {[]string{"--min-workers", "0"}, "minWorkers", float64(0)}, {[]string{"--idle-ttl", "120"}, "idleTtlSecs", float64(120)}, {[]string{"--scaling-delay", "15"}, "scalingDelaySecs", float64(15)}, - {[]string{"--concurrency", "4"}, "concurrency", float64(4)}, {[]string{"--gpu-type", testGPUType}, "gpuType", testGPUType}, {[]string{"--gpus-per-worker", "2"}, "gpusPerWorker", float64(2)}, {[]string{"--fallback-gpu-type", testGPUType}, "fallbackGpuType", testGPUType}, diff --git a/internal/cmd/serverless/deploy_endpoints_test.go b/internal/cmd/serverless/deploy_endpoints_test.go index cb022ce..1e3842d 100644 --- a/internal/cmd/serverless/deploy_endpoints_test.go +++ b/internal/cmd/serverless/deploy_endpoints_test.go @@ -25,7 +25,7 @@ func activeAppBody(activeVersionID string) string { } return fmt.Sprintf(`{ "appId":"my-app","appName":"My App","status":"active","activeVersionId":%s, - "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"concurrency":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, + "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, "environmentVariables":[],"secrets":[], "createdAt":"2026-07-30T12:00:00Z","updatedAt":"2026-07-30T12:00:00Z" }`, pin) diff --git a/internal/cmd/serverless/deploy_test.go b/internal/cmd/serverless/deploy_test.go index 00eb34b..60010fa 100644 --- a/internal/cmd/serverless/deploy_test.go +++ b/internal/cmd/serverless/deploy_test.go @@ -230,7 +230,7 @@ func TestExistingApp(t *testing.T) { "appId":"my-app", "appName":"My App", "status":"active", - "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"concurrency":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, + "configuration":{"maxWorkers":1,"idleTtlSecs":60,"scalingDelaySecs":10,"minWorkers":0,"gpusPerWorker":1,"gracefulStopTtlSecs":120,"computeType":"gpu"}, "environmentVariables":[], "secrets":[], "createdAt":"2026-07-30T12:00:00Z", diff --git a/internal/cmd/serverless/display.go b/internal/cmd/serverless/display.go index 1f6e8cd..838769b 100644 --- a/internal/cmd/serverless/display.go +++ b/internal/cmd/serverless/display.go @@ -47,7 +47,6 @@ const ( colAvailableWorkersPct = "Available workers %" colIdleTTL = "Idle TTL (s)" colScalingDelay = "Scaling delay (s)" - colConcurrency = "Concurrency" colEffectiveMaxWorkers = "Effective max workers" colActiveWorkers = "Active workers" colQueueDepth = "Queue depth" @@ -87,7 +86,6 @@ func (r appResult) Rows() [][]any { {colAvailableWorkersPct, formatOptionalInt32(cfg.AvailableWorkersPct)}, {colIdleTTL, cfg.IdleTtlSecs}, {colScalingDelay, cfg.ScalingDelaySecs}, - {colConcurrency, cfg.Concurrency}, {colEffectiveMaxWorkers, formatOptionalInt32(r.EffectiveMaxWorkers)}, {colActiveWorkers, r.Runtime.ActiveWorkers}, {colQueueDepth, formatOptionalInt64(r.Runtime.QueueDepth)}, diff --git a/internal/cmd/serverless/display_test.go b/internal/cmd/serverless/display_test.go index b342855..80ff968 100644 --- a/internal/cmd/serverless/display_test.go +++ b/internal/cmd/serverless/display_test.go @@ -328,7 +328,6 @@ func TestAppResult_IncludesConfiguration(t *testing.T) { MaxWorkers: 2, IdleTtlSecs: 60, ScalingDelaySecs: 10, - Concurrency: 1, }, } if got := r.Headers(); len(got) != 2 || got[0] != colField || got[1] != colValue { @@ -353,7 +352,6 @@ func TestAppResult_IncludesConfiguration(t *testing.T) { colAvailableWorkersPct: "", colIdleTTL: int32(60), colScalingDelay: int32(10), - colConcurrency: int32(1), colEffectiveMaxWorkers: "", colActiveWorkers: int64(0), colQueueDepth: "",