Skip to content

Nightly - PD Disaggregation E2E (GKE TPU) #47

Nightly - PD Disaggregation E2E (GKE TPU)

Nightly - PD Disaggregation E2E (GKE TPU) #47

name: Nightly - PD Disaggregation E2E (GKE TPU)
on:
workflow_dispatch:
inputs:
decode_pods:
description: 'Number of decode (vllm "standalone") pods (default "auto", means what was configured on the scenario)'
required: false
type: 'string'
default: 'auto'
prefill_pods:
description: 'Number of prefill pods (default "auto", means what was configured on the scenario)'
required: false
type: 'string'
default: 'auto'
gateway_class:
description: 'Class of gateway used ("istio", "agentgateway", "epponly" (i.e., "standalone"), "none")'
required: false
default: 'epponly'
monitoring_enabled:
description: 'Enabled monitoring ("true", "false")'
required: true
type: 'string'
default: 'false'
dry_run:
description: 'Execute workflow in "dry-run" mode ("true", "false")'
required: false
default: 'false'
verbose:
description: 'Execute workflow in "verbose" mode ("true", "false")'
required: false
default: 'false'
type: 'string'
harness:
description: 'Harness to be used during "run" operation ("inference-perf", "guidellm", "inferencemax", "vllm-benchmark", "nop")'
required: false
default: 'inference-perf'
type: 'string'
workload:
description: 'Workload profile to be used during "run" operation (check list under "workload/profiles")'
required: false
default: 'sanity_random.yaml'
# default: 'guide_pd-disaggregation_1.yaml'
type: 'string'
cleanup:
description: 'Cleanup the llm-d stack stood up by this workflow'
required: false
default: 'true'
type: string
update_badge:
description: 'Update the badge on gh-pages ("true", "false")'
required: false
default: 'true'
type: string
tpu_chips:
description: 'Total TPU chips to request (0 for simulated)'
required: false
type: 'string'
default: '16'
tpu_topology:
description: 'GKE TPU topology (e.g. 2x4, 2x2x1)'
required: false
type: 'string'
default: '2x4'
# push:
# branches:
# - main
schedule:
- cron: '0 3 * * *'
permissions:
contents: write
actions: read
concurrency:
group: nightly-e2e-pd-disaggregation-gke-tpu
cancel-in-progress: true
jobs:
nightly:
uses: llm-d/llm-d-infra/.github/workflows/reusable-ci-nightly-benchmark.yaml@main
with:
scenario_dir: ${{ inputs.scenario_dir || 'guides' }}
standup_method: ${{ inputs.standup_method || 'kustomize' }}
standup_scenario: ${{ inputs.standup_scenario || 'pd-disaggregation' }}
decode_pods: ${{ inputs.decode_pods || 'auto' }}
prefill_pods: ${{ inputs.prefill_pods || 'auto' }}
accelerator_type: ${{ inputs.accelerator_type || 'tpu/v6' }}
backend_type: ${{ inputs.backend_type || 'vllm' }}
infra_provider: ${{ inputs.infra_provider || '' }}
connector: ${{ inputs.connector || '' }}
cluster_namespace: ${{ inputs.cluster_namespace || 'llm-d-nightly-pd-disaggregation-gke-tpu' }}
helm_release: ${{ inputs.helm_release || 'llmdbenchcicdr-gke' }}
workspace_dir: ${{ inputs.workspace_dir || '/tmp/llmdbenchcicdk-gke-tpu' }}
bucket_project: ${{ inputs.bucket_project || 'llm-d-scale' }}
bucket_provider: ${{ inputs.bucket_provider || 'gcs' }}
bucket_path: ${{ inputs.bucket_path || 'llm-d-benchmarks/regressions/pd-disaggregation' }}
gateway_class: ${{ inputs.gateway_class || 'epponly' }}
monitoring_enabled: ${{ inputs.monitoring_enabled || 'false' }}
dry_run: ${{ inputs.dry_run || 'false' }}
verbose: ${{ inputs.verbose || 'false' }}
harness: ${{ inputs.harness || 'inference-perf' }}
workload: ${{ inputs.workload || 'sanity_random.yaml' }}
# workload: ${{ inputs.workload || 'guide_pd_disaggregation_1.yaml' }}
cleanup: ${{ inputs.cleanup || 'true' }}
tpu_chips: ${{ inputs.tpu_chips || '16' }}
tpu_topology: ${{ inputs.tpu_topology || '2x4' }}
data_access_timeout: '3600'
wait_timeout: '3600'
job_timeout: '3600s'
secrets: inherit
update-badge:
needs: [nightly]
if: always() && inputs.update_badge != 'false'
uses: llm-d/llm-d-infra/.github/workflows/reusable-update-badge.yaml@main
with:
badge_name: pd-disaggregation-gke-tpu
badge_label: "VLLM TPU"
result: ${{ needs.nightly.result }}
dry_run: ${{ inputs.dry_run || 'false' }}
failure_category: ${{ needs.nightly.outputs.failure_category }}