diff --git a/.github/workflows/CI.yml b/.github/workflows/CI.yml index 10b57780..90a3bb2c 100644 --- a/.github/workflows/CI.yml +++ b/.github/workflows/CI.yml @@ -211,9 +211,9 @@ jobs: location=${{ env.AZURE_LOCATION }} \ azureAiServiceLocation=${{ env.AZURE_LOCATION }} \ deploymentType="GlobalStandard" \ - gptModelName="gpt-4.1-mini" \ + gptModelName="gpt-5-mini" \ gptDeploymentCapacity=${{ env.GPT_CAPACITY }} \ - gptModelVersion="2025-04-14" \ + gptModelVersion="2025-08-07" \ embeddingModelName="text-embedding-3-large" \ embeddingDeploymentCapacity=${{ env.TEXT_EMBEDDING_CAPACITY }} \ embeddingModelVersion="1" \ diff --git a/Deployment/checkquota.ps1 b/Deployment/checkquota.ps1 index c16a0b85..b8265ad9 100644 --- a/Deployment/checkquota.ps1 +++ b/Deployment/checkquota.ps1 @@ -41,7 +41,7 @@ Write-Host "✅ Azure subscription set successfully." # Define models and their minimum required capacities $MIN_CAPACITY = @{ - "OpenAI.GlobalStandard.gpt4.1-mini" = $GPT_MIN_CAPACITY + "OpenAI.GlobalStandard.gpt-5-mini" = $GPT_MIN_CAPACITY "OpenAI.GlobalStandard.text-embedding-3-large" = $TEXT_EMBEDDING_MIN_CAPACITY } diff --git a/Deployment/quota_check_params.sh b/Deployment/quota_check_params.sh index 28b78981..0e0c7574 100644 --- a/Deployment/quota_check_params.sh +++ b/Deployment/quota_check_params.sh @@ -47,7 +47,7 @@ log_verbose() { } # Default Models and Capacities (Comma-separated in "model:capacity" format) -DEFAULT_MODEL_CAPACITY="gpt4.1-mini:150,text-embedding-3-large:100" +DEFAULT_MODEL_CAPACITY="gpt-5-mini:150,text-embedding-3-large:100" # Convert the comma-separated string into an array IFS=',' read -r -a MODEL_CAPACITY_PAIRS <<< "$DEFAULT_MODEL_CAPACITY" diff --git a/docs/AVMPostDeploymentGuide.md b/docs/AVMPostDeploymentGuide.md index 3a92ddaa..cd3539ea 100644 --- a/docs/AVMPostDeploymentGuide.md +++ b/docs/AVMPostDeploymentGuide.md @@ -147,7 +147,7 @@ Upon successful completion, you'll see a success message with important informat | Model Name | Recommended TPM | Minimum TPM | |------------------------|----------------|-------------| -| gpt-4.1-mini | 100K TPM | 10K TPM | +| gpt-5-mini | 100K TPM | 10K TPM | | text-embedding-3-large | 200K TPM | 50K TPM | > **⚠️ Warning**: Insufficient quota will cause failures during document upload and processing. Ensure adequate capacity before proceeding. diff --git a/docs/CustomizingAzdParameters.md b/docs/CustomizingAzdParameters.md index eeed5166..e0abc960 100644 --- a/docs/CustomizingAzdParameters.md +++ b/docs/CustomizingAzdParameters.md @@ -12,9 +12,9 @@ By default this template will use the environment name as the prefix to prevent | `AZURE_LOCATION` | string | `` | Location of the Azure resources. Controls where the infrastructure will be deployed. | | `AZURE_ENV_AI_SERVICE_LOCATION` | string | `` | Location for Azure OpenAI resources. Can be different from AZURE_LOCATION for optimized AI service placement. | | `AZURE_ENV_MODEL_DEPLOYMENT_TYPE` | string | `GlobalStandard` | Defines the deployment type for the AI model (e.g., Standard, GlobalStandard). | -| `AZURE_ENV_GPT_MODEL_NAME` | string | `gpt-4.1-mini` | Specifies the name of the GPT model to be deployed. | +| `AZURE_ENV_GPT_MODEL_NAME` | string | `gpt-5-mini` | Specifies the name of the GPT model to be deployed. | | `AZURE_ENV_GPT_MODEL_CAPACITY` | int | `100` | Sets the GPT model capacity (in thousands of tokens per minute). | -| `AZURE_ENV_GPT_MODEL_VERSION` | string | `2025-04-14` | Version of the GPT model to be used for deployment. | +| `AZURE_ENV_GPT_MODEL_VERSION` | string | `2025-08-07` | Version of the GPT model to be used for deployment. | | `AZURE_ENV_EMBEDDING_MODEL_NAME` | string | `text-embedding-3-large` | Sets the name of the embedding model to use. | | `AZURE_ENV_EMBEDDING_MODEL_VERSION` | string | `1` | Version of the embedding model to be used for deployment. | | `AZURE_ENV_EMBEDDING_DEPLOYMENT_CAPACITY` | int | `100` | Capacity for embedding model deployment (in thousands of tokens per minute). | diff --git a/docs/QuotaCheck.md b/docs/QuotaCheck.md index ca0f5e01..973317d6 100644 --- a/docs/QuotaCheck.md +++ b/docs/QuotaCheck.md @@ -1,7 +1,7 @@ ## Check Quota Availability Before Deployment Before deploying the accelerator, **ensure sufficient quota availability** for the required model. -> **For Global Standard | gpt4.1-mini - increase the capacity to at least 150K tokens for optimal performance.** +> **For Global Standard | gpt-5-mini - increase the capacity to at least 150K tokens for optimal performance.** ### Login if you have not done so already ``` @@ -11,7 +11,7 @@ azd auth login ### 📌 Default Models & Capacities: ``` -gpt4.1-mini:150, text-embedding-3-large:100 +gpt-5-mini:150, text-embedding-3-large:100 ``` ### 📌 Default Regions: ``` @@ -37,7 +37,7 @@ eastus, uksouth, eastus2, northcentralus, swedencentral, westus, westus2, southc ``` ✔️ Check specific model(s) in default regions: ``` - ./quota_check_params.sh --models gpt4.1-mini:150,text-embedding-3-large:100 + ./quota_check_params.sh --models gpt-5-mini:150,text-embedding-3-large:100 ``` ✔️ Check default models in specific region(s): ``` @@ -45,11 +45,11 @@ eastus, uksouth, eastus2, northcentralus, swedencentral, westus, westus2, southc ``` ✔️ Passing Both models and regions: ``` - ./quota_check_params.sh --models gpt4.1-mini:150 --regions eastus,westus2 + ./quota_check_params.sh --models gpt-5-mini:150 --regions eastus,westus2 ``` ✔️ All parameters combined: ``` - ./quota_check_params.sh --models gpt4.1-mini:150,text-embedding-3-large:100 --regions eastus,westus --verbose + ./quota_check_params.sh --models gpt-5-mini:150,text-embedding-3-large:100 --regions eastus,westus --verbose ``` ### **Sample Output** diff --git a/infra/main.bicep b/infra/main.bicep index 7fb16a4d..59381049 100644 --- a/infra/main.bicep +++ b/infra/main.bicep @@ -34,12 +34,12 @@ param deploymentType string = 'GlobalStandard' @minLength(1) @description('Optional. Name of the GPT model to deploy:') @allowed([ - 'gpt-4.1-mini' + 'gpt-5-mini' ]) -param gptModelName string = 'gpt-4.1-mini' +param gptModelName string = 'gpt-5-mini' @description('Optional. Version of the GPT model to deploy.') -param gptModelVersion string = '2025-04-14' +param gptModelVersion string = '2025-08-07' @description('Optional. Capacity of the GPT model deployment:') @minValue(10) @@ -95,7 +95,7 @@ param enableScalability bool = false azd: { type: 'location' usageName: [ - 'OpenAI.GlobalStandard.gpt4.1-mini,150' + 'OpenAI.GlobalStandard.gpt-5-mini,150' 'OpenAI.GlobalStandard.text-embedding-3-large,100' ] } @@ -984,6 +984,9 @@ module managedCluster 'br/public:avm/res/container-service/managed-cluster:0.13. minCount: 1 maxCount: 2 + // Disable zonal placement; not all regions/subscriptions expose availability zones for the agent pool SKU + availabilityZones: [] + // WAF aligned configuration for Private Networking enableAutoScaling: true scaleSetEvictionPolicy: 'Delete' diff --git a/infra/main.json b/infra/main.json index df6a00d0..895bf0dd 100644 --- a/infra/main.json +++ b/infra/main.json @@ -6,7 +6,7 @@ "_generator": { "name": "bicep", "version": "0.43.8.12551", - "templateHash": "9930092765515882543" + "templateHash": "5797490851931362105" } }, "parameters": { @@ -48,9 +48,9 @@ }, "gptModelName": { "type": "string", - "defaultValue": "gpt-4.1-mini", + "defaultValue": "gpt-5-mini", "allowedValues": [ - "gpt-4.1-mini" + "gpt-5-mini" ], "minLength": 1, "metadata": { @@ -59,7 +59,7 @@ }, "gptModelVersion": { "type": "string", - "defaultValue": "2025-04-14", + "defaultValue": "2025-08-07", "metadata": { "description": "Optional. Version of the GPT model to deploy." } @@ -177,7 +177,7 @@ "azd": { "type": "location", "usageName": [ - "OpenAI.GlobalStandard.gpt4.1-mini,150", + "OpenAI.GlobalStandard.gpt-5-mini,150", "OpenAI.GlobalStandard.text-embedding-3-large,100" ] }, @@ -43817,8 +43817,8 @@ } }, "dependsOn": [ - "[format('avmPrivateDnsZones[{0}]', variables('dnsZoneIndex').storageBlob)]", "[format('avmPrivateDnsZones[{0}]', variables('dnsZoneIndex').storageQueue)]", + "[format('avmPrivateDnsZones[{0}]', variables('dnsZoneIndex').storageBlob)]", "userAssignedIdentity", "virtualNetwork" ] @@ -52360,6 +52360,7 @@ "type": "VirtualMachineScaleSets", "minCount": 1, "maxCount": 2, + "availabilityZones": [], "enableAutoScaling": true, "scaleSetEvictionPolicy": "Delete", "scaleSetPriority": "Regular",