feat: add provider-key concurrent limit (#352)

* feat: add provider-key concurrent limit

* Fix provider key concurrent limit checks

---------

Co-authored-by: fawney19 <[email protected]>
This commit is contained in:
Kayphoon
2026-05-03 01:55:23 +08:00
committed by GitHub
co-authored by fawney19
parent 11c5884d4f
commit fe27fb17fb
23 changed files with 1827 additions and 62 deletions
@@ -25,6 +25,8 @@ pub(crate) struct AdminProviderKeyCreateRequest {
#[serde(default)]
pub(crate) rpm_limit: Option<u32>,
#[serde(default)]
pub(crate) concurrent_limit: Option<i32>,
#[serde(default)]
pub(crate) allowed_models: Option<Vec<String>>,
#[serde(default)]
pub(crate) capabilities: Option<serde_json::Value>,
@@ -69,6 +71,8 @@ pub(crate) struct AdminProviderKeyUpdateRequest {
#[serde(default)]
pub(crate) rpm_limit: Option<u32>,
#[serde(default)]
pub(crate) concurrent_limit: Option<i32>,
#[serde(default)]
pub(crate) allowed_models: Option<Vec<String>>,
#[serde(default)]
pub(crate) capabilities: Option<serde_json::Value>,
@@ -9,6 +9,7 @@ use crate::handlers::admin::shared::{
decrypt_catalog_secret_with_fallbacks, encrypt_catalog_secret_with_fallbacks,
normalize_json_object, normalize_string_list, parse_catalog_auth_config_json,
};
use crate::handlers::shared::normalize_optional_api_key_concurrent_limit;
use aether_data_contracts::repository::provider_catalog::{
StoredProviderCatalogKey, StoredProviderCatalogProvider,
};
@@ -178,6 +179,7 @@ pub(crate) async fn build_admin_create_provider_key_record(
.filter(|value| !value.is_empty());
key.internal_priority = payload.internal_priority.unwrap_or(50);
key.rpm_limit = payload.rpm_limit;
key.concurrent_limit = normalize_optional_api_key_concurrent_limit(payload.concurrent_limit)?;
key.cache_ttl_minutes = payload.cache_ttl_minutes.unwrap_or(5);
key.max_probe_interval_minutes = payload.max_probe_interval_minutes.unwrap_or(32);
key.request_count = Some(0);
@@ -9,6 +9,7 @@ use crate::handlers::admin::shared::{
decrypt_catalog_secret_with_fallbacks, encrypt_catalog_secret_with_fallbacks, json_string_list,
normalize_json_object, normalize_string_list, parse_catalog_auth_config_json,
};
use crate::handlers::shared::normalize_optional_api_key_concurrent_limit;
use crate::provider_key_auth::provider_key_is_oauth_managed;
use aether_data_contracts::repository::provider_catalog::{
StoredProviderCatalogKey, StoredProviderCatalogProvider,
@@ -277,6 +278,10 @@ pub(crate) async fn build_admin_update_provider_key_record(
updated.learned_rpm_limit = None;
}
}
if fields.contains("concurrent_limit") {
updated.concurrent_limit =
normalize_optional_api_key_concurrent_limit(payload.concurrent_limit)?;
}
if fields.contains("allowed_models") {
updated.allowed_models =
normalize_string_list(payload.allowed_models).map(|value| json!(value));
@@ -1351,6 +1351,7 @@ pub(crate) fn build_admin_provider_key_response(
json!(key.global_priority_by_format),
);
payload.insert("rpm_limit".to_string(), json!(key.rpm_limit));
payload.insert("concurrent_limit".to_string(), json!(key.concurrent_limit));
payload.insert(
"allowed_models".to_string(),
serde_json::Value::Array(
@@ -1,5 +1,6 @@
use aether_scheduler_core::{
build_scheduler_affinity_cache_key_for_api_key_id, SchedulerAffinityTarget,
build_scheduler_affinity_cache_key_for_api_key_id, candidate_affinity_hash, candidate_key,
matches_affinity_target, SchedulerAffinityTarget,
};
use crate::data::auth::GatewayAuthApiKeySnapshot;
@@ -21,6 +22,20 @@ pub(super) fn build_scheduler_affinity_cache_key(
build_scheduler_affinity_cache_key_for_api_key_id(api_key_id, api_format, global_model_name)
}
pub(super) fn scheduler_candidate_affinity_hash(
affinity_key: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> u64 {
candidate_affinity_hash(affinity_key, candidate)
}
pub(super) fn scheduler_candidate_matches_affinity_target(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
target: &SchedulerAffinityTarget,
) -> bool {
matches_affinity_target(candidate, target)
}
#[cfg_attr(not(test), allow(dead_code))]
pub(super) fn remember_scheduler_affinity(
affinity_cache_key: Option<&str>,
@@ -30,13 +45,14 @@ pub(super) fn remember_scheduler_affinity(
let Some(cache_key) = affinity_cache_key else {
return;
};
let (provider_id, endpoint_id, key_id) = candidate_key(candidate);
state.remember_scheduler_affinity_target(
cache_key,
SchedulerAffinityTarget {
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
provider_id,
endpoint_id,
key_id,
},
SCHEDULER_AFFINITY_TTL,
SCHEDULER_AFFINITY_MAX_ENTRIES,
@@ -1,12 +1,15 @@
use aether_scheduler_core::{
apply_scheduler_candidate_ranking, candidate_affinity_hash,
effective_provider_key_health_score, matches_affinity_target, provider_key_health_bucket,
requested_capability_priority_for_candidate, SchedulerAffinityTarget,
SchedulerRankableCandidate, SchedulerRankingContext, SchedulerRankingMode,
apply_scheduler_candidate_ranking, effective_provider_key_health_score,
provider_key_health_bucket, requested_capability_priority_for_candidate,
SchedulerAffinityTarget, SchedulerRankableCandidate, SchedulerRankingContext,
SchedulerRankingMode,
};
use crate::scheduler::config::{SchedulerOrderingConfig, SchedulerSchedulingMode};
use super::affinity::{
scheduler_candidate_affinity_hash, scheduler_candidate_matches_affinity_target,
};
use super::runtime::CandidateRuntimeSelectionSnapshot;
use super::SchedulerMinimalCandidateSelectionCandidate;
@@ -31,12 +34,12 @@ pub(super) fn rank_scheduler_candidates(
required_capabilities,
candidate,
))
.with_cached_affinity_match(
cached_affinity_target
.is_some_and(|target| matches_affinity_target(candidate, target)),
)
.with_cached_affinity_match(cached_affinity_target.is_some_and(|target| {
scheduler_candidate_matches_affinity_target(candidate, target)
}))
.with_affinity_hash(
priority_affinity_key.map(|key| candidate_affinity_hash(key, candidate)),
priority_affinity_key
.map(|key| scheduler_candidate_affinity_hash(key, candidate)),
)
.with_health(
provider_key.and_then(|key| {
@@ -5,10 +5,13 @@ use aether_data::repository::candidate_selection::InMemoryMinimalCandidateSelect
use aether_data::repository::candidates::InMemoryRequestCandidateRepository;
use aether_data::repository::provider_catalog::InMemoryProviderCatalogReadRepository;
use aether_data::repository::quota::InMemoryProviderQuotaRepository;
use aether_data_contracts::repository::candidate_selection::StoredProviderModelMapping;
use aether_data_contracts::repository::candidate_selection::{
StoredMinimalCandidateSelectionRow, StoredProviderModelMapping,
};
use aether_data_contracts::repository::candidates::{
RequestCandidateStatus, StoredRequestCandidate,
};
use aether_data_contracts::repository::provider_catalog::StoredProviderCatalogKey;
use aether_data_contracts::repository::quota::StoredProviderQuotaSnapshot;
use aether_scheduler_core::SchedulerMinimalCandidateSelectionCandidate;
use serde_json::json;
@@ -99,6 +102,102 @@ async fn collect_selectable_candidates_with_skip_reasons(
.await
}
fn provider_key_concurrency_row(
provider_id: &str,
endpoint_id: &str,
key_id: &str,
key_name: &str,
provider_priority: i32,
key_priority: i32,
) -> StoredMinimalCandidateSelectionRow {
let mut row = sample_row();
row.provider_id = provider_id.to_string();
row.provider_name = provider_id.to_string();
row.endpoint_id = endpoint_id.to_string();
row.key_id = key_id.to_string();
row.key_name = key_name.to_string();
row.provider_priority = provider_priority;
row.key_internal_priority = key_priority;
row.key_global_priority_by_format = Some(serde_json::json!({"openai:chat": key_priority}));
row
}
fn provider_key_with_concurrent_limit(
key_id: &str,
provider_id: &str,
concurrent_limit: Option<i32>,
) -> StoredProviderCatalogKey {
let mut key = sample_key(key_id, provider_id, Some(10));
key.concurrent_limit = concurrent_limit;
key
}
fn active_provider_key_candidate(
candidate_id: &str,
request_id: &str,
provider_id: &str,
endpoint_id: &str,
key_id: &str,
status: RequestCandidateStatus,
) -> StoredRequestCandidate {
StoredRequestCandidate::new(
candidate_id.to_string(),
request_id.to_string(),
None,
None,
None,
None,
0,
0,
Some(provider_id.to_string()),
Some(endpoint_id.to_string()),
Some(key_id.to_string()),
status,
None,
false,
None,
None,
None,
None,
None,
None,
None,
95_000,
Some(95_000),
None,
)
.expect("candidate should build")
}
fn provider_key_concurrency_state(
rows: Vec<StoredMinimalCandidateSelectionRow>,
keys: Vec<StoredProviderCatalogKey>,
request_candidates: Vec<StoredRequestCandidate>,
) -> AppState {
let candidates = Arc::new(InMemoryMinimalCandidateSelectionReadRepository::seed(rows));
let provider_catalog = Arc::new(InMemoryProviderCatalogReadRepository::seed(
vec![
sample_provider("test-provider-a", None),
sample_provider("test-provider-b", None),
],
Vec::new(),
keys,
));
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
let request_candidates = Arc::new(InMemoryRequestCandidateRepository::seed(request_candidates));
AppState::new()
.expect("state should build")
.with_data_state_for_tests(
GatewayDataState::with_candidate_selection_provider_catalog_quota_and_request_candidates_for_tests(
candidates,
provider_catalog,
quotas,
request_candidates,
),
)
}
#[test]
fn skips_only_exhausted_monthly_quota_provider() {
let inactive = StoredProviderQuotaSnapshot::new(
@@ -860,6 +959,294 @@ async fn selects_next_candidate_when_first_provider_concurrent_limit_is_reached(
assert_eq!(selected.key_id, "key-b");
}
#[tokio::test]
async fn provider_key_concurrency_selects_next_key_when_first_provider_key_concurrent_limit_is_reached(
) {
let state = provider_key_concurrency_state(
vec![
provider_key_concurrency_row(
"test-provider-a",
"endpoint-a",
"provider-key-a",
"alpha",
0,
0,
),
provider_key_concurrency_row(
"test-provider-a",
"endpoint-a",
"provider-key-b",
"beta",
0,
1,
),
],
vec![
provider_key_with_concurrent_limit("provider-key-a", "test-provider-a", Some(1)),
provider_key_with_concurrent_limit("provider-key-b", "test-provider-a", Some(1)),
],
vec![active_provider_key_candidate(
"cand-provider-key-a",
"req-provider-key-a",
"test-provider-a",
"endpoint-a",
"provider-key-a",
RequestCandidateStatus::Streaming,
)],
);
let selected = select_candidate(
state.data.as_ref(),
&state,
"openai:chat",
"gpt-4.1",
false,
None,
100,
)
.await
.expect("selection should succeed")
.expect("candidate should exist");
assert_eq!(selected.provider_id, "test-provider-a");
assert_eq!(selected.key_id, "provider-key-b");
}
#[tokio::test]
async fn provider_key_concurrency_selects_next_provider_when_all_provider_keys_concurrent_limit_reached(
) {
let state = provider_key_concurrency_state(
vec![
provider_key_concurrency_row(
"test-provider-a",
"endpoint-a",
"provider-key-a",
"alpha",
0,
0,
),
provider_key_concurrency_row(
"test-provider-a",
"endpoint-a",
"provider-key-b",
"beta",
0,
1,
),
provider_key_concurrency_row(
"test-provider-b",
"endpoint-b",
"provider-key-c",
"gamma",
1,
0,
),
],
vec![
provider_key_with_concurrent_limit("provider-key-a", "test-provider-a", Some(1)),
provider_key_with_concurrent_limit("provider-key-b", "test-provider-a", Some(1)),
provider_key_with_concurrent_limit("provider-key-c", "test-provider-b", Some(1)),
],
vec![
active_provider_key_candidate(
"cand-provider-key-a",
"req-provider-key-a",
"test-provider-a",
"endpoint-a",
"provider-key-a",
RequestCandidateStatus::Pending,
),
active_provider_key_candidate(
"cand-provider-key-b",
"req-provider-key-b",
"test-provider-a",
"endpoint-a",
"provider-key-b",
RequestCandidateStatus::Streaming,
),
],
);
let selected = select_candidate(
state.data.as_ref(),
&state,
"openai:chat",
"gpt-4.1",
false,
None,
100,
)
.await
.expect("selection should succeed")
.expect("candidate should exist");
assert_eq!(selected.provider_id, "test-provider-b");
assert_eq!(selected.key_id, "provider-key-c");
}
#[tokio::test]
async fn provider_key_concurrency_returns_none_when_all_provider_keys_concurrent_limit_reached() {
let state = provider_key_concurrency_state(
vec![
provider_key_concurrency_row(
"test-provider-a",
"endpoint-a",
"provider-key-a",
"alpha",
0,
0,
),
provider_key_concurrency_row(
"test-provider-a",
"endpoint-a",
"provider-key-b",
"beta",
0,
1,
),
provider_key_concurrency_row(
"test-provider-b",
"endpoint-b",
"provider-key-c",
"gamma",
1,
0,
),
],
vec![
provider_key_with_concurrent_limit("provider-key-a", "test-provider-a", Some(1)),
provider_key_with_concurrent_limit("provider-key-b", "test-provider-a", Some(1)),
provider_key_with_concurrent_limit("provider-key-c", "test-provider-b", Some(1)),
],
vec![
active_provider_key_candidate(
"cand-provider-key-a",
"req-provider-key-a",
"test-provider-a",
"endpoint-a",
"provider-key-a",
RequestCandidateStatus::Pending,
),
active_provider_key_candidate(
"cand-provider-key-b",
"req-provider-key-b",
"test-provider-a",
"endpoint-a",
"provider-key-b",
RequestCandidateStatus::Streaming,
),
active_provider_key_candidate(
"cand-provider-key-c",
"req-provider-key-c",
"test-provider-b",
"endpoint-b",
"provider-key-c",
RequestCandidateStatus::Streaming,
),
],
);
let selected = select_candidate(
state.data.as_ref(),
&state,
"openai:chat",
"gpt-4.1",
false,
None,
100,
)
.await
.expect("selection should succeed");
assert!(selected.is_none());
let (selected_candidates, skipped_candidates) =
collect_selectable_candidates_with_skip_reasons(
state.data.as_ref(),
&state,
"openai:chat",
"gpt-4.1",
false,
None,
100,
)
.await
.expect("selection should succeed");
assert!(selected_candidates.is_empty());
assert_eq!(skipped_candidates.len(), 3);
assert!(skipped_candidates
.iter()
.all(|skipped| { skipped.skip_reason == "provider_key_concurrency_limit_reached" }));
assert!(!is_exact_all_skipped_by_auth_limit(
&selected_candidates,
&skipped_candidates,
));
}
#[tokio::test]
async fn provider_key_concurrency_collects_exact_skip_reason_for_saturated_provider_keys() {
let state = provider_key_concurrency_state(
vec![
provider_key_concurrency_row(
"test-provider-a",
"endpoint-a",
"provider-key-a",
"alpha",
0,
0,
),
provider_key_concurrency_row(
"test-provider-a",
"endpoint-a",
"provider-key-b",
"beta",
0,
1,
),
],
vec![
provider_key_with_concurrent_limit("provider-key-a", "test-provider-a", Some(1)),
provider_key_with_concurrent_limit("provider-key-b", "test-provider-a", Some(1)),
],
vec![active_provider_key_candidate(
"cand-provider-key-a",
"req-provider-key-a",
"test-provider-a",
"endpoint-a",
"provider-key-a",
RequestCandidateStatus::Pending,
)],
);
let (selected_candidates, skipped_candidates) =
collect_selectable_candidates_with_skip_reasons(
state.data.as_ref(),
&state,
"openai:chat",
"gpt-4.1",
false,
None,
100,
)
.await
.expect("selection should succeed");
assert_eq!(selected_candidates.len(), 1);
assert_eq!(selected_candidates[0].provider_id, "test-provider-a");
assert_eq!(selected_candidates[0].key_id, "provider-key-b");
assert_eq!(skipped_candidates.len(), 1);
assert_eq!(
skipped_candidates[0].candidate.provider_id,
"test-provider-a"
);
assert_eq!(skipped_candidates[0].candidate.key_id, "provider-key-a");
assert_eq!(
skipped_candidates[0].skip_reason,
"provider_key_concurrency_limit_reached",
);
}
#[tokio::test]
async fn returns_none_when_auth_api_key_concurrent_limit_is_reached() {
let candidates = Arc::new(InMemoryMinimalCandidateSelectionReadRepository::seed(vec![
@@ -90,7 +90,17 @@ pub(super) fn sample_key(
true,
)
.expect("key should build")
.with_rate_limit_fields(rpm_limit, None, None, None, None, None, Some(20), Some(20))
.with_rate_limit_fields(
rpm_limit,
None,
None,
None,
None,
None,
None,
Some(20),
Some(20),
)
}
pub(super) fn sample_auth_snapshot(api_key_id: &str) -> GatewayAuthApiKeySnapshot {
@@ -282,6 +282,232 @@ async fn gateway_creates_admin_provider_key_locally_with_trusted_admin_principal
upstream_handle.abort();
}
#[tokio::test]
async fn provider_key_concurrent_limit_create_and_list_responses() {
let provider_catalog_repository = Arc::new(InMemoryProviderCatalogReadRepository::seed(
vec![sample_provider("provider-openai", "openai", 10)],
vec![],
vec![],
));
let gateway = build_router_with_state(
AppState::new()
.expect("gateway should build")
.with_data_state_for_tests(
GatewayDataState::with_provider_catalog_repository_for_tests(
provider_catalog_repository.clone(),
)
.with_encryption_key_for_tests(DEVELOPMENT_ENCRYPTION_KEY),
),
);
let (gateway_url, gateway_handle) = start_server(gateway).await;
let create_with_limit_response = reqwest::Client::new()
.post(format!(
"{gateway_url}/api/admin/endpoints/providers/provider-openai/keys"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.json(&json!({
"api_formats": ["openai:chat"],
"api_key": "sk-created-openai-concurrent",
"name": "created key with concurrency",
"rpm_limit": 60,
"concurrent_limit": 3
}))
.send()
.await
.expect("request should succeed");
assert_eq!(create_with_limit_response.status(), StatusCode::OK);
let create_payload: serde_json::Value = create_with_limit_response
.json()
.await
.expect("json body should parse");
let create_with_limit_id = create_payload["id"]
.as_str()
.expect("created key id should be returned")
.to_string();
assert_eq!(create_payload["rpm_limit"], 60);
assert_eq!(create_payload["concurrent_limit"], 3);
let create_null_response = reqwest::Client::new()
.post(format!(
"{gateway_url}/api/admin/endpoints/providers/provider-openai/keys"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.json(&json!({
"api_formats": ["openai:chat"],
"api_key": "sk-created-openai-concurrent-null",
"name": "created key with null concurrency",
"concurrent_limit": null
}))
.send()
.await
.expect("request should succeed");
assert_eq!(create_null_response.status(), StatusCode::OK);
let null_payload: serde_json::Value = create_null_response
.json()
.await
.expect("json body should parse");
let create_null_id = null_payload["id"]
.as_str()
.expect("created null-limit key id should be returned")
.to_string();
assert_eq!(null_payload["concurrent_limit"], serde_json::Value::Null);
let create_negative_response = reqwest::Client::new()
.post(format!(
"{gateway_url}/api/admin/endpoints/providers/provider-openai/keys"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.json(&json!({
"api_formats": ["openai:chat"],
"api_key": "sk-created-openai-concurrent-negative",
"name": "created key with negative concurrency",
"concurrent_limit": -1
}))
.send()
.await
.expect("request should succeed");
assert_eq!(create_negative_response.status(), StatusCode::BAD_REQUEST);
let negative_payload: serde_json::Value = create_negative_response
.json()
.await
.expect("json body should parse");
assert!(negative_payload["detail"]
.as_str()
.expect("detail should be string")
.contains("concurrent_limit"));
let list_response = reqwest::Client::new()
.get(format!(
"{gateway_url}/api/admin/endpoints/providers/provider-openai/keys?skip=0&limit=50"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.send()
.await
.expect("request should succeed");
assert_eq!(list_response.status(), StatusCode::OK);
let list_payload: serde_json::Value =
list_response.json().await.expect("json body should parse");
let items = list_payload.as_array().expect("payload should be an array");
assert_eq!(items.len(), 2);
let with_limit = items
.iter()
.find(|item| item["name"].as_str() == Some("created key with concurrency"))
.expect("created key with concurrency should be listed");
let with_null = items
.iter()
.find(|item| item["name"].as_str() == Some("created key with null concurrency"))
.expect("created key with null concurrency should be listed");
assert_eq!(with_limit["concurrent_limit"], 3);
assert_eq!(with_null["concurrent_limit"], serde_json::Value::Null);
let read_back = provider_catalog_repository
.list_keys_by_ids(&[create_with_limit_id, create_null_id])
.await
.expect("created keys should read by id");
assert_eq!(read_back.len(), 2);
assert!(read_back.iter().any(|key| {
key.name == "created key with concurrency" && key.concurrent_limit == Some(3)
}));
assert!(read_back.iter().any(|key| {
key.name == "created key with null concurrency" && key.concurrent_limit.is_none()
}));
let keys = provider_catalog_repository
.list_keys_by_provider_ids(&["provider-openai".to_string()])
.await
.expect("keys should read");
assert_eq!(keys.len(), 2);
assert!(keys
.iter()
.any(|key| key.name == "created key with concurrency" && key.concurrent_limit == Some(3)));
assert!(keys.iter().any(
|key| key.name == "created key with null concurrency" && key.concurrent_limit.is_none()
));
gateway_handle.abort();
}
#[tokio::test]
async fn provider_key_concurrent_limit_reads_existing_list_response() {
let mut key_a = sample_key(
"provider-key-a",
"test-provider-a",
"openai:chat",
"sk-provider-key-a",
);
key_a.concurrent_limit = Some(1);
let mut key_b = sample_key(
"provider-key-b",
"test-provider-a",
"openai:chat",
"sk-provider-key-b",
);
key_b.concurrent_limit = None;
let provider_catalog_repository = Arc::new(InMemoryProviderCatalogReadRepository::seed(
vec![sample_provider("test-provider-a", "openai", 10)],
vec![],
vec![key_a, key_b],
));
let gateway = build_router_with_state(
AppState::new()
.expect("gateway should build")
.with_data_state_for_tests(GatewayDataState::with_provider_catalog_reader_for_tests(
provider_catalog_repository,
)),
);
let (gateway_url, gateway_handle) = start_server(gateway).await;
let response = reqwest::Client::new()
.get(format!(
"{gateway_url}/api/admin/endpoints/providers/test-provider-a/keys?skip=0&limit=50"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.send()
.await
.expect("request should succeed");
assert_eq!(response.status(), StatusCode::OK);
let payload: serde_json::Value = response.json().await.expect("json body should parse");
let items = payload.as_array().expect("payload should be an array");
assert_eq!(items.len(), 2);
let provider_key_a = items
.iter()
.find(|item| item["id"].as_str() == Some("provider-key-a"))
.expect("provider-key-a should be listed");
let provider_key_b = items
.iter()
.find(|item| item["id"].as_str() == Some("provider-key-b"))
.expect("provider-key-b should be listed");
assert_eq!(provider_key_a["concurrent_limit"], 1);
assert_eq!(provider_key_b["concurrent_limit"], serde_json::Value::Null);
gateway_handle.abort();
}
#[tokio::test]
async fn gateway_fetches_allowed_models_immediately_when_creating_key_with_auto_fetch() {
let execution_runtime_hits = Arc::new(Mutex::new(0usize));
@@ -885,6 +1111,132 @@ async fn gateway_updates_admin_provider_key_locally_with_trusted_admin_principal
upstream_handle.abort();
}
#[tokio::test]
async fn provider_key_concurrent_limit_update_presence_semantics() {
let mut key = sample_key(
"key-openai-a",
"provider-openai",
"openai:chat",
"sk-test-a",
);
key.concurrent_limit = Some(4);
let provider_catalog_repository = Arc::new(InMemoryProviderCatalogReadRepository::seed(
vec![sample_provider("provider-openai", "openai", 10)],
vec![],
vec![key],
));
let gateway = build_router_with_state(
AppState::new()
.expect("gateway should build")
.with_data_state_for_tests(
GatewayDataState::with_provider_catalog_repository_for_tests(
provider_catalog_repository.clone(),
)
.with_encryption_key_for_tests(DEVELOPMENT_ENCRYPTION_KEY),
),
);
let (gateway_url, gateway_handle) = start_server(gateway).await;
let omitted_response = reqwest::Client::new()
.put(format!(
"{gateway_url}/api/admin/endpoints/keys/key-openai-a"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.json(&json!({
"name": "renamed without concurrency"
}))
.send()
.await
.expect("request should succeed");
assert_eq!(omitted_response.status(), StatusCode::OK);
let omitted_payload: serde_json::Value = omitted_response
.json()
.await
.expect("json body should parse");
assert_eq!(omitted_payload["name"], "renamed without concurrency");
assert_eq!(omitted_payload["concurrent_limit"], 4);
let set_response = reqwest::Client::new()
.put(format!(
"{gateway_url}/api/admin/endpoints/keys/key-openai-a"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.json(&json!({
"concurrent_limit": 7
}))
.send()
.await
.expect("request should succeed");
assert_eq!(set_response.status(), StatusCode::OK);
let set_payload: serde_json::Value = set_response.json().await.expect("json body should parse");
assert_eq!(set_payload["concurrent_limit"], 7);
let clear_response = reqwest::Client::new()
.put(format!(
"{gateway_url}/api/admin/endpoints/keys/key-openai-a"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.json(&json!({
"concurrent_limit": null
}))
.send()
.await
.expect("request should succeed");
assert_eq!(clear_response.status(), StatusCode::OK);
let clear_payload: serde_json::Value =
clear_response.json().await.expect("json body should parse");
assert_eq!(clear_payload["concurrent_limit"], serde_json::Value::Null);
let negative_response = reqwest::Client::new()
.put(format!(
"{gateway_url}/api/admin/endpoints/keys/key-openai-a"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.json(&json!({
"concurrent_limit": -1
}))
.send()
.await
.expect("request should succeed");
assert_eq!(negative_response.status(), StatusCode::BAD_REQUEST);
let negative_payload: serde_json::Value = negative_response
.json()
.await
.expect("json body should parse");
assert!(negative_payload["detail"]
.as_str()
.expect("detail should be string")
.contains("concurrent_limit"));
let reloaded = provider_catalog_repository
.list_keys_by_ids(&["key-openai-a".to_string()])
.await
.expect("keys should read");
assert_eq!(reloaded.len(), 1);
assert_eq!(reloaded[0].name, "renamed without concurrency");
assert_eq!(reloaded[0].concurrent_limit, None);
gateway_handle.abort();
}
#[tokio::test]
async fn gateway_clears_allowed_models_when_disabling_auto_fetch_on_provider_key_update() {
let upstream_hits = Arc::new(Mutex::new(0usize));
@@ -334,7 +334,7 @@ async fn gateway_handles_admin_key_health_locally_with_trusted_admin_principal()
)],
vec![
sample_key("key-openai", "provider-openai", "openai:chat", "sk-test")
.with_rate_limit_fields(None, None, None, None, None, None, Some(10), Some(7))
.with_rate_limit_fields(None, None, None, None, None, None, None, Some(10), Some(7))
.with_usage_fields(Some(3), Some(2100))
.with_health_fields(
Some(json!({"openai:chat": {
@@ -1424,7 +1424,7 @@ async fn gateway_handles_admin_key_rpm_locally_with_trusted_admin_principal() {
)],
vec![
sample_key("key-openai", "provider-openai", "openai:chat", "sk-test")
.with_rate_limit_fields(Some(60), None, None, None, None, None, None, None),
.with_rate_limit_fields(Some(60), None, None, None, None, None, None, None, None),
],
));
let now_unix_secs = SystemTime::now()
@@ -1511,7 +1511,7 @@ async fn gateway_resets_admin_key_rpm_locally_with_trusted_admin_principal() {
)],
vec![
sample_key("key-openai", "provider-openai", "openai:chat", "sk-test")
.with_rate_limit_fields(Some(60), None, None, None, None, None, None, None),
.with_rate_limit_fields(Some(60), None, None, None, None, None, None, None, None),
],
));
let now_unix_secs = SystemTime::now()
@@ -263,6 +263,7 @@ pub struct StoredProviderCatalogKey {
pub proxy: Option<serde_json::Value>,
pub fingerprint: Option<serde_json::Value>,
pub rpm_limit: Option<u32>,
pub concurrent_limit: Option<i32>,
pub learned_rpm_limit: Option<u32>,
pub concurrent_429_count: Option<u32>,
pub rpm_429_count: Option<u32>,
@@ -338,6 +339,7 @@ impl StoredProviderCatalogKey {
proxy: None,
fingerprint: None,
rpm_limit: None,
concurrent_limit: None,
learned_rpm_limit: None,
concurrent_429_count: None,
rpm_429_count: None,
@@ -410,6 +412,7 @@ impl StoredProviderCatalogKey {
pub fn with_rate_limit_fields(
mut self,
rpm_limit: Option<u32>,
concurrent_limit: Option<i32>,
learned_rpm_limit: Option<u32>,
concurrent_429_count: Option<u32>,
rpm_429_count: Option<u32>,
@@ -419,6 +422,7 @@ impl StoredProviderCatalogKey {
success_count: Option<u32>,
) -> Self {
self.rpm_limit = rpm_limit;
self.concurrent_limit = concurrent_limit;
self.learned_rpm_limit = learned_rpm_limit;
self.concurrent_429_count = concurrent_429_count;
self.rpm_429_count = rpm_429_count;
@@ -460,6 +464,53 @@ impl StoredProviderCatalogKey {
}
}
#[cfg(test)]
mod transport_tests {
use super::StoredProviderCatalogKey;
#[test]
fn provider_catalog_key_defaults_concurrent_limit_to_none() {
let key = StoredProviderCatalogKey::new(
"key-1".to_string(),
"provider-1".to_string(),
"default".to_string(),
"api_key".to_string(),
None,
true,
)
.expect("key should build");
assert_eq!(key.concurrent_limit, None);
}
#[test]
fn provider_catalog_key_rate_limit_builder_sets_concurrent_limit() {
let key = StoredProviderCatalogKey::new(
"key-1".to_string(),
"provider-1".to_string(),
"default".to_string(),
"api_key".to_string(),
None,
true,
)
.expect("key should build")
.with_rate_limit_fields(
Some(120),
Some(3),
None,
None,
None,
None,
None,
None,
None,
);
assert_eq!(key.rpm_limit, Some(120));
assert_eq!(key.concurrent_limit, Some(3));
}
}
#[derive(Debug, Clone, PartialEq, Eq, Default)]
pub enum ProviderCatalogKeyListOrder {
#[default]
@@ -845,6 +845,36 @@ mod tests {
assert_eq!(stored[0].provider_id, "provider-1");
}
#[tokio::test]
async fn provider_api_keys_concurrent_limit_defaults_and_round_trips_in_memory() {
let repository = InMemoryProviderCatalogReadRepository::seed(
vec![sample_provider("test-provider-a")],
vec![],
vec![],
);
let mut key = sample_key("provider-key-a", "test-provider-a");
assert_eq!(key.concurrent_limit, None);
key.concurrent_limit = Some(1);
let created = repository
.create_key(&key)
.await
.expect("key should create");
assert_eq!(created.concurrent_limit, Some(1));
let mut updated = created.clone();
updated.concurrent_limit = None;
repository
.update_key(&updated)
.await
.expect("key should update");
let reloaded = repository
.list_keys_by_ids(&["provider-key-a".to_string()])
.await
.expect("keys should read");
assert_eq!(reloaded[0].concurrent_limit, None);
}
#[tokio::test]
async fn creates_endpoint() {
let repository = InMemoryProviderCatalogReadRepository::seed(
@@ -155,6 +155,7 @@ SELECT
proxy,
fingerprint,
rpm_limit,
concurrent_limit,
learned_rpm_limit,
concurrent_429_count,
rpm_429_count,
@@ -213,6 +214,7 @@ SELECT
proxy,
fingerprint,
rpm_limit,
concurrent_limit,
learned_rpm_limit,
concurrent_429_count,
rpm_429_count,
@@ -273,6 +275,7 @@ SELECT
NULL::jsonb AS proxy,
NULL::jsonb AS fingerprint,
NULL::integer AS rpm_limit,
NULL::integer AS concurrent_limit,
NULL::integer AS learned_rpm_limit,
NULL::integer AS concurrent_429_count,
NULL::integer AS rpm_429_count,
@@ -611,6 +614,7 @@ SELECT
proxy,
fingerprint,
rpm_limit,
concurrent_limit,
learned_rpm_limit,
concurrent_429_count,
rpm_429_count,
@@ -1199,6 +1203,7 @@ INSERT INTO provider_api_keys (
internal_priority,
global_priority_by_format,
rpm_limit,
concurrent_limit,
learned_rpm_limit,
allowed_models,
capabilities,
@@ -1331,6 +1336,7 @@ INSERT INTO provider_api_keys (
.bind(key.internal_priority)
.bind(&key.global_priority_by_format)
.bind(key.rpm_limit.map(|value| value as i32))
.bind(key.concurrent_limit)
.bind(key.learned_rpm_limit.map(|value| value as i32))
.bind(&key.allowed_models)
.bind(&key.capabilities)
@@ -1726,8 +1732,8 @@ UPDATE provider_api_keys
SET
provider_id = $2,
api_formats = $3,
auth_type_by_format = $39,
allow_auth_channel_mismatch_formats = $40,
auth_type_by_format = $40,
allow_auth_channel_mismatch_formats = $41,
auth_type = $4,
api_key = $5,
auth_config = $6,
@@ -1737,46 +1743,47 @@ SET
internal_priority = $10,
global_priority_by_format = $11,
rpm_limit = $12,
learned_rpm_limit = $13,
allowed_models = $14,
capabilities = $15,
cache_ttl_minutes = $16,
max_probe_interval_minutes = $17,
auto_fetch_models = $18,
locked_models = $19,
model_include_patterns = $20,
model_exclude_patterns = $21,
proxy = $22,
fingerprint = $23,
upstream_metadata = $24,
concurrent_limit = $13,
learned_rpm_limit = $14,
allowed_models = $15,
capabilities = $16,
cache_ttl_minutes = $17,
max_probe_interval_minutes = $18,
auto_fetch_models = $19,
locked_models = $20,
model_include_patterns = $21,
model_exclude_patterns = $22,
proxy = $23,
fingerprint = $24,
upstream_metadata = $25,
expires_at = CASE
WHEN $38::double precision IS NULL THEN NULL
ELSE TO_TIMESTAMP($38::double precision)
WHEN $39::double precision IS NULL THEN NULL
ELSE TO_TIMESTAMP($39::double precision)
END,
oauth_invalid_at = CASE
WHEN $25::double precision IS NULL THEN NULL
ELSE TO_TIMESTAMP($25::double precision)
WHEN $26::double precision IS NULL THEN NULL
ELSE TO_TIMESTAMP($26::double precision)
END,
oauth_invalid_reason = $26,
status_snapshot = $27,
concurrent_429_count = COALESCE($28, 0),
rpm_429_count = COALESCE($29, 0),
oauth_invalid_reason = $27,
status_snapshot = $28,
concurrent_429_count = COALESCE($29, 0),
rpm_429_count = COALESCE($30, 0),
last_429_at = CASE
WHEN $30::double precision IS NULL THEN NULL
ELSE TO_TIMESTAMP($30::double precision)
WHEN $31::double precision IS NULL THEN NULL
ELSE TO_TIMESTAMP($31::double precision)
END,
last_429_type = $31,
adjustment_history = $32,
utilization_samples = $33,
last_429_type = $32,
adjustment_history = $33,
utilization_samples = $34,
last_probe_increase_at = CASE
WHEN $34::double precision IS NULL THEN NULL
ELSE TO_TIMESTAMP($34::double precision)
WHEN $35::double precision IS NULL THEN NULL
ELSE TO_TIMESTAMP($35::double precision)
END,
last_rpm_peak = $35,
is_active = $36,
last_rpm_peak = $36,
is_active = $37,
updated_at = CASE
WHEN $37::double precision IS NULL THEN NOW()
ELSE TO_TIMESTAMP($37::double precision)
WHEN $38::double precision IS NULL THEN NOW()
ELSE TO_TIMESTAMP($38::double precision)
END
WHERE id = $1
"#,
@@ -1793,6 +1800,7 @@ WHERE id = $1
.bind(key.internal_priority)
.bind(&key.global_priority_by_format)
.bind(key.rpm_limit.map(|value| value as i32))
.bind(key.concurrent_limit)
.bind(key.learned_rpm_limit.map(|value| value as i32))
.bind(&key.allowed_models)
.bind(&key.capabilities)
@@ -2301,6 +2309,7 @@ fn map_key_row(row: &PgRow) -> Result<StoredProviderCatalogKey, DataLayerError>
})
})
.transpose()?;
let concurrent_limit = row_get::<Option<i32>>(row, "concurrent_limit")?;
let learned_rpm_limit = row_get::<Option<i32>>(row, "learned_rpm_limit")?
.map(|value| {
u32::try_from(value).map_err(|_| {
@@ -2466,6 +2475,7 @@ fn map_key_row(row: &PgRow) -> Result<StoredProviderCatalogKey, DataLayerError>
let mut key = key
.with_rate_limit_fields(
rpm_limit,
concurrent_limit,
learned_rpm_limit,
concurrent_429_count,
rpm_429_count,
@@ -2545,4 +2555,41 @@ mod tests {
assert!(sql.contains("total_cost_usd"));
}
}
#[test]
fn provider_api_keys_concurrent_limit_queries_include_field() {
for sql in [
super::LIST_KEYS_BY_IDS_PREFIX,
super::LIST_KEYS_BY_PROVIDER_IDS_PREFIX,
super::LIST_KEY_SUMMARIES_BY_PROVIDER_IDS_PREFIX,
] {
assert!(sql.contains("concurrent_limit"));
}
let source = include_str!("sql.rs");
assert!(source.contains("concurrent_limit,"));
assert!(source.contains("concurrent_limit = $13"));
assert!(source.contains(".bind(key.concurrent_limit)"));
assert!(source.contains("row_get::<Option<i32>>(row, \"concurrent_limit\")"));
}
#[test]
fn provider_api_keys_concurrent_limit_schema_is_nullable_without_default() {
let migration = include_str!(
"../../../migrations/20260502000000_add_provider_key_auth_channel_mismatch_formats.sql"
);
let concurrent_limit_line = migration
.lines()
.find(|line| line.contains("ADD COLUMN IF NOT EXISTS concurrent_limit"))
.expect("concurrent_limit migration line should exist")
.to_ascii_lowercase();
assert_eq!(
concurrent_limit_line.trim(),
"add column if not exists concurrent_limit integer;"
);
let baseline = include_str!("../../../bootstrap/20260413020000_baseline_v2.sql");
assert!(baseline.contains("CREATE TABLE IF NOT EXISTS public.provider_api_keys"));
assert!(baseline.contains("concurrent_limit integer,"));
}
}
@@ -32,9 +32,9 @@ mod tests {
use super::{
auth_api_key_concurrency_limit_reached, candidate_is_selectable_with_runtime_state,
candidate_supports_required_capability, collect_global_model_names_for_required_capability,
CandidateRuntimeSelectabilityInput, EnumerateMinimalCandidateSelectionInput,
SchedulerMinimalCandidateSelectionCandidate,
candidate_runtime_skip_reason_with_state, candidate_supports_required_capability,
collect_global_model_names_for_required_capability, CandidateRuntimeSelectabilityInput,
EnumerateMinimalCandidateSelectionInput, SchedulerMinimalCandidateSelectionCandidate,
};
use crate::SchedulerAuthConstraints;
@@ -118,6 +118,15 @@ mod tests {
key
}
fn sample_key_with_concurrent_limit(
id: &str,
concurrent_limit: Option<i32>,
) -> StoredProviderCatalogKey {
let mut key = sample_key(id, 1.0);
key.concurrent_limit = concurrent_limit;
key
}
fn stored_candidate(
id: &str,
status: RequestCandidateStatus,
@@ -304,6 +313,198 @@ mod tests {
));
}
#[test]
fn provider_key_concurrency_limit_unset_or_zero_is_unlimited() {
let recent_candidates = vec![stored_candidate("one", RequestCandidateStatus::Pending, 95)];
for concurrent_limit in [None, Some(0)] {
let provider_key_rpm_states = BTreeMap::from([(
"key-1".to_string(),
sample_key_with_concurrent_limit("1", concurrent_limit),
)]);
assert_eq!(
candidate_runtime_skip_reason_with_state(CandidateRuntimeSelectabilityInput {
candidate: &sample_candidate("1", None),
recent_candidates: &recent_candidates,
provider_concurrent_limits: &BTreeMap::new(),
provider_key_rpm_states: &provider_key_rpm_states,
now_unix_secs: 100,
cached_affinity_target: None,
provider_quota_blocks_requests: false,
account_quota_exhausted: false,
oauth_invalid: false,
rpm_reset_at: None,
}),
None
);
}
}
#[test]
fn provider_key_concurrency_limit_rejects_pending_active_with_exact_skip_reason() {
let recent_candidates = vec![stored_candidate("one", RequestCandidateStatus::Pending, 95)];
let provider_key_rpm_states = BTreeMap::from([(
"key-1".to_string(),
sample_key_with_concurrent_limit("1", Some(1)),
)]);
assert_eq!(
candidate_runtime_skip_reason_with_state(CandidateRuntimeSelectabilityInput {
candidate: &sample_candidate("1", None),
recent_candidates: &recent_candidates,
provider_concurrent_limits: &BTreeMap::new(),
provider_key_rpm_states: &provider_key_rpm_states,
now_unix_secs: 100,
cached_affinity_target: None,
provider_quota_blocks_requests: false,
account_quota_exhausted: false,
oauth_invalid: false,
rpm_reset_at: None,
}),
Some("provider_key_concurrency_limit_reached")
);
}
#[test]
fn provider_key_concurrency_limit_rejects_streaming_active() {
let recent_candidates = vec![stored_candidate(
"streaming",
RequestCandidateStatus::Streaming,
95,
)];
let provider_key_rpm_states = BTreeMap::from([(
"key-1".to_string(),
sample_key_with_concurrent_limit("1", Some(1)),
)]);
assert_eq!(
candidate_runtime_skip_reason_with_state(CandidateRuntimeSelectabilityInput {
candidate: &sample_candidate("1", None),
recent_candidates: &recent_candidates,
provider_concurrent_limits: &BTreeMap::new(),
provider_key_rpm_states: &provider_key_rpm_states,
now_unix_secs: 100,
cached_affinity_target: None,
provider_quota_blocks_requests: false,
account_quota_exhausted: false,
oauth_invalid: false,
rpm_reset_at: None,
}),
Some("provider_key_concurrency_limit_reached")
);
}
#[test]
fn provider_key_concurrency_limit_ignores_finished_and_stale_active_requests() {
let recent_candidates = vec![
stored_candidate("finished", RequestCandidateStatus::Success, 95),
stored_candidate("failed", RequestCandidateStatus::Failed, 96),
stored_candidate("cancelled", RequestCandidateStatus::Cancelled, 97),
stored_candidate("stale", RequestCandidateStatus::Pending, 699_000),
];
let provider_key_rpm_states = BTreeMap::from([(
"key-1".to_string(),
sample_key_with_concurrent_limit("1", Some(1)),
)]);
assert_eq!(
candidate_runtime_skip_reason_with_state(CandidateRuntimeSelectabilityInput {
candidate: &sample_candidate("1", None),
recent_candidates: &recent_candidates,
provider_concurrent_limits: &BTreeMap::new(),
provider_key_rpm_states: &provider_key_rpm_states,
now_unix_secs: 1_000,
cached_affinity_target: None,
provider_quota_blocks_requests: false,
account_quota_exhausted: false,
oauth_invalid: false,
rpm_reset_at: None,
}),
None
);
}
#[test]
fn provider_key_concurrency_limit_missing_state_does_not_skip() {
let recent_candidates = vec![stored_candidate("one", RequestCandidateStatus::Pending, 95)];
assert_eq!(
candidate_runtime_skip_reason_with_state(CandidateRuntimeSelectabilityInput {
candidate: &sample_candidate("1", None),
recent_candidates: &recent_candidates,
provider_concurrent_limits: &BTreeMap::new(),
provider_key_rpm_states: &BTreeMap::new(),
now_unix_secs: 100,
cached_affinity_target: None,
provider_quota_blocks_requests: false,
account_quota_exhausted: false,
oauth_invalid: false,
rpm_reset_at: None,
}),
None
);
}
#[test]
fn provider_key_concurrency_limit_preserves_key_circuit_and_rpm_checks() {
let mut circuit_open_key = sample_key_with_concurrent_limit("1", Some(2));
circuit_open_key.circuit_breaker_by_format = Some(serde_json::json!({
"openai:chat": {"open": true}
}));
let provider_key_rpm_states = BTreeMap::from([("key-1".to_string(), circuit_open_key)]);
assert_eq!(
candidate_runtime_skip_reason_with_state(CandidateRuntimeSelectabilityInput {
candidate: &sample_candidate("1", None),
recent_candidates: &[],
provider_concurrent_limits: &BTreeMap::new(),
provider_key_rpm_states: &provider_key_rpm_states,
now_unix_secs: 100,
cached_affinity_target: None,
provider_quota_blocks_requests: false,
account_quota_exhausted: false,
oauth_invalid: false,
rpm_reset_at: None,
}),
Some("key_circuit_open")
);
let recent_candidates = vec![stored_candidate(
"one",
RequestCandidateStatus::Pending,
95_000,
)];
let provider_key_rpm_states = BTreeMap::from([(
"key-1".to_string(),
sample_key_with_concurrent_limit("1", Some(2)).with_rate_limit_fields(
Some(1),
Some(2),
None,
None,
None,
None,
None,
None,
None,
),
)]);
assert_eq!(
candidate_runtime_skip_reason_with_state(CandidateRuntimeSelectabilityInput {
candidate: &sample_candidate("1", None),
recent_candidates: &recent_candidates,
provider_concurrent_limits: &BTreeMap::new(),
provider_key_rpm_states: &provider_key_rpm_states,
now_unix_secs: 100,
cached_affinity_target: None,
provider_quota_blocks_requests: false,
account_quota_exhausted: false,
oauth_invalid: false,
rpm_reset_at: None,
}),
Some("key_rpm_exhausted")
);
}
#[test]
fn candidate_selectability_rejects_quota_or_zero_health() {
let provider_key_rpm_states = BTreeMap::from([("key-1".to_string(), sample_key("1", 0.0))]);
@@ -86,9 +86,27 @@ pub fn candidate_runtime_skip_reason_with_state(
return Some("provider_concurrency_limit_reached");
}
let provider_key = provider_key_rpm_states.get(&candidate.key_id);
if let Some(provider_key) = provider_key {
if let Some(limit) = provider_key
.concurrent_limit
.filter(|limit| *limit > 0)
.and_then(|limit| usize::try_from(limit).ok())
{
if crate::count_recent_active_requests_for_provider_key(
recent_candidates,
candidate.key_id.as_str(),
now_unix_secs,
) >= limit
{
return Some("provider_key_concurrency_limit_reached");
}
}
}
let is_cached_user = cached_affinity_target
.is_some_and(|target| crate::matches_affinity_target(candidate, target));
if let Some(provider_key) = provider_key_rpm_states.get(&candidate.key_id) {
if let Some(provider_key) = provider_key {
if crate::is_provider_key_circuit_open(provider_key, candidate.endpoint_api_format.as_str())
{
return Some("key_circuit_open");
+219 -1
View File
@@ -100,6 +100,18 @@ pub fn count_recent_active_requests_for_provider(
.count()
}
pub fn count_recent_active_requests_for_provider_key(
recent_candidates: &[StoredRequestCandidate],
key_id: &str,
now_unix_secs: u64,
) -> usize {
recent_candidates
.iter()
.filter(|candidate| candidate.key_id.as_deref() == Some(key_id))
.filter(|candidate| is_recently_active(candidate, now_unix_secs))
.count()
}
pub fn count_recent_active_requests_for_api_key(
recent_candidates: &[StoredRequestCandidate],
api_key_id: &str,
@@ -598,7 +610,8 @@ mod tests {
use super::{
aggregate_provider_key_health_score, count_recent_active_requests_for_api_key,
count_recent_active_requests_for_provider, count_recent_rpm_requests_for_provider_key,
count_recent_active_requests_for_provider, count_recent_active_requests_for_provider_key,
count_recent_rpm_requests_for_provider_key,
count_recent_rpm_requests_for_provider_key_since, effective_provider_key_health_score,
effective_provider_key_rpm_limit, is_candidate_in_recent_failure_cooldown,
is_provider_key_circuit_open, provider_key_health_bucket, provider_key_health_score,
@@ -782,10 +795,211 @@ mod tests {
);
}
#[test]
fn provider_key_concurrency_counts_only_recent_active_requests() {
let recent_candidates = vec![
StoredRequestCandidate::new(
"pending".to_string(),
"req-pending".to_string(),
None,
Some("api-key-1".to_string()),
None,
None,
0,
0,
Some("provider-a".to_string()),
Some("endpoint-a".to_string()),
Some("key-a".to_string()),
RequestCandidateStatus::Pending,
None,
false,
None,
None,
None,
None,
None,
None,
None,
900_000,
Some(900_000),
None,
)
.expect("candidate should build"),
StoredRequestCandidate::new(
"streaming".to_string(),
"req-streaming".to_string(),
None,
Some("api-key-1".to_string()),
None,
None,
0,
0,
Some("provider-a".to_string()),
Some("endpoint-a".to_string()),
Some("key-a".to_string()),
RequestCandidateStatus::Streaming,
None,
false,
None,
None,
None,
None,
None,
None,
None,
950_000,
Some(950_000),
None,
)
.expect("candidate should build"),
StoredRequestCandidate::new(
"finished".to_string(),
"req-finished".to_string(),
None,
Some("api-key-1".to_string()),
None,
None,
0,
0,
Some("provider-a".to_string()),
Some("endpoint-a".to_string()),
Some("key-a".to_string()),
RequestCandidateStatus::Success,
None,
false,
None,
None,
None,
None,
None,
None,
None,
975_000,
Some(975_000),
Some(976_000),
)
.expect("candidate should build"),
StoredRequestCandidate::new(
"failed".to_string(),
"req-failed".to_string(),
None,
Some("api-key-1".to_string()),
None,
None,
0,
0,
Some("provider-a".to_string()),
Some("endpoint-a".to_string()),
Some("key-a".to_string()),
RequestCandidateStatus::Failed,
None,
false,
Some(429),
None,
Some("upstream failure".to_string()),
Some(20),
None,
None,
None,
980_000,
Some(980_000),
Some(981_000),
)
.expect("candidate should build"),
StoredRequestCandidate::new(
"cancelled".to_string(),
"req-cancelled".to_string(),
None,
Some("api-key-1".to_string()),
None,
None,
0,
0,
Some("provider-a".to_string()),
Some("endpoint-a".to_string()),
Some("key-a".to_string()),
RequestCandidateStatus::Cancelled,
None,
false,
Some(499),
None,
Some("client cancelled".to_string()),
Some(10),
None,
None,
None,
982_000,
Some(982_000),
Some(983_000),
)
.expect("candidate should build"),
StoredRequestCandidate::new(
"stale".to_string(),
"req-stale".to_string(),
None,
Some("api-key-1".to_string()),
None,
None,
0,
0,
Some("provider-a".to_string()),
Some("endpoint-a".to_string()),
Some("key-a".to_string()),
RequestCandidateStatus::Pending,
None,
false,
None,
None,
None,
None,
None,
None,
None,
699_000,
Some(699_000),
None,
)
.expect("candidate should build"),
StoredRequestCandidate::new(
"other-key".to_string(),
"req-other-key".to_string(),
None,
Some("api-key-1".to_string()),
None,
None,
0,
0,
Some("provider-a".to_string()),
Some("endpoint-a".to_string()),
Some("key-b".to_string()),
RequestCandidateStatus::Pending,
None,
false,
None,
None,
None,
None,
None,
None,
None,
990_000,
Some(990_000),
None,
)
.expect("candidate should build"),
];
assert_eq!(
count_recent_active_requests_for_provider_key(&recent_candidates, "key-a", 1_000),
2
);
}
#[test]
fn fixed_provider_key_rpm_limit_takes_precedence() {
let key = provider_catalog_key("key-a").with_rate_limit_fields(
Some(120),
None,
Some(80),
None,
None,
@@ -801,6 +1015,7 @@ mod tests {
#[test]
fn learned_provider_key_rpm_limit_requires_confidence() {
let low_confidence = provider_catalog_key("key-a").with_rate_limit_fields(
None,
None,
Some(80),
Some(0),
@@ -813,6 +1028,7 @@ mod tests {
assert_eq!(effective_provider_key_rpm_limit(&low_confidence, 100), None);
let mut high_confidence = provider_catalog_key("key-a").with_rate_limit_fields(
None,
None,
Some(80),
Some(0),
@@ -840,6 +1056,7 @@ mod tests {
#[test]
fn learned_provider_key_rpm_limit_uses_confirmed_observations_as_fallback_confidence() {
let key = provider_catalog_key("key-a").with_rate_limit_fields(
None,
None,
Some(80),
Some(0),
@@ -1010,6 +1227,7 @@ mod tests {
None,
None,
None,
None,
Some(5),
Some(5),
);
+6 -6
View File
@@ -26,12 +26,12 @@ pub use candidate::{
};
pub use health::{
aggregate_provider_key_health_score, count_recent_active_requests_for_api_key,
count_recent_active_requests_for_provider, count_recent_rpm_requests_for_provider_key,
count_recent_rpm_requests_for_provider_key_since, effective_provider_key_health_score,
effective_provider_key_rpm_limit, is_candidate_in_recent_failure_cooldown,
is_provider_key_circuit_open, provider_key_health_bucket, provider_key_health_score,
provider_key_rpm_allows_request, provider_key_rpm_allows_request_since,
ProviderKeyHealthBucket, PROVIDER_KEY_RPM_WINDOW_SECS,
count_recent_active_requests_for_provider, count_recent_active_requests_for_provider_key,
count_recent_rpm_requests_for_provider_key, count_recent_rpm_requests_for_provider_key_since,
effective_provider_key_health_score, effective_provider_key_rpm_limit,
is_candidate_in_recent_failure_cooldown, is_provider_key_circuit_open,
provider_key_health_bucket, provider_key_health_score, provider_key_rpm_allows_request,
provider_key_rpm_allows_request_since, ProviderKeyHealthBucket, PROVIDER_KEY_RPM_WINDOW_SECS,
};
pub use model::{
candidate_model_names, extract_global_priority_for_format, matches_model_mapping,
+2
View File
@@ -146,6 +146,7 @@ export async function addProviderKey(
rate_multipliers?: Record<string, number> | null // 按 API 格式的成本倍率
internal_priority?: number
rpm_limit?: number | null // RPM 限制(留空=自适应模式)
concurrent_limit?: number | null // 并发请求上限(留空或 0=不限制)
cache_ttl_minutes?: number
max_probe_interval_minutes?: number
allowed_models?: AllowedModels
@@ -177,6 +178,7 @@ export async function updateProviderKey(
internal_priority: number
global_priority_by_format: Record<string, number> | null // 按 API 格式的全局优先级
rpm_limit: number | null // RPM 限制(留空=自适应模式)
concurrent_limit: number | null // 并发请求上限(留空或 0=不限制)
cache_ttl_minutes: number
max_probe_interval_minutes: number
allowed_models: AllowedModels
@@ -243,6 +243,7 @@ export interface EndpointAPIKey {
internal_priority: number // Key 内部优先级
global_priority_by_format?: Record<string, number> | null // 按 endpoint signature 的全局优先级
rpm_limit?: number | null // RPM 速率限制 (1-10000),null 表示自适应模式
concurrent_limit?: number | null // 并发请求上限,null/0 表示不限制
allowed_models?: AllowedModels // 允许使用的模型列表(null=不限制)
capabilities?: Record<string, boolean> | null // 能力标签配置(如 cache_1h, context_1m)
// 缓存与熔断配置
@@ -395,6 +396,7 @@ export interface EndpointAPIKeyUpdate {
internal_priority?: number
global_priority_by_format?: Record<string, number> | null // 按 API 格式的全局优先级
rpm_limit?: number | null // RPM 速率限制 (1-10000),null 表示切换为自适应模式
concurrent_limit?: number | null // 并发请求上限,null/0 表示不限制
allowed_models?: AllowedModels
capabilities?: Record<string, boolean> | null
cache_ttl_minutes?: number
@@ -217,6 +217,24 @@
留空自适应
</p>
</div>
<div>
<Label
for="concurrent_limit"
class="text-xs"
>并发请求上限</Label>
<Input
id="concurrent_limit"
:model-value="form.concurrent_limit ?? ''"
type="number"
min="0"
placeholder="不限制"
class="h-8"
@update:model-value="(v) => form.concurrent_limit = parseNullableNumberInput(v, { min: 0 })"
/>
<p class="text-xs text-muted-foreground mt-0.5">
同一时间允许使用该 Key 的最大请求数,留空或 0 表示不限制
</p>
</div>
<div>
<Label
for="cache_ttl_minutes"
@@ -679,6 +697,7 @@ const form = ref({
rate_multipliers: {} as Record<string, number>, // 按 API 格式的成本倍率
internal_priority: 10,
rpm_limit: undefined as number | null | undefined, // RPM 限制(null=自适应,undefined=保持原值)
concurrent_limit: undefined as number | null | undefined, // 并发请求上限(null/0=不限制,undefined=保持原值)
cache_ttl_minutes: 5,
max_probe_interval_minutes: 32,
note: '',
@@ -786,6 +805,7 @@ function resetForm() {
rate_multipliers: {},
internal_priority: 10,
rpm_limit: undefined,
concurrent_limit: undefined,
cache_ttl_minutes: 5,
max_probe_interval_minutes: 32,
note: '',
@@ -837,6 +857,7 @@ function loadKeyData() {
internal_priority: props.editingKey.internal_priority ?? 10,
// 保留原始的 null/undefined 状态,null 表示自适应模式
rpm_limit: props.editingKey.rpm_limit ?? undefined,
concurrent_limit: props.editingKey.concurrent_limit ?? undefined,
cache_ttl_minutes: props.editingKey.cache_ttl_minutes ?? 5,
max_probe_interval_minutes: props.editingKey.max_probe_interval_minutes ?? 32,
note: props.editingKey.note || '',
@@ -969,6 +990,7 @@ async function handleSave() {
rate_multipliers: rateMultipliersData,
internal_priority: form.value.internal_priority,
rpm_limit: form.value.rpm_limit,
concurrent_limit: form.value.concurrent_limit,
cache_ttl_minutes: form.value.cache_ttl_minutes,
max_probe_interval_minutes: form.value.max_probe_interval_minutes,
note: form.value.note,
@@ -1003,6 +1025,7 @@ async function handleSave() {
rate_multipliers: rateMultipliersData,
internal_priority: form.value.internal_priority,
rpm_limit: form.value.rpm_limit,
concurrent_limit: form.value.concurrent_limit,
cache_ttl_minutes: form.value.cache_ttl_minutes,
max_probe_interval_minutes: form.value.max_probe_interval_minutes,
note: form.value.note,
@@ -73,6 +73,24 @@
留空自适应
</p>
</div>
<div>
<Label
for="concurrent_limit"
class="text-xs"
>并发请求上限</Label>
<Input
id="concurrent_limit"
:model-value="form.concurrent_limit ?? ''"
type="number"
min="0"
placeholder="不限制"
class="h-8"
@update:model-value="(v) => form.concurrent_limit = parseNullableNumberInput(v, { min: 0 })"
/>
<p class="text-xs text-muted-foreground mt-0.5">
同一时间允许使用该 Key 的最大请求数,留空或 0 表示不限制
</p>
</div>
<div>
<Label
for="cache_ttl_minutes"
@@ -243,6 +261,7 @@ const form = ref({
name: '',
internal_priority: 10,
rpm_limit: undefined as number | null | undefined,
concurrent_limit: undefined as number | null | undefined,
cache_ttl_minutes: 5,
max_probe_interval_minutes: 32,
note: '',
@@ -278,6 +297,7 @@ function resetForm() {
name: '',
internal_priority: 10,
rpm_limit: undefined,
concurrent_limit: undefined,
cache_ttl_minutes: 5,
max_probe_interval_minutes: 32,
note: '',
@@ -295,6 +315,7 @@ function loadKeyData() {
name: props.editingKey.name,
internal_priority: props.editingKey.internal_priority ?? 10,
rpm_limit: props.editingKey.rpm_limit ?? undefined,
concurrent_limit: props.editingKey.concurrent_limit ?? undefined,
cache_ttl_minutes: props.editingKey.cache_ttl_minutes ?? 5,
max_probe_interval_minutes: props.editingKey.max_probe_interval_minutes ?? 32,
note: props.editingKey.note || '',
@@ -359,6 +380,7 @@ async function handleSave() {
name: form.value.name,
internal_priority: form.value.internal_priority,
rpm_limit: form.value.rpm_limit,
concurrent_limit: form.value.concurrent_limit,
cache_ttl_minutes: form.value.cache_ttl_minutes,
max_probe_interval_minutes: form.value.max_probe_interval_minutes,
note: form.value.note,
@@ -0,0 +1,371 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { createApp, nextTick, type App, type Component } from 'vue'
import KeyFormDialog from '@/features/providers/components/KeyFormDialog.vue'
import OAuthKeyEditDialog from '@/features/providers/components/OAuthKeyEditDialog.vue'
import type { EndpointAPIKey } from '@/api/endpoints'
const endpointMocks = vi.hoisted(() => ({
addProviderKey: vi.fn(),
updateProviderKey: vi.fn(),
getAllCapabilities: vi.fn(),
sortApiFormats: vi.fn((formats: string[]) => [...formats].sort()),
}))
vi.mock('@/api/endpoints', () => ({
addProviderKey: endpointMocks.addProviderKey,
updateProviderKey: endpointMocks.updateProviderKey,
getAllCapabilities: endpointMocks.getAllCapabilities,
sortApiFormats: endpointMocks.sortApiFormats,
}))
vi.mock('@/components/ui', async () => {
const { defineComponent, h } = await import('vue')
const passthrough = (name: string, tag = 'div') => defineComponent({
name,
setup(_, { slots }) {
return () => h(tag, slots.default?.())
},
})
const Dialog = defineComponent({
name: 'DialogStub',
props: {
modelValue: Boolean,
},
setup(props, { slots }) {
return () => props.modelValue
? h('section', [slots.default?.(), slots.footer?.()])
: null
},
})
const Input = defineComponent({
name: 'InputStub',
inheritAttrs: false,
props: {
modelValue: {
type: [String, Number],
default: '',
},
masked: Boolean,
},
emits: ['update:modelValue'],
setup(props, { attrs, emit }) {
return () => h('input', {
...attrs,
value: props.modelValue ?? '',
onInput: (event: Event) => emit('update:modelValue', (event.target as HTMLInputElement).value),
})
},
})
const Label = defineComponent({
name: 'LabelStub',
inheritAttrs: false,
props: {
for: String,
},
setup(props, { attrs, slots }) {
return () => h('label', { ...attrs, for: props.for }, slots.default?.())
},
})
const Button = defineComponent({
name: 'ButtonStub',
inheritAttrs: false,
props: {
disabled: Boolean,
variant: String,
},
setup(props, { attrs, slots }) {
return () => h('button', {
...attrs,
disabled: props.disabled,
type: attrs.type ?? 'button',
}, slots.default?.())
},
})
const Switch = defineComponent({
name: 'SwitchStub',
inheritAttrs: false,
props: {
modelValue: Boolean,
},
emits: ['update:modelValue'],
setup(props, { attrs, emit }) {
return () => h('input', {
...attrs,
type: 'checkbox',
checked: props.modelValue,
onChange: (event: Event) => emit('update:modelValue', (event.target as HTMLInputElement).checked),
})
},
})
return {
Dialog,
Button,
Input,
Label,
Switch,
Select: passthrough('SelectStub'),
SelectTrigger: passthrough('SelectTriggerStub'),
SelectValue: passthrough('SelectValueStub', 'span'),
SelectContent: passthrough('SelectContentStub'),
SelectItem: passthrough('SelectItemStub'),
}
})
vi.mock('@/components/common/JsonImportInput.vue', async () => {
const { defineComponent, h } = await import('vue')
return {
default: defineComponent({
name: 'JsonImportInputStub',
setup() {
return () => h('textarea')
},
}),
}
})
vi.mock('@/composables/useToast', () => ({
useToast: () => ({
success: vi.fn(),
error: vi.fn(),
}),
}))
vi.mock('@/composables/useConfirm', () => ({
useConfirm: () => ({
confirmWarning: vi.fn().mockResolvedValue(true),
}),
}))
vi.mock('lucide-vue-next', async () => {
const { defineComponent, h } = await import('vue')
const Icon = defineComponent({
name: 'IconStub',
setup() {
return () => h('span')
},
})
return {
CircleHelp: Icon,
Key: Icon,
SquarePen: Icon,
}
})
const mountedApps: Array<{ app: App, root: HTMLElement }> = []
function createProviderKey(overrides: Partial<EndpointAPIKey> = {}): EndpointAPIKey {
return {
id: 'provider-key-1',
provider_id: 'provider-1',
api_formats: ['openai:chat'],
api_key_masked: 'sk-***',
auth_type: 'api_key',
name: 'Primary key',
rate_multipliers: null,
internal_priority: 10,
rpm_limit: 30,
concurrent_limit: null,
allowed_models: null,
capabilities: null,
cache_ttl_minutes: 5,
max_probe_interval_minutes: 32,
health_score: 100,
consecutive_failures: 0,
request_count: 0,
success_count: 0,
error_count: 0,
success_rate: 1,
avg_response_time_ms: 0,
is_active: true,
note: '',
created_at: '2026-04-27T00:00:00Z',
updated_at: '2026-04-27T00:00:00Z',
auto_fetch_models: false,
model_include_patterns: [],
model_exclude_patterns: [],
...overrides,
}
}
function mountDialog(component: Component, props: Record<string, unknown>) {
const root = document.createElement('div')
document.body.appendChild(root)
const app = createApp(component, props)
app.mount(root)
mountedApps.push({ app, root })
return root
}
async function settle() {
await nextTick()
await Promise.resolve()
await nextTick()
}
function findInput(root: HTMLElement, id: string) {
const input = root.querySelector<HTMLInputElement>(`#${id}`)
expect(input).not.toBeNull()
return input as HTMLInputElement
}
function updateInput(input: HTMLInputElement, value: string) {
input.value = value
input.dispatchEvent(new Event('input', { bubbles: true }))
}
async function submit(root: HTMLElement) {
const form = root.querySelector('form')
expect(form).not.toBeNull()
form?.dispatchEvent(new Event('submit', { bubbles: true, cancelable: true }))
await settle()
}
function lastUpdatePayload() {
const calls = endpointMocks.updateProviderKey.mock.calls
expect(calls.length).toBeGreaterThan(0)
return calls[calls.length - 1][1] as Record<string, unknown>
}
beforeEach(() => {
endpointMocks.addProviderKey.mockReset()
endpointMocks.updateProviderKey.mockReset()
endpointMocks.getAllCapabilities.mockReset()
endpointMocks.sortApiFormats.mockClear()
endpointMocks.addProviderKey.mockResolvedValue(createProviderKey())
endpointMocks.updateProviderKey.mockResolvedValue(createProviderKey())
endpointMocks.getAllCapabilities.mockResolvedValue([])
})
afterEach(() => {
for (const { app, root } of mountedApps.splice(0)) {
app.unmount()
root.remove()
}
})
describe('provider key concurrent_limit form behavior', () => {
it('hydrates and serializes a positive concurrent_limit number from the normal key form', async () => {
const root = mountDialog(KeyFormDialog, {
open: true,
endpoint: null,
editingKey: createProviderKey({ rpm_limit: 42, concurrent_limit: 3 }),
providerId: 'provider-1',
providerType: 'openai',
availableApiFormats: ['openai:chat'],
})
await settle()
const concurrentLimitInput = findInput(root, 'concurrent_limit')
expect(concurrentLimitInput.value).toBe('3')
expect(findInput(root, 'rpm_limit').value).toBe('42')
updateInput(concurrentLimitInput, '5')
await submit(root)
const payload = lastUpdatePayload()
expect(payload.concurrent_limit).toBe(5)
expect(typeof payload.concurrent_limit).toBe('number')
expect(payload.concurrent_limit).not.toBe('')
expect(payload.rpm_limit).toBe(42)
})
it('serializes cleared normal key concurrent_limit as null instead of an empty string', async () => {
const root = mountDialog(KeyFormDialog, {
open: true,
endpoint: null,
editingKey: createProviderKey({ rpm_limit: 24, concurrent_limit: 6 }),
providerId: 'provider-1',
providerType: 'openai',
availableApiFormats: ['openai:chat'],
})
await settle()
updateInput(findInput(root, 'concurrent_limit'), '')
await submit(root)
const payload = lastUpdatePayload()
expect(payload).toHaveProperty('concurrent_limit', null)
expect(payload.concurrent_limit).not.toBe('')
expect(payload.rpm_limit).toBe(24)
})
it('hydrates and serializes a positive concurrent_limit number from the OAuth edit form', async () => {
const root = mountDialog(OAuthKeyEditDialog, {
open: true,
editingKey: createProviderKey({
id: 'oauth-key-1',
auth_type: 'oauth',
name: 'OAuth account',
rpm_limit: 35,
concurrent_limit: 3,
}),
})
await settle()
const concurrentLimitInput = findInput(root, 'concurrent_limit')
expect(concurrentLimitInput.value).toBe('3')
expect(findInput(root, 'rpm_limit').value).toBe('35')
updateInput(concurrentLimitInput, '7')
await submit(root)
const payload = lastUpdatePayload()
expect(endpointMocks.updateProviderKey).toHaveBeenCalledWith('oauth-key-1', expect.any(Object))
expect(payload.concurrent_limit).toBe(7)
expect(typeof payload.concurrent_limit).toBe('number')
expect(payload.concurrent_limit).not.toBe('')
expect(payload.rpm_limit).toBe(35)
})
it('serializes cleared OAuth concurrent_limit as null instead of an empty string', async () => {
const root = mountDialog(OAuthKeyEditDialog, {
open: true,
editingKey: createProviderKey({
id: 'oauth-key-2',
auth_type: 'oauth',
rpm_limit: 18,
concurrent_limit: 4,
}),
})
await settle()
updateInput(findInput(root, 'concurrent_limit'), '')
await submit(root)
const payload = lastUpdatePayload()
expect(payload).toHaveProperty('concurrent_limit', null)
expect(payload.concurrent_limit).not.toBe('')
expect(payload.rpm_limit).toBe(18)
})
it('keeps zero concurrent_limit as a numeric unlimited value', async () => {
const root = mountDialog(OAuthKeyEditDialog, {
open: true,
editingKey: createProviderKey({
id: 'oauth-key-zero',
auth_type: 'oauth',
rpm_limit: 11,
concurrent_limit: 2,
}),
})
await settle()
updateInput(findInput(root, 'concurrent_limit'), '0')
await submit(root)
const payload = lastUpdatePayload()
expect(payload.concurrent_limit).toBe(0)
expect(typeof payload.concurrent_limit).toBe('number')
expect(payload.rpm_limit).toBe(11)
})
})