feat(routing): consolidate scheduling strategy configuration

This commit is contained in:
fawney
2026-09-03 11:05:59 +08:00
parent e8d9877b79
commit 2cb4d554aa
88 changed files with 1836 additions and 3635 deletions
@@ -1,7 +1,6 @@
use self::selection::{
collect_selectable_candidates, collect_selectable_candidates_with_skip_reasons_and_ordering,
collect_selectable_enumerated_candidates_with_skip_reasons,
resolve_preselection_ordering_config,
};
use super::config::SchedulerOrderingConfig;
use super::state::SchedulerRuntimeState;
@@ -56,8 +55,7 @@ enum RequiredCapabilityMatchMode {
}
/// `ordering_config` carries the request's routing-policy derived scheduler
/// config. `None` falls back to the runtime default (system-default routing
/// group, then legacy system-config keys).
/// config. Every production scheduling pass must provide this snapshot.
#[allow(clippy::too_many_arguments)]
pub(crate) async fn list_selectable_candidates(
selection_row_source: &(impl MinimalCandidateSelectionRowSource + Sync),
@@ -70,7 +68,7 @@ pub(crate) async fn list_selectable_candidates(
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
ordering_config: Option<SchedulerOrderingConfig>,
ordering_config: SchedulerOrderingConfig,
) -> Result<Vec<SchedulerMinimalCandidateSelectionCandidate>, GatewayError> {
collect_selectable_candidates(
selection_row_source,
@@ -107,7 +105,7 @@ pub(crate) async fn list_selectable_candidates_with_skip_reasons(
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
ordering_config: Option<SchedulerOrderingConfig>,
ordering_config: SchedulerOrderingConfig,
) -> Result<
(
Vec<SchedulerMinimalCandidateSelectionCandidate>,
@@ -145,7 +143,7 @@ pub(crate) async fn list_selectable_candidates_with_skip_reasons_for_request_ope
now_unix_secs: u64,
enable_model_directives: bool,
request_operation: Option<&str>,
ordering_config: Option<SchedulerOrderingConfig>,
ordering_config: SchedulerOrderingConfig,
) -> Result<
(
Vec<SchedulerMinimalCandidateSelectionCandidate>,
@@ -180,7 +178,7 @@ pub(crate) async fn list_selectable_enumerated_candidates_with_skip_reasons(
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
ordering_config: Option<SchedulerOrderingConfig>,
ordering_config: SchedulerOrderingConfig,
) -> Result<
(
Vec<SchedulerMinimalCandidateSelectionCandidate>,
@@ -188,8 +186,6 @@ pub(crate) async fn list_selectable_enumerated_candidates_with_skip_reasons(
),
GatewayError,
> {
let ordering_config =
resolve_preselection_ordering_config(runtime_state, ordering_config).await?;
let priority_affinity_key = selection::scheduling_priority_affinity_key(
auth_snapshot,
client_session_affinity,
@@ -220,7 +216,7 @@ pub(crate) async fn list_selectable_candidates_for_required_capability_without_r
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
ordering_config: Option<SchedulerOrderingConfig>,
ordering_config: SchedulerOrderingConfig,
) -> Result<Vec<SchedulerMinimalCandidateSelectionCandidate>, GatewayError> {
Ok(
list_selectable_candidates_for_required_capability_without_requested_model_with_auth_limit_signal(
@@ -249,7 +245,7 @@ pub(crate) async fn list_selectable_candidates_for_required_capability_without_r
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
ordering_config: Option<SchedulerOrderingConfig>,
ordering_config: SchedulerOrderingConfig,
) -> Result<(Vec<SchedulerMinimalCandidateSelectionCandidate>, bool), GatewayError> {
let normalized_api_format = normalize_api_format(candidate_api_format);
if normalized_api_format.is_empty() {
@@ -47,7 +47,7 @@ pub(super) fn is_exact_all_skipped_by_auth_limit(
.all(|candidate| is_auth_api_key_concurrency_limit_skip_reason(candidate.skip_reason))
}
#[cfg_attr(not(test), allow(dead_code))]
#[cfg(test)]
pub(super) async fn select_minimal_candidate(
selection_row_source: &(impl MinimalCandidateSelectionRowSource + Sync),
runtime_state: &impl SchedulerRuntimeState,
@@ -59,20 +59,8 @@ pub(super) async fn select_minimal_candidate(
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
ordering_config: SchedulerOrderingConfig,
) -> Result<Option<SchedulerMinimalCandidateSelectionCandidate>, GatewayError> {
let affinity_epoch = runtime_state.scheduler_affinity_epoch();
let ordering_config = runtime_state.read_scheduler_ordering_config().await?;
let affinity_cache_key = build_scheduler_affinity_cache_key(
auth_snapshot,
api_format,
global_model_name,
client_session_affinity,
);
let priority_affinity_key = scheduling_priority_affinity_key(
auth_snapshot,
client_session_affinity,
ordering_config.scheduling_mode,
);
let candidates = enumerate_scheduler_candidates(
selection_row_source,
api_format,
@@ -84,7 +72,7 @@ pub(super) async fn select_minimal_candidate(
None,
)
.await?;
let selected = collect_selectable_enumerated_candidates_with_skip_reasons(
Ok(collect_selectable_enumerated_candidates_with_skip_reasons(
runtime_state,
api_format,
global_model_name,
@@ -94,25 +82,16 @@ pub(super) async fn select_minimal_candidate(
client_session_affinity,
now_unix_secs,
ordering_config,
priority_affinity_key,
scheduling_priority_affinity_key(
auth_snapshot,
client_session_affinity,
ordering_config.scheduling_mode,
),
)
.await?
.0
.into_iter()
.next();
if ordering_config.scheduling_mode == SchedulerSchedulingMode::CacheAffinity
&& has_explicit_session_affinity(client_session_affinity)
{
if let Some(candidate) = selected.as_ref() {
remember_scheduler_affinity(
affinity_cache_key.as_deref(),
runtime_state,
candidate,
Some(affinity_epoch),
);
}
}
Ok(selected)
.next())
}
#[allow(clippy::too_many_arguments)]
@@ -127,7 +106,7 @@ pub(super) async fn collect_selectable_candidates(
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
ordering_config: Option<SchedulerOrderingConfig>,
ordering_config: SchedulerOrderingConfig,
) -> Result<Vec<SchedulerMinimalCandidateSelectionCandidate>, GatewayError> {
Ok(
collect_selectable_candidates_with_skip_reasons_and_ordering(
@@ -149,10 +128,8 @@ pub(super) async fn collect_selectable_candidates(
)
}
/// Legacy-shaped entrypoint that resolves the ordering config from the
/// runtime state. Prefer `collect_selectable_candidates_with_skip_reasons_and_ordering`
/// and pass the request's routing-policy config explicitly.
#[allow(clippy::too_many_arguments)]
#[cfg(test)]
pub(super) async fn collect_selectable_candidates_with_skip_reasons(
selection_row_source: &(impl MinimalCandidateSelectionRowSource + Sync),
runtime_state: &impl SchedulerRuntimeState,
@@ -184,24 +161,11 @@ pub(super) async fn collect_selectable_candidates_with_skip_reasons(
now_unix_secs,
enable_model_directives,
request_operation,
None,
SchedulerOrderingConfig::default(),
)
.await
}
/// Resolve the ordering config for a preselection pass: the routing-policy
/// derived config wins when the caller has one; otherwise fall back to the
/// runtime default (system-default routing group, then legacy keys).
pub(super) async fn resolve_preselection_ordering_config(
runtime_state: &impl SchedulerRuntimeState,
ordering_config: Option<SchedulerOrderingConfig>,
) -> Result<SchedulerOrderingConfig, GatewayError> {
match ordering_config {
Some(config) => Ok(config),
None => runtime_state.read_scheduler_ordering_config().await,
}
}
#[allow(clippy::too_many_arguments)]
pub(super) async fn collect_selectable_candidates_with_skip_reasons_and_ordering(
selection_row_source: &(impl MinimalCandidateSelectionRowSource + Sync),
@@ -215,7 +179,7 @@ pub(super) async fn collect_selectable_candidates_with_skip_reasons_and_ordering
now_unix_secs: u64,
enable_model_directives: bool,
request_operation: Option<&str>,
ordering_config: Option<SchedulerOrderingConfig>,
ordering_config: SchedulerOrderingConfig,
) -> Result<
(
Vec<SchedulerMinimalCandidateSelectionCandidate>,
@@ -223,8 +187,6 @@ pub(super) async fn collect_selectable_candidates_with_skip_reasons_and_ordering
),
GatewayError,
> {
let ordering_config =
resolve_preselection_ordering_config(runtime_state, ordering_config).await?;
let priority_affinity_key = scheduling_priority_affinity_key(
auth_snapshot,
client_session_affinity,
@@ -23,6 +23,7 @@ use crate::data::candidate_selection::{
read_requested_model_rows, MinimalCandidateSelectionRowSource,
};
use crate::data::GatewayDataState;
use crate::scheduler::config::SchedulerOrderingConfig;
use crate::{AppState, GatewayError};
use super::super::affinity::build_scheduler_affinity_cache_key;
@@ -50,6 +51,7 @@ async fn select_candidate(
client_session_affinity,
now_unix_secs,
false,
SchedulerOrderingConfig::default(),
)
.await
}
@@ -65,7 +65,7 @@ async fn compatible_required_capability_prefers_matching_keys_without_hard_filte
None,
None,
100,
None,
crate::scheduler::config::SchedulerOrderingConfig::default(),
)
.await
.expect("selection should succeed");
@@ -121,7 +121,7 @@ async fn exclusive_required_capability_keeps_hard_filtering_only_matching_keys()
None,
None,
100,
None,
crate::scheduler::config::SchedulerOrderingConfig::default(),
)
.await
.expect("selection should succeed");
@@ -198,7 +198,7 @@ async fn required_capability_without_model_uses_session_scoped_affinity() {
Some(&auth_snapshot),
Some(&client_session_affinity),
100,
None,
crate::scheduler::config::SchedulerOrderingConfig::default(),
)
.await
.expect("selection should succeed");
@@ -276,7 +276,7 @@ async fn required_capability_reports_auth_limit_signal_when_every_model_is_block
Some(&auth_snapshot),
None,
100,
None,
crate::scheduler::config::SchedulerOrderingConfig::default(),
)
.await
.expect("selection should succeed");
@@ -5,6 +5,7 @@ use aether_data::repository::candidate_selection::InMemoryMinimalCandidateSelect
use aether_data::repository::candidates::InMemoryRequestCandidateRepository;
use aether_data::repository::provider_catalog::InMemoryProviderCatalogReadRepository;
use aether_data::repository::quota::InMemoryProviderQuotaRepository;
use aether_data::repository::routing_profiles::InMemoryRoutingGroupRepository;
use aether_data_contracts::repository::candidate_selection::{
StoredMinimalCandidateSelectionRow, StoredProviderModelMapping,
};
@@ -13,6 +14,9 @@ use aether_data_contracts::repository::candidates::{
};
use aether_data_contracts::repository::provider_catalog::StoredProviderCatalogKey;
use aether_data_contracts::repository::quota::StoredProviderQuotaSnapshot;
use aether_data_contracts::repository::routing_profiles::{
CreateRoutingGroupRecord, RoutingGroupWriteRepository,
};
use aether_scheduler_core::{ClientSessionAffinity, SchedulerMinimalCandidateSelectionCandidate};
use serde_json::json;
@@ -20,6 +24,7 @@ use crate::cache::SchedulerAffinityTarget;
use crate::data::auth::GatewayAuthApiKeySnapshot;
use crate::data::candidate_selection::MinimalCandidateSelectionRowSource;
use crate::data::GatewayDataState;
use crate::scheduler::config::SchedulerOrderingConfig;
use crate::{AppState, GatewayError};
use super::super::affinity::build_scheduler_affinity_cache_key;
@@ -31,6 +36,39 @@ use super::super::selection::{
};
use super::support::{sample_auth_snapshot, sample_key, sample_provider, sample_row};
async fn state_with_routing_default_policy(
data_state: GatewayDataState,
default_policy: serde_json::Value,
) -> AppState {
let repository = Arc::new(InMemoryRoutingGroupRepository::default());
repository
.create_routing_group(CreateRoutingGroupRecord {
id: "selection-test-default".to_string(),
name: "selection-test-default".to_string(),
description: None,
enabled: true,
is_system_default: true,
sort_order: 0,
config_json: json!({"default_policy": default_policy}),
version: 1,
created_at: 1,
updated_at: 1,
published_at: None,
})
.await
.expect("routing strategy should be created");
AppState::new()
.expect("state should build")
.with_data_state_for_tests(data_state.with_routing_group_repository_for_tests(repository))
}
async fn ordering_config(state: &AppState) -> SchedulerOrderingConfig {
crate::scheduler::config::read_system_default_routing_ordering_config(state)
.await
.expect("routing strategy should load")
.unwrap_or_default()
}
async fn select_candidate(
selection_row_source: &(impl MinimalCandidateSelectionRowSource + Sync),
runtime_state: &AppState,
@@ -40,6 +78,7 @@ async fn select_candidate(
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
now_unix_secs: u64,
) -> Result<Option<SchedulerMinimalCandidateSelectionCandidate>, GatewayError> {
let ordering_config = ordering_config(runtime_state).await;
select_candidate_impl(
selection_row_source,
runtime_state,
@@ -51,6 +90,7 @@ async fn select_candidate(
None,
now_unix_secs,
false,
ordering_config,
)
.await
}
@@ -64,6 +104,7 @@ async fn collect_selectable_candidates(
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
now_unix_secs: u64,
) -> Result<Vec<SchedulerMinimalCandidateSelectionCandidate>, GatewayError> {
let ordering_config = ordering_config(runtime_state).await;
collect_selectable_candidates_impl(
selection_row_source,
runtime_state,
@@ -75,7 +116,7 @@ async fn collect_selectable_candidates(
None,
now_unix_secs,
false,
None,
ordering_config,
)
.await
}
@@ -289,15 +330,11 @@ async fn selects_by_provider_priority_when_priority_mode_is_provider() {
global_key_first,
]));
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
let state = AppState::new()
.expect("state should build")
.with_data_state_for_tests(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas)
.with_system_config_values_for_tests(vec![(
"provider_priority_mode".to_string(),
json!("provider"),
)]),
);
let state = state_with_routing_default_policy(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas),
json!({"priority_mode": "provider"}),
)
.await;
let selected = select_candidate(
state.data.as_ref(),
@@ -343,15 +380,11 @@ async fn selects_by_global_key_priority_when_priority_mode_is_global_key() {
global_key_first,
]));
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
let state = AppState::new()
.expect("state should build")
.with_data_state_for_tests(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas)
.with_system_config_values_for_tests(vec![(
"provider_priority_mode".to_string(),
json!("global_key"),
)]),
);
let state = state_with_routing_default_policy(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas),
json!({"priority_mode": "global_key"}),
)
.await;
let selected = select_candidate(
state.data.as_ref(),
@@ -415,6 +448,7 @@ async fn scheduler_selection_prefers_required_capability_matches_before_priority
None,
100,
false,
SchedulerOrderingConfig::default(),
)
.await
.expect("selection should succeed")
@@ -450,15 +484,11 @@ async fn fixed_order_ignores_cached_scheduler_affinity_promotion() {
first, second,
]));
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
let state = AppState::new()
.expect("state should build")
.with_data_state_for_tests(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas)
.with_system_config_values_for_tests(vec![(
"scheduling_mode".to_string(),
json!("fixed_order"),
)]),
);
let state = state_with_routing_default_policy(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas),
json!({"scheduling_mode": "fixed_order"}),
)
.await;
let auth_snapshot = sample_auth_snapshot("affinity-key-1");
state.remember_scheduler_affinity_target(
@@ -515,15 +545,11 @@ async fn fixed_order_disables_same_priority_affinity_hash_tiebreaker() {
first, second,
]));
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
let state = AppState::new()
.expect("state should build")
.with_data_state_for_tests(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas)
.with_system_config_values_for_tests(vec![(
"scheduling_mode".to_string(),
json!("fixed_order"),
)]),
);
let state = state_with_routing_default_policy(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas),
json!({"scheduling_mode": "fixed_order"}),
)
.await;
let auth_snapshot = sample_auth_snapshot("affinity-key-1");
let selection = collect_selectable_candidates(
@@ -569,15 +595,11 @@ async fn cache_affinity_promotes_cached_scheduler_affinity_candidate_when_enable
first, second,
]));
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
let state = AppState::new()
.expect("state should build")
.with_data_state_for_tests(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas)
.with_system_config_values_for_tests(vec![(
"scheduling_mode".to_string(),
json!("cache_affinity"),
)]),
);
let state = state_with_routing_default_policy(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas),
json!({"scheduling_mode": "cache_affinity"}),
)
.await;
let auth_snapshot = sample_auth_snapshot("affinity-key-1");
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
@@ -610,6 +632,7 @@ async fn cache_affinity_promotes_cached_scheduler_affinity_candidate_when_enable
Some(&client_session_affinity),
100,
false,
ordering_config(&state).await,
)
.await
.expect("selection should succeed")
@@ -645,15 +668,11 @@ async fn cache_affinity_ignores_cached_scheduler_affinity_without_client_session
first, second,
]));
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
let state = AppState::new()
.expect("state should build")
.with_data_state_for_tests(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas)
.with_system_config_values_for_tests(vec![(
"scheduling_mode".to_string(),
json!("cache_affinity"),
)]),
);
let state = state_with_routing_default_policy(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas),
json!({"scheduling_mode": "cache_affinity"}),
)
.await;
let auth_snapshot = sample_auth_snapshot("affinity-key-1");
state.remember_scheduler_affinity_target(
@@ -691,15 +710,11 @@ async fn load_balance_selection_does_not_remember_scheduler_affinity() {
row,
]));
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
let state = AppState::new()
.expect("state should build")
.with_data_state_for_tests(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas)
.with_system_config_values_for_tests(vec![(
"scheduling_mode".to_string(),
json!("load_balance"),
)]),
);
let state = state_with_routing_default_policy(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas),
json!({"scheduling_mode": "load_balance"}),
)
.await;
let auth_snapshot = sample_auth_snapshot("affinity-key-1");
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cache_key = build_scheduler_affinity_cache_key(
@@ -721,6 +736,7 @@ async fn load_balance_selection_does_not_remember_scheduler_affinity() {
Some(&client_session_affinity),
100,
false,
ordering_config(&state).await,
)
.await
.expect("selection should succeed")
@@ -758,15 +774,11 @@ async fn load_balance_ignores_provider_priority_and_cached_affinity() {
first, second,
]));
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
let state = AppState::new()
.expect("state should build")
.with_data_state_for_tests(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas)
.with_system_config_values_for_tests(vec![(
"scheduling_mode".to_string(),
json!("load_balance"),
)]),
);
let state = state_with_routing_default_policy(
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas),
json!({"scheduling_mode": "load_balance"}),
)
.await;
let auth_snapshot = sample_auth_snapshot("affinity-key-1");
state.remember_scheduler_affinity_target(
+61 -125
View File
@@ -55,7 +55,7 @@ impl Default for SchedulerOrderingConfig {
impl SchedulerOrderingConfig {
/// Ordering config derived from a resolved routing policy. The policy is
/// the single source of truth: no legacy system-config value is merged in.
/// the single source of truth for request scheduling.
pub(crate) fn from_routing_policy(policy: &ResolvedRoutingPolicy) -> Self {
Self {
priority_mode: scheduler_priority_mode_from_routing(policy.priority_mode),
@@ -87,6 +87,7 @@ impl SchedulerOrderingConfig {
},
keep_priority_on_conversion: self.keep_priority_on_conversion,
sticky_key_attempts: self.sticky_key_attempts,
execution_policy: aether_routing_core::RoutingExecutionPolicy::default(),
}
}
@@ -114,61 +115,6 @@ fn scheduler_scheduling_mode_from_routing(mode: RoutingSchedulingMode) -> Schedu
}
}
pub(crate) fn parse_scheduler_priority_mode(
value: Option<&serde_json::Value>,
) -> SchedulerPriorityMode {
match value
.and_then(serde_json::Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
.map(|value| value.to_ascii_lowercase())
.as_deref()
{
Some("global_key") => SchedulerPriorityMode::GlobalKey,
_ => SchedulerPriorityMode::Provider,
}
}
pub(crate) fn parse_keep_priority_on_conversion(value: Option<&serde_json::Value>) -> bool {
value.and_then(serde_json::Value::as_bool).unwrap_or(false)
}
pub(crate) fn parse_scheduler_scheduling_mode(
value: Option<&serde_json::Value>,
) -> SchedulerSchedulingMode {
match value
.and_then(serde_json::Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
.map(|value| value.to_ascii_lowercase())
.as_deref()
{
Some("fixed_order") => SchedulerSchedulingMode::FixedOrder,
Some("load_balance") => SchedulerSchedulingMode::LoadBalance,
_ => SchedulerSchedulingMode::CacheAffinity,
}
}
/// Effective scheduler ordering config for requests that carry no resolved
/// routing policy.
///
/// Resolution order:
/// 1. the enabled system-default routing group's `default_policy`;
/// 2. the legacy system-config keys (`provider_priority_mode`,
/// `scheduling_mode`, `keep_priority_on_conversion`).
///
/// Step 2 only exists so deployments that never created a routing group keep
/// their behaviour; once the legacy keys are removed this function collapses
/// to step 1 plus `SchedulerOrderingConfig::default()`.
pub(crate) async fn read_scheduler_ordering_config(
state: &AppState,
) -> Result<SchedulerOrderingConfig, GatewayError> {
if let Some(config) = read_system_default_routing_ordering_config(state).await? {
return Ok(config);
}
read_legacy_scheduler_ordering_config(state).await
}
/// Ordering config from the enabled system-default routing group, if any.
pub(crate) async fn read_system_default_routing_ordering_config(
state: &AppState,
@@ -201,38 +147,6 @@ pub(crate) async fn read_system_default_routing_ordering_config(
)))
}
/// Legacy system-config based ordering config. Kept only as a migration
/// fallback; see `read_scheduler_ordering_config`.
pub(crate) async fn read_legacy_scheduler_ordering_config(
state: &AppState,
) -> Result<SchedulerOrderingConfig, GatewayError> {
let priority_mode = parse_scheduler_priority_mode(
state
.read_system_config_json_value("provider_priority_mode")
.await?
.as_ref(),
);
let scheduling_mode = parse_scheduler_scheduling_mode(
state
.read_system_config_json_value("scheduling_mode")
.await?
.as_ref(),
);
let keep_priority_on_conversion = parse_keep_priority_on_conversion(
state
.read_system_config_json_value("keep_priority_on_conversion")
.await?
.as_ref(),
);
Ok(SchedulerOrderingConfig {
priority_mode,
scheduling_mode,
keep_priority_on_conversion,
// Legacy config never carried a sticky-key setting; use the routing default.
sticky_key_attempts: DEFAULT_STICKY_KEY_ATTEMPTS,
})
}
#[cfg(test)]
mod tests {
use std::sync::Arc;
@@ -247,14 +161,6 @@ mod tests {
use super::*;
use crate::data::GatewayDataState;
fn legacy_values() -> [(String, serde_json::Value); 3] {
[
("provider_priority_mode".to_string(), json!("global_key")),
("scheduling_mode".to_string(), json!("load_balance")),
("keep_priority_on_conversion".to_string(), json!(true)),
]
}
async fn create_system_default(
repository: &InMemoryRoutingGroupRepository,
enabled: bool,
@@ -267,6 +173,7 @@ mod tests {
description: None,
enabled,
is_system_default: true,
sort_order: 0,
config_json,
version: 1,
created_at: 1,
@@ -278,7 +185,7 @@ mod tests {
}
#[tokio::test]
async fn system_default_routing_group_overrides_legacy_keys() {
async fn system_default_routing_group_exposes_strategy_ordering() {
let repository = Arc::new(InMemoryRoutingGroupRepository::default());
create_system_default(
&repository,
@@ -293,12 +200,13 @@ mod tests {
)
.await;
let state = AppState::new().unwrap().with_data_state_for_tests(
GatewayDataState::disabled()
.with_system_config_values_for_tests(legacy_values())
.with_routing_group_repository_for_tests(repository),
GatewayDataState::disabled().with_routing_group_repository_for_tests(repository),
);
let config = read_scheduler_ordering_config(&state).await.unwrap();
let config = read_system_default_routing_ordering_config(&state)
.await
.unwrap()
.unwrap();
assert_eq!(config.priority_mode, SchedulerPriorityMode::Provider);
assert_eq!(config.scheduling_mode, SchedulerSchedulingMode::FixedOrder);
@@ -310,18 +218,19 @@ mod tests {
let repository = Arc::new(InMemoryRoutingGroupRepository::default());
create_system_default(&repository, true, json!({})).await;
let state = AppState::new().unwrap().with_data_state_for_tests(
GatewayDataState::disabled()
.with_system_config_values_for_tests(legacy_values())
.with_routing_group_repository_for_tests(repository),
GatewayDataState::disabled().with_routing_group_repository_for_tests(repository),
);
let config = read_scheduler_ordering_config(&state).await.unwrap();
let config = read_system_default_routing_ordering_config(&state)
.await
.unwrap()
.unwrap();
assert_eq!(config, SchedulerOrderingConfig::default());
}
#[tokio::test]
async fn disabled_or_missing_system_default_group_falls_back_to_legacy_keys() {
async fn disabled_or_missing_system_default_group_uses_routing_defaults() {
let repository = Arc::new(InMemoryRoutingGroupRepository::default());
create_system_default(
&repository,
@@ -330,28 +239,25 @@ mod tests {
)
.await;
let with_disabled_group = AppState::new().unwrap().with_data_state_for_tests(
GatewayDataState::disabled()
.with_system_config_values_for_tests(legacy_values())
.with_routing_group_repository_for_tests(repository),
);
let without_repository = AppState::new().unwrap().with_data_state_for_tests(
GatewayDataState::disabled().with_system_config_values_for_tests(legacy_values()),
GatewayDataState::disabled().with_routing_group_repository_for_tests(repository),
);
let without_repository = AppState::new()
.unwrap()
.with_data_state_for_tests(GatewayDataState::disabled());
for state in [with_disabled_group, without_repository] {
let config = read_scheduler_ordering_config(&state).await.unwrap();
assert_eq!(config.priority_mode, SchedulerPriorityMode::GlobalKey);
assert_eq!(config.scheduling_mode, SchedulerSchedulingMode::LoadBalance);
assert!(config.keep_priority_on_conversion);
let config = read_system_default_routing_ordering_config(&state)
.await
.unwrap();
assert!(config.is_none());
}
}
#[tokio::test]
async fn bootstrap_creates_system_default_group_from_legacy_keys_once() {
async fn bootstrap_creates_system_default_group_from_routing_defaults_once() {
let repository = Arc::new(InMemoryRoutingGroupRepository::default());
let state = AppState::new().unwrap().with_data_state_for_tests(
GatewayDataState::disabled()
.with_system_config_values_for_tests(legacy_values())
.with_routing_group_repository_for_tests(repository.clone()),
);
@@ -365,9 +271,9 @@ mod tests {
assert_eq!(
created.config_json["default_policy"],
json!({
"priority_mode": "global_key",
"scheduling_mode": "load_balance",
"keep_priority_on_conversion": true,
"priority_mode": "provider",
"scheduling_mode": "cache_affinity",
"keep_priority_on_conversion": false,
"sticky_key_attempts": DEFAULT_STICKY_KEY_ATTEMPTS
})
);
@@ -386,9 +292,39 @@ mod tests {
Some(created.id)
);
let config = read_scheduler_ordering_config(&state).await.unwrap();
assert_eq!(config.priority_mode, SchedulerPriorityMode::GlobalKey);
assert_eq!(config.scheduling_mode, SchedulerSchedulingMode::LoadBalance);
assert!(config.keep_priority_on_conversion);
let config = read_system_default_routing_ordering_config(&state)
.await
.unwrap()
.unwrap();
assert_eq!(config, SchedulerOrderingConfig::default());
}
#[tokio::test]
async fn bootstrap_does_not_migrate_legacy_scheduler_keys() {
let repository = Arc::new(InMemoryRoutingGroupRepository::default());
let state = AppState::new().unwrap().with_data_state_for_tests(
GatewayDataState::disabled()
.with_system_config_values_for_tests([
("provider_priority_mode".to_string(), json!("global_key")),
("scheduling_mode".to_string(), json!("load_balance")),
("keep_priority_on_conversion".to_string(), json!(true)),
])
.with_routing_group_repository_for_tests(repository),
);
let created = state
.ensure_system_default_routing_group_inner()
.await
.unwrap()
.expect("bootstrap should create the strategy");
assert_eq!(
created.config_json["default_policy"],
json!({
"priority_mode": "provider",
"scheduling_mode": "cache_affinity",
"keep_priority_on_conversion": false,
"sticky_key_attempts": DEFAULT_STICKY_KEY_ATTEMPTS
})
);
}
}
@@ -10,8 +10,6 @@ use async_trait::async_trait;
use crate::GatewayError;
use super::config::SchedulerOrderingConfig;
#[async_trait]
pub(crate) trait SchedulerRuntimeState {
async fn read_provider_quota_snapshot(
@@ -60,7 +58,4 @@ pub(crate) trait SchedulerRuntimeState {
max_entries: usize,
expected_epoch: Option<u64>,
) -> bool;
async fn read_scheduler_ordering_config(&self)
-> Result<SchedulerOrderingConfig, GatewayError>;
}