mirror of
https://github.com/fawney19/Aether.git
synced 2026-10-09 02:47:45 +08:00
Improve gateway scheduling and runtime admission
This commit is contained in:
@@ -17,6 +17,9 @@ pub(super) fn build_scheduler_affinity_cache_key(
|
||||
global_model_name: &str,
|
||||
client_session_affinity: Option<&ClientSessionAffinity>,
|
||||
) -> Option<String> {
|
||||
if !has_explicit_session_affinity(client_session_affinity) {
|
||||
return None;
|
||||
}
|
||||
let api_key_id = auth_snapshot
|
||||
.map(|snapshot| snapshot.api_key_id.trim())
|
||||
.filter(|value| !value.is_empty())?;
|
||||
@@ -28,6 +31,12 @@ pub(super) fn build_scheduler_affinity_cache_key(
|
||||
)
|
||||
}
|
||||
|
||||
pub(super) fn has_explicit_session_affinity(
|
||||
client_session_affinity: Option<&ClientSessionAffinity>,
|
||||
) -> bool {
|
||||
client_session_affinity.is_some_and(ClientSessionAffinity::has_session_key)
|
||||
}
|
||||
|
||||
pub(super) fn scheduler_candidate_affinity_hash(
|
||||
affinity_key: &str,
|
||||
candidate: &SchedulerMinimalCandidateSelectionCandidate,
|
||||
|
||||
@@ -138,8 +138,11 @@ pub(crate) async fn list_selectable_enumerated_candidates_with_skip_reasons(
|
||||
GatewayError,
|
||||
> {
|
||||
let ordering_config = runtime_state.read_scheduler_ordering_config().await?;
|
||||
let priority_affinity_key =
|
||||
selection::scheduling_priority_affinity_key(auth_snapshot, ordering_config.scheduling_mode);
|
||||
let priority_affinity_key = selection::scheduling_priority_affinity_key(
|
||||
auth_snapshot,
|
||||
client_session_affinity,
|
||||
ordering_config.scheduling_mode,
|
||||
);
|
||||
collect_selectable_enumerated_candidates_with_skip_reasons(
|
||||
runtime_state,
|
||||
api_format,
|
||||
|
||||
@@ -5,7 +5,9 @@ use crate::scheduler::config::SchedulerSchedulingMode;
|
||||
use crate::GatewayError;
|
||||
use aether_scheduler_core::ClientSessionAffinity;
|
||||
|
||||
use super::affinity::{build_scheduler_affinity_cache_key, remember_scheduler_affinity};
|
||||
use super::affinity::{
|
||||
build_scheduler_affinity_cache_key, has_explicit_session_affinity, remember_scheduler_affinity,
|
||||
};
|
||||
use super::enumeration::enumerate_scheduler_candidates;
|
||||
use super::ranking::rank_scheduler_candidates;
|
||||
use super::resolution::resolve_scheduler_candidate_selectability;
|
||||
@@ -66,8 +68,11 @@ pub(super) async fn select_minimal_candidate(
|
||||
global_model_name,
|
||||
client_session_affinity,
|
||||
);
|
||||
let priority_affinity_key =
|
||||
scheduling_priority_affinity_key(auth_snapshot, ordering_config.scheduling_mode);
|
||||
let priority_affinity_key = scheduling_priority_affinity_key(
|
||||
auth_snapshot,
|
||||
client_session_affinity,
|
||||
ordering_config.scheduling_mode,
|
||||
);
|
||||
let candidates = enumerate_scheduler_candidates(
|
||||
selection_row_source,
|
||||
api_format,
|
||||
@@ -94,7 +99,9 @@ pub(super) async fn select_minimal_candidate(
|
||||
.0
|
||||
.into_iter()
|
||||
.next();
|
||||
if ordering_config.scheduling_mode == SchedulerSchedulingMode::CacheAffinity {
|
||||
if ordering_config.scheduling_mode == SchedulerSchedulingMode::CacheAffinity
|
||||
&& has_explicit_session_affinity(client_session_affinity)
|
||||
{
|
||||
if let Some(candidate) = selected.as_ref() {
|
||||
remember_scheduler_affinity(
|
||||
affinity_cache_key.as_deref(),
|
||||
@@ -154,8 +161,11 @@ pub(super) async fn collect_selectable_candidates_with_skip_reasons(
|
||||
GatewayError,
|
||||
> {
|
||||
let ordering_config = runtime_state.read_scheduler_ordering_config().await?;
|
||||
let priority_affinity_key =
|
||||
scheduling_priority_affinity_key(auth_snapshot, ordering_config.scheduling_mode);
|
||||
let priority_affinity_key = scheduling_priority_affinity_key(
|
||||
auth_snapshot,
|
||||
client_session_affinity,
|
||||
ordering_config.scheduling_mode,
|
||||
);
|
||||
let candidates = enumerate_scheduler_candidates(
|
||||
selection_row_source,
|
||||
api_format,
|
||||
@@ -262,11 +272,17 @@ pub(super) async fn collect_selectable_enumerated_candidates_with_skip_reasons(
|
||||
|
||||
pub(super) fn scheduling_priority_affinity_key<'a>(
|
||||
auth_snapshot: Option<&'a GatewayAuthApiKeySnapshot>,
|
||||
client_session_affinity: Option<&ClientSessionAffinity>,
|
||||
scheduling_mode: SchedulerSchedulingMode,
|
||||
) -> Option<&'a str> {
|
||||
if scheduling_mode == SchedulerSchedulingMode::FixedOrder {
|
||||
return None;
|
||||
}
|
||||
if scheduling_mode == SchedulerSchedulingMode::CacheAffinity
|
||||
&& !has_explicit_session_affinity(client_session_affinity)
|
||||
{
|
||||
return None;
|
||||
}
|
||||
|
||||
auth_snapshot
|
||||
.map(|snapshot| snapshot.api_key_id.trim())
|
||||
|
||||
@@ -13,7 +13,7 @@ use aether_data_contracts::repository::candidates::{
|
||||
};
|
||||
use aether_data_contracts::repository::provider_catalog::StoredProviderCatalogKey;
|
||||
use aether_data_contracts::repository::quota::StoredProviderQuotaSnapshot;
|
||||
use aether_scheduler_core::SchedulerMinimalCandidateSelectionCandidate;
|
||||
use aether_scheduler_core::{ClientSessionAffinity, SchedulerMinimalCandidateSelectionCandidate};
|
||||
use serde_json::json;
|
||||
|
||||
use crate::cache::SchedulerAffinityTarget;
|
||||
@@ -559,9 +559,85 @@ async fn cache_affinity_promotes_cached_scheduler_affinity_candidate_when_enable
|
||||
second.endpoint_id = "endpoint-b".to_string();
|
||||
second.key_id = "key-b".to_string();
|
||||
second.key_name = "beta".to_string();
|
||||
second.provider_priority = 0;
|
||||
second.key_internal_priority = 0;
|
||||
second.key_global_priority_by_format = Some(json!({"openai:chat": 0}));
|
||||
second.provider_priority = 10;
|
||||
second.key_internal_priority = 10;
|
||||
second.key_global_priority_by_format = Some(json!({"openai:chat": 10}));
|
||||
|
||||
let candidates = Arc::new(InMemoryMinimalCandidateSelectionReadRepository::seed(vec![
|
||||
first, second,
|
||||
]));
|
||||
let quotas = Arc::new(InMemoryProviderQuotaRepository::seed(vec![]));
|
||||
let state = AppState::new()
|
||||
.expect("state should build")
|
||||
.with_data_state_for_tests(
|
||||
GatewayDataState::with_candidate_selection_and_quota_for_tests(candidates, quotas)
|
||||
.with_system_config_values_for_tests(vec![(
|
||||
"scheduling_mode".to_string(),
|
||||
json!("cache_affinity"),
|
||||
)]),
|
||||
);
|
||||
|
||||
let auth_snapshot = sample_auth_snapshot("affinity-key-1");
|
||||
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
|
||||
let cache_key = build_scheduler_affinity_cache_key(
|
||||
Some(&auth_snapshot),
|
||||
"openai:chat",
|
||||
"gpt-4.1",
|
||||
Some(&client_session_affinity),
|
||||
)
|
||||
.expect("scheduler affinity cache key should build");
|
||||
state.remember_scheduler_affinity_target(
|
||||
&cache_key,
|
||||
SchedulerAffinityTarget {
|
||||
provider_id: "provider-b".to_string(),
|
||||
endpoint_id: "endpoint-b".to_string(),
|
||||
key_id: "key-b".to_string(),
|
||||
},
|
||||
Duration::from_secs(300),
|
||||
100,
|
||||
);
|
||||
|
||||
let selected = select_candidate_impl(
|
||||
state.data.as_ref(),
|
||||
&state,
|
||||
"openai:chat",
|
||||
"gpt-4.1",
|
||||
false,
|
||||
None,
|
||||
Some(&auth_snapshot),
|
||||
Some(&client_session_affinity),
|
||||
100,
|
||||
false,
|
||||
)
|
||||
.await
|
||||
.expect("selection should succeed")
|
||||
.expect("candidate should exist");
|
||||
|
||||
assert_eq!(selected.provider_id, "provider-b");
|
||||
assert_eq!(selected.key_id, "key-b");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn cache_affinity_ignores_cached_scheduler_affinity_without_client_session() {
|
||||
let mut first = sample_row();
|
||||
first.provider_id = "provider-a".to_string();
|
||||
first.provider_name = "provider-a".to_string();
|
||||
first.endpoint_id = "endpoint-a".to_string();
|
||||
first.key_id = "key-a".to_string();
|
||||
first.key_name = "alpha".to_string();
|
||||
first.provider_priority = 0;
|
||||
first.key_internal_priority = 0;
|
||||
first.key_global_priority_by_format = Some(json!({"openai:chat": 0}));
|
||||
|
||||
let mut second = sample_row();
|
||||
second.provider_id = "provider-b".to_string();
|
||||
second.provider_name = "provider-b".to_string();
|
||||
second.endpoint_id = "endpoint-b".to_string();
|
||||
second.key_id = "key-b".to_string();
|
||||
second.key_name = "beta".to_string();
|
||||
second.provider_priority = 10;
|
||||
second.key_internal_priority = 10;
|
||||
second.key_global_priority_by_format = Some(json!({"openai:chat": 10}));
|
||||
|
||||
let candidates = Arc::new(InMemoryMinimalCandidateSelectionReadRepository::seed(vec![
|
||||
first, second,
|
||||
@@ -602,8 +678,8 @@ async fn cache_affinity_promotes_cached_scheduler_affinity_candidate_when_enable
|
||||
.expect("selection should succeed")
|
||||
.expect("candidate should exist");
|
||||
|
||||
assert_eq!(selected.provider_id, "provider-b");
|
||||
assert_eq!(selected.key_id, "key-b");
|
||||
assert_eq!(selected.provider_id, "provider-a");
|
||||
assert_eq!(selected.key_id, "key-a");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
@@ -623,18 +699,26 @@ async fn load_balance_selection_does_not_remember_scheduler_affinity() {
|
||||
)]),
|
||||
);
|
||||
let auth_snapshot = sample_auth_snapshot("affinity-key-1");
|
||||
let cache_key =
|
||||
build_scheduler_affinity_cache_key(Some(&auth_snapshot), "openai:chat", "gpt-4.1", None)
|
||||
.expect("scheduler affinity cache key should build");
|
||||
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
|
||||
let cache_key = build_scheduler_affinity_cache_key(
|
||||
Some(&auth_snapshot),
|
||||
"openai:chat",
|
||||
"gpt-4.1",
|
||||
Some(&client_session_affinity),
|
||||
)
|
||||
.expect("scheduler affinity cache key should build");
|
||||
|
||||
let selected = select_candidate(
|
||||
let selected = select_candidate_impl(
|
||||
state.data.as_ref(),
|
||||
&state,
|
||||
"openai:chat",
|
||||
"gpt-4.1",
|
||||
false,
|
||||
None,
|
||||
Some(&auth_snapshot),
|
||||
Some(&client_session_affinity),
|
||||
100,
|
||||
false,
|
||||
)
|
||||
.await
|
||||
.expect("selection should succeed")
|
||||
|
||||
Reference in New Issue
Block a user