Improve gateway transport and usage runtime

This commit is contained in:
elky
2026-06-25 22:36:27 +08:00
parent d336d1a7fa
commit 6f00e9fc67
112 changed files with 12456 additions and 1387 deletions
@@ -17,9 +17,6 @@ pub(super) fn build_scheduler_affinity_cache_key(
global_model_name: &str,
client_session_affinity: Option<&ClientSessionAffinity>,
) -> Option<String> {
if !has_explicit_session_affinity(client_session_affinity) {
return None;
}
let api_key_id = auth_snapshot
.map(|snapshot| snapshot.api_key_id.trim())
.filter(|value| !value.is_empty())?;
@@ -225,6 +225,7 @@ pub(super) async fn collect_selectable_enumerated_candidates_with_skip_reasons(
);
let cached_affinity_target = if ordering_config.scheduling_mode
== SchedulerSchedulingMode::CacheAffinity
&& has_explicit_session_affinity(client_session_affinity)
{
affinity_cache_key.as_deref().and_then(|cache_key| {
runtime_state.read_cached_scheduler_affinity_target(cache_key, SCHEDULER_AFFINITY_TTL)
@@ -36,6 +36,7 @@ async fn select_candidate(
global_model_name: &str,
require_streaming: bool,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
) -> Result<Option<SchedulerMinimalCandidateSelectionCandidate>, GatewayError> {
select_candidate_impl(
@@ -46,7 +47,7 @@ async fn select_candidate(
require_streaming,
None,
auth_snapshot,
None,
client_session_affinity,
now_unix_secs,
false,
)
@@ -207,9 +208,14 @@ async fn reuses_cached_scheduler_affinity_candidate_before_sorted_fallback() {
);
let auth_snapshot = sample_auth_snapshot("affinity-key-1");
let cache_key =
build_scheduler_affinity_cache_key(Some(&auth_snapshot), "openai:chat", "gpt-4.1", None)
.expect("cache key should build");
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cache_key = build_scheduler_affinity_cache_key(
Some(&auth_snapshot),
"openai:chat",
"gpt-4.1",
Some(&client_session_affinity),
)
.expect("cache key should build");
state.remember_scheduler_affinity_target(
&cache_key,
SchedulerAffinityTarget {
@@ -228,6 +234,7 @@ async fn reuses_cached_scheduler_affinity_candidate_before_sorted_fallback() {
"gpt-4.1",
false,
Some(&auth_snapshot),
Some(&client_session_affinity),
100,
)
.await
@@ -312,9 +319,14 @@ async fn cached_affinity_candidate_cannot_use_reserved_provider_key_rpm_capacity
);
let auth_snapshot = sample_auth_snapshot("api-key-cached-user");
let cache_key =
build_scheduler_affinity_cache_key(Some(&auth_snapshot), "openai:chat", "gpt-4.1", None)
.expect("cache key should build");
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cache_key = build_scheduler_affinity_cache_key(
Some(&auth_snapshot),
"openai:chat",
"gpt-4.1",
Some(&client_session_affinity),
)
.expect("cache key should build");
state.remember_scheduler_affinity_target(
&cache_key,
SchedulerAffinityTarget {
@@ -333,6 +345,7 @@ async fn cached_affinity_candidate_cannot_use_reserved_provider_key_rpm_capacity
"gpt-4.1",
false,
Some(&auth_snapshot),
Some(&client_session_affinity),
100,
)
.await