refactor: 大规模模块拆分与代码精简,新增 ai-pipeline/data-contracts 独立 crate

- 新增 aether-ai-pipeline 和 aether-data-contracts crate,将 pipeline 逻辑与数据契约从 gateway 中解耦
- 重构 admin handlers:拆分单体模块为 auth/billing/endpoint/features/model/observability/provider/system 等独立子模块
- 合并 chat/cli 重复代码路径:精简 conversion、finalize、planner 中的 sync/chat/cli 分支
- 重构 scheduler/executor/data 层,引入 facade 模式降低模块间耦合
- 移除冗余的 intent 模块,将 plan_fallback/policy/stream_path/sync_path 迁移至 executor
- 前端适配:调整 admin API 调用和 provider 模型测试对话框
This commit is contained in:
fawney19
2026-04-07 02:50:19 +08:00
parent 763ff03a7b
commit 5d96d6673b
732 changed files with 28589 additions and 20662 deletions
@@ -0,0 +1,182 @@
use std::collections::{BTreeMap, BTreeSet};
use aether_data_contracts::repository::candidates::StoredRequestCandidate;
use aether_data_contracts::repository::provider_catalog::StoredProviderCatalogKey;
use aether_scheduler_core::{
auth_api_key_concurrency_limit_reached, build_provider_concurrent_limit_map,
candidate_is_selectable_with_runtime_state, SchedulerAffinityTarget,
};
use crate::data::auth::GatewayAuthApiKeySnapshot;
use crate::GatewayError;
use super::{SchedulerMinimalCandidateSelectionCandidate, SchedulerRuntimeState};
pub(super) use aether_scheduler_core::should_skip_provider_quota;
pub(super) struct CandidateRuntimeSelectionSnapshot {
pub(super) recent_candidates: Vec<StoredRequestCandidate>,
pub(super) provider_concurrent_limits: BTreeMap<String, usize>,
pub(super) provider_key_rpm_states: BTreeMap<String, StoredProviderCatalogKey>,
provider_quota_blocks_requests: BTreeMap<String, bool>,
provider_key_rpm_reset_ats: BTreeMap<String, Option<u64>>,
}
pub(super) async fn read_candidate_runtime_selection_snapshot(
state: &(impl SchedulerRuntimeState + ?Sized),
candidates: &[SchedulerMinimalCandidateSelectionCandidate],
now_unix_secs: u64,
) -> Result<CandidateRuntimeSelectionSnapshot, GatewayError> {
let recent_candidates = state.read_recent_request_candidates(128).await?;
let provider_concurrent_limits = read_provider_concurrent_limits(state, candidates).await?;
let provider_key_rpm_states = read_provider_key_rpm_states(state, candidates).await?;
let provider_quota_blocks_requests =
read_provider_quota_block_map(state, candidates, now_unix_secs).await?;
let provider_key_rpm_reset_ats =
read_provider_key_rpm_reset_at_map(state, candidates, now_unix_secs);
Ok(CandidateRuntimeSelectionSnapshot {
recent_candidates,
provider_concurrent_limits,
provider_key_rpm_states,
provider_quota_blocks_requests,
provider_key_rpm_reset_ats,
})
}
pub(super) fn auth_snapshot_concurrency_limit_reached(
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
snapshot: &CandidateRuntimeSelectionSnapshot,
now_unix_secs: u64,
) -> bool {
auth_snapshot
.and_then(|snapshot| {
usize::try_from(snapshot.api_key_concurrent_limit?)
.ok()
.and_then(|limit| {
if limit == 0 {
return None;
}
Some((snapshot.api_key_id.as_str(), limit))
})
})
.is_some_and(|(api_key_id, limit)| {
auth_api_key_concurrency_limit_reached(
&snapshot.recent_candidates,
now_unix_secs,
api_key_id,
limit,
)
})
}
pub(super) fn is_candidate_selectable(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
snapshot: &CandidateRuntimeSelectionSnapshot,
now_unix_secs: u64,
cached_affinity_target: Option<&SchedulerAffinityTarget>,
) -> bool {
let provider_quota_blocks_requests = snapshot
.provider_quota_blocks_requests
.get(candidate.provider_id.as_str())
.copied()
.unwrap_or(false);
let rpm_reset_at = snapshot
.provider_key_rpm_reset_ats
.get(candidate.key_id.as_str())
.copied()
.flatten();
candidate_is_selectable_with_runtime_state(
candidate,
&snapshot.recent_candidates,
&snapshot.provider_concurrent_limits,
&snapshot.provider_key_rpm_states,
now_unix_secs,
cached_affinity_target,
provider_quota_blocks_requests,
rpm_reset_at,
)
}
pub(super) async fn read_provider_concurrent_limits(
state: &(impl SchedulerRuntimeState + ?Sized),
candidates: &[SchedulerMinimalCandidateSelectionCandidate],
) -> Result<BTreeMap<String, usize>, GatewayError> {
let provider_ids = candidates
.iter()
.map(|candidate| candidate.provider_id.clone())
.collect::<BTreeSet<_>>()
.into_iter()
.collect::<Vec<_>>();
if provider_ids.is_empty() {
return Ok(BTreeMap::new());
}
let providers = state
.read_provider_catalog_providers_by_ids(&provider_ids)
.await?;
Ok(build_provider_concurrent_limit_map(providers))
}
pub(super) async fn read_provider_key_rpm_states(
state: &(impl SchedulerRuntimeState + ?Sized),
candidates: &[SchedulerMinimalCandidateSelectionCandidate],
) -> Result<BTreeMap<String, StoredProviderCatalogKey>, GatewayError> {
let key_ids = candidates
.iter()
.map(|candidate| candidate.key_id.clone())
.collect::<BTreeSet<_>>()
.into_iter()
.collect::<Vec<_>>();
if key_ids.is_empty() {
return Ok(BTreeMap::new());
}
let keys = state.read_provider_catalog_keys_by_ids(&key_ids).await?;
Ok(keys
.into_iter()
.map(|key| (key.id.clone(), key))
.collect::<BTreeMap<_, _>>())
}
async fn read_provider_quota_block_map(
state: &(impl SchedulerRuntimeState + ?Sized),
candidates: &[SchedulerMinimalCandidateSelectionCandidate],
now_unix_secs: u64,
) -> Result<BTreeMap<String, bool>, GatewayError> {
let provider_ids = candidates
.iter()
.map(|candidate| candidate.provider_id.clone())
.collect::<BTreeSet<_>>()
.into_iter()
.collect::<Vec<_>>();
let mut quota_blocks = BTreeMap::new();
for provider_id in provider_ids {
let blocks_requests = state
.read_provider_quota_snapshot(&provider_id)
.await?
.as_ref()
.is_some_and(|quota| should_skip_provider_quota(quota, now_unix_secs));
quota_blocks.insert(provider_id, blocks_requests);
}
Ok(quota_blocks)
}
fn read_provider_key_rpm_reset_at_map(
state: &(impl SchedulerRuntimeState + ?Sized),
candidates: &[SchedulerMinimalCandidateSelectionCandidate],
now_unix_secs: u64,
) -> BTreeMap<String, Option<u64>> {
candidates
.iter()
.map(|candidate| {
(
candidate.key_id.clone(),
state.provider_key_rpm_reset_at(candidate.key_id.as_str(), now_unix_secs),
)
})
.collect::<BTreeMap<_, _>>()
}