Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -92,20 +92,17 @@ public CustomSceneDefinition generate(
String normalizedInput = requiredInput(sceneInput);
String prompt = buildPrompt(normalizedInput, currentPreference, profile);
BusinessException lastFailure = null;
List<String> models = List.of(
AiProviderRegistry.QWEN_LLM_FLASH,
AiProviderRegistry.QWEN_LLM_PLUS);
for (int index = 0; index < models.size(); index++) {
int maximumAttempts = 2;
for (int index = 0; index < maximumAttempts; index++) {
int attempt = index + 1;
String modelId = models.get(index);
String attemptPrompt = attempt == 1
? prompt
: prompt + "\n\nA prior generation attempt did not satisfy the JSON contract. "
+ "Return a corrected JSON object only.";
try {
long llmStartedAt = System.nanoTime();
String content = providerRegistry.executeLlmTask(
modelId,
null,
attemptPrompt,
null,
LlmResponseFormat.JSON_OBJECT);
Expand All @@ -118,9 +115,8 @@ public CustomSceneDefinition generate(
catch (BusinessException exception) {
if ("CUSTOM_SCENE_LLM_RESPONSE_INVALID".equals(exception.code())) {
LOGGER.warn(
"custom scene LLM response rejected sceneId={} model={} attempt={} llmMs={} parseMs={} responseChars={}",
"custom scene LLM response rejected sceneId={} route=default attempt={} llmMs={} parseMs={} responseChars={}",
sceneId,
modelId,
attempt,
llmMillis,
elapsedMillis(parseStartedAt),
Expand All @@ -129,22 +125,20 @@ public CustomSceneDefinition generate(
throw exception;
}
LOGGER.info(
"custom scene LLM completed sceneId={} model={} attempt={} llmMs={} parseMs={}",
"custom scene LLM completed sceneId={} route=default attempt={} llmMs={} parseMs={}",
sceneId,
modelId,
attempt,
llmMillis,
elapsedMillis(parseStartedAt));
return definition;
}
catch (BusinessException exception) {
lastFailure = exception;
if (index + 1 < models.size()) {
if (index + 1 < maximumAttempts) {
LOGGER.warn(
"custom scene LLM falling back sceneId={} failedModel={} nextModel={} code={}",
"custom scene LLM retrying configured route sceneId={} attempt={} code={}",
sceneId,
modelId,
models.get(index + 1),
attempt,
exception.code());
continue;
}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,8 @@ public class QwenLlmProvider extends LlmProvider {
private final ObjectMapper objectMapper;
private final String apiKey;
private final URI endpoint;
private String workspaceId = "";
private String region = "";
private final String model;
private final Duration readTimeout;
private final int maxResponseBytes;
Expand All @@ -63,6 +65,8 @@ public QwenLlmProvider(
model,
positiveDuration(readTimeoutSeconds, "Qwen LLM read timeout"),
maxResponseBytes);
this.workspaceId = trim(workspaceId);
this.region = trim(region);
}

public QwenLlmProvider(
Expand Down Expand Up @@ -131,7 +135,8 @@ private AiProviderResponse<String> callForContent(
if (prompt.isBlank()) {
throw nonRetryableFailure("INVALID_LLM_PROMPT", "LLM task prompt is required");
}
requireHttpsEndpoint();
URI requestEndpoint = resolveEndpoint();
requireHttpsEndpoint(requestEndpoint);

try {
Map<String, Object> body = new LinkedHashMap<>();
Expand All @@ -142,7 +147,7 @@ private AiProviderResponse<String> callForContent(
body.put("response_format", Map.of("type", "json_object"));
}
HttpRequest httpRequest = HttpRequest.newBuilder()
.uri(endpoint)
.uri(requestEndpoint)
.timeout(readTimeout)
.header("Authorization", "Bearer " + credential)
.header("Content-Type", "application/json")
Expand Down Expand Up @@ -225,19 +230,27 @@ private Object parseContent(String content) {
}
}

private void requireHttpsEndpoint() {
String host = endpoint == null || endpoint.getHost() == null
private URI resolveEndpoint() {
String effectiveWorkspaceId = ProviderCredentialOverride.currentOr("workspaceId", workspaceId);
if (!effectiveWorkspaceId.isBlank()) {
return buildEndpoint(effectiveWorkspaceId, region);
}
return endpoint;
}

private void requireHttpsEndpoint(URI requestEndpoint) {
String host = requestEndpoint == null || requestEndpoint.getHost() == null
? ""
: endpoint.getHost().toLowerCase(java.util.Locale.ROOT);
if (endpoint == null
|| !endpoint.isAbsolute()
|| !"https".equalsIgnoreCase(endpoint.getScheme())
: requestEndpoint.getHost().toLowerCase(java.util.Locale.ROOT);
if (requestEndpoint == null
|| !requestEndpoint.isAbsolute()
|| !"https".equalsIgnoreCase(requestEndpoint.getScheme())
|| !host.endsWith(".maas.aliyuncs.com")
|| endpoint.getUserInfo() != null
|| endpoint.getPort() != -1
|| !"/compatible-mode/v1/chat/completions".equals(endpoint.getPath())
|| endpoint.getRawQuery() != null
|| endpoint.getRawFragment() != null) {
|| requestEndpoint.getUserInfo() != null
|| requestEndpoint.getPort() != -1
|| !"/compatible-mode/v1/chat/completions".equals(requestEndpoint.getPath())
|| requestEndpoint.getRawQuery() != null
|| requestEndpoint.getRawFragment() != null) {
throw retryableFailure(
"QWEN_LLM_ENDPOINT_INVALID",
"Qwen LLM endpoint must be the trusted Aliyun compatible-mode URL");
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -115,9 +115,7 @@ public AiProviderRegistry(
@Value("${AI_PROVIDER_ROUTE_TTS:}")
String ttsRoute,
@Value("${AI_PROVIDER_ROUTE_TRANSCRIPTION:}")
String transcriptionRoute,
@Value("${AI_QINIU_MODELS_ENABLED:false}")
boolean qiniuModelsEnabled) {
String transcriptionRoute) {
this(
realtimeProviders,
llmProviders,
Expand All @@ -129,8 +127,7 @@ AiCapability.REALTIME, parseRoute(realtimeRoute),
AiCapability.LLM, parseRoute(llmRoute),
AiCapability.SCORING, parseRoute(scoringRoute),
AiCapability.TTS, parseRoute(ttsRoute),
AiCapability.TRANSCRIPTION, parseRoute(transcriptionRoute)),
qiniuModelsEnabled);
AiCapability.TRANSCRIPTION, parseRoute(transcriptionRoute)));
}

public AiProviderRegistry(
Expand All @@ -155,30 +152,6 @@ public AiProviderRegistry(
List<TtsProvider> ttsProviders,
List<TranscriptionProvider> transcriptionProviders,
Map<AiCapability, List<String>> configuredRoutes) {
this(
realtimeProviders,
llmProviders,
scoringProviders,
ttsProviders,
transcriptionProviders,
configuredRoutes,
true);
}

AiProviderRegistry(
List<RealtimeProvider> realtimeProviders,
List<LlmProvider> llmProviders,
List<ScoringProvider> scoringProviders,
List<TtsProvider> ttsProviders,
List<TranscriptionProvider> transcriptionProviders,
Map<AiCapability, List<String>> configuredRoutes,
boolean qiniuModelsEnabled) {
if (!qiniuModelsEnabled) {
realtimeProviders = withoutProvider(realtimeProviders, "qiniu");
llmProviders = withoutProvider(llmProviders, "qiniu-maas");
configuredRoutes = withoutQiniuModels(configuredRoutes);
LOGGER.info("Qiniu AI models are disabled; RTI and MaaS adapters will not be registered");
}
this.realtimeProviders = registerProviders(realtimeProviders, AiCapability.REALTIME);
this.llmProviders = registerProviders(llmProviders, AiCapability.LLM);
this.scoringProviders = registerProviders(scoringProviders, AiCapability.SCORING);
Expand All @@ -191,29 +164,6 @@ public AiProviderRegistry(
this.models = List.copyOf(modelDefinitions.values());
}

private static <T extends AbstractAiProvider> List<T> withoutProvider(
List<T> providers,
String providerId) {
return providers.stream()
.filter(provider -> !provider.providerId().equalsIgnoreCase(providerId))
.toList();
}

private static Map<AiCapability, List<String>> withoutQiniuModels(
Map<AiCapability, List<String>> configuredRoutes) {
if (configuredRoutes == null || configuredRoutes.isEmpty()) return Map.of();
Map<AiCapability, List<String>> filtered = new EnumMap<>(AiCapability.class);
configuredRoutes.forEach((capability, route) -> filtered.put(
capability,
route.stream()
.map(AbstractAiProvider::normalizeModelId)
.filter(modelId -> !modelId.equals(QINIU_REALTIME_PLUS))
.filter(modelId -> !modelId.equals(QINIU_MAAS_QWEN_PLUS))
.filter(modelId -> !modelId.equals(QINIU_MAAS_DEEPSEEK_FLASH))
.toList()));
return Map.copyOf(filtered);
}

public List<AiModelDefinition> models() {
AiRuntimeConfiguration runtime = runtimeConfiguration();
if (!runtime.databaseBacked()) return models;
Expand Down Expand Up @@ -426,9 +376,16 @@ public byte[] generateSpeechAudioBytes(
String token,
String voice) {
if (modelId == null || modelId.isBlank()) {
throw new BusinessException(
"AI_TTS_MODEL_REQUIRED",
"A voice-specific TTS request requires an explicit model");
return unboxAudio(invokeRouteWithResult(
context,
AiCapability.TTS,
id -> {
AiProviderResponse<byte[]> measured = getTtsProvider(id)
.generateSpeechAudioMeasured(
text, credential(id, token), voice);
return new AiProviderResponse<>(boxAudio(measured.response()),
measured.providerRequestId(), measured.usage());
}).response());
}
return invokeExplicitMeasured(context, AiCapability.TTS, modelId,
id -> getTtsProvider(id).generateSpeechAudioMeasured(
Expand All @@ -444,6 +401,9 @@ public String executeLlmTask(
String prompt,
String token,
LlmResponseFormat responseFormat) {
if (modelId == null || modelId.isBlank()) {
return executeLlmTaskRouted(prompt, token, responseFormat).response();
}
return invokeExplicitMeasured(
automaticContext("llm"),
AiCapability.LLM,
Expand Down Expand Up @@ -855,7 +815,14 @@ private AiModelConfiguration modelConfigurationForLedger(String modelId, AiCapab

private AiInvocationContext automaticContext(String businessScene) {
AiInvocationContext scoped = AiInvocationContexts.current();
if (scoped != null) return scoped;
if (scoped != null) {
return new AiInvocationContext(
UUID.randomUUID(),
scoped.userId(),
scoped.sessionId(),
scoped.businessScene(),
scoped.routeKey());
}
String userId = null;
try {
if (authService != null) userId = authService.currentUserIdOrNull();
Expand All @@ -877,11 +844,16 @@ private static long elapsedMillis(long startedNanos) {
}

private boolean shouldFailOver(BusinessException exception) {
String code = exception.code() == null ? "" : exception.code();
// Authentication and account-policy failures cannot be retried against the
// same Qiniu model, but they must not prevent the configured route fallback.
if ("QINIU_MAAS_LLM_REQUEST_FAILED".equals(code)) {
return true;
}
Boolean classifiedRetryable = AbstractAiProvider.retryable(exception);
if (classifiedRetryable != null) {
return classifiedRetryable;
}
String code = exception.code() == null ? "" : exception.code();
return !code.startsWith("INVALID_")
&& !code.startsWith("UNSUPPORTED_")
&& !code.endsWith("_INTERRUPTED")
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -105,7 +105,7 @@ public byte[] synthesizeSpeech(String sceneId, String text, String model) {
}
UserProfile profile = profileService.getProfile(definition.userId());
byte[] audio = providerRegistry.generateSpeechAudioBytes(
AiProviderRegistry.QWEN_TTS,
model,
text.strip(),
null,
profile == null ? null : profile.voiceId());
Expand Down
Loading
Loading