From 40d3f033a4fdfdc06ab6816f8f622edb85a5a862 Mon Sep 17 00:00:00 2001 From: lb <542828+lukebuehler@users.noreply.github.com> Date: Mon, 28 Sep 2026 23:03:11 +0200 Subject: [PATCH 01/12] model defaults --- clients/typescript/schema/api.schema.json | 124 ++++- clients/typescript/src/errors.ts | 3 + clients/typescript/src/generated/methods.ts | 54 +- clients/typescript/src/generated/types.ts | 64 ++- clients/typescript/test/client.test.ts | 8 +- crates/api/contract/api-reference.md | 28 +- crates/api/contract/api.schema.json | 124 ++++- crates/api/contract/methods.json | 52 +- crates/api/contract/openrpc.json | 182 ++++++- crates/api/src/constants.rs | 2 + crates/api/src/lib.rs | 2 + crates/api/src/model_defaults.rs | 150 ++++++ crates/api/src/rpc.rs | 24 +- crates/api/src/schema_export.rs | 2 +- crates/api/src/service.rs | 14 + crates/api/src/sessions.rs | 4 +- crates/api/tests/schema_artifacts.rs | 15 + crates/cli/src/administration_cli.rs | 3 + crates/cli/src/main.rs | 1 + crates/cli/src/model_defaults_cli.rs | 192 +++++++ crates/cli/tests/resource_commands.rs | 98 ++++ .../migrations/010_model_defaults.sql | 8 + crates/store-pg/src/lib.rs | 2 + crates/store-pg/src/migrations.rs | 8 +- crates/store-pg/src/model_defaults.rs | 90 ++++ crates/store-pg/tests/model_defaults_live.rs | 157 ++++++ crates/temporal-server/src/config.rs | 32 +- crates/temporal-server/src/gateway/mod.rs | 4 +- .../src/gateway/service/api_config.rs | 20 +- .../src/gateway/service/mod.rs | 49 +- .../src/gateway/service/model_defaults.rs | 153 ++++++ .../src/gateway/service/profiles.rs | 31 +- .../src/gateway/service/session_lifecycle.rs | 4 +- crates/temporal-server/src/lib.rs | 4 +- crates/temporal-server/src/main.rs | 1 + crates/temporal-server/src/worker/reaper.rs | 4 +- crates/temporal-server/tests/bots_live.rs | 11 +- crates/temporal-server/tests/channels_live.rs | 1 + .../tests/environment_provider_live.rs | 6 +- crates/temporal-server/tests/mcp_live.rs | 21 +- .../temporal-server/tests/preprocess_live.rs | 9 +- crates/temporal-server/tests/profiles_live.rs | 10 +- crates/temporal-server/tests/runs_live.rs | 22 +- .../temporal-server/tests/runs_live_slow.rs | 7 +- crates/temporal-server/tests/sessions_live.rs | 135 +++-- .../temporal-server/tests/subagents_live.rs | 5 +- crates/temporal-server/tests/support/live.rs | 28 +- .../tests/vfs_transfer_live.rs | 2 +- .../tests/workflow_tool_plugins_live.rs | 4 +- crates/temporal-workflow/src/config.rs | 1 - crates/temporal-workflow/src/lib.rs | 2 +- ...iverse-model-defaults-and-transcription.md | 488 ++++++++++++++++++ .../configurator-mcp/src/generated/tools.ts | 100 +++- platform/server/src/routes/gateway.ts | 23 +- platform/server/src/routes/method-roles.ts | 2 + .../server/src/routes/model-defaults.test.ts | 75 +++ platform/shared/src/index.ts | 1 + platform/shared/src/model-defaults.ts | 17 + platform/web/src/api.ts | 1 + .../models/add-model-provider-dialog.tsx | 3 + .../src/components/models/model-defaults.tsx | 138 +++++ .../components/provider-readiness-banner.tsx | 38 +- .../session/session-config-editor.tsx | 10 +- .../src/demo/fixtures/personal-assistant.ts | 1 + .../web/src/demo/fixtures/software-factory.ts | 1 + .../src/demo/fixtures/technical-support.ts | 1 + platform/web/src/demo/router.test.ts | 63 +++ platform/web/src/demo/routes/bots.ts | 25 +- platform/web/src/demo/routes/secrets.ts | 20 + platform/web/src/demo/routes/sessions.ts | 14 +- platform/web/src/demo/store.ts | 3 + platform/web/src/lib/model-defaults.ts | 45 ++ .../web/src/lib/profile-config-reference.ts | 2 +- .../web/src/lib/provider-readiness.test.ts | 68 ++- platform/web/src/lib/provider-readiness.ts | 66 +-- .../web/src/lib/sessions/editor-options.ts | 11 +- platform/web/src/pages/BotCreatePage.tsx | 11 +- platform/web/src/pages/ModelsPage.test.tsx | 104 +++- platform/web/src/pages/ModelsPage.tsx | 21 +- platform/web/src/pages/ProfilesPage.tsx | 3 +- .../pages/SessionsPage.permissions.test.tsx | 56 +- platform/web/src/pages/SessionsPage.tsx | 38 +- release/metadata.env | 2 +- scripts/dev/cli-connection.mjs | 41 ++ scripts/dev/cli-connection.test.mjs | 41 ++ scripts/dev/stack.mjs | 8 +- 86 files changed, 3221 insertions(+), 297 deletions(-) create mode 100644 crates/api/src/model_defaults.rs create mode 100644 crates/cli/src/model_defaults_cli.rs create mode 100644 crates/store-pg/migrations/010_model_defaults.sql create mode 100644 crates/store-pg/src/model_defaults.rs create mode 100644 crates/store-pg/tests/model_defaults_live.rs create mode 100644 crates/temporal-server/src/gateway/service/model_defaults.rs create mode 100644 docs/roadmap/p184-universe-model-defaults-and-transcription.md create mode 100644 platform/server/src/routes/model-defaults.test.ts create mode 100644 platform/shared/src/model-defaults.ts create mode 100644 platform/web/src/components/models/model-defaults.tsx create mode 100644 platform/web/src/lib/model-defaults.ts diff --git a/clients/typescript/schema/api.schema.json b/clients/typescript/schema/api.schema.json index 59004bb8f..11f14af3b 100644 --- a/clients/typescript/schema/api.schema.json +++ b/clients/typescript/schema/api.schema.json @@ -81,6 +81,17 @@ }, "message": { "type": "string" + }, + "modelDefaultSlot": { + "anyOf": [ + { + "$ref": "#/definitions/ModelDefaultSlot" + }, + { + "type": "null" + } + ], + "description": "Present for model_default_unset; clients need not parse the message." } }, "required": [ @@ -121,6 +132,11 @@ "description": "The authenticated caller lacks permission for this operation or target.", "type": "string" }, + { + "const": "model_default_unset", + "description": "No model was supplied and this universe has no default for the requested use.", + "type": "string" + }, { "const": "session_bootstrap_failed", "description": "The session's agent workflow exists but failed during bootstrap\n(rehydration) and cannot serve runs. Distinct from `NotFound` (no\nworkflow) so clients/bridges treat it as a session recovery problem\nrather than an ordinary \"answer this message\" failure.", @@ -1719,6 +1735,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfModelDefaultsResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/definitions/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/definitions/ModelDefaultsResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfModelListResponse": { "properties": { "notifications": { @@ -11823,6 +11856,95 @@ ], "type": "object" }, + "ModelDefaultSlot": { + "description": "A universe's model selection for a particular use. Protocol and purpose\nare separate: several purposes may use the same provider API.", + "enum": [ + "agentRun", + "speechToText" + ], + "type": "string" + }, + "ModelDefaults": { + "additionalProperties": false, + "description": "Persisted universe defaults. Revision zero means no update has been made.\nClearing a slot still advances the revision, so setup cannot undo a clear.", + "properties": { + "agentRun": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "revision": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "speechToText": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + } + }, + "required": [ + "revision" + ], + "type": "object" + }, + "ModelDefaultsPutParams": { + "additionalProperties": false, + "properties": { + "expectedRevision": { + "description": "Revision returned by read/put; zero for a universe with no updates.", + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ], + "description": "Complete selection, or explicit null to clear this slot. Required." + }, + "slot": { + "$ref": "#/definitions/ModelDefaultSlot" + } + }, + "required": [ + "slot", + "model", + "expectedRevision" + ], + "type": "object" + }, + "ModelDefaultsReadParams": { + "additionalProperties": false, + "type": "object" + }, + "ModelDefaultsResponse": { + "properties": { + "defaults": { + "$ref": "#/definitions/ModelDefaults" + } + }, + "required": [ + "defaults" + ], + "type": "object" + }, "ModelEndpointConfig": { "properties": { "apiKinds": { @@ -13452,7 +13574,7 @@ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" diff --git a/clients/typescript/src/errors.ts b/clients/typescript/src/errors.ts index eb3105f33..a00766d1e 100644 --- a/clients/typescript/src/errors.ts +++ b/clients/typescript/src/errors.ts @@ -11,6 +11,7 @@ export type LightspeedRpcErrorKind = | "conflict" | "rejected" | "environment_not_ready" + | "model_default_unset" | "internal" | "unknown"; @@ -34,6 +35,8 @@ export function lightspeedRpcErrorKind(code: number): LightspeedRpcErrorKind { return "environment_not_ready"; case -32603: return "internal"; + case -32014: + return "model_default_unset"; default: return "unknown"; } diff --git a/clients/typescript/src/generated/methods.ts b/clients/typescript/src/generated/methods.ts index f0837d6f7..98b6acc60 100644 --- a/clients/typescript/src/generated/methods.ts +++ b/clients/typescript/src/generated/methods.ts @@ -54,6 +54,8 @@ export const METHODS = [ "environments/registration-keys/read", "environments/registration-keys/list", "environments/registration-keys/revoke", + "models/defaults/read", + "models/defaults/put", "models/list", "profiles/create", "profiles/read", @@ -179,7 +181,7 @@ export const METHOD_INFO = { scope: "universe", access: {"action":"control_session","kind":"universe"}, summary: "Replace session configuration", - description: "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + description: "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", }, "session/rename": { scope: "universe", @@ -433,6 +435,18 @@ export const METHOD_INFO = { summary: "Revoke an environment registration key", description: "Stops the key from admitting new daemon identities; already registered daemons keep reconnecting. With closeEnvironments, also closes every non-closed environment the key admitted. Idempotent.", }, + "models/defaults/read": { + scope: "universe", + access: {"action":"read","kind":"universe"}, + summary: "Read universe model defaults", + description: "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers.", + }, + "models/defaults/put": { + scope: "universe", + access: {"action":"configure_resource","kind":"universe"}, + summary: "Set a universe model default", + description: "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery.", + }, "models/list": { scope: "universe", access: {"action":"read","kind":"universe"}, @@ -997,7 +1011,7 @@ export interface MethodMap { /** * Replace session configuration * - * Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op. + * Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op. */ "session/config/put": { params: Api.SessionConfigPutParams; @@ -1381,6 +1395,24 @@ export interface MethodMap { params: Api.EnvironmentRegistrationKeyRevokeParams; result: Api.AgentApiOutcomeOfEnvironmentRegistrationKeyRevokeResponse; }; + /** + * Read universe model defaults + * + * Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers. + */ + "models/defaults/read": { + params: Api.ModelDefaultsReadParams; + result: Api.AgentApiOutcomeOfModelDefaultsResponse; + }; + /** + * Set a universe model default + * + * Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery. + */ + "models/defaults/put": { + params: Api.ModelDefaultsPutParams; + result: Api.AgentApiOutcomeOfModelDefaultsResponse; + }; /** * Discover available models * @@ -2180,7 +2212,7 @@ export const rpc = { /** * Replace session configuration * - * Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op. + * Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op. */ sessionConfigPut(client: RpcCaller, params: Api.SessionConfigPutParams): Promise { return client.call("session/config/put", params); @@ -2521,6 +2553,22 @@ export const rpc = { environmentsRegistrationKeysRevoke(client: RpcCaller, params: Api.EnvironmentRegistrationKeyRevokeParams): Promise { return client.call("environments/registration-keys/revoke", params); }, + /** + * Read universe model defaults + * + * Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers. + */ + modelsDefaultsRead(client: RpcCaller, params: Api.ModelDefaultsReadParams): Promise { + return client.call("models/defaults/read", params); + }, + /** + * Set a universe model default + * + * Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery. + */ + modelsDefaultsPut(client: RpcCaller, params: Api.ModelDefaultsPutParams): Promise { + return client.call("models/defaults/put", params); + }, /** * Discover available models * diff --git a/clients/typescript/src/generated/types.ts b/clients/typescript/src/generated/types.ts index 0a092efa7..67f4ffa6a 100644 --- a/clients/typescript/src/generated/types.ts +++ b/clients/typescript/src/generated/types.ts @@ -95,9 +95,18 @@ export type AgentApiErrorKind = | "rejected" | "unauthenticated" | "forbidden" + | "model_default_unset" | "session_bootstrap_failed" | "environment_not_ready" | "response_too_large"; +/** + * A universe's model selection for a particular use. Protocol and purpose + * are separate: several purposes may use the same provider API. + * + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaultSlot". + */ +export type ModelDefaultSlot = "agentRun" | "speechToText"; /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "AgentNotification". @@ -1865,6 +1874,10 @@ export interface ToolView { export interface AgentApiError { kind: AgentApiErrorKind; message: string; + /** + * Present for model_default_unset; clients need not parse the message. + */ + modelDefaultSlot?: ModelDefaultSlot | null; } /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema @@ -2154,7 +2167,9 @@ export interface SessionConfig { generation?: GenerationConfig | null; limits?: LimitsConfig | null; /** - * Absent on input means the deployment default model. Documents read + * At creation, omission uses the profile model or universe agentRun + * default. On configuration replacement or profile application to an + * existing session, omission preserves its current model. Documents read * back from a session always carry the model. Provider identity and API * kind are fixed for the session lifetime; the model name may change. */ @@ -5496,6 +5511,33 @@ export interface McpToolAnnotationsView { openWorldHint?: boolean | null; readOnlyHint?: boolean | null; } +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "AgentApiOutcomeOfModelDefaultsResponse". + */ +export interface AgentApiOutcomeOfModelDefaultsResponse { + notifications?: AgentNotification[]; + result: ModelDefaultsResponse; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaultsResponse". + */ +export interface ModelDefaultsResponse { + defaults: ModelDefaults; +} +/** + * Persisted universe defaults. Revision zero means no update has been made. + * Clearing a slot still advances the revision, so setup cannot undo a clear. + * + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaults". + */ +export interface ModelDefaults { + agentRun?: ModelConfig | null; + revision: number; + speechToText?: ModelConfig | null; +} /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "AgentApiOutcomeOfModelListResponse". @@ -7556,6 +7598,26 @@ export interface McpServerReadParams { export interface McpServerToolsDiscoverParams { serverId: string; } +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaultsPutParams". + */ +export interface ModelDefaultsPutParams { + /** + * Revision returned by read/put; zero for a universe with no updates. + */ + expectedRevision: number; + /** + * Complete selection, or explicit null to clear this slot. Required. + */ + model: ModelConfig | null; + slot: ModelDefaultSlot; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaultsReadParams". + */ +export interface ModelDefaultsReadParams {} /** * Direct provider model discovery. Results may be served from a brief * process-local cache; clients refresh by calling this method. diff --git a/clients/typescript/test/client.test.ts b/clients/typescript/test/client.test.ts index 3337f67c9..4b40f8277 100644 --- a/clients/typescript/test/client.test.ts +++ b/clients/typescript/test/client.test.ts @@ -20,6 +20,12 @@ function decodeBody(init: RequestInit | undefined): Record { } describe("LightspeedClient", () => { + it("classifies missing model defaults and retains the affected slot", () => { + const error = new LightspeedRpcError({ code: -32014, message: "Choose a model", data: { kind: "model_default_unset", message: "Choose a model", modelDefaultSlot: "agentRun" } }); + expect(error.kind).toBe("model_default_unset"); + expect(error.data?.modelDefaultSlot).toBe("agentRun"); + }); + it("omits universe selection on connection and deployment calls", async () => { const captured: Headers[] = []; const fetchImpl = vi.fn(async (_input: RequestInfo | URL, init?: RequestInit) => { @@ -40,7 +46,7 @@ describe("LightspeedClient", () => { access: { kind: "universe", action: "control_session" }, summary: "Replace session configuration", description: - "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", }); expect(METHOD_INFO["deployment/api-keys/create"]).toMatchObject({ scope: "deployment", diff --git a/crates/api/contract/api-reference.md b/crates/api/contract/api-reference.md index ae515f8e6..330c2e834 100644 --- a/crates/api/contract/api-reference.md +++ b/crates/api/contract/api-reference.md @@ -94,7 +94,7 @@ Returns a cursor-paginated summary list ordered by most recent update, optionall **Replace session configuration** -Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op. +Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op. - Access: `{"kind":"universe","action":"control_session"}` - Group: `session` @@ -649,6 +649,32 @@ Stops the key from admitting new daemon identities; already registered daemons k - Params: `EnvironmentRegistrationKeyRevokeParams` - Result: `AgentApiOutcome` +### `models/defaults/read` + +**Read universe model defaults** + +Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers. + +- Access: `{"kind":"universe","action":"read"}` +- Group: `models` +- Role: `viewer` +- Target: `none` +- Params: `ModelDefaultsReadParams` +- Result: `AgentApiOutcome` + +### `models/defaults/put` + +**Set a universe model default** + +Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery. + +- Access: `{"kind":"universe","action":"configure_resource"}` +- Group: `models` +- Role: `operator` +- Target: `none` +- Params: `ModelDefaultsPutParams` +- Result: `AgentApiOutcome` + ### `models/list` **Discover available models** diff --git a/crates/api/contract/api.schema.json b/crates/api/contract/api.schema.json index 59004bb8f..11f14af3b 100644 --- a/crates/api/contract/api.schema.json +++ b/crates/api/contract/api.schema.json @@ -81,6 +81,17 @@ }, "message": { "type": "string" + }, + "modelDefaultSlot": { + "anyOf": [ + { + "$ref": "#/definitions/ModelDefaultSlot" + }, + { + "type": "null" + } + ], + "description": "Present for model_default_unset; clients need not parse the message." } }, "required": [ @@ -121,6 +132,11 @@ "description": "The authenticated caller lacks permission for this operation or target.", "type": "string" }, + { + "const": "model_default_unset", + "description": "No model was supplied and this universe has no default for the requested use.", + "type": "string" + }, { "const": "session_bootstrap_failed", "description": "The session's agent workflow exists but failed during bootstrap\n(rehydration) and cannot serve runs. Distinct from `NotFound` (no\nworkflow) so clients/bridges treat it as a session recovery problem\nrather than an ordinary \"answer this message\" failure.", @@ -1719,6 +1735,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfModelDefaultsResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/definitions/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/definitions/ModelDefaultsResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfModelListResponse": { "properties": { "notifications": { @@ -11823,6 +11856,95 @@ ], "type": "object" }, + "ModelDefaultSlot": { + "description": "A universe's model selection for a particular use. Protocol and purpose\nare separate: several purposes may use the same provider API.", + "enum": [ + "agentRun", + "speechToText" + ], + "type": "string" + }, + "ModelDefaults": { + "additionalProperties": false, + "description": "Persisted universe defaults. Revision zero means no update has been made.\nClearing a slot still advances the revision, so setup cannot undo a clear.", + "properties": { + "agentRun": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "revision": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "speechToText": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + } + }, + "required": [ + "revision" + ], + "type": "object" + }, + "ModelDefaultsPutParams": { + "additionalProperties": false, + "properties": { + "expectedRevision": { + "description": "Revision returned by read/put; zero for a universe with no updates.", + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ], + "description": "Complete selection, or explicit null to clear this slot. Required." + }, + "slot": { + "$ref": "#/definitions/ModelDefaultSlot" + } + }, + "required": [ + "slot", + "model", + "expectedRevision" + ], + "type": "object" + }, + "ModelDefaultsReadParams": { + "additionalProperties": false, + "type": "object" + }, + "ModelDefaultsResponse": { + "properties": { + "defaults": { + "$ref": "#/definitions/ModelDefaults" + } + }, + "required": [ + "defaults" + ], + "type": "object" + }, "ModelEndpointConfig": { "properties": { "apiKinds": { @@ -13452,7 +13574,7 @@ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" diff --git a/crates/api/contract/methods.json b/crates/api/contract/methods.json index 16bebb4f2..f9c1d80b6 100644 --- a/crates/api/contract/methods.json +++ b/crates/api/contract/methods.json @@ -154,7 +154,7 @@ "action": "control_session", "kind": "universe" }, - "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", "group": "session", "method": "session/config/put", "params": { @@ -1224,6 +1224,56 @@ "summary": "Revoke an environment registration key", "target": null }, + { + "access": { + "action": "read", + "kind": "universe" + }, + "description": "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers.", + "group": "models", + "method": "models/defaults/read", + "params": { + "schema": { + "$ref": "#/definitions/ModelDefaultsReadParams" + }, + "type": "ModelDefaultsReadParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfModelDefaultsResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "viewer", + "scope": "universe", + "summary": "Read universe model defaults", + "target": null + }, + { + "access": { + "action": "configure_resource", + "kind": "universe" + }, + "description": "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery.", + "group": "models", + "method": "models/defaults/put", + "params": { + "schema": { + "$ref": "#/definitions/ModelDefaultsPutParams" + }, + "type": "ModelDefaultsPutParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfModelDefaultsResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "operator", + "scope": "universe", + "summary": "Set a universe model default", + "target": null + }, { "access": { "action": "read", diff --git a/crates/api/contract/openrpc.json b/crates/api/contract/openrpc.json index d900f24c6..2d8bcc43e 100644 --- a/crates/api/contract/openrpc.json +++ b/crates/api/contract/openrpc.json @@ -81,6 +81,17 @@ }, "message": { "type": "string" + }, + "modelDefaultSlot": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelDefaultSlot" + }, + { + "type": "null" + } + ], + "description": "Present for model_default_unset; clients need not parse the message." } }, "required": [ @@ -121,6 +132,11 @@ "description": "The authenticated caller lacks permission for this operation or target.", "type": "string" }, + { + "const": "model_default_unset", + "description": "No model was supplied and this universe has no default for the requested use.", + "type": "string" + }, { "const": "session_bootstrap_failed", "description": "The session's agent workflow exists but failed during bootstrap\n(rehydration) and cannot serve runs. Distinct from `NotFound` (no\nworkflow) so clients/bridges treat it as a session recovery problem\nrather than an ordinary \"answer this message\" failure.", @@ -1719,6 +1735,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfModelDefaultsResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/components/schemas/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/components/schemas/ModelDefaultsResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfModelListResponse": { "properties": { "notifications": { @@ -11823,6 +11856,95 @@ ], "type": "object" }, + "ModelDefaultSlot": { + "description": "A universe's model selection for a particular use. Protocol and purpose\nare separate: several purposes may use the same provider API.", + "enum": [ + "agentRun", + "speechToText" + ], + "type": "string" + }, + "ModelDefaults": { + "additionalProperties": false, + "description": "Persisted universe defaults. Revision zero means no update has been made.\nClearing a slot still advances the revision, so setup cannot undo a clear.", + "properties": { + "agentRun": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "revision": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "speechToText": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelConfig" + }, + { + "type": "null" + } + ] + } + }, + "required": [ + "revision" + ], + "type": "object" + }, + "ModelDefaultsPutParams": { + "additionalProperties": false, + "properties": { + "expectedRevision": { + "description": "Revision returned by read/put; zero for a universe with no updates.", + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "model": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelConfig" + }, + { + "type": "null" + } + ], + "description": "Complete selection, or explicit null to clear this slot. Required." + }, + "slot": { + "$ref": "#/components/schemas/ModelDefaultSlot" + } + }, + "required": [ + "slot", + "model", + "expectedRevision" + ], + "type": "object" + }, + "ModelDefaultsReadParams": { + "additionalProperties": false, + "type": "object" + }, + "ModelDefaultsResponse": { + "properties": { + "defaults": { + "$ref": "#/components/schemas/ModelDefaults" + } + }, + "required": [ + "defaults" + ], + "type": "object" + }, "ModelEndpointConfig": { "properties": { "apiKinds": { @@ -13452,7 +13574,7 @@ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -18074,7 +18196,7 @@ "x-lightspeed-target": null }, { - "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", "name": "session/config/put", "paramStructure": "by-name", "params": [ @@ -19277,6 +19399,62 @@ "x-lightspeed-role": "operator", "x-lightspeed-target": null }, + { + "description": "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers.", + "name": "models/defaults/read", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/ModelDefaultsReadParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfModelDefaultsResponse" + } + }, + "summary": "Read universe model defaults", + "x-lightspeed-access": { + "action": "read", + "kind": "universe" + }, + "x-lightspeed-group": "models", + "x-lightspeed-role": "viewer", + "x-lightspeed-target": null + }, + { + "description": "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery.", + "name": "models/defaults/put", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/ModelDefaultsPutParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfModelDefaultsResponse" + } + }, + "summary": "Set a universe model default", + "x-lightspeed-access": { + "action": "configure_resource", + "kind": "universe" + }, + "x-lightspeed-group": "models", + "x-lightspeed-role": "operator", + "x-lightspeed-target": null + }, { "description": "Queries supported providers directly, with a brief process-local burst cache, and returns best-effort selectable routes. One provider failure does not discard successful results from others.", "name": "models/list", diff --git a/crates/api/src/constants.rs b/crates/api/src/constants.rs index 7832921f4..dc6a1482e 100644 --- a/crates/api/src/constants.rs +++ b/crates/api/src/constants.rs @@ -60,6 +60,8 @@ pub const METHOD_ENVIRONMENTS_CREDENTIALS_UNBIND: &str = "environments/credentia // ── Universe: direct provider model discovery ─────────────────────────────── +pub const METHOD_MODELS_DEFAULTS_READ: &str = "models/defaults/read"; +pub const METHOD_MODELS_DEFAULTS_PUT: &str = "models/defaults/put"; pub const METHOD_MODELS_LIST: &str = "models/list"; pub const METHOD_PROFILES_CREATE: &str = "profiles/create"; diff --git a/crates/api/src/lib.rs b/crates/api/src/lib.rs index 183679f3c..ebfd0c89d 100644 --- a/crates/api/src/lib.rs +++ b/crates/api/src/lib.rs @@ -27,6 +27,7 @@ mod handshake; mod ids; mod mcp; mod model; +mod model_defaults; mod models; mod notifications; mod profiles; @@ -50,6 +51,7 @@ pub use handshake::*; pub use ids::*; pub use mcp::*; pub use model::*; +pub use model_defaults::*; pub use models::*; pub use notifications::*; pub use profiles::*; diff --git a/crates/api/src/model_defaults.rs b/crates/api/src/model_defaults.rs new file mode 100644 index 000000000..1a25625e8 --- /dev/null +++ b/crates/api/src/model_defaults.rs @@ -0,0 +1,150 @@ +use super::*; + +/// A universe's model selection for a particular use. Protocol and purpose +/// are separate: several purposes may use the same provider API. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub enum ModelDefaultSlot { + AgentRun, + SpeechToText, +} + +impl ModelDefaultSlot { + pub fn as_str(self) -> &'static str { + match self { + Self::AgentRun => "agentRun", + Self::SpeechToText => "speechToText", + } + } + + pub fn validate_model(self, model: &ModelConfig) -> Result<(), AgentApiError> { + for (field, value) in [("providerId", &model.provider_id), ("model", &model.model)] { + if value.trim().is_empty() || value.trim() != value || value.len() > 512 { + return Err(AgentApiError::invalid_request(format!( + "{field} must contain 1..=512 bytes without surrounding whitespace" + ))); + } + } + let supported = match self { + Self::AgentRun => matches!( + model.api_kind.as_str(), + "openai:responses" | "openai:completions" | "anthropic:messages" + ), + Self::SpeechToText => model.api_kind == "openai:audio-transcriptions", + }; + if !supported { + return Err(AgentApiError::invalid_request(format!( + "API kind {} cannot be used for {}", + model.api_kind, + self.as_str() + ))); + } + Ok(()) + } +} + +/// Persisted universe defaults. Revision zero means no update has been made. +/// Clearing a slot still advances the revision, so setup cannot undo a clear. +#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct ModelDefaults { + pub revision: u64, + pub agent_run: Option, + pub speech_to_text: Option, +} + +impl ModelDefaults { + pub fn model(&self, slot: ModelDefaultSlot) -> Option<&ModelConfig> { + match slot { + ModelDefaultSlot::AgentRun => self.agent_run.as_ref(), + ModelDefaultSlot::SpeechToText => self.speech_to_text.as_ref(), + } + } +} + +#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct ModelDefaultsReadParams {} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct ModelDefaultsPutParams { + pub slot: ModelDefaultSlot, + /// Complete selection, or explicit null to clear this slot. Required. + #[serde(deserialize_with = "Option::deserialize")] + #[schemars(required, schema_with = "required_nullable_model_schema")] + pub model: Option, + /// Revision returned by read/put; zero for a universe with no updates. + pub expected_revision: u64, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct ModelDefaultsResponse { + pub defaults: ModelDefaults, +} + +fn required_nullable_model_schema(generator: &mut schemars::SchemaGenerator) -> schemars::Schema { + Option::::json_schema(generator) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn slots_validate_protocols_without_requiring_a_catalog_model() { + for (kind, slot) in [ + ("openai:responses", ModelDefaultSlot::AgentRun), + ("anthropic:messages", ModelDefaultSlot::AgentRun), + ("openai:completions", ModelDefaultSlot::AgentRun), + ( + "openai:audio-transcriptions", + ModelDefaultSlot::SpeechToText, + ), + ] { + let model = ModelConfig { + provider_id: "custom".into(), + api_kind: kind.into(), + model: "private-model".into(), + }; + slot.validate_model(&model).unwrap(); + let other = match slot { + ModelDefaultSlot::AgentRun => ModelDefaultSlot::SpeechToText, + ModelDefaultSlot::SpeechToText => ModelDefaultSlot::AgentRun, + }; + assert_eq!( + other.validate_model(&model).unwrap_err().kind, + AgentApiErrorKind::InvalidRequest + ); + } + } + + #[test] + fn clearing_requires_an_explicit_null_and_a_revision() { + let request = serde_json::json!({"slot":"agentRun", "model":null, "expectedRevision":4}); + assert_eq!( + serde_json::from_value::(request.clone()) + .unwrap() + .model, + None + ); + for field in ["model", "expectedRevision"] { + let mut missing = request.clone(); + missing.as_object_mut().unwrap().remove(field); + assert!(serde_json::from_value::(missing).is_err()); + } + } + + #[test] + fn missing_default_error_identifies_its_slot() { + let error = AgentApiError::model_default_unset(ModelDefaultSlot::AgentRun); + let json = serde_json::to_value(&error).unwrap(); + assert_eq!(json["kind"], "model_default_unset"); + assert_eq!(json["modelDefaultSlot"], "agentRun"); + assert_eq!( + serde_json::from_value::(json).unwrap(), + error + ); + } +} diff --git a/crates/api/src/rpc.rs b/crates/api/src/rpc.rs index 396a83ef6..93fa40df7 100644 --- a/crates/api/src/rpc.rs +++ b/crates/api/src/rpc.rs @@ -13,6 +13,8 @@ pub enum AgentApiErrorKind { Unauthenticated, /// The authenticated caller lacks permission for this operation or target. Forbidden, + /// No model was supplied and this universe has no default for the requested use. + ModelDefaultUnset, UnsupportedAudioMime, AudioBlobTooLarge, AudioDurationTooLong, @@ -42,6 +44,9 @@ pub enum AgentApiErrorKind { pub struct AgentApiError { pub kind: AgentApiErrorKind, pub message: String, + /// Present for model_default_unset; clients need not parse the message. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_default_slot: Option, } impl AgentApiError { @@ -49,6 +54,18 @@ impl AgentApiError { Self { kind, message: message.into(), + model_default_slot: None, + } + } + + pub fn model_default_unset(slot: ModelDefaultSlot) -> Self { + Self { + kind: AgentApiErrorKind::ModelDefaultUnset, + message: format!( + "no universe model default is configured for {}; choose a model or set it with models/defaults/put", + slot.as_str() + ), + model_default_slot: Some(slot), } } @@ -136,6 +153,7 @@ impl AgentApiError { AgentApiErrorKind::SessionBootstrapFailed => -32011, AgentApiErrorKind::EnvironmentNotReady => -32012, AgentApiErrorKind::ResponseTooLarge => -32013, + AgentApiErrorKind::ModelDefaultUnset => -32014, AgentApiErrorKind::Internal => -32603, } } @@ -358,7 +376,7 @@ api_methods! { METHOD_SESSION_LIST => list_sessions(SessionListParams) -> SessionListResponse => ["List sessions", "Returns a cursor-paginated summary list ordered by most recent update, optionally narrowed by the audience of each session's root: createdBy, visibility, or visibleTo (shared with the universe or created by that actor). Pages may shift while sessions are changing."], access: MethodAccess::Universe(UniverseAction::Read), METHOD_SESSION_CONFIG_PUT => put_session_config(SessionConfigPutParams) -> SessionConfigPutResponse => - ["Replace session configuration", "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op."], access: MethodAccess::Universe(UniverseAction::ControlSession), + ["Replace session configuration", "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op."], access: MethodAccess::Universe(UniverseAction::ControlSession), METHOD_SESSION_RENAME => rename_session(SessionRenameParams) -> SessionRenameResponse => ["Rename a session", "Sets the display name, or clears it when displayName is omitted."], access: MethodAccess::Universe(UniverseAction::ControlSession), METHOD_SESSION_METADATA_PUT => put_session_metadata(SessionMetadataPutParams) -> SessionMetadataPutResponse => @@ -443,6 +461,10 @@ api_methods! { ["List environment registration keys", "Lists this universe's registration keys with policy, status, and derived counts. Each key is the group of the environments it admitted."], access: MethodAccess::Universe(UniverseAction::ConfigureResource), METHOD_ENVIRONMENTS_REGISTRATION_KEYS_REVOKE => revoke_environment_registration_key(EnvironmentRegistrationKeyRevokeParams) -> EnvironmentRegistrationKeyRevokeResponse => ["Revoke an environment registration key", "Stops the key from admitting new daemon identities; already registered daemons keep reconnecting. With closeEnvironments, also closes every non-closed environment the key admitted. Idempotent."], access: MethodAccess::Universe(UniverseAction::ConfigureResource), + METHOD_MODELS_DEFAULTS_READ => read_model_defaults(ModelDefaultsReadParams) -> ModelDefaultsResponse => + ["Read universe model defaults", "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers."], access: MethodAccess::Universe(UniverseAction::Read), + METHOD_MODELS_DEFAULTS_PUT => put_model_defaults(ModelDefaultsPutParams) -> ModelDefaultsResponse => + ["Set a universe model default", "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery."], access: MethodAccess::Universe(UniverseAction::ConfigureResource), METHOD_MODELS_LIST => list_models(ModelListParams) -> ModelListResponse => ["Discover available models", "Queries supported providers directly, with a brief process-local burst cache, and returns best-effort selectable routes. One provider failure does not discard successful results from others."], access: MethodAccess::Universe(UniverseAction::Read), METHOD_PROFILES_CREATE => create_profile(ProfileCreateParams) -> ProfileCreateResponse => diff --git a/crates/api/src/schema_export.rs b/crates/api/src/schema_export.rs index ffa10ec31..49408dd9e 100644 --- a/crates/api/src/schema_export.rs +++ b/crates/api/src/schema_export.rs @@ -205,7 +205,7 @@ mod tests { methods.sort_unstable(); methods.dedup(); assert_eq!(methods.len(), total, "duplicate method in manifest"); - assert_eq!(total, 131); + assert_eq!(total, 133); assert_eq!( manifest .iter() diff --git a/crates/api/src/service.rs b/crates/api/src/service.rs index 09cb6ce52..16a37cac2 100644 --- a/crates/api/src/service.rs +++ b/crates/api/src/service.rs @@ -15,6 +15,20 @@ pub trait AgentApiService: Send + Sync { params: InitializeParams, ) -> Result, AgentApiError>; + async fn read_model_defaults( + &self, + _params: ModelDefaultsReadParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("model defaults are unavailable")) + } + + async fn put_model_defaults( + &self, + _params: ModelDefaultsPutParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("model defaults are unavailable")) + } + async fn list_models( &self, params: ModelListParams, diff --git a/crates/api/src/sessions.rs b/crates/api/src/sessions.rs index 08300dc3c..5e47da7e0 100644 --- a/crates/api/src/sessions.rs +++ b/crates/api/src/sessions.rs @@ -264,7 +264,9 @@ fn default_feature_version() -> u32 { #[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] #[serde(rename_all = "camelCase", deny_unknown_fields)] pub struct SessionConfig { - /// Absent on input means the deployment default model. Documents read + /// At creation, omission uses the profile model or universe agentRun + /// default. On configuration replacement or profile application to an + /// existing session, omission preserves its current model. Documents read /// back from a session always carry the model. Provider identity and API /// kind are fixed for the session lifetime; the model name may change. #[serde(default, skip_serializing_if = "Option::is_none")] diff --git a/crates/api/tests/schema_artifacts.rs b/crates/api/tests/schema_artifacts.rs index d77fdcf52..147f9d5ac 100644 --- a/crates/api/tests/schema_artifacts.rs +++ b/crates/api/tests/schema_artifacts.rs @@ -36,6 +36,21 @@ fn committed_text(name: &str) -> String { }) } +#[test] +fn model_default_updates_accept_explicit_null_in_the_public_contract() { + let bundle = api::export_schemas().schema_bundle; + for model in [ + Value::Null, + json!({"providerId":"private", "apiKind":"openai:completions", "model":"custom"}), + ] { + let request = json!({"slot":"agentRun", "model":model, "expectedRevision":0}); + assert_validates(&bundle, "ModelDefaultsPutParams", &request); + serde_json::from_value::(request).unwrap(); + } + let required = &bundle["definitions"]["ModelDefaultsPutParams"]["required"]; + assert!(required.as_array().unwrap().contains(&json!("model"))); +} + fn assert_validates(bundle: &Value, definition: &str, instance: &Value) { let schema = json!({ "$schema": "http://json-schema.org/draft-07/schema#", diff --git a/crates/cli/src/administration_cli.rs b/crates/cli/src/administration_cli.rs index ab7bcc35c..688d48a32 100644 --- a/crates/cli/src/administration_cli.rs +++ b/crates/cli/src/administration_cli.rs @@ -281,6 +281,8 @@ pub struct ModelsArgs { } #[derive(Debug, Subcommand)] enum ModelsCommand { + /// Inspect, set, or clear universe model defaults. + Defaults(crate::model_defaults_cli::ModelDefaultsArgs), /// Configure provider endpoints and their API-key or OAuth credentials. #[command(visible_alias = "providers")] Provider(crate::auth_cli::AuthModelArgs), @@ -293,6 +295,7 @@ enum ModelsCommand { } pub async fn models(args: ModelsArgs) -> Result<()> { let (json, all) = match args.command { + ModelsCommand::Defaults(args) => return crate::model_defaults_cli::run(args).await, ModelsCommand::Provider(args) => return crate::auth_cli::model(args).await, ModelsCommand::List { json, all } => (json, all), }; diff --git a/crates/cli/src/main.rs b/crates/cli/src/main.rs index 46c282334..6bc8870f5 100644 --- a/crates/cli/src/main.rs +++ b/crates/cli/src/main.rs @@ -5,6 +5,7 @@ mod chat; mod connection; mod env_cli; mod mcp_cli; +mod model_defaults_cli; mod output; mod profile_cli; mod session_cli; diff --git a/crates/cli/src/model_defaults_cli.rs b/crates/cli/src/model_defaults_cli.rs new file mode 100644 index 000000000..843220e04 --- /dev/null +++ b/crates/cli/src/model_defaults_cli.rs @@ -0,0 +1,192 @@ +use anyhow::Result; +use clap::{Args, Subcommand, ValueEnum}; + +use crate::api_client::HttpAgentApi; + +#[derive(Debug, Args)] +pub struct ModelDefaultsArgs { + #[arg(long, global = true)] + json: bool, + #[command(subcommand)] + command: ModelDefaultsCommand, +} + +#[derive(Debug, Subcommand)] +enum ModelDefaultsCommand { + /// Read this universe's model selections and revision. + Read, + /// Set a complete provider/API/model route for one use. + Set { + #[arg(value_enum)] + slot: Slot, + #[arg(long)] + provider: String, + #[arg(long)] + api_kind: String, + #[arg(long)] + model: String, + /// Use this revision; otherwise read it immediately before updating. + #[arg(long)] + expected_revision: Option, + }, + /// Clear one default; existing sessions keep their model. + Clear { + #[arg(value_enum)] + slot: Slot, + #[arg(long)] + expected_revision: Option, + }, +} + +#[derive(Clone, Copy, Debug, ValueEnum)] +enum Slot { + #[value(alias = "agentRun")] + AgentRun, + #[value(alias = "speechToText")] + SpeechToText, +} + +impl From for api::ModelDefaultSlot { + fn from(slot: Slot) -> Self { + match slot { + Slot::AgentRun => Self::AgentRun, + Slot::SpeechToText => Self::SpeechToText, + } + } +} + +pub async fn run(args: ModelDefaultsArgs) -> Result<()> { + let client = HttpAgentApi::new(""); + let defaults = match args.command { + ModelDefaultsCommand::Read => read(&client).await?, + ModelDefaultsCommand::Set { + slot, + provider, + api_kind, + model, + expected_revision, + } => { + let slot = slot.into(); + let model = api::ModelConfig { + provider_id: provider, + api_kind, + model, + }; + api::ModelDefaultSlot::validate_model(slot, &model)?; + put(&client, slot, Some(model), expected_revision).await? + } + ModelDefaultsCommand::Clear { + slot, + expected_revision, + } => put(&client, slot.into(), None, expected_revision).await?, + }; + if args.json { + println!("{}", serde_json::to_string_pretty(&defaults)?); + } else { + println!("Model defaults (revision {})", defaults.revision); + for slot in [ + api::ModelDefaultSlot::AgentRun, + api::ModelDefaultSlot::SpeechToText, + ] { + match defaults.model(slot) { + Some(model) => println!( + "{}: {} / {} / {}", + slot.as_str(), + model.provider_id, + model.api_kind, + model.model + ), + None => println!("{}: unset", slot.as_str()), + } + } + } + Ok(()) +} + +async fn read(client: &HttpAgentApi) -> Result { + let response: api::AgentApiOutcome = client + .request( + api::METHOD_MODELS_DEFAULTS_READ, + api::ModelDefaultsReadParams {}, + ) + .await?; + Ok(response.result.defaults) +} + +async fn put( + client: &HttpAgentApi, + slot: api::ModelDefaultSlot, + model: Option, + revision: Option, +) -> Result { + let expected_revision = match revision { + Some(revision) => revision, + None => read(client).await?.revision, + }; + let response: api::AgentApiOutcome = client + .request( + api::METHOD_MODELS_DEFAULTS_PUT, + api::ModelDefaultsPutParams { + slot, + model, + expected_revision, + }, + ) + .await?; + // A concurrent change is returned to the caller; never retry a write + // against a newer revision without the caller reviewing that state. + Ok(response.result.defaults) +} + +#[cfg(test)] +mod tests { + use crate::Cli; + use clap::Parser; + + #[test] + fn defaults_commands_require_complete_routes_and_accept_revision_guards() { + for args in [ + vec!["lightspeed", "model", "defaults", "read", "--json"], + vec![ + "lightspeed", + "model", + "defaults", + "set", + "agent-run", + "--provider", + "custom", + "--api-kind", + "openai:completions", + "--model", + "model", + "--expected-revision", + "0", + ], + vec![ + "lightspeed", + "models", + "defaults", + "clear", + "speech-to-text", + "--json", + ], + ] { + Cli::try_parse_from(args).unwrap(); + } + assert!( + Cli::try_parse_from([ + "lightspeed", + "model", + "defaults", + "set", + "agent-run", + "--model", + "model" + ]) + .is_err() + ); + assert!( + Cli::try_parse_from(["lightspeed", "model", "defaults", "clear", "unknown"]).is_err() + ); + } +} diff --git a/crates/cli/tests/resource_commands.rs b/crates/cli/tests/resource_commands.rs index 1d6d49a47..3fe1e2859 100644 --- a/crates/cli/tests/resource_commands.rs +++ b/crates/cli/tests/resource_commands.rs @@ -617,3 +617,101 @@ fn config_replacement_checks_the_requested_revision_and_outputs_one_document() { ]); assert_eq!(output["session"]["configRevision"], 43); } + +#[test] +fn model_defaults_commands_round_trip_routes_clear_explicitly_and_guard_revisions() { + let mut defaults = json!({"revision":0, "agentRun":null, "speechToText":null}); + let runtime = Runtime::start(move |method, params| { + match method { + "models/defaults/read" => {} + "models/defaults/put" => { + assert_eq!(params["expectedRevision"], defaults["revision"]); + let slot = params["slot"].as_str().unwrap(); + assert!( + params.get("model").is_some(), + "clearing must send explicit null" + ); + defaults[slot] = params["model"].clone(); + defaults["revision"] = json!(defaults["revision"].as_u64().unwrap() + 1); + } + other => panic!("unexpected method {other}"), + } + json!({"defaults":defaults}) + }); + let agent = runtime.json(&[ + "model", + "defaults", + "set", + "agent-run", + "--provider", + "anthropic", + "--api-kind", + "anthropic:messages", + "--model", + "chosen", + "--json", + ]); + assert_eq!(agent["agentRun"]["apiKind"], "anthropic:messages"); + assert_eq!(agent["revision"], 1); + let speech = runtime.json(&[ + "model", + "defaults", + "set", + "speech-to-text", + "--provider", + "custom-speech", + "--api-kind", + "openai:audio-transcriptions", + "--model", + "transcriber", + "--expected-revision", + "1", + "--json", + ]); + assert_eq!(speech["agentRun"], agent["agentRun"]); + let cleared = runtime.json(&["model", "defaults", "clear", "agent-run", "--json"]); + assert_eq!(cleared["agentRun"], Value::Null); + assert_eq!(cleared["speechToText"], speech["speechToText"]); + assert_eq!( + runtime.json(&["model", "defaults", "read", "--json"]), + cleared + ); + let requests = runtime.requests.lock().unwrap(); + assert_eq!( + requests + .iter() + .filter(|r| r["method"] == "models/defaults/read") + .count(), + 3 + ); +} + +#[test] +fn model_defaults_conflict_is_not_retried_with_a_new_revision() { + let runtime = Runtime::start_api(|method, _params| { + assert_eq!(method, "models/defaults/put"); + Err( + json!({"code":-32009, "message":"defaults changed", "data":{"kind":"conflict","message":"defaults changed"}}), + ) + }); + let output = runtime.run(&[ + "model", + "defaults", + "clear", + "agent-run", + "--expected-revision", + "4", + ]); + assert!(!output.status.success()); + assert!(String::from_utf8_lossy(&output.stderr).contains("defaults changed")); + assert_eq!( + runtime + .requests + .lock() + .unwrap() + .iter() + .filter(|r| r["method"] == "models/defaults/put") + .count(), + 1 + ); +} diff --git a/crates/store-pg/migrations/010_model_defaults.sql b/crates/store-pg/migrations/010_model_defaults.sql new file mode 100644 index 000000000..38f3e3dec --- /dev/null +++ b/crates/store-pg/migrations/010_model_defaults.sql @@ -0,0 +1,8 @@ +-- Defaults are universe policy; provider transport and credentials stay in +-- auth_providers. Retain a row after clears to preserve its revision. +CREATE TABLE universe_model_defaults ( + universe_id uuid PRIMARY KEY REFERENCES universes (universe_id) ON DELETE CASCADE, + revision bigint NOT NULL CHECK (revision > 0), + agent_run jsonb, + speech_to_text jsonb +); diff --git a/crates/store-pg/src/lib.rs b/crates/store-pg/src/lib.rs index c52f66e1f..73ab8cdc7 100644 --- a/crates/store-pg/src/lib.rs +++ b/crates/store-pg/src/lib.rs @@ -16,6 +16,7 @@ mod environment; mod environment_registration; mod mcp; mod migrations; +mod model_defaults; mod oauth; mod object; mod profile; @@ -36,6 +37,7 @@ use uuid::Uuid; pub use access::AccessFilter; pub use access::{AccessStoreError, PgAccessStore, ResourceAccess}; +pub use model_defaults::ModelDefaultsStoreError; /// A session page with the access summary each view carries. #[derive(Clone, Debug)] diff --git a/crates/store-pg/src/migrations.rs b/crates/store-pg/src/migrations.rs index d6632b55d..a8ac3f92b 100644 --- a/crates/store-pg/src/migrations.rs +++ b/crates/store-pg/src/migrations.rs @@ -44,6 +44,7 @@ const LIGHTSPEED_TABLES: &[&str] = &[ "session_checkpoints", "session_events", "sessions", + "universe_model_defaults", "universes", "vfs_snapshots", "vfs_workspaces", @@ -102,9 +103,14 @@ pub const MIGRATIONS: &[EmbeddedMigration] = &[ name: "channels", sql: include_str!("../migrations/009_channels.sql"), }, + EmbeddedMigration { + version: 10, + name: "model_defaults", + sql: include_str!("../migrations/010_model_defaults.sql"), + }, ]; -pub const REQUIRED_SCHEMA_REVISION: i64 = 9; +pub const REQUIRED_SCHEMA_REVISION: i64 = 10; #[derive(Clone, Debug, PartialEq, Eq)] pub struct SchemaStatus { diff --git a/crates/store-pg/src/model_defaults.rs b/crates/store-pg/src/model_defaults.rs new file mode 100644 index 000000000..f2eaf9dbd --- /dev/null +++ b/crates/store-pg/src/model_defaults.rs @@ -0,0 +1,90 @@ +use api::{ModelDefaultSlot, ModelDefaults, ModelDefaultsPutParams}; +use sqlx::Row; +use thiserror::Error; + +use crate::PgStore; + +#[derive(Debug, Error)] +pub enum ModelDefaultsStoreError { + #[error("model defaults revision conflict (expected {expected})")] + Conflict { expected: u64 }, + #[error(transparent)] + Invalid(#[from] api::AgentApiError), + #[error("model defaults storage failure: {0}")] + Postgres(#[from] sqlx::Error), + #[error("invalid stored model defaults: {0}")] + Decode(#[from] serde_json::Error), +} + +impl PgStore { + pub async fn read_model_defaults(&self) -> Result { + let row = sqlx::query("SELECT revision, agent_run, speech_to_text FROM universe_model_defaults WHERE universe_id = $1") + .bind(self.config.universe_id) + .fetch_optional(&self.pool) + .await?; + row.as_ref() + .map(decode) + .transpose() + .map(Option::unwrap_or_default) + } + + pub async fn put_model_defaults( + &self, + params: ModelDefaultsPutParams, + ) -> Result { + if let Some(model) = ¶ms.model { + params.slot.validate_model(model)?; + } + let expected = i64::try_from(params.expected_revision) + .ok() + .filter(|value| *value < i64::MAX) + .ok_or_else(|| { + api::AgentApiError::invalid_request("expectedRevision is out of range") + })?; + let model = params + .model + .as_ref() + .map(serde_json::to_value) + .transpose()?; + // An untouched universe uses a unique insert. Later writes use a + // revision guard; neither path can overwrite a concurrent winner. + let statement = if expected == 0 { + "INSERT INTO universe_model_defaults (universe_id, revision, agent_run, speech_to_text) + SELECT $1, 1, CASE WHEN $2 THEN $3::jsonb END, CASE WHEN NOT $2 THEN $3::jsonb END + WHERE $4::bigint = 0 + ON CONFLICT (universe_id) DO NOTHING + RETURNING revision, agent_run, speech_to_text" + } else { + "UPDATE universe_model_defaults SET revision = revision + 1, + agent_run = CASE WHEN $2 THEN $3::jsonb ELSE agent_run END, + speech_to_text = CASE WHEN NOT $2 THEN $3::jsonb ELSE speech_to_text END + WHERE universe_id = $1 AND revision = $4 + RETURNING revision, agent_run, speech_to_text" + }; + let row = sqlx::query(statement) + .bind(self.config.universe_id) + .bind(params.slot == ModelDefaultSlot::AgentRun) + .bind(model) + .bind(expected) + .fetch_optional(&self.pool) + .await? + .ok_or(ModelDefaultsStoreError::Conflict { + expected: params.expected_revision, + })?; + decode(&row) + } +} + +fn decode(row: &sqlx::postgres::PgRow) -> Result { + Ok(ModelDefaults { + revision: row.try_get::("revision")? as u64, + agent_run: row + .try_get::, _>("agent_run")? + .map(serde_json::from_value) + .transpose()?, + speech_to_text: row + .try_get::, _>("speech_to_text")? + .map(serde_json::from_value) + .transpose()?, + }) +} diff --git a/crates/store-pg/tests/model_defaults_live.rs b/crates/store-pg/tests/model_defaults_live.rs new file mode 100644 index 000000000..eddc318b4 --- /dev/null +++ b/crates/store-pg/tests/model_defaults_live.rs @@ -0,0 +1,157 @@ +use futures_util::FutureExt as _; +use sqlx::{ + Executor as _, + postgres::{PgConnectOptions, PgPoolOptions}, +}; +use std::{panic::AssertUnwindSafe, str::FromStr}; +use store_pg::{PgStore, PgStoreConfig}; +use uuid::Uuid; + +#[tokio::test(flavor = "current_thread")] +#[ignore = "requires local LIGHTSPEED_TEST_POSTGRES_URL; isolated schema"] +async fn universe_defaults_are_revision_safe_and_isolated() { + let database_url = std::env::var("LIGHTSPEED_TEST_POSTGRES_URL") + .expect("LIGHTSPEED_TEST_POSTGRES_URL must be set; run ./dev.sh infra and source scripts/dev/env.sh"); + let admin = PgPoolOptions::new() + .max_connections(1) + .connect(&database_url) + .await + .expect("connect to live Postgres"); + let schema = format!("lightspeed_model_defaults_test_{}", Uuid::new_v4().simple()); + admin + .execute(format!("CREATE SCHEMA \"{schema}\"").as_str()) + .await + .expect("create isolated schema"); + let pool = PgPoolOptions::new() + .max_connections(2) + .connect_with( + PgConnectOptions::from_str(&database_url) + .expect("parse Postgres URL") + .options([("search_path", schema.as_str())]), + ) + .await + .expect("connect to isolated schema"); + let outcome = AssertUnwindSafe(exercise(&pool)).catch_unwind().await; + pool.close().await; + admin + .execute(format!("DROP SCHEMA \"{schema}\" CASCADE").as_str()) + .await + .expect("drop isolated schema"); + admin.close().await; + if let Err(panic) = outcome { + std::panic::resume_unwind(panic); + } +} + +async fn exercise(pool: &sqlx::PgPool) { + use api::{ModelConfig, ModelDefaultSlot as Slot, ModelDefaultsPutParams as Put}; + use store_pg::ModelDefaultsStoreError; + PgStore::migrate(pool).await.unwrap(); + let first = PgStore::new(pool.clone(), PgStoreConfig::new(Uuid::new_v4())); + let second = PgStore::new(pool.clone(), PgStoreConfig::new(Uuid::new_v4())); + first.ensure_universe().await.unwrap(); + second.ensure_universe().await.unwrap(); + assert_eq!( + first.read_model_defaults().await.unwrap(), + api::ModelDefaults::default() + ); + let route = |provider: &str, kind: &str| ModelConfig { + provider_id: provider.into(), + api_kind: kind.into(), + model: "private-model".into(), + }; + let request = Put { + slot: Slot::AgentRun, + model: Some(route("anthropic", "anthropic:messages")), + expected_revision: 0, + }; + let (left, right) = tokio::join!( + first.put_model_defaults(request.clone()), + first.put_model_defaults(request) + ); + let winner = match (left, right) { + (Ok(value), Err(ModelDefaultsStoreError::Conflict { expected: 0 })) + | (Err(ModelDefaultsStoreError::Conflict { expected: 0 }), Ok(value)) => value, + other => panic!("exactly one first writer must win: {other:?}"), + }; + assert_eq!(winner.revision, 1); + let other = second + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: Some(route("local", "openai:completions")), + expected_revision: 0, + }) + .await + .unwrap(); + let speech = first + .put_model_defaults(Put { + slot: Slot::SpeechToText, + model: Some(route("speech", "openai:audio-transcriptions")), + expected_revision: 1, + }) + .await + .unwrap(); + assert_eq!(speech.agent_run, winner.agent_run); + assert!(speech.speech_to_text.is_some()); + assert_eq!(second.read_model_defaults().await.unwrap(), other); + + let stale = first + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: None, + expected_revision: 1, + }) + .await + .unwrap_err(); + assert!(matches!( + stale, + ModelDefaultsStoreError::Conflict { expected: 1 } + )); + let invalid = first + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: speech.speech_to_text.clone(), + expected_revision: 2, + }) + .await + .unwrap_err(); + assert!( + matches!(invalid, ModelDefaultsStoreError::Invalid(error) if error.kind == api::AgentApiErrorKind::InvalidRequest) + ); + assert_eq!(first.read_model_defaults().await.unwrap(), speech); + + let cleared = first + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: None, + expected_revision: 2, + }) + .await + .unwrap(); + assert_eq!(cleared.revision, 3); + assert_eq!(cleared.agent_run, None); + assert_eq!(cleared.speech_to_text, speech.speech_to_text); + let retry_seed = first + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: winner.agent_run, + expected_revision: 0, + }) + .await + .unwrap_err(); + assert!(matches!( + retry_seed, + ModelDefaultsStoreError::Conflict { expected: 0 } + )); + assert_eq!(first.read_model_defaults().await.unwrap(), cleared); + sqlx::query("DELETE FROM universes WHERE universe_id = $1") + .bind(first.config().universe_id) + .execute(pool) + .await + .unwrap(); + assert_eq!( + first.read_model_defaults().await.unwrap(), + api::ModelDefaults::default() + ); + assert_eq!(second.read_model_defaults().await.unwrap(), other); +} diff --git a/crates/temporal-server/src/config.rs b/crates/temporal-server/src/config.rs index f0545b483..1cc633425 100644 --- a/crates/temporal-server/src/config.rs +++ b/crates/temporal-server/src/config.rs @@ -1,21 +1,29 @@ use std::{env, sync::Arc, time::Duration}; -use engine::{ModelSelection, ProviderApiKind}; use object_store::ObjectStore; use sqlx::{PgPool, postgres::PgPoolOptions}; use store_pg::{ BlobCache, PgStore, PgStoreConfig, PgStoreError, S3ObjectStoreConfig, SchemaStatus, SecretsMasterKey, build_s3_object_store, }; -use temporal_workflow::{DEFAULT_MODEL, DEFAULT_TASK_QUEUE, bots::DEFAULT_BOTS_TASK_QUEUE}; +use temporal_workflow::{DEFAULT_TASK_QUEUE, bots::DEFAULT_BOTS_TASK_QUEUE}; use uuid::Uuid; -pub fn default_model_from_env() -> ModelSelection { - ModelSelection { - api_kind: ProviderApiKind::OpenAiResponses, - provider_id: env::var("LIGHTSPEED_CHAT_PROVIDER").unwrap_or_else(|_| "openai".to_owned()), - model: env::var("LIGHTSPEED_CHAT_MODEL").unwrap_or_else(|_| DEFAULT_MODEL.to_owned()), +/// Retired deployment model policy must be migrated to universe defaults. +/// Credentials and provider transport still use their existing configuration. +pub fn validate_model_environment() -> anyhow::Result<()> { + validate_model_environment_with(|name| env::var_os(name).is_some()) +} + +fn validate_model_environment_with(is_set: impl Fn(&str) -> bool) -> anyhow::Result<()> { + for name in ["LIGHTSPEED_CHAT_PROVIDER", "LIGHTSPEED_CHAT_MODEL"] { + if is_set(name) { + anyhow::bail!( + "{name} is retired; remove it and configure each universe with `lightspeed model defaults set agent-run --provider --api-kind --model `" + ); + } } + Ok(()) } pub fn universe_id_from_env() -> anyhow::Result { @@ -413,6 +421,16 @@ fn optional_env(key: &str) -> Option { mod tests { use super::*; + #[test] + fn retired_model_environment_is_rejected_with_a_configuration_action() { + validate_model_environment_with(|_| false).unwrap(); + for name in ["LIGHTSPEED_CHAT_MODEL", "LIGHTSPEED_CHAT_PROVIDER"] { + let error = validate_model_environment_with(|candidate| candidate == name).unwrap_err(); + assert!(error.to_string().contains(name)); + assert!(error.to_string().contains("model defaults set")); + } + } + #[test] fn unledgered_schema_bypass_is_narrow() { let relations = vec!["sessions".to_owned(), "universes".to_owned()]; diff --git a/crates/temporal-server/src/gateway/mod.rs b/crates/temporal-server/src/gateway/mod.rs index e8ecd246e..29d320d88 100644 --- a/crates/temporal-server/src/gateway/mod.rs +++ b/crates/temporal-server/src/gateway/mod.rs @@ -8,7 +8,7 @@ pub mod registration; pub mod request_context; pub(crate) mod service; -pub use crate::config::{default_model_from_env, pg_store_from_env}; +pub use crate::config::pg_store_from_env; pub use deployment::GatewayDeploymentApi; pub use http::{ ACTOR_HEADER, DEFAULT_GATEWAY_BIND, DEFAULT_MAX_REQUEST_BODY_BYTES, GatewayRoutes, @@ -21,6 +21,6 @@ pub use service::{ }; pub use temporal_workflow::{ AgentAdmission, AgentAdmissionFailure, AgentAdmissionFailureKind, AgentCompletedRunSummary, - AgentSessionArgs, AgentSessionStatus, AgentSessionWorkflow, DEFAULT_MODEL, DEFAULT_TASK_QUEUE, + AgentSessionArgs, AgentSessionStatus, AgentSessionWorkflow, DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, connect_temporal, default_session_config, }; diff --git a/crates/temporal-server/src/gateway/service/api_config.rs b/crates/temporal-server/src/gateway/service/api_config.rs index d4260c370..68fda99dd 100644 --- a/crates/temporal-server/src/gateway/service/api_config.rs +++ b/crates/temporal-server/src/gateway/service/api_config.rs @@ -5,10 +5,15 @@ impl GatewayAgentApi { &self, api_config: Option, ) -> Result { - let config = engine_session_config_from_api( - api_config.unwrap_or_default(), - self.default_model.clone(), - )?; + let api_config = api_config.unwrap_or_default(); + let model = model_defaults::creation_model(api_config.model.clone(), async { + self.store + .read_model_defaults() + .await + .map_err(model_defaults::map_store_error) + }) + .await?; + let config = engine_session_config_from_api(api_config, model)?; config .validate() .map_err(|error| AgentApiError::invalid_request(error.to_string()))?; @@ -67,14 +72,15 @@ pub(super) fn run_config_for_start( } /// Translate the wire config document into the engine document. An absent -/// `model` falls back to the deployment default; everything else maps 1:1. +/// `model` uses the caller's resolved creation model or current session model; +/// everything else maps 1:1. pub(super) fn engine_session_config_from_api( api_config: api::SessionConfig, - default_model: ModelSelection, + resolved_model: ModelSelection, ) -> Result { let model = match api_config.model { Some(model) => model_selection_from_api(model)?, - None => default_model, + None => resolved_model, }; let generation = generation_from_api(api_config.generation, &model)?; Ok(SessionConfig { diff --git a/crates/temporal-server/src/gateway/service/mod.rs b/crates/temporal-server/src/gateway/service/mod.rs index 562fadec3..630adc7eb 100644 --- a/crates/temporal-server/src/gateway/service/mod.rs +++ b/crates/temporal-server/src/gateway/service/mod.rs @@ -19,6 +19,7 @@ mod github_api; mod input; mod mcp_api; pub(crate) mod mcp_discovery; +mod model_defaults; mod models_api; mod oauth_api; mod parse; @@ -137,7 +138,7 @@ use vfs::{ use super::{ AgentAdmission, AgentAdmissionFailure, AgentAdmissionFailureKind, AgentSessionArgs, AgentSessionStatus, AgentSessionWorkflow, DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, - DEFAULT_TEMPORAL_TARGET, connect_temporal, default_model_from_env, pg_store_from_env, + DEFAULT_TEMPORAL_TARGET, connect_temporal, pg_store_from_env, }; const DEFAULT_POLL_INTERVAL: Duration = Duration::from_millis(500); @@ -553,7 +554,6 @@ pub struct GatewayAgentApiBuilder { task_queue: String, bot_task_queue: String, channel_task_queue: String, - default_model: ModelSelection, continue_as_new_history_threshold: Option, poll_interval: Duration, operation_timeout: Duration, @@ -635,11 +635,6 @@ impl GatewayAgentApiBuilder { self } - pub fn with_default_model(mut self, model: ModelSelection) -> Self { - self.default_model = model; - self - } - pub fn with_environment_gateway( mut self, gateway: crate::environments::gateway::EnvironmentGatewayClientConfig, @@ -754,7 +749,6 @@ impl GatewayAgentApiBuilder { task_queue: self.task_queue, bot_task_queue: self.bot_task_queue, channel_task_queue: self.channel_task_queue, - default_model: self.default_model, continue_as_new_history_threshold: self.continue_as_new_history_threshold, poll_interval: self.poll_interval, operation_timeout: self.operation_timeout, @@ -781,7 +775,6 @@ pub struct GatewayAgentApi { task_queue: String, pub(crate) bot_task_queue: String, pub(crate) channel_task_queue: String, - default_model: ModelSelection, continue_as_new_history_threshold: Option, poll_interval: Duration, operation_timeout: Duration, @@ -818,7 +811,6 @@ impl GatewayAgentApi { task_queue: DEFAULT_TASK_QUEUE.to_owned(), bot_task_queue: temporal_workflow::bots::DEFAULT_BOTS_TASK_QUEUE.to_owned(), channel_task_queue: crate::config::DEFAULT_CHANNELS_TASK_QUEUE.to_owned(), - default_model: default_model_from_env(), continue_as_new_history_threshold: None, poll_interval: DEFAULT_POLL_INTERVAL, operation_timeout: DEFAULT_OPERATION_TIMEOUT, @@ -2022,6 +2014,38 @@ impl AgentApiService for GatewayAgentApi { .map(AgentApiOutcome::new) } + async fn read_model_defaults( + &self, + _params: api::ModelDefaultsReadParams, + ) -> Result, AgentApiError> { + self.authorize_method(api::METHOD_MODELS_DEFAULTS_READ, None) + .await?; + let defaults = self + .store + .read_model_defaults() + .await + .map_err(model_defaults::map_store_error)?; + Ok(AgentApiOutcome::new(api::ModelDefaultsResponse { + defaults, + })) + } + + async fn put_model_defaults( + &self, + params: api::ModelDefaultsPutParams, + ) -> Result, AgentApiError> { + self.authorize_method(api::METHOD_MODELS_DEFAULTS_PUT, None) + .await?; + let defaults = self + .store + .put_model_defaults(params) + .await + .map_err(model_defaults::map_store_error)?; + Ok(AgentApiOutcome::new(api::ModelDefaultsResponse { + defaults, + })) + } + async fn list_models( &self, params: ModelListParams, @@ -2227,7 +2251,10 @@ impl AgentApiService for GatewayAgentApi { ))); } } - let config = engine_session_config_from_api(params.config, self.default_model.clone())?; + let config = engine_session_config_from_api( + params.config, + model_defaults::current_session_model(&loaded.state)?, + )?; config .validate() .map_err(|error| AgentApiError::invalid_request(error.to_string()))?; diff --git a/crates/temporal-server/src/gateway/service/model_defaults.rs b/crates/temporal-server/src/gateway/service/model_defaults.rs new file mode 100644 index 000000000..f7a7daf85 --- /dev/null +++ b/crates/temporal-server/src/gateway/service/model_defaults.rs @@ -0,0 +1,153 @@ +use super::*; + +pub(super) fn map_store_error(error: store_pg::ModelDefaultsStoreError) -> AgentApiError { + match error { + store_pg::ModelDefaultsStoreError::Conflict { .. } => { + AgentApiError::conflict(error.to_string()) + } + store_pg::ModelDefaultsStoreError::Invalid(error) => error, + other => AgentApiError::internal(other.to_string()), + } +} + +/// An explicit (already profile-merged) model never consults universe policy. +pub(super) async fn creation_model( + explicit: Option, + defaults: impl std::future::Future>, +) -> Result { + let model = match explicit { + Some(model) => model, + None => defaults + .await? + .agent_run + .ok_or_else(|| AgentApiError::model_default_unset(api::ModelDefaultSlot::AgentRun))?, + }; + api::ModelDefaultSlot::AgentRun.validate_model(&model)?; + api_config::model_selection_from_api(model) +} + +pub(super) fn current_session_model( + state: &engine::CoreAgentState, +) -> Result { + state + .lifecycle + .config + .as_ref() + .map(|config| config.model.clone()) + .ok_or_else(|| AgentApiError::rejected("session has no configuration")) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn route(provider: &str, kind: &str, model: &str) -> ModelConfig { + ModelConfig { + provider_id: provider.into(), + api_kind: kind.into(), + model: model.into(), + } + } + + #[tokio::test(flavor = "current_thread")] + async fn explicit_selection_does_not_read_universe_defaults() { + let explicit = route("private", "openai:completions", "custom"); + let model = creation_model(Some(explicit), async { + panic!("explicit model must not read defaults") + }) + .await + .unwrap(); + assert_eq!(model.provider_id, "private"); + assert_eq!(model.api_kind, ProviderApiKind::OpenAiCompletions); + } + + #[tokio::test(flavor = "current_thread")] + async fn profile_merge_precedes_universe_resolution() { + let profile = api::SessionConfig { + model: Some(route("anthropic", "anthropic:messages", "profile")), + ..Default::default() + }; + for explicit in [None, Some(route("local", "openai:completions", "explicit"))] { + let expected = explicit.clone().or(profile.model.clone()).unwrap(); + let merged = GatewayAgentApi::merge_profile_start_config( + Some(profile.clone()), + Some(api::SessionConfig { + model: explicit, + ..Default::default() + }), + ) + .unwrap(); + let model = creation_model(merged.model, async { panic!("merged selection must win") }) + .await + .unwrap(); + assert_eq!(model.provider_id, expected.provider_id); + assert_eq!(model.model, expected.model); + } + } + + #[test] + fn applying_sparse_profiles_preserves_model_but_replaces_features() { + let model = + api_config::model_selection_from_api(route("local", "openai:completions", "pinned")) + .unwrap(); + let mut previous = + engine_session_config_from_api(api::SessionConfig::default(), model.clone()).unwrap(); + previous.features.web = Some(engine::WebFeature { + version: 1, + fetch: Some(engine::WebFetchFeature {}), + search: None, + }); + let profile = api::ProfileDocument { + config: Some(api::SessionConfig::default()), + ..Default::default() + }; + let applied = + GatewayAgentApi::profile_intent(&profile, Some(previous.model.clone())).unwrap(); + let config = applied.config.unwrap(); + assert_eq!(config.model, model); + assert_eq!(config.features.web, None); + let replaced = + engine_session_config_from_api(api::SessionConfig::default(), previous.model).unwrap(); + assert_eq!(replaced, config); + assert!( + GatewayAgentApi::profile_intent(&profile, None) + .unwrap() + .config + .is_none(), + "creation already merged its profile configuration" + ); + } + + #[tokio::test(flavor = "current_thread")] + async fn universe_routes_are_independent_and_resolution_is_a_snapshot() { + let mut first = api::ModelDefaults { + revision: 1, + agent_run: Some(route("anthropic", "anthropic:messages", "first")), + speech_to_text: None, + }; + let second = api::ModelDefaults { + revision: 1, + agent_run: Some(route("private", "openai:completions", "second")), + speech_to_text: None, + }; + let first_model = creation_model(None, async { Ok(first.clone()) }) + .await + .unwrap(); + let second_model = creation_model(None, async { Ok(second) }).await.unwrap(); + let existing = + engine_session_config_from_api(api::SessionConfig::default(), first_model).unwrap(); + first.agent_run = None; + let error = creation_model(None, async { Ok(first) }).await.unwrap_err(); + assert_eq!(error.kind, AgentApiErrorKind::ModelDefaultUnset); + assert_eq!( + error.model_default_slot, + Some(api::ModelDefaultSlot::AgentRun) + ); + assert_eq!(existing.model.provider_id, "anthropic"); + assert_eq!(second_model.provider_id, "private"); + let replaced = + engine_session_config_from_api(api::SessionConfig::default(), existing.model.clone()) + .unwrap(); + assert_eq!(replaced.model, existing.model); + } +} diff --git a/crates/temporal-server/src/gateway/service/profiles.rs b/crates/temporal-server/src/gateway/service/profiles.rs index 8e653e5a6..43d92698d 100644 --- a/crates/temporal-server/src/gateway/service/profiles.rs +++ b/crates/temporal-server/src/gateway/service/profiles.rs @@ -91,13 +91,25 @@ impl GatewayAgentApi { AgentApiError::invalid_request(format!("invalid session id: {error}")) })?; let resolved = self.resolve_profile_source(params.profile).await?; - let profile = self.profile_intent(&resolved, true)?; + let loaded = self.load_session_state(&session_id).await?; + let revision = loaded.state.lifecycle.config_revision; + if let Some(expected) = params.expected_config_revision + && expected != revision + { + return Err(AgentApiError::conflict(format!( + "expected config revision {expected}, got {revision}" + ))); + } + let profile = Self::profile_intent( + &resolved, + Some(model_defaults::current_session_model(&loaded.state)?), + )?; let applied = self .prepare_session_operation( &session_id, temporal_workflow::SessionOperation::ApplyProfile { profile, - expected_config_revision: params.expected_config_revision, + expected_config_revision: Some(revision), expected_tools_revision: params.expected_tools_revision, }, ) @@ -125,7 +137,6 @@ impl GatewayAgentApi { } pub(super) fn merge_profile_start_config( - &self, profile_config: Option, explicit_config: Option, ) -> Option { @@ -144,20 +155,18 @@ impl GatewayAgentApi { }) } - /// With `apply_config`, the profile's own configuration is applied and - /// its default attachment is the environment fill candidate. Without it - /// (session start), the caller has merged the effective configuration - /// and sets the candidate itself. + /// With a current model, apply the profile configuration and preserve an + /// omitted model. On creation the caller has already merged configuration + /// and sets the environment candidate itself. pub(super) fn profile_intent( - &self, profile: &ProfileDocument, - apply_config: bool, + current_model: Option, ) -> Result { - let config = if apply_config { + let config = if let Some(current_model) = current_model { profile .config .clone() - .map(|config| engine_session_config_from_api(config, self.default_model.clone())) + .map(|config| engine_session_config_from_api(config, current_model)) .transpose()? } else { None diff --git a/crates/temporal-server/src/gateway/service/session_lifecycle.rs b/crates/temporal-server/src/gateway/service/session_lifecycle.rs index 879b94762..e7405e673 100644 --- a/crates/temporal-server/src/gateway/service/session_lifecycle.rs +++ b/crates/temporal-server/src/gateway/service/session_lifecycle.rs @@ -333,7 +333,7 @@ impl GatewayAgentApi { delete_after_close_ms, ); validate_delete_after_close_ms(effective_delete_after_close_ms)?; - let start_config = self.merge_profile_start_config( + let start_config = Self::merge_profile_start_config( resolved_profile .as_ref() .and_then(|profile| profile.config.clone()), @@ -358,7 +358,7 @@ impl GatewayAgentApi { ); args.setup = match resolved_profile.as_ref() { Some(profile) => { - let mut intent = self.profile_intent(profile, false)?; + let mut intent = Self::profile_intent(profile, None)?; intent.environment = setup_environment; Some(intent) } diff --git a/crates/temporal-server/src/lib.rs b/crates/temporal-server/src/lib.rs index 8eb12d690..1390551f2 100644 --- a/crates/temporal-server/src/lib.rs +++ b/crates/temporal-server/src/lib.rs @@ -17,7 +17,7 @@ pub mod universe; pub mod worker; pub use config::{ - DeploymentStores, GatewayAuthMode, default_model_from_env, gateway_auth_mode_from_env, - pg_store_from_env, task_queue_from_env, universe_id_from_env, + DeploymentStores, GatewayAuthMode, gateway_auth_mode_from_env, pg_store_from_env, + task_queue_from_env, universe_id_from_env, }; pub use universe::{UniverseError, UniverseRuntime, UniverseState}; diff --git a/crates/temporal-server/src/main.rs b/crates/temporal-server/src/main.rs index dc65ec14d..f7244b2e5 100644 --- a/crates/temporal-server/src/main.rs +++ b/crates/temporal-server/src/main.rs @@ -517,6 +517,7 @@ async fn mint( /// Temporal worker on its own task queue; the gateway role adds the HTTP /// server and the deployment reconcilers. async fn run_roles(args: RunArgs) -> anyhow::Result<()> { + temporal_server::config::validate_model_environment()?; let roles = args.roles()?; let task_types = args.task_types()?; let task_queues = args.task_queues()?; diff --git a/crates/temporal-server/src/worker/reaper.rs b/crates/temporal-server/src/worker/reaper.rs index 142f9f025..d4b582d00 100644 --- a/crates/temporal-server/src/worker/reaper.rs +++ b/crates/temporal-server/src/worker/reaper.rs @@ -1115,7 +1115,7 @@ mod tests { ActiveRun, ModelSelection, ProviderApiKind, RunId, RunSource, RunStatus, storage::{BlobGraphStore as _, BlobStore as _}, }; - use temporal_workflow::{DEFAULT_MODEL, default_run_config, default_session_config}; + use temporal_workflow::{default_run_config, default_session_config}; use super::*; @@ -1200,7 +1200,7 @@ mod tests { ModelSelection { api_kind: ProviderApiKind::OpenAiResponses, provider_id: "openai".to_owned(), - model: DEFAULT_MODEL.to_owned(), + model: "test-model".to_owned(), } } diff --git a/crates/temporal-server/tests/bots_live.rs b/crates/temporal-server/tests/bots_live.rs index fd8c6c031..5469acac0 100644 --- a/crates/temporal-server/tests/bots_live.rs +++ b/crates/temporal-server/tests/bots_live.rs @@ -30,8 +30,7 @@ use api::{ use bots::ids::{bot_controller_workflow_id, bot_main_session_id, bot_schedule_id}; use engine::{CoreAgentLlm, CoreAgentTools, storage::BlobStore}; use support::live::{ - LIVE_TEST_LOCK, live_universe_id, openai_live_model, require_openai_live_env, - require_storage_live_env, + LIVE_TEST_LOCK, live_universe_id, require_openai_live_env, require_storage_live_env, }; use temporal_server::{ config::TaskQueues, @@ -66,6 +65,7 @@ where let _guard = LIVE_TEST_LOCK.lock().await; let universe = live_universe_id()?; let store = pg_store_from_env().await?; + support::live::seed_agent_default(&store, &support::live::openai_live_model()).await?; let queues = TaskQueues::derived_from(format!( "lightspeed-bots-live-{}", uuid::Uuid::new_v4().simple() @@ -77,7 +77,7 @@ where let runtime = core_runtime()?; let client = connect_temporal(&temporal_target, &namespace).await?; - let mut builder = GatewayAgentApi::builder(client.clone(), store.clone()) + let builder = GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(queues.sessions.clone()) .with_bot_task_queue(queues.bots.clone()) .with_channel_task_queue(queues.channels.clone()); @@ -89,10 +89,7 @@ where let tools = Arc::new(FakeTools::new(blobs)) as Arc; ActivityState::from_pg_store(store.clone(), llm, tools) } - Llm::Real => { - builder = builder.with_default_model(openai_live_model()); - ActivityState::from_pg_store_with_default_runtime(store.clone())? - } + Llm::Real => ActivityState::from_pg_store_with_default_runtime(store.clone())?, }; let api = Arc::new(builder.build()); diff --git a/crates/temporal-server/tests/channels_live.rs b/crates/temporal-server/tests/channels_live.rs index e62c0427e..3f9921d13 100644 --- a/crates/temporal-server/tests/channels_live.rs +++ b/crates/temporal-server/tests/channels_live.rs @@ -136,6 +136,7 @@ where let _guard = LIVE_TEST_LOCK.lock().await; let universe = live_universe_id()?; let store = pg_store_from_env().await?; + support::live::seed_agent_default(&store, &support::live::openai_live_model()).await?; let queues = TaskQueues::derived_from(format!( "lightspeed-channels-live-{}", uuid::Uuid::new_v4().simple() diff --git a/crates/temporal-server/tests/environment_provider_live.rs b/crates/temporal-server/tests/environment_provider_live.rs index f5ef6da0b..5b1bb6e3e 100644 --- a/crates/temporal-server/tests/environment_provider_live.rs +++ b/crates/temporal-server/tests/environment_provider_live.rs @@ -25,7 +25,7 @@ use support::live::{ run_with_live_worker, wait_for_environment_status, }; use temporal_server::{ - DeploymentStores, UniverseRuntime, default_model_from_env, + DeploymentStores, UniverseRuntime, gateway::{GatewayAgentApi, GatewayDeploymentApi}, pg_store_from_env, }; @@ -361,10 +361,10 @@ async fn run_environment_power_live_client( }; let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let suffix = uuid::Uuid::new_v4().simple().to_string(); let provider_id = format!("fake-power-{suffix}"); diff --git a/crates/temporal-server/tests/mcp_live.rs b/crates/temporal-server/tests/mcp_live.rs index 30633c41f..3fc25f85d 100644 --- a/crates/temporal-server/tests/mcp_live.rs +++ b/crates/temporal-server/tests/mcp_live.rs @@ -38,7 +38,6 @@ use support::live::{ run_with_live_worker, wait_for_terminal_run, }; use temporal_server::{ - default_model_from_env, gateway::{DEFAULT_MAX_REQUEST_BODY_BYTES, GatewayAgentApi, GatewayState, gateway_router}, pg_store_from_env, worker::{ActivityState, FakeTools, SessionTools, WorkerActivities}, @@ -433,10 +432,10 @@ async fn run_matrix_client( ids: MatrixServerIds, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); put_fixture_server( @@ -1379,10 +1378,10 @@ async fn run_approval_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { access: None, @@ -1520,11 +1519,11 @@ async fn run_native_mcp_live_client( server_id: String, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = Arc::new( GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(), ); let configurator = LiveConfigurator::start(api.clone()).await?; @@ -1633,11 +1632,11 @@ async fn run_mcp_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = Arc::new( GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(), ); let configurator = LiveConfigurator::start(api.clone()).await?; @@ -2186,11 +2185,11 @@ async fn run_mixed_batch_live_client( server_id: String, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = Arc::new( GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(), ); let configurator = LiveConfigurator::start(api.clone()).await?; diff --git a/crates/temporal-server/tests/preprocess_live.rs b/crates/temporal-server/tests/preprocess_live.rs index 0b3c47613..17dc34565 100644 --- a/crates/temporal-server/tests/preprocess_live.rs +++ b/crates/temporal-server/tests/preprocess_live.rs @@ -17,7 +17,6 @@ use support::live::{ require_storage_live_env, run_with_live_worker, wait_for_terminal_run, }; use temporal_server::{ - default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env, worker::{ @@ -85,10 +84,10 @@ async fn run_audio_preprocess_live_client( transcriber: Arc, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -201,10 +200,10 @@ async fn run_transcodable_audio_preprocess_live_client( transcoded_bytes: Vec, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { diff --git a/crates/temporal-server/tests/profiles_live.rs b/crates/temporal-server/tests/profiles_live.rs index 1dcb89b88..79fe0971a 100644 --- a/crates/temporal-server/tests/profiles_live.rs +++ b/crates/temporal-server/tests/profiles_live.rs @@ -16,7 +16,7 @@ use support::live::{ read_session_view, require_storage_live_env, run_with_live_worker, wait_for_environment_status, wait_for_terminal_run, }; -use temporal_server::{default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env}; +use temporal_server::{gateway::GatewayAgentApi, pg_store_from_env}; use temporalio_client::{Client, WorkflowTerminateOptions}; #[tokio::test(flavor = "current_thread")] @@ -58,10 +58,10 @@ async fn run_profile_environment_selection_live_client( }; let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let suffix = uuid::Uuid::new_v4().simple().to_string(); let provider_id = format!("fake-profile-{suffix}"); @@ -242,10 +242,10 @@ async fn run_profiles_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let profile_id = ProfileId::new(format!("live_profile_{}", uuid::Uuid::new_v4().simple())); let server_id = format!("profile_crm_{}", uuid::Uuid::new_v4().simple()); diff --git a/crates/temporal-server/tests/runs_live.rs b/crates/temporal-server/tests/runs_live.rs index f488c7e62..913dcf421 100644 --- a/crates/temporal-server/tests/runs_live.rs +++ b/crates/temporal-server/tests/runs_live.rs @@ -19,7 +19,7 @@ use support::live::{ read_run, require_storage_live_env, run_with_live_worker, start_text_run, terminate_live_session, wait_for_terminal_run, wait_until, }; -use temporal_server::{default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env}; +use temporal_server::{gateway::GatewayAgentApi, pg_store_from_env}; use temporal_workflow::LLM_RETRY_MAX_ATTEMPTS; use temporalio_client::{Client, WorkflowDescribeOptions, WorkflowTerminateOptions}; @@ -220,10 +220,10 @@ async fn run_control_api( with_tools: bool, ) -> anyhow::Result { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let config = if with_tools { let workspace = api @@ -725,10 +725,10 @@ async fn run_parallel_tool_batch_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); // A read-only workspace attachment derives parallel-safe function tools @@ -853,10 +853,10 @@ async fn run_transient_llm_retry_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model) .build(); api.start_session(SessionStartParams { @@ -934,10 +934,10 @@ async fn run_llm_retry_exhaustion_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model) .build(); api.start_session(SessionStartParams { @@ -1042,10 +1042,10 @@ async fn run_unbounded_hosted_run_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { diff --git a/crates/temporal-server/tests/runs_live_slow.rs b/crates/temporal-server/tests/runs_live_slow.rs index a89a20a74..6c31cac4d 100644 --- a/crates/temporal-server/tests/runs_live_slow.rs +++ b/crates/temporal-server/tests/runs_live_slow.rs @@ -27,10 +27,7 @@ use support::live::{ live_workflow_handle, require_storage_live_env, run_with_live_worker_timeout, wait_for_terminal_run, }; -use temporal_server::{ - default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env, - worker::FakeRuntimeCounters, -}; +use temporal_server::{gateway::GatewayAgentApi, pg_store_from_env, worker::FakeRuntimeCounters}; use temporal_workflow::{LLM_SCHEDULE_TO_CLOSE, LLM_START_TO_CLOSE}; use temporalio_client::{Client, WorkflowDescribeOptions, WorkflowTerminateOptions}; @@ -67,9 +64,9 @@ async fn run_llm_timeout_live_client( counters: FakeRuntimeCounters, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; + support::live::seed_agent_default(&store, &support::live::openai_live_model()).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(default_model_from_env()) .build(); api.start_session(SessionStartParams { diff --git a/crates/temporal-server/tests/sessions_live.rs b/crates/temporal-server/tests/sessions_live.rs index 064df7805..2cefd34a5 100644 --- a/crates/temporal-server/tests/sessions_live.rs +++ b/crates/temporal-server/tests/sessions_live.rs @@ -23,9 +23,7 @@ use support::live::{ require_storage_live_env, run_with_live_worker, start_text_run, wait_for_admission_failure, wait_for_session_status, wait_for_terminal_run, }; -use temporal_server::{ - default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env, worker::WorkerActivities, -}; +use temporal_server::{gateway::GatewayAgentApi, pg_store_from_env, worker::WorkerActivities}; use temporal_workflow::{AgentAdmission, AgentAdmissionFailureKind, AgentSessionWorkflow}; use temporalio_client::{ Client, WorkflowDescribeOptions, WorkflowSignalOptions, WorkflowTerminateOptions, @@ -179,10 +177,10 @@ async fn run_checkpoint_and_bounded_reads_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store.clone()) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -372,10 +370,10 @@ async fn run_fake_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let initialized = api.initialize(InitializeParams::default()).await?; @@ -662,10 +660,10 @@ async fn run_lifecycle_delete_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store) .with_task_queue(task_queue) - .with_default_model(model) .build(); api.start_session(SessionStartParams { @@ -751,10 +749,10 @@ async fn run_continue_as_new_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .with_continue_as_new_history_threshold(1) .build(); @@ -851,10 +849,10 @@ async fn run_missing_session_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store) .with_task_queue(task_queue) - .with_default_model(model) .build(); let error = api @@ -882,10 +880,10 @@ async fn run_context_append_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -1101,10 +1099,10 @@ async fn run_admission_failure_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -1217,25 +1215,23 @@ async fn run_openai_live_client( let store = pg_store_from_env().await?; let instructions = "You are Agent in a live integration test. Do not call tools for this test. Reply with the exact phrase requested by the user."; let model = openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); - api.start_session(SessionStartParams { + let start = SessionStartParams { access: None, metadata: Default::default(), session_id: Some(session_id.as_str().to_owned()), display_name: None, - config: Some(SessionConfig { - model: Some(model_to_api(&model)), - ..SessionConfig::default() - }), + config: None, profile: Some(ProfileSource::Inline { profile: Box::new(InlineAgentProfile { display_name: Some("OpenAI live test".to_owned()), description: None, document: ProfileDocument { + config: Some(SessionConfig::default()), instructions: Some(ProfileInstructions::Text { text: instructions.to_owned(), }), @@ -1244,8 +1240,66 @@ async fn run_openai_live_client( }), }), delete_after_close_ms: None, + }; + api.start_session(start.clone()).await?; + let session = read_session_view(&api, &session_id).await?; + assert_eq!( + session + .config + .as_ref() + .and_then(|config| config.model.as_ref()), + Some(&model_to_api(&model)), + "creation must persist the universe default" + ); + + let defaults = api + .read_model_defaults(api::ModelDefaultsReadParams {}) + .await? + .result + .defaults; + api.put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::AgentRun, + model: None, + expected_revision: defaults.revision, + }) + .await?; + let missing = api + .start_session(SessionStartParams { + session_id: Some(format!("{session_id}_without_default")), + ..start.clone() + }) + .await + .unwrap_err(); + assert_eq!(missing.kind, AgentApiErrorKind::ModelDefaultUnset); + assert_eq!( + missing.model_default_slot, + Some(api::ModelDefaultSlot::AgentRun) + ); + + // Reopening and sparse updates use the stored model after policy is cleared. + api.start_session(start.clone()).await?; + let replaced = api + .put_session_config(SessionConfigPutParams { + session_id: session_id.as_str().to_owned(), + config: SessionConfig::default(), + expected_config_revision: Some(session.config_revision), + }) + .await? + .result + .session; + assert_eq!( + read_session_view(&api, &session_id).await?.config, + session.config + ); + api.apply_profile(api::ProfileApplyParams { + session_id: session_id.as_str().to_owned(), + profile: start.profile.expect("inline test profile"), + expected_config_revision: Some(replaced.config_revision), + expected_tools_revision: None, }) .await?; + let reapplied = read_session_view(&api, &session_id).await?; + assert_eq!(reapplied.config, session.config); let run = api .start_run(RunStartParams { @@ -1290,10 +1344,18 @@ async fn run_builtin_tool_live_client( session_id: SessionId, model: engine::ModelSelection, ) -> anyhow::Result<()> { + // gpt-6-sol requires reasoning to be disabled when combining function + // tools with Chat Completions. This fixture tests tool execution. + let generation = (model.api_kind == engine::ProviderApiKind::OpenAiCompletions).then(|| { + api::GenerationConfig { + reasoning_effort: Some("none".to_owned()), + ..Default::default() + } + }); let store = pg_store_from_env().await?; + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -1303,6 +1365,7 @@ async fn run_builtin_tool_live_client( display_name: None, config: Some(SessionConfig { model: Some(model_to_api(&model)), + generation, features: Some(FeaturesConfig { timers: Some(TimersFeature { version: api::CURRENT_FEATURE_VERSION, @@ -1341,11 +1404,6 @@ async fn run_builtin_tool_live_client( }) .await?; let run = wait_for_terminal_run(&api, &session_id, &run.result.run.id).await?; - let output = final_assistant_text(&run).expect("assistant output"); - assert!( - output.to_lowercase().contains("temporal tool ok"), - "expected completion marker: {output}" - ); let events = api .read_session_events(SessionEventsReadParams { direction: Default::default(), @@ -1358,6 +1416,21 @@ async fn run_builtin_tool_live_client( .await? .result .events; + anyhow::ensure!( + run.status == api::RunStatus::Completed, + "provider tool run ended with {:?}: {:?}", + run.status, + events + .iter() + .filter(|event| matches!(&event.kind, api::SessionEventKindView::RunFailed { run_id, .. } if run_id == &run.id)) + .map(|event| &event.kind) + .collect::>() + ); + let output = final_assistant_text(&run).expect("assistant output"); + assert!( + output.to_lowercase().contains("temporal tool ok"), + "expected completion marker: {output}" + ); let content = events .iter() .find_map(|event| match &event.kind { @@ -1438,10 +1511,10 @@ async fn run_session_metadata_live_client( use std::collections::BTreeMap; let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store) .with_task_queue(task_queue) - .with_default_model(model) .build(); let pair = |key: &str, value: &str| (key.to_owned(), value.to_owned()); // The job value is unique per run so the filter isolates this session diff --git a/crates/temporal-server/tests/subagents_live.rs b/crates/temporal-server/tests/subagents_live.rs index 3c25e903e..4b4f09487 100644 --- a/crates/temporal-server/tests/subagents_live.rs +++ b/crates/temporal-server/tests/subagents_live.rs @@ -23,7 +23,6 @@ use support::live::{ terminate_live_session, wait_for_terminal_run, wait_until, }; use temporal_server::{ - default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env, subagents::AgentApiSubagentRuntime, @@ -495,11 +494,11 @@ where let runtime = core_runtime()?; let client = connect_temporal(&temporal_target, &namespace).await?; let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = Arc::new( GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(task_queue.clone()) - .with_default_model(model.clone()) .build(), ); diff --git a/crates/temporal-server/tests/support/live.rs b/crates/temporal-server/tests/support/live.rs index 014e0cccc..0b2d92ae8 100644 --- a/crates/temporal-server/tests/support/live.rs +++ b/crates/temporal-server/tests/support/live.rs @@ -610,7 +610,6 @@ pub fn openai_live_model() -> ModelSelection { model: env::var("LIGHTSPEED_OPENAI_MODEL") .or_else(|_| env::var("OPENAI_RESPONSES_MODEL")) .or_else(|_| env::var("OPENAI_LIVE_MODEL")) - .or_else(|_| env::var("LIGHTSPEED_CHAT_MODEL")) .unwrap_or_else(|_| "gpt-5.5".to_owned()), } } @@ -622,7 +621,6 @@ pub fn openai_completions_live_model() -> ModelSelection { model: env::var("LIGHTSPEED_OPENAI_MODEL") .or_else(|_| env::var("OPENAI_COMPLETIONS_MODEL")) .or_else(|_| env::var("OPENAI_LIVE_MODEL")) - .or_else(|_| env::var("LIGHTSPEED_CHAT_MODEL")) .unwrap_or_else(|_| "gpt-5.5".to_owned()), } } @@ -643,3 +641,29 @@ mod tests { assert!(error.to_string().contains("stuck request")); } } + +/// Tests that exercise omitted models configure the same durable policy as clients. +pub async fn seed_agent_default( + store: &store_pg::PgStore, + model: &ModelSelection, +) -> anyhow::Result<()> { + store.ensure_universe().await?; + let current = store.read_model_defaults().await?; + store + .put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::AgentRun, + model: Some(api::ModelConfig { + provider_id: model.provider_id.clone(), + api_kind: match model.api_kind { + ProviderApiKind::OpenAiResponses => "openai:responses", + ProviderApiKind::OpenAiCompletions => "openai:completions", + ProviderApiKind::AnthropicMessages => "anthropic:messages", + } + .into(), + model: model.model.clone(), + }), + expected_revision: current.revision, + }) + .await?; + Ok(()) +} diff --git a/crates/temporal-server/tests/vfs_transfer_live.rs b/crates/temporal-server/tests/vfs_transfer_live.rs index fd747c162..2e1856974 100644 --- a/crates/temporal-server/tests/vfs_transfer_live.rs +++ b/crates/temporal-server/tests/vfs_transfer_live.rs @@ -334,7 +334,7 @@ async fn run_case( features["environments"]["environments"][0]["workingDirectory"] = json!(root); features["environments"]["prompts"] = json!({"roots":[".agents/prompts"]}); } - let mut model = temporal_server::default_model_from_env(); + let mut model = support::live::openai_live_model(); model.api_kind = provider; let profile = api.create_profile(api::ProfileCreateParams { profile: api::AgentProfileInput { profile_id: api::ProfileId::new(format!("profile_{session}")), display_name: None, description: None, diff --git a/crates/temporal-server/tests/workflow_tool_plugins_live.rs b/crates/temporal-server/tests/workflow_tool_plugins_live.rs index 4b41cec3a..78b4f4ff9 100644 --- a/crates/temporal-server/tests/workflow_tool_plugins_live.rs +++ b/crates/temporal-server/tests/workflow_tool_plugins_live.rs @@ -835,9 +835,9 @@ where let store = pg_store_from_env().await?; let blobs: Arc = store.clone(); + support::live::seed_agent_default(&store, &support::live::openai_live_model()).await?; let mut api_builder = GatewayAgentApi::builder(client.clone(), store.clone()) - .with_task_queue(session_queue.clone()) - .with_default_model(temporal_server::default_model_from_env()); + .with_task_queue(session_queue.clone()); if let Some(threshold) = continue_as_new_history_threshold { api_builder = api_builder.with_continue_as_new_history_threshold(threshold); } diff --git a/crates/temporal-workflow/src/config.rs b/crates/temporal-workflow/src/config.rs index e3c3d5e9e..4a0e667eb 100644 --- a/crates/temporal-workflow/src/config.rs +++ b/crates/temporal-workflow/src/config.rs @@ -8,7 +8,6 @@ use temporalio_sdk::{ActivityCloseTimeouts, ActivityOptions}; pub const DEFAULT_TASK_QUEUE: &str = "lightspeed-sessions"; pub const DEFAULT_TEMPORAL_TARGET: &str = "localhost:7233"; pub const DEFAULT_TEMPORAL_NAMESPACE: &str = "default"; -pub const DEFAULT_MODEL: &str = "gpt-5.5"; pub const DEFAULT_CONTINUE_AS_NEW_HISTORY_THRESHOLD: u32 = 10_000; pub const DEFAULT_ACTIVITY_START_TO_CLOSE_TIMEOUT: Duration = Duration::from_secs(360); diff --git a/crates/temporal-workflow/src/lib.rs b/crates/temporal-workflow/src/lib.rs index 441a7997f..823be6bb1 100644 --- a/crates/temporal-workflow/src/lib.rs +++ b/crates/temporal-workflow/src/lib.rs @@ -26,7 +26,7 @@ pub use activities::{ pub use config::{ ACTIVITY_CANCELLATION_HEARTBEAT_INTERVAL, ACTIVITY_CANCELLATION_HEARTBEAT_TIMEOUT, DEFAULT_BOOTSTRAP_PAYLOAD_BUDGET_BYTES, DEFAULT_CONTINUE_AS_NEW_HISTORY_THRESHOLD, - DEFAULT_MODEL, DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, + DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, ENVIRONMENT_READY_GRACE, ENVIRONMENT_READY_HEARTBEAT_TIMEOUT, ENVIRONMENT_READY_POLL_INTERVAL, ENVIRONMENT_READY_WAIT, FAKE_TOOL_NAME, LLM_RETRY_MAX_ATTEMPTS, LLM_RETRY_MAX_INTERVAL, LLM_SCHEDULE_TO_CLOSE, LLM_START_TO_CLOSE, MAX_CONCURRENT_TOOL_CALLS_PER_BATCH, diff --git a/docs/roadmap/p184-universe-model-defaults-and-transcription.md b/docs/roadmap/p184-universe-model-defaults-and-transcription.md new file mode 100644 index 000000000..1688ece4e --- /dev/null +++ b/docs/roadmap/p184-universe-model-defaults-and-transcription.md @@ -0,0 +1,488 @@ +# P184 — Universe model defaults and standalone transcription + +**Status:** First slice implemented, 2026-09-28: universe defaults, session +resolution, CLI configuration, and development seeding. Platform settings, +route readiness, and standalone transcription remain to be implemented. +Builds on [CLI session model routing](cli-session-model-routing.md) and +[the runtime CLI](p183-first-class-runtime-cli.md). Supersedes the placement +of transcription inside session admission in +[audio transcription preprocessing](archive/p72-audio-transcription-preprocessing.md). + +## Outcome + +A universe chooses its default agent model and speech-to-text model through +the public runtime API. These choices work for the web, CLI, bots, and direct +API clients, including installations using custom provider endpoints. + +Transcription is an independent operation. Channels transcribe voice messages +before delivering prepared input to a bot. Web dictation produces an editable +draft; only the user's ordinary send action submits it to a session. + +The design stays small: two default slots, shared provider resolution, and one +typed transcription workflow. Future capabilities, such as image generation, +can add a slot, adapter, and operation without extending session orchestration. + +## Baseline before implementation + +- `temporal-server/src/config.rs` resolves an omitted session model from + `LIGHTSPEED_CHAT_PROVIDER` and `LIGHTSPEED_CHAT_MODEL`, with built-in defaults + of `openai` and `gpt-5.5`. The API kind is fixed to OpenAI Responses. +- Session creation, configuration replacement, and profile application use + that deployment fallback. Existing sessions pin their provider identity and + API kind; model changes within the route remain supported. +- Session admission detects audio in run requests and context upserts, awaits + preprocessing, and admits the rewritten transcript. Steering follows a + separate path. Audio preparation therefore has inconsistent entry points and + runs inside the session's admission loop. +- The transcription activity uses the built-in `openai` provider and the audio + client's `gpt-4o-transcribe` default. It resolves a stored credential but does + not apply a stored custom endpoint. Provider failures become ordinary failed + preprocessing outcomes rather than retryable activity failures. +- Provider records and generation clients already support custom endpoints, + API-kind declarations, and credential resolution. Custom endpoint validation + currently admits only Responses and Chat Completions protocols. +- The web composer sends text. Its provider-readiness helper considers any + provider with a configured credential sufficient, independently of the route + a session will actually use. + +## Decisions + +### 1. Universe-owned defaults with a focused API + +Store a small, typed model-defaults record in runtime PostgreSQL, keyed by +universe and protected by an optimistic revision. Platform reads and changes +this record through the runtime; it does not keep another copy of the policy. + +The initial slots are: + +| Slot | Selection | When the default is resolved | +| --- | --- | --- | +| `agentRun` | Agent generation route | Creation of a new session without an explicit or profile model | +| `speechToText` | Transcription route | Admission of a new transcription job without an explicit model | + +Every selection contains `{ providerId, apiKind, model }`. Endpoint URLs, +headers, and credentials remain in the existing provider configuration. +Slots can be unset independently; a universe may use agent runs without +configuring transcription, or transcription with no default agent model. + +Expose: + +- `models/defaults/read`: returns the revision, selections, and per-selection + configuration status. Keep persisted choices distinct from current provider + diagnostics. +- `models/defaults/put`: sets or clears one named slot, using an expected + revision. Clearing uses `null`; unknown slots and incompatible API kinds are + rejected. An update must not overwrite another slot inadvertently. + +Defaults use the existing model-resource policy: members can inspect them; +Operators and Admins can configure them through Platform. Direct runtime keys +continue to use method-group authority. Define these rules in the method +manifest and regenerate its consumers. + +A missing default fails with a typed `model_default_unset` error identifying +the slot, but only when the request needs that default. A valid explicit model +does not require a configured default. Invalid or unavailable selections never +silently switch to another provider or model. + +### 2. Resolve agent defaults at session creation + +New-session precedence is: + +1. Explicit session model. +2. Model supplied by the selected profile. +3. Universe `agentRun` default. + +Persist the resulting complete model selection in the session. Run precedence +remains explicit run override, then stored session model. Changing or clearing +a universe default does not change an existing session or an admitted run. +Provider identity and API kind remain pinned for the session's lifetime. + +Use the same resolution boundary for web, CLI, bot, and sub-agent creation. +Preserve established profile and inheritance rules around that boundary. +Reopening an existing session recovers its stored configuration before reading +mutable defaults. Clones and forks retain the configuration they already copy +or inherit; they do not acquire a new default as a side effect. + +For existing sessions, an omitted model in `session/config/put` or an applied +profile means **preserve the current model**. Make this an explicit contract +rule and resolve it against the current configuration revision. The remainder +of configuration replacement keeps its existing semantics, including revoking +omitted features. An explicit incompatible provider route remains an error; +bot profile reconciliation retains its existing session-rotation behavior. + +Remove `LIGHTSPEED_CHAT_PROVIDER` and `LIGHTSPEED_CHAT_MODEL` as runtime model +fallbacks. Setup and a deliberate upgrade/import step may write the old +effective selection into specific universe records. Do not populate every +universe with an assumed OpenAI route or overwrite configured values during +startup. Retired variables should produce an actionable configuration message. +Existing deployment credential fallbacks are a separate concern and retain +their current provider-resolution policy. + +The development launcher explicitly seeds defaults for its development +universes through the API, without resetting subsequent user choices. The CLI +must support reading, setting, and clearing defaults so setup stays usable +without Platform. Adding a provider in the web UI offers selection of a +default; adding a credential alone does not silently choose a model. + +### 3. Separate purpose, protocol, and provider + +Purpose names the use within Lightspeed. API kind names the protocol spoken by +an adapter. Provider ID names an endpoint and its authentication configuration. +Several purposes may use the same protocol and still have different defaults. + +Keep broader route selection and purpose validation outside the deterministic +engine. The engine's model types continue to describe agent generation only. +Share provider resolution and route validation where useful; provider requests +and responses stay native to their adapters. + +The first transcription protocol is the existing +`openai:audio-transcriptions`. Support both the built-in provider and configured +compatible endpoints, including explicit credentialless endpoints. Extend +endpoint validation, the resolver's API-kind check, and the audio client's +per-request transport support together. A custom provider must never fall back +to the built-in endpoint or credential when its configuration is missing or +unusable. FFmpeg remains an optional adapter. + +Add a purpose filter to model discovery and use it in the relevant pickers. +An endpoint's protocol declaration does not establish that every model it lists +supports that protocol. Use provider metadata or conservative suggestions and +retain manual model entry. Discovery failure is distinct from missing +configuration and must not prevent saving an otherwise valid explicit route. +Transition existing `selectableOnly` consumers deliberately. + +### 4. One standalone transcription workflow + +Expose a universe-scoped API independent of sessions: + +| Method | Behavior | +| --- | --- | +| `transcriptions/start` | Accepts an audio CAS reference with MIME/name, optional model, language/prompt options, and an idempotency key; returns a transcription ID, resolved model, and status | +| `transcriptions/read` | Returns pending/running/succeeded/failed/cancelled status and, on success, transcript reference and text | +| `transcriptions/cancel` | Requests cancellation of an unfinished job; repeated requests are safe | + +The gateway admits the request and starts a `TranscriptionWorkflow`. The +workflow owns validation, optional transcoding, provider execution, result +storage, and cancellation. Activities perform all provider, filesystem, and +store I/O. Keep its dependencies and registration separate from session +admission. + +Initially host it in the existing server deployment and sessions worker role +and queue. This is a hosting choice, not session-workflow ownership. Channel +controllers start/join the transcription workflow through the workflow +boundary; they do not dispatch its activities onto another role's queue. +A new worker role or independent capacity can be introduced when operationally +needed, without changing the public operation. + +Use bounded provider attempts and an overall deadline, with activity and HTTP +timeouts aligned. Classify transient provider failures for durable retry; +configuration, authentication, unsupported input, and other terminal failures +return typed outcomes. Preserve the existing byte/duration admission limits +and optional transcoder behavior, with provider-specific format support in +the adapter. Cancellation must reach in-flight activities; terminal completion +and cancellation races must converge on one recorded outcome. + +### 5. Explicit request identity and result lifetime + +Persist the admitted request identity, attribution, immutable resolved model, +and input/result references in a small transcription record. Temporal owns the +execution lifecycle. This record supports recovery, API lookup, and CAS +retention; it is not a generic job framework or a transcript cache. + +Within the caller's universe and ownership scope, an idempotency key identifies +one request. A matching retry rejoins that job; conflicting input or options +produce a conflict. Look up the original admission before resolving current +defaults. Fingerprint the submitted request, including whether its model was +omitted, rather than recomputing its identity from today's default. + +Persist the resolved route once. Credentials and endpoint configuration are +resolved through the provider record at execution time, following the same +policy as generation. Changes to defaults cannot select another model during +retry. A recoverable workflow-start boundary must handle crashes between +record creation and Temporal start. + +Do not use an audio-content hash as the public job identity. Explicitly +repeating a transcription can be intentional, and identical audio may belong +to different callers. Reuse an already persisted result on retry; do not claim +exactly-once upstream billing after an ambiguous provider response. + +Audio and transcript content live in CAS; workflow payloads carry bounded +metadata and references. A transcript artifact records text, source audio, +resolved model, and relevant options. Retain provider-native response material +separately if needed, without embedding credentials or transport headers. + +Active jobs root their required blobs. Completed jobs retain results and their +idempotency records for an explicit bounded interval, exposed through expiry +metadata. Define and test that interval before delivery. Expired results must +be distinguishable from provider failure. Once a transcript is admitted to a +session, session retention roots its content and source-audio provenance +independently of job expiry. Extend CAS reference traversal accordingly. + +Add a dedicated `transcriptions` method group. Platform Contributors can start +jobs; reads and cancellation respect requester ownership and administrator +access, including audio/result download paths. Dictation drafts are not exposed +as a universe-wide job listing. Preserve the existing distinction between +Platform person-level access and direct universe-key authority; a CAS hash is +not a substitute for an access check. + +### 6. Prepared input is the session boundary + +The target session APIs accept ordinary text or an explicit transcript +reference for transcribed speech. Add `InputItem::Transcript { transcriptRef }` +to materialize the existing transcript context representation and source-audio +provenance. Validate and retain the artifact at admission. The engine receives +prepared content and references and performs no transcription orchestration. + +After callers migrate, raw audio `Media` is rejected consistently by run +start, context append, and steering, with an actionable typed error directing +callers to transcription. Context append retains its per-entry failure +semantics. Existing transcript history remains readable. Native audio model +input, if added later, will be a separate explicit capability. + +Do not build a general ingress coordinator merely to preserve the old +single-call audio convenience. Audit existing clients before cutover. If a +specific compatibility requirement emerges, document its scope and retirement +separately rather than making it part of the target architecture. + +For Telegram and WhatsApp, the conversation workflow starts transcription after +channel authorization and media download/upload, before emitting the bot +event. Retain the original message and attach the transcript explicitly so +downstream routing and filters can inspect its words. Spoken text must not +automatically become a channel-management command. + +The conversation controller owns ordering, failure reporting, cancellation, +and delivery retries. Stable inbound identity recovers the same transcription +job. A retry after transcription succeeds reuses the result and does not emit +a duplicate event or run. Preserve the conversation's ordering contract when +audio and text arrive together. Connectors stay transport-only and acquire no +model credentials or routing authority. + +### 7. Web dictation edits a draft + +The interaction is: + +```text +Record → stop → upload/transcribe → review and edit → send +``` + +The microphone uses browser-supported recording formats selected by feature +detection. Recording and transcription never start, queue, or steer a run. +Successful transcription inserts text into the composer while preserving +existing draft content and edits made during the request. Late completion +after cancellation, navigation, draft submission, or universe/session change +must not restore or modify a stale draft. Release microphone resources on all +exit paths; failures preserve the user's existing text and support retry. + +The ordinary send controls determine whether reviewed text starts a run, +queues it, or steers it. It is sent as text, without representing the edited +words as the original audio transcript. Keep temporary job retention separate +from the session's history. + +Show transcription availability and an actionable reason when unavailable. +Readiness follows the selected transcription route, including providers that +need no credential. For new sessions, check the effective explicit/profile/ +universe selection; for existing sessions, check their stored route. A key on +some other provider does not make the selected route ready. Unknown provider +health remains distinct from invalid configuration. + +Add matching demo routes, fixtures, and composer behavior so the browser demo +exercises the same draft flow without requiring a real microphone or provider. + +## Migration and rollout + +Introduce defaults and the standalone operation before removing old audio +admission. Migrate first-party channel and web consumers to prepared input, +then cut over the raw-audio API contract. CLI and direct-client migration +guidance must describe the upload, transcription, and submission steps. + +Treat removal of the old workflow activity calls as a Temporal compatibility +change. Inventory existing histories and pending admissions, then use a +supported workflow-versioning or drain/transition strategy with replay +coverage. Keep historical decoding and any required legacy activity handlers +until that transition is complete. Deleting PostgreSQL data is not a rollout +strategy and does not resolve Temporal histories. + +Existing sessions keep their persisted models. Upgrade tooling imports model +defaults only for explicitly selected universes. Fresh universes remain +unconfigured until setup supplies their choices. Document failures caused by +unset defaults and retired environment variables before release. + +Regenerate public API artifacts, TypeScript clients, method roles, and the +workflow integration contract when changed. Keep schema revision metadata +aligned. Update affected product/development documentation with user review; +this roadmap does not authorize unrelated documentation or root README edits. + +## Implementation progress + +- [x] Add revisioned universe model defaults, APIs, typed failures, purpose + validation, and CLI read/set/clear commands. +- [x] Apply creation-time default resolution across session entry points; + preserve omitted models on existing-session changes and retain route pins. +- [x] Remove environment model fallbacks; support explicit universe setup with + the CLI and seed untouched development universes through the API. +- [x] Add Platform Models settings, per-selection configuration diagnostics, + and effective-route readiness. +- [ ] Extend provider configuration/resolution and the audio client for + compatible transcription endpoints, including credentialless transport. +- [ ] Add the transcription record, workflow, start/read/cancel APIs, request + deduplication, access rules, and CAS lifetime handling. +- [ ] Add transcript artifacts and prepared transcript input with provenance. +- [ ] Add web dictation with draft preview, cancellation, and demo coverage. +- [ ] Move channel voice-message preparation before bot-event delivery and + validate ordering, failure handling, and retries. +- [ ] Audit raw-audio clients, implement the workflow-history transition, and + remove session preprocessing from new execution paths. +- [ ] Regenerate affected contracts, complete scoped checks, and record + migration and validation results here. + +### First slice + +Migration `010_model_defaults.sql` adds one row per configured universe, with +an optimistic revision and independent nullable selections. An untouched +universe reads as revision zero. Setting or clearing either slot advances the +shared revision, and stale writes fail without changing either selection. +Clearing retains the row so development setup cannot undo an intentional +clear. No model policy is inferred or backfilled by migration or server startup. + +The read/put APIs use the model method group's existing authority, with read +and configure-resource actions respectively. Reads return persisted selections +and the revision; the web combines these with separate provider diagnostics. +The `speechToText` slot can be configured now, but the existing transcription +path will begin consuming it only when standalone transcription is implemented. + +Creation merges explicit and profile configuration before consulting the +universe default. An explicit or profile model bypasses the defaults lookup. +A missing required default returns `model_default_unset` with +`modelDefaultSlot`. Existing-session replacement and profile application resolve +omitted models from current session state and guard the resulting write with +that configuration revision. The engine's deterministic behavior is unchanged. + +The CLI exposes: + +```bash +lightspeed model defaults read --json +lightspeed model defaults set agent-run \ + --provider --api-kind --model +lightspeed model defaults clear agent-run +``` + +Use `speech-to-text` for the other slot. Set and clear accept +`--expected-revision`; otherwise the CLI reads the current revision once before +writing. A conflict is reported without automatically retrying over a newer +choice. `--json` returns the defaults record for each command. + +Before starting an upgraded runtime, apply schema revision 10 with +`cargo run -p temporal-server -- migrate`. Remove `LIGHTSPEED_CHAT_PROVIDER` +and `LIGHTSPEED_CHAT_MODEL`; their presence now produces an actionable startup +error. Select each universe explicitly in the CLI and set its intended route. +Existing sessions remain usable with their stored models, and explicit-model +creation remains available before a default is configured. + +The development launcher seeds its development universe and, when enabled, the +Platform Test universe after readiness. This fixture chooses the existing +OpenAI development route only at revision zero. Subsequent settings or clears +are preserved, including concurrent changes. Runtime startup itself never +chooses a default. Live-test fixtures now seed their model policy explicitly. + +Validation covers API wire schemas, protocol/purpose checks, model precedence, +omission semantics, CLI round trips and conflicts, launcher behavior, generated +TypeScript consumers, and the runtime library suite. PostgreSQL tests use +temporary isolated schemas to check migrations, concurrent first writes, +universe isolation, slot preservation, clears, stale revisions, and deletion. +Public API contracts and TypeScript consumers were regenerated; the workflow +contract exporter produced no contract change. + +Follow-up live validation ran seven selected Temporal tests serially against +temporary local databases and a separate test universe. All passed: the fake +session lifecycle, OpenAI default-backed session execution, OpenAI Responses +and Chat Completions tool round trips, an Anthropic Messages tool round trip, +and both profile integration tests. OpenAI calls used `gpt-6-sol`, the model +now explicitly seeded by the development launcher. Each temporary database +was migrated to revision 10 and removed after testing. + +The OpenAI session test now verifies creation from the universe default, +clearing through the API, the typed missing-default failure for new sessions, +and preservation of the original model across reopening, configuration +replacement, and profile application before a real model call. + +OpenAI rejected `gpt-6-sol` function tools with its default reasoning setting +on Chat Completions. That protocol's tool fixture now explicitly sets +`reasoning_effort: "none"`, and failure diagnostics report the recorded run +error. Responses passed with its default reasoning settings; production route +selection and generation policy were not changed to hide the provider error. +The slow live suite remains unrun. + +### Web defaults slice + +Setup → Models (`/u/:slug/models`) now places Defaults below Providers. The +Agent runs selection supports choosing a discovered model, entering a manual +provider/API/model route, and explicitly clearing the slot. Operators and +universe admins can edit; contributors and viewers can inspect. Platform +admins retain their admin access. The Platform routes use the member-scoped +runtime client and its method permission gate. + +Edits retain the revision loaded when the dialog opened. A conflict preserves +the draft and requires an explicit reload and review before another write; +there is no automatic overwrite. Discovery failures leave manual selection +available. Adding a provider offers default selection as a separate action. + +New-session creation previews the effective session, profile, or universe +model without copying the preview into the request. An unset universe default +blocks creation only when no model was supplied. Existing-session readiness +uses the session's stored model, including embedded bot sessions. Profile and +bot setup editors label omitted selections as the universe default. + +Readiness checks the selected provider and API, distinguishes missing or +disabled credentials from unknown availability, and accepts credentialless +providers and models absent from discovery. This reports configuration status, +not proof that a future model call will succeed. The speech-to-text slot stays +out of this UI until standalone transcription consumes it. + +The demo uses the same defaults editor, revision checks, and creation policy, +including bot sessions. Clearing a default leaves existing session models +intact. Tests cover API permissions and typed failures, manual selection +during discovery failure, conflicts, provider setup, effective-model previews, +unset defaults, and demo isolation and preservation. The full web, Platform +server, and TypeScript client suites, workspace typechecks, and production and +demo builds pass. The Models page was also visually checked in the browser. + +## Acceptance and validation + +Use offline tests with fake provider transports for the default validation +loop. Add focused Temporal/PostgreSQL integration coverage where storage or +workflow boundaries require it; live/credentialed suites remain explicit. + +- Defaults are universe-isolated and revision-safe. Explicit and profile + models take precedence; unset defaults only reject requests that need them. + Updating defaults leaves existing sessions and admitted jobs unchanged. +- Session creation, reopening, profile application, configuration replacement, + clones/forks, bot rotation, and sub-agent creation preserve the documented + model semantics. Include engine replay coverage if deterministic behavior + or event handling changes. +- Custom transcription routes use their configured endpoint, model, headers, + and authentication. Anonymous endpoints work; missing/disabled custom + providers never fall back to OpenAI. Discovery absence does not prohibit + valid manual selection. +- Matching job retries survive default changes and gateway/workflow restarts; + conflicting payloads fail. Exercise the record/start crash boundary, + transient versus terminal failures, deadlines, and cancellation races. +- Active and retained jobs keep blobs alive. Job expiry releases temporary + roots while session-admitted transcripts retain their source audio. Verify + access isolation for job reads and content downloads. +- Channel redelivery and delivery failure after successful transcription do + not repeat admitted work. Voice/text ordering and mixed-media failures are + explicit. Authorization precedes model work, and transcripts are available + to downstream routing without invoking channel-control commands. +- Web recording/transcription never sends a message automatically. Test + preserved edits, cancellation, late completion after send/navigation, + microphone cleanup, permission/format failures, and unavailable defaults. +- Run start, context append, and steering consistently enforce prepared input. + Old transcript history remains readable and recorded workflow histories + remain replayable across the selected rollout strategy. + +## Scope boundary + +This delivery implements agent defaults and batch speech-to-text. Image +generation, realtime transcription, diarization, automatic model failover, +cross-request transcript caching, and a general processing-pipeline framework +remain outside it. Shared abstractions cover model selection and provider +resolution; new capabilities get typed operations as their requirements arise. diff --git a/platform/configurator-mcp/src/generated/tools.ts b/platform/configurator-mcp/src/generated/tools.ts index 91f5b5063..6ea2d5906 100644 --- a/platform/configurator-mcp/src/generated/tools.ts +++ b/platform/configurator-mcp/src/generated/tools.ts @@ -772,7 +772,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -1232,7 +1232,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "method": "session/config/put", "group": "session", "summary": "Replace session configuration", - "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", "paramsType": "SessionConfigPutParams", "resultType": "AgentApiOutcome", "inputSchema": { @@ -1756,7 +1756,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -3971,7 +3971,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -5033,6 +5033,94 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "object" } }, + { + "name": "lightspeed_models_defaults_read", + "method": "models/defaults/read", + "group": "models", + "summary": "Read universe model defaults", + "description": "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers.", + "paramsType": "ModelDefaultsReadParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "type": "object" + } + }, + { + "name": "lightspeed_models_defaults_put", + "method": "models/defaults/put", + "group": "models", + "summary": "Set a universe model default", + "description": "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery.", + "paramsType": "ModelDefaultsPutParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "properties": { + "expectedRevision": { + "description": "Revision returned by read/put; zero for a universe with no updates.", + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ], + "description": "Complete selection, or explicit null to clear this slot. Required." + }, + "slot": { + "$ref": "#/definitions/ModelDefaultSlot" + } + }, + "required": [ + "slot", + "model", + "expectedRevision" + ], + "type": "object", + "definitions": { + "ModelConfig": { + "properties": { + "apiKind": { + "type": "string" + }, + "model": { + "type": "string" + }, + "providerId": { + "type": "string" + } + }, + "required": [ + "providerId", + "apiKind", + "model" + ], + "type": "object" + }, + "ModelDefaultSlot": { + "description": "A universe's model selection for a particular use. Protocol and purpose\nare separate: several purposes may use the same provider API.", + "enum": [ + "agentRun", + "speechToText" + ], + "type": "string" + } + } + } + }, { "name": "lightspeed_models_list", "method": "models/list", @@ -5685,7 +5773,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -6703,7 +6791,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" diff --git a/platform/server/src/routes/gateway.ts b/platform/server/src/routes/gateway.ts index c797b8a52..9341d466f 100644 --- a/platform/server/src/routes/gateway.ts +++ b/platform/server/src/routes/gateway.ts @@ -25,7 +25,7 @@ import { type SessionConfig, } from "@lightspeed-ai/agent-client"; import { schema } from "@lightspeed/platform-db"; -import { roleAtLeast, slugify, workspaceCreateSchema } from "@lightspeed/platform-shared"; +import { modelDefaultsPutSchema, roleAtLeast, slugify, workspaceCreateSchema } from "@lightspeed/platform-shared"; import type { AppContext, ApiVariables } from "../context.js"; import { parseBody } from "../http.js"; import { @@ -939,6 +939,26 @@ export function gatewayRoutes(ctx: AppContext) { }); }); + app.get("/:id/models/defaults", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("models/defaults/read", {}); + return c.json(response.result.defaults); + }); + }); + + app.put("/:id/models/defaults", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + const body = await parseBody(c, modelDefaultsPutSchema); + if (!body.ok) return body.response; + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("models/defaults/put", body.data); + return c.json(response.result.defaults); + }); + }); + /// Provider-discovered model routes for the session-config model picker. /// Lightspeed owns credential injection and sanitizes per-provider errors. app.get("/:id/models", async (c) => { @@ -2103,6 +2123,7 @@ export async function withGateway( const code = (error as { cause?: { code?: string } } | null)?.cause?.code; if (code === "23505") return c.json({ error: "record already exists" }, 409); if (error instanceof LightspeedRpcError) { + if (error.kind === "model_default_unset") return c.json({ error: error.message, kind: error.kind, modelDefaultSlot: error.data?.modelDefaultSlot }, 400); if (error.kind === "invalid_request") return c.json({ error: error.message }, 400); // Core refusing the Platform means its own key or gate is wrong: the // member was already admitted here. diff --git a/platform/server/src/routes/method-roles.ts b/platform/server/src/routes/method-roles.ts index 422587325..020646fc4 100644 --- a/platform/server/src/routes/method-roles.ts +++ b/platform/server/src/routes/method-roles.ts @@ -78,6 +78,8 @@ export const METHOD_ROLES: Readonly> = { "mcp/servers/put": "operator", "mcp/servers/read": "viewer", "mcp/servers/tools/discover": "operator", + "models/defaults/put": "operator", + "models/defaults/read": "viewer", "models/list": "viewer", "profiles/create": "operator", "profiles/delete": "operator", diff --git a/platform/server/src/routes/model-defaults.test.ts b/platform/server/src/routes/model-defaults.test.ts new file mode 100644 index 000000000..c3da70a12 --- /dev/null +++ b/platform/server/src/routes/model-defaults.test.ts @@ -0,0 +1,75 @@ +import { Hono } from "hono"; +import { afterEach, beforeEach, expect, it, vi } from "vitest"; +import type { ModelDefaults } from "@lightspeed-ai/agent-client"; +import type { ApiVariables, AppContext } from "../context.js"; +import { gatewayRoutes, withGateway } from "./gateway.js"; +import { LightspeedRpcError } from "@lightspeed-ai/agent-client"; + +const auth = vi.hoisted(() => ({ role: "operator" })); +vi.mock("./universes.js", () => ({ universeForSession: vi.fn(async (_ctx, _c, id: string) => ({ + universe: { lightspeedUniverseId: id, gatewayUrl: "https://engine.example/rpc" }, + slug: "test", role: auth.role, member: { userId: "member", role: auth.role }, +})) })); +beforeEach(() => { auth.role = "operator"; }); +afterEach(() => vi.unstubAllGlobals()); + +function fixture() { + const requests: { method: string; params: Record }[] = []; + let defaults: ModelDefaults = { revision: 2, agentRun: null, speechToText: null }; + const fetch = vi.fn(async (_url: unknown, init: RequestInit) => { + expect(new Headers(init.headers).get("x-lightspeed-universe")).toBe("universe"); + expect(new Headers(init.headers).get("x-lightspeed-actor")).toBe("member"); + const rpc = JSON.parse(String(init.body)); + requests.push(rpc); + if (rpc.method === "models/defaults/put") { + if (rpc.params.expectedRevision !== defaults.revision) return Response.json({ id: rpc.id, + error: { code: -32009, message: "defaults changed", data: { kind: "conflict", message: "defaults changed" } } }); + defaults = { ...defaults, [rpc.params.slot]: rpc.params.model, revision: defaults.revision + 1 }; + } + return Response.json({ id: rpc.id, result: { result: { defaults }, notifications: [] } }); + }); + vi.stubGlobal("fetch", fetch); + const app = new Hono<{ Variables: ApiVariables }>(); + app.use("*", async (c, next) => { c.set("session", { user: { id: "member" } } as ApiVariables["session"]); await next(); }); + app.route("/", gatewayRoutes({ env: { lightspeedApiUrl: "https://engine.example/rpc", lightspeedApiKey: "lsk_fixture" } } as AppContext)); + const call = (method = "GET", body?: unknown) => app.request("/universe/models/defaults", { + method, headers: { "content-type": "application/json" }, ...(body === undefined ? {} : { body: JSON.stringify(body) }), + }); + return { app, call, requests, fetch }; +} + +it("reads and writes defaults through the member-scoped runtime client", async () => { + const f = fixture(); + expect(await (await f.call()).json()).toMatchObject({ revision: 2, agentRun: null }); + const model = { providerId: "private", apiKind: "openai:completions", model: "unlisted" }; + expect(await (await f.call("PUT", { slot: "agentRun", model, expectedRevision: 2 })).json()).toMatchObject({ revision: 3, agentRun: model }); + expect((await f.call("PUT", { slot: "agentRun", model: null, expectedRevision: 2 })).status).toBe(409); + expect(await (await f.call("PUT", { slot: "agentRun", model: null, expectedRevision: 3 })).json()).toMatchObject({ revision: 4, agentRun: null }); + expect(f.requests.map((request) => request.method)).toEqual(["models/defaults/read", "models/defaults/put", "models/defaults/put", "models/defaults/put"]); +}); + +it.each(["viewer", "contributor"])("allows %s to inspect but not configure defaults", async (role) => { + auth.role = role; + const f = fixture(); + expect((await f.call()).status).toBe(200); + expect((await f.call("PUT", { slot: "agentRun", model: null, expectedRevision: 2 })).status).toBe(403); + expect(f.fetch).toHaveBeenCalledTimes(1); +}); + +it.each([ + { slot: "agentRun", expectedRevision: 2 }, + { slot: "agentRun", model: null }, + { slot: "agentRun", model: { providerId: "openai", apiKind: "openai:audio-transcriptions", model: "speech" }, expectedRevision: 2 }, +])("rejects incomplete or incompatible writes before sending them upstream", async (body) => { + const f = fixture(); + expect((await f.call("PUT", body)).status).toBe(400); + expect(f.fetch).not.toHaveBeenCalled(); +}); + +it("preserves the typed missing-default error for web clients", async () => { + const app = new Hono(); + app.get("/", (c) => withGateway(c, async () => { throw new LightspeedRpcError({ code: -32014, message: "Choose a model", data: { kind: "model_default_unset", message: "Choose a model", modelDefaultSlot: "agentRun" } }); })); + const response = await app.request("/"); + expect(response.status).toBe(400); + expect(await response.json()).toMatchObject({ kind: "model_default_unset", modelDefaultSlot: "agentRun" }); +}); diff --git a/platform/shared/src/index.ts b/platform/shared/src/index.ts index bc3c6c3de..819af6589 100644 --- a/platform/shared/src/index.ts +++ b/platform/shared/src/index.ts @@ -136,3 +136,4 @@ export function mergeFeatureOverrides(stored: unknown, changes: FeatureOverrides } return merged; } +export { AGENT_MODEL_API_KINDS, modelDefaultsPutSchema } from "./model-defaults.js"; diff --git a/platform/shared/src/model-defaults.ts b/platform/shared/src/model-defaults.ts new file mode 100644 index 000000000..09e8a2981 --- /dev/null +++ b/platform/shared/src/model-defaults.ts @@ -0,0 +1,17 @@ +import { z } from "zod"; + +export const AGENT_MODEL_API_KINDS = ["openai:responses", "openai:completions", "anthropic:messages"] as const; +const routeName = z.string().min(1).refine( + (value) => value.trim() === value && new TextEncoder().encode(value).length <= 512, + "Use 1–512 bytes without surrounding whitespace.", +); + +export const modelDefaultsPutSchema = z.object({ + slot: z.enum(["agentRun", "speechToText"]), + model: z.object({ providerId: routeName, apiKind: z.string(), model: routeName }).strict().nullable(), + expectedRevision: z.number().int().min(0).max(Number.MAX_SAFE_INTEGER), +}).strict().refine(({ slot, model }) => !model || (slot === "agentRun" + ? (AGENT_MODEL_API_KINDS as readonly string[]).includes(model.apiKind) + : model.apiKind === "openai:audio-transcriptions"), { + message: "The selected API does not support this use.", path: ["model", "apiKind"], +}); diff --git a/platform/web/src/api.ts b/platform/web/src/api.ts index 013e9db17..6fe988965 100644 --- a/platform/web/src/api.ts +++ b/platform/web/src/api.ts @@ -1,4 +1,5 @@ import type { FeatureStates, UniverseRole } from "@lightspeed/platform-shared"; +export type { ModelConfig, ModelDefaults, ModelDefaultsPutParams } from "@lightspeed-ai/agent-client"; import type { Attribution, ResourceAccessSummary, diff --git a/platform/web/src/components/models/add-model-provider-dialog.tsx b/platform/web/src/components/models/add-model-provider-dialog.tsx index cce485aca..25cceab72 100644 --- a/platform/web/src/components/models/add-model-provider-dialog.tsx +++ b/platform/web/src/components/models/add-model-provider-dialog.tsx @@ -32,6 +32,7 @@ export function AddModelProviderDialog({ initialKind = null, onOpenChange, onAdded, + onChooseDefault, }: { universeId: string; open: boolean; @@ -40,6 +41,7 @@ export function AddModelProviderDialog({ initialKind?: ModelProviderKind | null; onOpenChange: (open: boolean) => void; onAdded: () => void; + onChooseDefault?: () => void; }) { const [selected, setSelected] = useState(initialKind); useEffect(() => { @@ -167,6 +169,7 @@ export function AddModelProviderDialog({ )} + {done.type === "modelKey" && onChooseDefault && } diff --git a/platform/web/src/components/models/model-defaults.tsx b/platform/web/src/components/models/model-defaults.tsx new file mode 100644 index 000000000..4ea976ca6 --- /dev/null +++ b/platform/web/src/components/models/model-defaults.tsx @@ -0,0 +1,138 @@ +import { useId, useState } from "react"; +import { useMutation, useQueryClient } from "@tanstack/react-query"; +import { AGENT_MODEL_API_KINDS, modelDefaultsPutSchema } from "@lightspeed/platform-shared"; +import { api, ApiError, type ModelConfig, type ModelDefaults, type ModelDefaultsPutParams } from "@/api"; +import { modelDefaultsKey, modelLabel, useModelDefaults, useModelDiscovery } from "@/lib/model-defaults"; +import { summarizeProviderReadiness } from "@/lib/provider-readiness"; +import { Button } from "@/components/ui/button"; +import { Input } from "@/components/ui/input"; +import { Field, FieldLabel, FieldDescription } from "@/components/ui/field"; +import { Dialog, DialogContent, DialogDescription, DialogFooter, DialogHeader, DialogTitle } from "@/components/ui/dialog"; +import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select"; +import { Combobox, ComboboxContent, ComboboxEmpty, ComboboxInput, ComboboxItem, ComboboxList } from "@/components/ui/combobox"; + +const apiLabels: Record = { + "openai:responses": "OpenAI Responses", + "openai:completions": "OpenAI Chat Completions", + "anthropic:messages": "Anthropic Messages", +}; + +export function ModelDefaultsSection({ universeId, writable, onEdit }: { universeId: string; writable: boolean; onEdit: (defaults: ModelDefaults) => void }) { + const defaults = useModelDefaults(universeId); + const discovery = useModelDiscovery(universeId); + const queryClient = useQueryClient(); + const clear = useMutation({ + mutationFn: (revision: number) => api("PUT", `/api/v1/universes/${universeId}/models/defaults`, { + slot: "agentRun", model: null, expectedRevision: revision, + } satisfies ModelDefaultsPutParams), + onSuccess: (value) => queryClient.setQueryData(modelDefaultsKey(universeId), value), + onError: () => { void queryClient.invalidateQueries({ queryKey: modelDefaultsKey(universeId) }); }, + }); + const model = defaults.data?.agentRun; + const readiness = summarizeProviderReadiness(model, discovery.error ? undefined : discovery.data?.providers); + return ( +
+
+

Defaults

+

Used when a new session and its profile leave the model unset. Existing sessions keep their model.

+
+
+
+
+

Agent runs

+

{defaults.isLoading ? "Loading…" : defaults.error ? "Default unavailable" : model ? modelLabel(model) : "No default selected"}

+ {model &&

{apiLabels[model.apiKind] ?? model.apiKind}

} + {model &&

+ {discovery.isLoading ? "Checking provider…" : readiness.message} +

} +
+ {writable && defaults.data && !defaults.error &&
+ + {model && } +
} +
+ {defaults.error &&
+ Could not load defaults: {defaults.error.message} + +
} + {clear.error &&

{clear.error instanceof ApiError && clear.error.status === 409 ? "Defaults changed elsewhere. Review the current selection before clearing it." : clear.error.message}

} +
+
+ ); +} + +export function DefaultModelDialog({ universeId, initial, onClose }: { universeId: string; initial: ModelDefaults; onClose: () => void }) { + const [revision, setRevision] = useState(initial.revision); + const [model, setModel] = useState(initial.agentRun ?? { providerId: "", apiKind: "openai:responses", model: "" }); + const [reloadError, setReloadError] = useState(null); + const [reloading, setReloading] = useState(false); + const discovery = useModelDiscovery(universeId); + const queryClient = useQueryClient(); + const id = useId(); + const [search, setSearch] = useState(""); + const routes = (discovery.data?.models ?? []).filter((route) => (AGENT_MODEL_API_KINDS as readonly string[]).includes(route.apiKind)); + const choices = routes.map((route) => JSON.stringify([route.providerId, route.apiKind, route.model])); + const routeFor = (key: string) => routes[choices.indexOf(key)]; + const clean = { providerId: model.providerId.trim(), apiKind: model.apiKind, model: model.model.trim() }; + const params: ModelDefaultsPutParams = { slot: "agentRun", model: clean, expectedRevision: revision }; + const valid = modelDefaultsPutSchema.safeParse(params).success; + const save = useMutation({ + mutationFn: () => api("PUT", `/api/v1/universes/${universeId}/models/defaults`, params), + onSuccess: (value) => { queryClient.setQueryData(modelDefaultsKey(universeId), value); onClose(); }, + }); + const conflict = save.error instanceof ApiError && save.error.status === 409; + const reload = async () => { + setReloading(true); + setReloadError(null); + try { + const latest = await api("GET", `/api/v1/universes/${universeId}/models/defaults`); + queryClient.setQueryData(modelDefaultsKey(universeId), latest); + setRevision(latest.revision); + setModel(latest.agentRun ?? { providerId: "", apiKind: "openai:responses", model: "" }); + save.reset(); + } catch (error) { setReloadError(error instanceof Error ? error.message : "Could not reload defaults."); } + finally { setReloading(false); } + }; + return { if (!open && !save.isPending) onClose(); }}> + + + Default model for agent runs + Choose a discovered model or enter your provider and model below. + +
{ event.preventDefault(); if (valid && !conflict && !save.isPending) save.mutate(); }}> +
+ + Find a model + items={choices} value={null} inputValue={search} onInputValueChange={setSearch} + itemToStringLabel={(key) => { const route = routeFor(key); return route ? `${modelLabel(route)} · ${apiLabels[route.apiKind]}` : key; }} + filter={(key, query) => key.toLowerCase().includes(query.toLowerCase())} + onValueChange={(key) => { const route = key ? routeFor(key) : undefined; if (route) { setModel({ providerId: route.providerId, apiKind: route.apiKind, model: route.model }); setSearch(""); } }}> + + No matching models. Enter a model below.{(key: string) => { + const route = routeFor(key); + return {route ? modelLabel(route) : key}{route && apiLabels[route.apiKind]}; + }} + + {discovery.isLoading ? "Loading suggestions…" : discovery.error ? "Suggestions are unavailable. You can still enter and save a model." : "Manual models do not need to appear in discovery."} + +
+ Provider setModel({ ...model, providerId: event.target.value })} /> + {discovery.data?.providers?.map((provider) => + API +
+ Model setModel({ ...model, model: event.target.value })} /> +
+ {save.error &&

{conflict ? "Defaults changed elsewhere. Reload and review the saved default before making another change." : save.error.message}

} + {reloadError &&

{reloadError}

} + + + {conflict ? + : } + +
+
+
; +} diff --git a/platform/web/src/components/provider-readiness-banner.tsx b/platform/web/src/components/provider-readiness-banner.tsx index 0a5638210..49371f7a0 100644 --- a/platform/web/src/components/provider-readiness-banner.tsx +++ b/platform/web/src/components/provider-readiness-banner.tsx @@ -1,38 +1,26 @@ import { useActionPermissions } from "@/lib/permissions"; import { Link } from "react-router-dom"; -import { KeyRound } from "lucide-react"; +import { Info } from "lucide-react"; +import type { ModelConfig } from "@/api"; import { Button } from "@/components/ui/button"; -import { addModelProviderHref, useProviderReadiness } from "@/lib/provider-readiness"; +import { useProviderReadiness } from "@/lib/provider-readiness"; -/// Nudges toward Models when no model provider has a usable credential. -/// Renders nothing while loading or when at least one provider is ready. -export function ProviderReadinessBanner({ - universeId, - slug, - className, -}: { +export function ProviderReadinessBanner({ universeId, slug, model, enabled = true, className }: { universeId: string; slug: string; + /// Omitted means universe policy; an existing session supplies its stored model. + model?: ModelConfig | null; + enabled?: boolean; className?: string; }) { - const readiness = useProviderReadiness(universeId); + const readiness = useProviderReadiness(universeId, model, enabled); const permissions = useActionPermissions(universeId); - if (readiness.isLoading || readiness.ready) return null; - const invalidOnly = readiness.missing.length === 0 && readiness.invalid.length > 0; + if (!enabled || readiness.isLoading || readiness.state === "configured") return null; return ( -
- - - {invalidOnly - ? "The configured model provider key was rejected. Sessions cannot run until a valid model provider API key is set." - : "No model provider is configured for this universe. Sessions cannot run until a model provider API key is added."} - - {permissions.can("configure_resource") && } +
+ + {readiness.message} + {permissions.can("configure_resource") && }
); } diff --git a/platform/web/src/components/session/session-config-editor.tsx b/platform/web/src/components/session/session-config-editor.tsx index 0d3a6c759..a97b4f13d 100644 --- a/platform/web/src/components/session/session-config-editor.tsx +++ b/platform/web/src/components/session/session-config-editor.tsx @@ -95,6 +95,7 @@ type Props = { workspaces?: WorkspaceOption[]; workspacesLoading?: boolean; models?: ModelOption[]; + defaultModelLabel?: string; profiles?: ProfileOption[]; environments?: EnvironmentOption[]; allowInherit?: boolean; @@ -526,6 +527,7 @@ export function SessionConfigEditor({ workspaces = [], workspacesLoading = false, models = [], + defaultModelLabel = "Universe default", profiles = [], environments = [], allowInherit = false, @@ -586,6 +588,7 @@ export function SessionConfigEditor({ void; pinnedApiKind?: string; @@ -799,8 +803,8 @@ function ModelFields({ config, models, manualModel, onManualModelChange, pinnedA const choices: ModelChoice[] = [ ...(!pinnedApiKind ? [{ key: "default", - label: "Deployment default", - search: "deployment default", + label: defaultModelLabel, + search: `universe default ${defaultModelLabel}`, }] : []), ...modelPickerOptions(models, currentModel, pinnedApiKind, pinnedProviderId) .map((option) => ({ diff --git a/platform/web/src/demo/fixtures/personal-assistant.ts b/platform/web/src/demo/fixtures/personal-assistant.ts index f42040bc5..0457155c9 100644 --- a/platform/web/src/demo/fixtures/personal-assistant.ts +++ b/platform/web/src/demo/fixtures/personal-assistant.ts @@ -1188,6 +1188,7 @@ function seedIntegrations(universe: UniverseState): void { const fetchedAtMs = ago(9 * MINUTE_MS); const efforts = ["none", "low", "medium", "high", "xhigh"]; + universe.modelDefaults = { revision: 1, agentRun: { ...OPUS }, speechToText: null }; universe.models = { models: [ modelOption(OPUS, "Claude Opus 5", { maxInputTokens: 1_000_000, maxOutputTokens: 128_000, parallelToolUse: true, reasoningEfforts: [...efforts, "max"] }, fetchedAtMs), diff --git a/platform/web/src/demo/fixtures/software-factory.ts b/platform/web/src/demo/fixtures/software-factory.ts index e5412fb2e..45b13d97a 100644 --- a/platform/web/src/demo/fixtures/software-factory.ts +++ b/platform/web/src/demo/fixtures/software-factory.ts @@ -1511,6 +1511,7 @@ function seedIntegrations(universe: UniverseState): void { const fetchedAtMs = ago(12 * MINUTE_MS); const efforts = ["none", "low", "medium", "high", "xhigh"]; + universe.modelDefaults = { revision: 1, agentRun: { ...OPUS }, speechToText: null }; universe.models = { models: [ modelOption(OPUS, "Claude Opus 5", { maxInputTokens: 1_000_000, maxOutputTokens: 128_000, parallelToolUse: true, reasoningEfforts: [...efforts, "max"] }, fetchedAtMs), diff --git a/platform/web/src/demo/fixtures/technical-support.ts b/platform/web/src/demo/fixtures/technical-support.ts index aca68b67f..b72407991 100644 --- a/platform/web/src/demo/fixtures/technical-support.ts +++ b/platform/web/src/demo/fixtures/technical-support.ts @@ -973,6 +973,7 @@ function seedIntegrations(universe: UniverseState): void { }; const fetchedAtMs = ago(6 * MINUTE_MS); + universe.modelDefaults = { revision: 1, agentRun: { ...OPUS }, speechToText: null }; universe.models = { models: [ modelOption(SONNET, "Claude Sonnet 5", { maxInputTokens: 200_000, maxOutputTokens: 64_000, parallelToolUse: true, reasoningEfforts: ["none", "low", "medium", "high"] }, fetchedAtMs), diff --git a/platform/web/src/demo/router.test.ts b/platform/web/src/demo/router.test.ts index a1187fe2e..a7679edc8 100644 --- a/platform/web/src/demo/router.test.ts +++ b/platform/web/src/demo/router.test.ts @@ -11,6 +11,7 @@ import type { SessionEventsPage, SessionListPage, SessionView, + ModelDefaults, } from "@/api"; import { SOFTWARE_FACTORY_UNIVERSE_ID } from "./fixtures/software-factory"; import { applyEvents, emptyTranscript } from "@/lib/sessions/transcript"; @@ -50,6 +51,7 @@ const universeReads = [ "secrets", "auth-grants", "models", + "models/defaults", "setups", "api-keys", "members", @@ -61,6 +63,67 @@ const universeReads = [ ]; describe("demo router", () => { + it("keeps model defaults universe-scoped and revision-safe while preserving existing session models", async () => { + const { store, call } = await boot(); + const base = `/api/v1/universes/${SOFTWARE_FACTORY_UNIVERSE_ID}`; + const other = [...store.universes.values()].find((state) => state.universe.id !== SOFTWARE_FACTORY_UNIVERSE_ID)!; + const otherDefaults = structuredClone(other.modelDefaults); + const original = (await call("GET", `${base}/models/defaults`)).json as ModelDefaults; + const model = { providerId: "private", apiKind: "openai:responses", model: "manual-model" }; + const updated = await call("PUT", `${base}/models/defaults`, { slot: "agentRun", model, expectedRevision: original.revision }); + expect(updated.status).toBe(200); + const revision = original.revision + 1; + expect(updated.json).toEqual({ ...original, agentRun: model, revision }); + expect((await call("PUT", `${base}/models/defaults`, { slot: "agentRun", model: null, expectedRevision: original.revision })).status).toBe(409); + expect((await call("PUT", `${base}/models/defaults`, { slot: "agentRun", expectedRevision: revision })).status).toBe(400); + const created = await call("POST", `${base}/sessions`, { profile: { kind: "inline", profile: {} } }); + expect(created.status).toBe(200); + const session = created.json as SessionView; + expect(session.config).toMatchObject({ model }); + const cleared = await call("PUT", `${base}/models/defaults`, { slot: "agentRun", model: null, expectedRevision: revision }); + expect(cleared.json).toMatchObject({ revision: revision + 1, agentRun: null }); + const missing = await call("POST", `${base}/sessions`, { profile: { kind: "inline", profile: {} } }); + expect(missing.status).toBe(400); + expect(missing.json).toMatchObject({ kind: "model_default_unset", modelDefaultSlot: "agentRun" }); + const existing = await call("PUT", `${base}/sessions/${session.id}/config`, { config: {}, expectedConfigRevision: session.configRevision }); + expect(existing.status).toBe(200); + expect(store.universe(SOFTWARE_FACTORY_UNIVERSE_ID)!.sessions.get(session.id)!.view.config).toMatchObject({ model }); + const explicit = await call("POST", `${base}/sessions`, { profile: { kind: "inline", profile: { config: { model } } } }); + expect(explicit.status).toBe(200); + expect(other.modelDefaults).toEqual(otherDefaults); + }); + + it("lets demo members read defaults while reserving changes for operators", async () => { + const { store, call } = await boot(); + store.currentUser.role = "user"; + const universe = store.universe(SOFTWARE_FACTORY_UNIVERSE_ID)!; + universe.universe.role = "viewer"; + const path = `/api/v1/universes/${SOFTWARE_FACTORY_UNIVERSE_ID}/models/defaults`; + expect((await call("GET", path)).status).toBe(200); + expect((await call("PUT", path, { slot: "agentRun", model: null, expectedRevision: universe.modelDefaults.revision })).status).toBe(403); + expect(universe.modelDefaults.agentRun).not.toBeNull(); + }); + + it("uses universe defaults for demo bot sessions and refuses rotation when that choice is cleared", async () => { + const { store, call } = await boot(); + const universe = store.universe(SOFTWARE_FACTORY_UNIVERSE_ID)!; + const base = `/api/v1/universes/${SOFTWARE_FACTORY_UNIVERSE_ID}`; + await call("PUT", `${base}/profiles/inherit`, { profileId: "inherit", config: {} }); + const created = await call("POST", `${base}/bots`, { bot: { botId: "inherit-model", profileId: "inherit" } }); + expect(created.status).toBe(201); + const record = universe.bots.get("inherit-model")!; + const sessionId = record.state.mainSessionId!; + const model = universe.modelDefaults.agentRun; + expect(universe.sessions.get(sessionId)!.view.config).toMatchObject({ model }); + await call("PUT", `${base}/models/defaults`, { slot: "agentRun", model: null, expectedRevision: universe.modelDefaults.revision }); + const refused = await call("POST", `${base}/bots/inherit-model/sessions/${sessionId}/rotate`); + expect(refused.status).toBe(400); + expect(refused.json).toMatchObject({ kind: "model_default_unset" }); + expect(universe.sessions.get(sessionId)!.view.status).not.toBe("closed"); + expect((await call("POST", `${base}/bots`, { bot: { botId: "no-model", profileId: "inherit" } })).status).toBe(400); + expect(universe.bots.has("no-model")).toBe(false); + }); + it("carries runtime slugs through creation and adoption without local suffixes", async () => { const { store, call } = await boot(); const created = await call("POST", "/api/v1/universes", { name: "Display", slug: "chosen" }); diff --git a/platform/web/src/demo/routes/bots.ts b/platform/web/src/demo/routes/bots.ts index d84ea6d79..6c8532a15 100644 --- a/platform/web/src/demo/routes/bots.ts +++ b/platform/web/src/demo/routes/bots.ts @@ -27,13 +27,13 @@ import type { } from "@lightspeed-ai/agent-client"; import type { ProfileDocument } from "@/api"; import { - DEFAULT_MODEL, EVENT_ORIGIN, activeRun, applyEntries, closeSession, contextMessage, newSession, + modelOf, startRun, steerRun, demoAccess, @@ -65,6 +65,7 @@ class BotConfigError extends Error { constructor( message: string, readonly status: 400 | 404 | 409 | 410, + readonly modelDefaultSlot?: "agentRun", ) { super(message); this.name = "BotConfigError"; @@ -72,7 +73,7 @@ class BotConfigError extends Error { } function configErrorResponse(c: Context, error: unknown): Response { - if (error instanceof BotConfigError) return c.json({ error: error.message }, error.status); + if (error instanceof BotConfigError) return c.json({ error: error.message, ...(error.modelDefaultSlot ? { kind: "model_default_unset", modelDefaultSlot: error.modelDefaultSlot } : {}) }, error.status); throw error; } @@ -169,9 +170,15 @@ function profileInstructions(profile: ProfileDocument | undefined): string | nul /// A session the bot's controller owns: managed, lifecycle-bound to the /// controller, configured from the bot's profile. -function botSession(store: DemoStore, universe: UniverseState, bot: BotView, label: string): SessionRecord { +function botSessionConfig(universe: UniverseState, bot: BotView): Record { + const config = asRecord(universe.profiles.get(bot.profileId)?.config) ?? {}; + const model = modelOf(config) ?? universe.modelDefaults.agentRun; + if (!model) throw new BotConfigError("No default agent model is selected. Choose a model in Models.", 400, "agentRun"); + return { ...structuredClone(config), model: { ...model } }; +} + +function botSession(store: DemoStore, universe: UniverseState, bot: BotView, label: string, config = botSessionConfig(universe, bot)): SessionRecord { const profile = universe.profiles.get(bot.profileId); - const config = asRecord(profile?.config); return newSession(store, universe, { displayName: `${bot.botId} · ${label}`, managed: true, @@ -180,7 +187,7 @@ function botSession(store: DemoStore, universe: UniverseState, bot: BotView, lab lifecycleController: { workflowId: `bot:v1:${bot.botId}`, workflowKind: "bot_controller_v1" }, tools: [], }, - config: { model: { ...DEFAULT_MODEL }, ...config }, + config, instructions: profileInstructions(profile), }); } @@ -724,6 +731,7 @@ function rotateSession( universe: UniverseState, record: BotRecord, managed: BotSessionSnapshot, + config: Record, ): void { const state = record.state; const old = universe.sessions.get(managed.sessionId); @@ -731,7 +739,7 @@ function rotateSession( abandonDeliveries(record, old.view.id, "session rotated"); closeSession(old, true); } - const fresh = botSession(store, universe, record.bot, managed.label); + const fresh = botSession(store, universe, record.bot, managed.label, config); const entry: BotSessionSnapshot = { sessionId: fresh.view.id, label: managed.label, @@ -1038,6 +1046,7 @@ function createBot( export function botRoutes(store: DemoStore): Hono { const app = new Hono(); + app.onError((error, c) => configErrorResponse(c, error)); app.get("/:id/bots", (c) => { const universe = universeFor(store, c); @@ -1141,7 +1150,8 @@ export function botRoutes(store: DemoStore): Hono { const managed = sessionsOf(record.state).find((entry) => entry.sessionId === sessionId); if (!managed) return c.json({ error: "session is not managed by this bot" }, 404); if (record.bot.closedAtMs != null) return conflict(c, "bot is closed"); - setTimeout(() => rotateSession(store, universe, record, managed), ROTATE_DELAY_MS); + const config = botSessionConfig(universe, record.bot); + setTimeout(() => rotateSession(store, universe, record, managed, config), ROTATE_DELAY_MS); return c.json({ accepted: true }); }); @@ -1411,6 +1421,7 @@ function universeByAnyId(store: DemoStore, id: string): UniverseState | null { /// indistinguishable from an unknown endpoint, like the core's route. export function hookRoutes(store: DemoStore): Hono { const app = new Hono(); + app.onError((error, c) => configErrorResponse(c, error)); app.post("/bots/:universeId/:botId/:triggerId/:token", async (c) => { const universe = universeByAnyId(store, c.req.param("universeId") ?? ""); diff --git a/platform/web/src/demo/routes/secrets.ts b/platform/web/src/demo/routes/secrets.ts index ca5f8a8ec..2d7317cd0 100644 --- a/platform/web/src/demo/routes/secrets.ts +++ b/platform/web/src/demo/routes/secrets.ts @@ -3,6 +3,7 @@ /// installation. Secret values are accepted and dropped; only the non-secret /// views the real gateway returns are kept. import { Hono } from "hono"; +import { modelDefaultsPutSchema } from "@lightspeed/platform-shared"; import type { GitHubApp, GitHubInstallation, @@ -272,6 +273,9 @@ function credentialStatus( provider: Pick, ): Pick { const secret = findModelProvider(universe, credentialIdFor(provider.providerId)); + if (secret && (secret.status !== "active" || !secret.usableForModels)) { + return { credential: "invalid", credentialSource: "universe", error: "Provider is disabled or unusable." }; + } if (secret?.status === "active" && secret.hasCredential) { return { credential: "configured", credentialSource: "universe", error: null }; } @@ -738,6 +742,22 @@ export function secretRoutes(store: DemoStore): Hono { return c.json(modelDiscovery(universe)); }); + app.get("/:id/models/defaults", (c) => { + const universe = universeFor(store, c); + return universe ? c.json(universe.modelDefaults) : notFound(c); + }); + + app.put("/:id/models/defaults", async (c) => { + const universe = universeFor(store, c); + if (!universe) return notFound(c); + if (store.currentUser.role !== "admin" && !["operator", "admin"].includes(universe.universe.role ?? "")) return c.json({ error: "operator role required" }, 403); + const body = modelDefaultsPutSchema.safeParse(await readBody(c)); + if (!body.success) return badRequest(c, body.error.issues[0]?.message ?? "Invalid defaults update"); + if (body.data.expectedRevision !== universe.modelDefaults.revision) return conflict(c, "Model defaults changed elsewhere."); + universe.modelDefaults = { ...universe.modelDefaults, [body.data.slot]: body.data.model, revision: universe.modelDefaults.revision + 1 }; + return c.json(universe.modelDefaults); + }); + /// Operators see the templates; installing stays with Admins. app.get("/:id/setups", (c) => { const universe = universeFor(store, c); diff --git a/platform/web/src/demo/routes/sessions.ts b/platform/web/src/demo/routes/sessions.ts index 508d93f6f..e75370e72 100644 --- a/platform/web/src/demo/routes/sessions.ts +++ b/platform/web/src/demo/routes/sessions.ts @@ -4,10 +4,9 @@ import { defaultEnvironmentAttachment, environmentAttachments, isEnvironmentAtta /// and status codes follow the platform server's gateway so the UI cannot /// tell the difference. import { Hono, type Context } from "hono"; -import type { Environment, ProfileSessionRetention, ProfileSource, SessionView } from "@/api"; +import type { Environment, ModelConfig, ProfileSessionRetention, ProfileSource, SessionView } from "@/api"; import type { ProfileInstructions } from "@lightspeed-ai/agent-client"; import { - DEFAULT_MODEL, PROFILE_INSTRUCTIONS_KEY, cancelRun, closeSession, @@ -110,7 +109,8 @@ export function sessionRoutes(store: DemoStore): Hono { if (!body.profile) return badRequest(c, "profile is required"); const profile = resolveProfile(universe, body.profile); if (!profile) return notFound(c, "not found in engine"); - const config = sessionConfig(profile.config); + const config = sessionConfig(profile.config, universe.modelDefaults.agentRun); + if (!config) return c.json({ error: "No default agent model is selected. Choose a model in Models.", kind: "model_default_unset", modelDefaultSlot: "agentRun" }, 400); const sessionId = store.nextId("session"); const resolved = resolveEnvironment(universe, profile); if ("error" in resolved) return conflict(c, `engine conflict: ${resolved.error}`); @@ -283,7 +283,8 @@ export function sessionRoutes(store: DemoStore): Hono { `engine conflict: expected config revision ${body.expectedConfigRevision}, got ${session.view.configRevision}`, ); } - const config = sessionConfig(body.config); + const config = sessionConfig(body.config, modelOf(session.view.config)); + if (!config) return badRequest(c, "Session has no model."); session.view.config = config; if (session.view.activeEnvironmentId && !isEnvironmentAttached(config, session.view.activeEnvironmentId)) session.view.activeEnvironmentId = null; session.view.configRevision += 1; @@ -488,9 +489,10 @@ function resolveProfile(universe: UniverseState, source: ProfileSource): Resolve /// The session's own copy of a profile config, with the model the demo /// answers as when the profile leaves it open. -function sessionConfig(config: Record): Record { +function sessionConfig(config: Record, fallback?: ModelConfig | null): Record | null { const copy = structuredClone(config); - return { ...copy, model: modelOf(copy) ?? { ...DEFAULT_MODEL } }; + const model = modelOf(copy) ?? fallback; + return model ? { ...copy, model: { ...model } } : null; } function instructionText(store: DemoStore, instructions: ProfileInstructions | null): string | null { diff --git a/platform/web/src/demo/store.ts b/platform/web/src/demo/store.ts index 7ae924eba..22bde5bd4 100644 --- a/platform/web/src/demo/store.ts +++ b/platform/web/src/demo/store.ts @@ -15,6 +15,7 @@ import type { McpServer, Member, ModelListResponse, + ModelDefaults, ProfileDocument, SecretsInventory, SessionSummary, @@ -148,6 +149,7 @@ export interface UniverseState { secrets: SecretsInventory; githubApps: GitHubApp[]; models: ModelListResponse; + modelDefaults: ModelDefaults; setups: UniverseSetup[]; bots: Map; /// Universe channel accounts (core wire shape), keyed by `accountId`. @@ -260,6 +262,7 @@ export class DemoStore { secrets: { providers: [], grants: [] }, githubApps: [], models: { models: [], providers: [] }, + modelDefaults: { revision: 0, agentRun: null, speechToText: null }, setups: [], bots: new Map(), channelAccounts: new Map(), diff --git a/platform/web/src/lib/model-defaults.ts b/platform/web/src/lib/model-defaults.ts new file mode 100644 index 000000000..c3a2c708b --- /dev/null +++ b/platform/web/src/lib/model-defaults.ts @@ -0,0 +1,45 @@ +import { useQuery } from "@tanstack/react-query"; +import { api, type ModelConfig, type ModelDefaults, type ModelListResponse } from "@/api"; + +export const modelDefaultsKey = (universeId: string) => ["model-defaults", universeId] as const; + +export function useModelDefaults(universeId: string, enabled = true) { + return useQuery({ + queryKey: modelDefaultsKey(universeId), + queryFn: () => api("GET", `/api/v1/universes/${universeId}/models/defaults`), + enabled, + }); +} + +export function useModelDiscovery(universeId: string, enabled = true) { + return useQuery({ + queryKey: ["models", universeId], + queryFn: () => api("GET", `/api/v1/universes/${universeId}/models`), + staleTime: 60_000, + enabled, + }); +} + +export function modelFromConfig(config: unknown): ModelConfig | null { + if (!config || typeof config !== "object") return null; + const model = (config as { model?: unknown }).model; + if (!model || typeof model !== "object") return null; + const route = model as Partial; + return typeof route.providerId === "string" && typeof route.apiKind === "string" && typeof route.model === "string" + ? route as ModelConfig : null; +} + +export function resolveCreationModel( + explicit: ModelConfig | null | undefined, + profile: ModelConfig | null | undefined, + defaults: ModelDefaults | undefined, +): { model: ModelConfig | null; source: "Session model" | "Profile model" | "Universe default" } { + if (explicit) return { model: explicit, source: "Session model" }; + if (profile) return { model: profile, source: "Profile model" }; + return { model: defaults?.agentRun ?? null, source: "Universe default" }; +} + +export function modelLabel(model: ModelConfig) { + const provider = model.providerId === "openai" ? "OpenAI" : model.providerId === "anthropic" ? "Anthropic" : model.providerId; + return `${provider} · ${model.model}`; +} diff --git a/platform/web/src/lib/profile-config-reference.ts b/platform/web/src/lib/profile-config-reference.ts index 289b28993..f5112a79f 100644 --- a/platform/web/src/lib/profile-config-reference.ts +++ b/platform/web/src/lib/profile-config-reference.ts @@ -131,7 +131,7 @@ export const PROFILE_CONFIG_REFERENCE = `// Every field is optional — omit any "maxToolRounds": 0, "maxTurns": 0, }, - // Absent on input means the deployment default model. Documents read back from a session always carry the model. Provider identity and API kind are fixed for the session lifetime; the model name may change. + // At creation, omission uses the profile model or universe agentRun default. On configuration replacement or profile application to an existing session, omission preserves its current model. Documents read back from a session always carry the model. Provider identity and API kind are fixed for the session lifetime; the model name may change. "model": { // (required when this object is present) "apiKind": "string", diff --git a/platform/web/src/lib/provider-readiness.test.ts b/platform/web/src/lib/provider-readiness.test.ts index 6305ca2ef..8ddd6892d 100644 --- a/platform/web/src/lib/provider-readiness.test.ts +++ b/platform/web/src/lib/provider-readiness.test.ts @@ -1,46 +1,44 @@ import { describe, expect, it } from "vitest"; +import type { ModelConfig, ModelProviderDiscovery } from "@/api"; import { addModelProviderHref, summarizeProviderReadiness } from "./provider-readiness"; +import { resolveCreationModel } from "./model-defaults"; -const provider = (providerId: string, credential: "configured" | "missing" | "invalid") => ({ - providerId, - apiKinds: [], - credential, - credentialSource: "none" as const, +const route: ModelConfig = { providerId: "openai", apiKind: "openai:responses", model: "private-model" }; +const provider = (providerId: string, credential: ModelProviderDiscovery["credential"], apiKinds = ["openai:responses"]): ModelProviderDiscovery => ({ + providerId, apiKinds, credential, credentialSource: "none", }); -describe("provider readiness", () => { - it("is ready when any provider has a usable credential", () => { - const summary = summarizeProviderReadiness([ - provider("openai", "missing"), - provider("anthropic", "configured"), - ]); - expect(summary.ready).toBe(true); - expect(summary.missing.map((p) => p.providerId)).toEqual(["openai"]); +describe("selected model readiness", () => { + it("does not use another provider's credential", () => { + expect(summarizeProviderReadiness(route, [provider("openai", "missing"), provider("anthropic", "configured")])) + .toMatchObject({ state: "missing", blocked: true }); }); - - it("is not ready when every provider is missing or invalid", () => { - const summary = summarizeProviderReadiness([ - provider("openai", "missing"), - provider("anthropic", "invalid"), - ]); - expect(summary.ready).toBe(false); - expect(summary.invalid.map((p) => p.providerId)).toEqual(["anthropic"]); + it("accepts credentialless and unlisted manual models", () => { + expect(summarizeProviderReadiness(route, [provider("openai", "notRequired")])) + .toMatchObject({ state: "configured", blocked: false }); }); - - it("does not nag while unknown", () => { - expect(summarizeProviderReadiness(undefined).ready).toBe(true); + it("checks the protocol as well as the provider", () => { + expect(summarizeProviderReadiness(route, [provider("openai", "configured", ["openai:completions"])]).state).toBe("unsupported"); + expect(summarizeProviderReadiness(route, [provider("openai", "invalid")]).state).toBe("invalid"); + expect(summarizeProviderReadiness(route, []).state).toBe("missing"); }); - - it("encodes reserved characters in add-provider kinds", () => { - const href = addModelProviderHref("acme", "custom provider/alpha?x=1&y=2#fragment"); - expect(href).toBe( - "/u/acme/models?add=custom%20provider%2Falpha%3Fx%3D1%26y%3D2%23fragment", - ); + it("distinguishes absent configuration from unknown provider health", () => { + expect(summarizeProviderReadiness(null, undefined)).toMatchObject({ state: "unset", blocked: true }); + expect(summarizeProviderReadiness(undefined, undefined)).toMatchObject({ state: "unknown", blocked: false }); + expect(summarizeProviderReadiness(route, undefined)).toMatchObject({ state: "unknown", blocked: false }); + expect(summarizeProviderReadiness(route, [{ ...provider("openai", "configured"), error: "discovery timed out" }])) + .toMatchObject({ state: "unknown", blocked: false }); }); - - it("builds the add-provider deep link", () => { - expect(addModelProviderHref("acme", "openAiApiKey")).toBe( - "/u/acme/models?add=openAiApiKey", - ); + it("resolves explicit, profile, and universe choices without requiring a default", () => { + const profile = { ...route, providerId: "profile" }; + const defaults = { revision: 1, agentRun: { ...route, providerId: "default" }, speechToText: null }; + expect(resolveCreationModel(route, profile, defaults)).toEqual({ model: route, source: "Session model" }); + expect(resolveCreationModel(null, profile, undefined)).toEqual({ model: profile, source: "Profile model" }); + expect(resolveCreationModel(null, null, defaults)).toEqual({ model: defaults.agentRun, source: "Universe default" }); + expect(resolveCreationModel(null, null, { ...defaults, agentRun: null }).model).toBeNull(); + }); + it("encodes add-provider deep links", () => { + expect(addModelProviderHref("acme", "custom provider/alpha?x=1&y=2#fragment")) + .toBe("/u/acme/models?add=custom%20provider%2Falpha%3Fx%3D1%26y%3D2%23fragment"); }); }); diff --git a/platform/web/src/lib/provider-readiness.ts b/platform/web/src/lib/provider-readiness.ts index 2df3dfa36..4850aa75d 100644 --- a/platform/web/src/lib/provider-readiness.ts +++ b/platform/web/src/lib/provider-readiness.ts @@ -1,41 +1,47 @@ -import { useQuery } from "@tanstack/react-query"; -import { api, type ModelListResponse, type ModelProviderDiscovery } from "@/api"; +import { AGENT_MODEL_API_KINDS } from "@lightspeed/platform-shared"; +import type { ModelConfig, ModelProviderDiscovery } from "@/api"; +import { useModelDefaults, useModelDiscovery } from "./model-defaults"; -/// Whether this universe can run sessions at all: at least one model provider -/// with a usable credential (universe key or deployment fallback). Shares the -/// `["models", universeId]` query with the session editor so it is fetched -/// once and invalidated when keys change. -export interface ProviderReadiness { - isLoading: boolean; - /// True while unknown (loading/error) so callers never nag prematurely. - ready: boolean; - missing: ModelProviderDiscovery[]; - invalid: ModelProviderDiscovery[]; -} +export type ModelReadiness = { + state: "unset" | "configured" | "missing" | "invalid" | "unsupported" | "unknown"; + blocked: boolean; + message: string; +}; +/// Configuration follows the exact provider/API route. Discovery is advisory: +/// an unlisted model or a network failure does not invalidate a manual choice. export function summarizeProviderReadiness( + model: ModelConfig | null | undefined, providers: ModelProviderDiscovery[] | undefined, -): Pick { - if (!providers) return { ready: true, missing: [], invalid: [] }; - return { - ready: providers.some((provider) => provider.credential === "configured"), - missing: providers.filter((provider) => provider.credential === "missing"), - invalid: providers.filter((provider) => provider.credential === "invalid"), - }; +): ModelReadiness { + if (model === undefined) return { state: "unknown", blocked: false, message: "Model settings could not be checked." }; + if (model === null) return { state: "unset", blocked: true, message: "No model selected. Choose a model or set the universe default." }; + if (!model.providerId || !model.model || !(AGENT_MODEL_API_KINDS as readonly string[]).includes(model.apiKind)) { + return { state: "unsupported", blocked: true, message: "This model selection cannot be used for agent runs." }; + } + if (!providers) return { state: "unknown", blocked: false, message: "Provider status could not be checked." }; + const provider = providers.find((entry) => entry.providerId === model.providerId); + if (!provider) return { state: "missing", blocked: true, message: `Provider ${model.providerId} is not configured.` }; + if (provider.credential === "invalid") return { state: "invalid", blocked: true, message: `Provider ${model.providerId} is disabled or its credential is unusable.` }; + if (provider.credential === "missing") return { state: "missing", blocked: true, message: `Provider ${model.providerId} needs a credential.` }; + if (!provider.apiKinds.includes(model.apiKind)) { + return { state: "unsupported", blocked: true, message: `Provider ${model.providerId} does not support ${model.apiKind}.` }; + } + if (provider.error) return { state: "unknown", blocked: false, message: "Provider configured; availability could not be checked." }; + return { state: "configured", blocked: false, message: provider.credential === "notRequired" ? "Provider configured · no credential required" : "Provider configured" }; } -export function useProviderReadiness(universeId: string, enabled = true): ProviderReadiness { - const models = useQuery({ - queryKey: ["models", universeId], - queryFn: () => api("GET", `/api/v1/universes/${universeId}/models`), - staleTime: 60_000, - enabled, - }); - const summary = summarizeProviderReadiness(models.error ? undefined : models.data?.providers); - return { isLoading: models.isLoading, ...summary }; +export function useProviderReadiness(universeId: string, model?: ModelConfig | null, enabled = true) { + const defaults = useModelDefaults(universeId, enabled && model === undefined); + const discovery = useModelDiscovery(universeId, enabled); + const selection = model === undefined ? defaults.data?.agentRun : model; + const readiness = summarizeProviderReadiness(selection, discovery.error ? undefined : discovery.data?.providers); + return { + ...readiness, + isLoading: enabled && (discovery.isLoading || (model === undefined && defaults.isLoading)), + }; } -/// Deep link that opens the Add-model-provider dialog on the given catalog entry. export function addModelProviderHref(slug: string, kind: string): string { return `/u/${slug}/models?add=${encodeURIComponent(kind)}`; } diff --git a/platform/web/src/lib/sessions/editor-options.ts b/platform/web/src/lib/sessions/editor-options.ts index 17d5f58cc..eb0180db1 100644 --- a/platform/web/src/lib/sessions/editor-options.ts +++ b/platform/web/src/lib/sessions/editor-options.ts @@ -1,10 +1,10 @@ import { useActionPermissions } from "@/lib/permissions"; import { useMcpToolDiscoverySource } from "@/lib/mcp/tool-discovery"; import { useQuery } from "@tanstack/react-query"; +import { modelLabel, useModelDefaults, useModelDiscovery } from "@/lib/model-defaults"; import { api, type Environment, - type ModelListResponse, type ProfileSummary, } from "@/api"; import type { @@ -28,12 +28,8 @@ export function useSessionConfigEditorOptions( api("GET", `/api/v1/universes/${universeId}/workspaces`), enabled, }); - const models = useQuery({ - queryKey: ["models", universeId], - queryFn: () => api("GET", `/api/v1/universes/${universeId}/models`), - staleTime: 60_000, - enabled, - }); + const models = useModelDiscovery(universeId, enabled); + const defaults = useModelDefaults(universeId, enabled); const profiles = useQuery({ queryKey: ["profiles", universeId], queryFn: () => api("GET", `/api/v1/universes/${universeId}/profiles`), @@ -51,6 +47,7 @@ export function useSessionConfigEditorOptions( workspaces: workspaces.data, workspacesLoading: workspaces.isLoading, models: models.data?.models, + defaultModelLabel: defaults.data?.agentRun ? `Universe default · ${modelLabel(defaults.data.agentRun)}` : "Universe default", profiles: profiles.data, environments: environments.data, mcpToolDiscovery: permissions.can("configure_resource") ? mcpToolDiscovery : undefined, diff --git a/platform/web/src/pages/BotCreatePage.tsx b/platform/web/src/pages/BotCreatePage.tsx index 7d7cc261a..2f328a797 100644 --- a/platform/web/src/pages/BotCreatePage.tsx +++ b/platform/web/src/pages/BotCreatePage.tsx @@ -32,6 +32,7 @@ import { type TriggerKind, } from "@/components/bot/triggers"; import { ProviderReadinessBanner } from "@/components/provider-readiness-banner"; +import { modelFromConfig } from "@/lib/model-defaults"; import { MetadataMapEditor } from "@/components/session/metadata-editor"; import { ProfileRetentionEditor } from "@/components/session/profile-retention-editor"; import { SessionConfigEditor } from "@/components/session/session-config-editor"; @@ -222,6 +223,11 @@ function Wizard({ queryFn: () => api("GET", `/api/v1/universes/${universeId}/bots`), }); const options = useSessionConfigEditorOptions(universeId, step === "profile" || step === "wakeups"); + const sharedProfile = useQuery({ + queryKey: ["profile", universeId, sharedProfileId], + queryFn: () => api("GET", `/api/v1/universes/${universeId}/profiles/${sharedProfileId}`), + enabled: step === "profile" && setupMode === "shared" && Boolean(sharedProfileId), + }); const defaultEnvironmentId = defaultEnvironmentAttachment(config)?.environmentId; const env: BotEnvStatus = @@ -628,7 +634,9 @@ function Wizard({ profile.

- +
({ api: vi.fn() })); +const mocks = vi.hoisted(() => ({ api: vi.fn(), role: "operator" })); vi.mock("@/api", async (original) => ({ ...await original(), api: mocks.api })); -vi.mock("@/lib/universes", () => ({ useActiveUniverse: () => ({ universe: { id: "universe", role: "operator", slug: "test", name: "Test" }, slug: "test", isLoading: false }) })); +vi.mock("@/lib/universes", () => ({ useActiveUniverse: () => ({ universe: { id: "universe", role: mocks.role, slug: "test", name: "Test" }, slug: "test", isLoading: false }) })); const legacy = { providerId: "anthropic", credentialId: "anthropic", usableForModels: false, providerKind: "modelApiKey", @@ -23,15 +24,33 @@ const subscription = { let root: Root; let container: HTMLDivElement; let client: QueryClient; +let defaults: ModelDefaults; +let actions: string[]; +let discoveryFails: boolean; beforeEach(() => { vi.useFakeTimers(); vi.stubGlobal("IS_REACT_ACT_ENVIRONMENT", true); vi.stubGlobal("PointerEvent", MouseEvent); - mocks.api.mockReset().mockImplementation(async (_method: string, path: string) => { - if (path.endsWith("/access")) return { actions: ["read", "configure_resource"], resources: [] }; + defaults = { revision: 7, agentRun: { providerId: "openai", apiKind: "openai:responses", model: "gpt-6-sol" }, speechToText: null }; + actions = ["read", "configure_resource"]; + discoveryFails = false; + mocks.role = "operator"; + mocks.api.mockReset().mockImplementation(async (method: string, path: string, body?: { expectedRevision: number; model: ModelDefaults["agentRun"] }) => { + if (path.endsWith("/access")) return { actions, resources: [] }; + if (path.endsWith("/models/defaults")) { + if (method === "PUT") { + if (body!.expectedRevision !== defaults.revision) throw new ApiError(409, { error: "defaults changed" }); + defaults = { ...defaults, revision: defaults.revision + 1, agentRun: body!.model }; + } + return defaults; + } if (path.endsWith("/secrets")) return { providers: [legacy], grants: [subscription] }; if (path.endsWith("/integrations/subscriptions")) return [subscription]; - if (path.endsWith("/models")) return { models: [], providers: [{ providerId: "openai", apiKinds: [], credential: "configured", credentialSource: "deployment" }] }; + if (path.endsWith("/integrations/model-keys")) return { ...legacy, providerId: "openai", credentialId: "model:openai", usableForModels: true }; + if (path.endsWith("/models")) { + if (discoveryFails) throw new Error("discovery unavailable"); + return { models: [], providers: [{ providerId: "openai", apiKinds: ["openai:responses"], credential: "configured", credentialSource: "deployment" }] }; + } throw new Error(`Unexpected request: ${path}`); }); client = new QueryClient({ defaultOptions: { queries: { retry: false, gcTime: Infinity } } }); @@ -61,7 +80,7 @@ function button(label: string) { it("lists model providers and subscriptions with the legacy-credential warning", async () => { await show(); expect(container.querySelector("h1")?.textContent).toBe("Models"); - expect(container.textContent).toContain("Operators add them; keys and logins are never shown again."); + expect(container.textContent).toContain("Operators manage them; keys and logins are never shown again."); expect(container.textContent).toContain("A legacy credential below has an incorrect internal ID"); const rows = [...container.querySelectorAll("tbody tr")].map((row) => row.textContent); expect(rows).toHaveLength(2); @@ -71,6 +90,62 @@ it("lists model providers and subscriptions with the legacy-credential warning", expect(rows[1]).toContain("Claude Code (subscription)"); }); +async function click(label: string) { + await act(async () => button(label)!.click()); + for (let step = 0; step < 4; step++) await act(async () => { await vi.advanceTimersByTimeAsync(5); }); +} +async function enterModel(value: string) { + const input = document.body.querySelector('input[placeholder="Model name"]')!; + await act(async () => { + Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, "value")!.set!.call(input, value); + input.dispatchEvent(new Event("input", { bubbles: true })); + }); +} + +it("saves an unlisted model while discovery is unavailable, then explicitly clears it", async () => { + discoveryFails = true; + await show(); + expect(container.textContent).toContain("OpenAI · gpt-6-sol"); + expect(container.textContent).toContain("Provider status could not be checked"); + await click("Change"); + await enterModel("private-model"); + await click("Save default"); + expect(defaults).toMatchObject({ revision: 8, agentRun: { model: "private-model" } }); + expect(container.textContent).toContain("OpenAI · private-model"); + await click("Clear"); + expect(defaults).toMatchObject({ revision: 9, agentRun: null }); + expect(container.textContent).toContain("No default selected"); + const writes = mocks.api.mock.calls.filter(([method]) => method === "PUT"); + expect(writes.map(([, , body]) => body.expectedRevision)).toEqual([7, 8]); + expect(writes[1]?.[2]).toEqual({ slot: "agentRun", model: null, expectedRevision: 8 }); +}); + +it("keeps a stale draft on conflict until the user reloads the saved choice", async () => { + await show(); + await click("Change"); + await enterModel("my-edit"); + defaults = { ...defaults, revision: 8, agentRun: { ...defaults.agentRun!, model: "concurrent-edit" } }; + await click("Save default"); + expect(document.body.textContent).toContain("Defaults changed elsewhere"); + expect(document.body.querySelector('input[placeholder="Model name"]')!.value).toBe("my-edit"); + expect(mocks.api.mock.calls.filter(([method]) => method === "PUT")).toHaveLength(1); + await click("Reload saved default"); + expect(document.body.querySelector('input[placeholder="Model name"]')!.value).toBe("concurrent-edit"); + await enterModel("reviewed-edit"); + await click("Save default"); + expect(defaults).toMatchObject({ revision: 9, agentRun: { model: "reviewed-edit" } }); +}); + +it("shows defaults without edit controls to a read-only member", async () => { + actions = ["read"]; + mocks.role = "viewer"; + await show(); + expect(container.textContent).toContain("OpenAI · gpt-6-sol"); + expect(button("Change")).toBeUndefined(); + expect(button("Clear")).toBeUndefined(); + expect(button("Add provider")).toBeUndefined(); +}); + it("offers model providers only when adding", async () => { await show(); await act(async () => button("Add provider")!.click()); @@ -89,3 +164,20 @@ it("opens the requested provider form from a readiness deep link", async () => { expect(dialog.textContent).toContain("OpenAI (API key)"); expect(dialog.querySelector("input[type=password]")).not.toBeNull(); }); + +it("offers default selection after adding a provider without changing it automatically", async () => { + await show("/?add=openAiApiKey"); + const input = document.body.querySelector('input[type="password"]')!; + await act(async () => { + Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, "value")!.set!.call(input, "test-key"); + input.dispatchEvent(new Event("input", { bubbles: true })); + }); + const form = input.closest("form")!; + await act(async () => form.dispatchEvent(new Event("submit", { bubbles: true, cancelable: true }))); + for (let step = 0; step < 4; step++) await act(async () => { await vi.advanceTimersByTimeAsync(5); }); + expect(button("Choose default model")).toBeDefined(); + expect(mocks.api.mock.calls.filter(([method]) => method === "PUT")).toHaveLength(0); + await click("Choose default model"); + expect(document.body.textContent).toContain("Default model for agent runs"); + expect(document.body.querySelector('input[placeholder="Model name"]')!.value).toBe("gpt-6-sol"); +}); diff --git a/platform/web/src/pages/ModelsPage.tsx b/platform/web/src/pages/ModelsPage.tsx index 7c1192faa..ddb753e79 100644 --- a/platform/web/src/pages/ModelsPage.tsx +++ b/platform/web/src/pages/ModelsPage.tsx @@ -13,7 +13,9 @@ import { type ConnectedModelProvider, } from "@/components/models/use-model-providers"; import { MODEL_PROVIDER_CATALOG, type ModelProviderKind } from "@/components/models/catalog"; -import { ProviderReadinessBanner } from "@/components/provider-readiness-banner"; +import { DefaultModelDialog, ModelDefaultsSection } from "@/components/models/model-defaults"; +import { useModelDefaults } from "@/lib/model-defaults"; +import type { ModelDefaults } from "@/api"; import { useActiveUniverse } from "@/lib/universes"; export function ModelsPage({ admin: _admin }: { admin: boolean }) { @@ -21,10 +23,12 @@ export function ModelsPage({ admin: _admin }: { admin: boolean }) { const permissions = useActionPermissions(universe?.id); if (isLoading) return ; if (!universe || !permissions.can("read")) return ; - return ; + return ; } -function Models({ universeId, slug }: { universeId: string; slug: string }) { +function Models({ universeId }: { universeId: string }) { + const defaults = useModelDefaults(universeId); + const [editingDefaults, setEditingDefaults] = useState(null); const [searchParams, setSearchParams] = useSearchParams(); const writable = useActionPermissions(universeId).can("configure_resource"); const requestedKind = parseModelProviderKind(searchParams.get("add")); @@ -53,7 +57,7 @@ function Models({ universeId, slug }: { universeId: string; slug: string }) { <> setAddOpen(true)}> @@ -61,11 +65,7 @@ function Models({ universeId, slug }: { universeId: string; slug: string }) { )} /> - +

Providers

{legacy && (

A legacy credential below has an incorrect internal ID and is not used for model calls. @@ -75,6 +75,7 @@ function Models({ universeId, slug }: { universeId: string; slug: string }) { {isLoading && } {error &&

{error.message}

} {!isLoading && } + {writable && ( void invalidate()} + onChooseDefault={defaults.data && !defaults.error ? () => setEditingDefaults(defaults.data!) : undefined} /> )} + {writable && editingDefaults && setEditingDefaults(null)} />} ({ api: vi.fn(), role: "contributor" })); vi.mock("@/api", async (original) => ({ ...await original(), api: mocks.api })); vi.mock("@/lib/universes", () => ({ useActiveUniverse: () => ({ universe: { id: "universe", role: mocks.role }, slug: "universe", isLoading: false }) })); vi.mock("@/components/provider-readiness-banner", () => ({ ProviderReadinessBanner: () => null })); +vi.mock("@/components/ui/select", () => { + // Exercise the page's selection behavior without popup positioning in jsdom. + const SelectItem = () => null; + const Select = ({ value, onValueChange, children }: { value: string; onValueChange: (value: string) => void; children: ReactNode }) => { + const options = (nodes: ReactNode): ReactNode => Children.map(nodes, (node) => { + if (!isValidElement<{ value?: string; children?: ReactNode }>(node)) return null; + return node.type === SelectItem ? : options(node.props.children); + }); + return ; + }; + return { Select, SelectItem, SelectContent: () => null, SelectTrigger: () => null, SelectValue: () => null }; +}); let root: Root; let container: HTMLDivElement; let client: QueryClient; +let defaults: ModelDefaults; const sessions = [ { id: "own", access: { visibility: "restricted", createdBy: { kind: "actor", id: "user" } } }, { id: "other", access: { visibility: "universe", createdBy: { kind: "actor", id: "someone" } } }, @@ -24,8 +38,13 @@ beforeEach(() => { vi.stubGlobal("PointerEvent", MouseEvent); window.localStorage.clear(); mocks.role = "contributor"; - mocks.api.mockReset().mockImplementation(async (_method: string, path: string) => { + defaults = { revision: 1, agentRun: { providerId: "openai", apiKind: "openai:responses", model: "gpt-6-sol" }, speechToText: null }; + mocks.api.mockReset().mockImplementation(async (method: string, path: string) => { if (path.includes("/sessions?")) return { sessions }; + if (path.endsWith("/models/defaults")) return defaults; + if (path.endsWith("/profiles")) return [{ profileId: "custom", displayName: "Custom model" }]; + if (path.endsWith("/profiles/custom")) return { profileId: "custom", config: { model: { providerId: "anthropic", apiKind: "anthropic:messages", model: "profile-model" } } }; + if (method === "POST" && path.endsWith("/sessions")) return { id: "created" }; throw new Error(`Unexpected request: ${path}`); }); client = new QueryClient({ defaultOptions: { queries: { retry: false, gcTime: Infinity } } }); @@ -72,3 +91,36 @@ it("offers bulk actions to a contributor over the listed sessions", async () => expect(container.textContent).toContain("2 selected"); expect([...container.querySelectorAll("button")].some((button) => button.textContent === "Close 2")).toBe(true); }); + +async function openCreate() { + await show(); + await act(async () => container.querySelector('[aria-label="New session"]')!.click()); + for (let step = 0; step < 4; step++) await act(async () => { await vi.advanceTimersByTimeAsync(5); }); + return document.body.querySelector('[role="dialog"]')!; +} + +it("previews the universe route while leaving default resolution to session creation", async () => { + const dialog = await openCreate(); + expect(dialog.textContent).toContain("Universe default: OpenAI · gpt-6-sol"); + await act(async () => [...dialog.querySelectorAll("button")].find((button) => button.textContent === "Create")!.click()); + const created = mocks.api.mock.calls.find(([method]) => method === "POST"); + expect(created?.[2]).toEqual({ profile: { kind: "inline", profile: {} } }); +}); + +it("blocks creation without a default but allows a profile's own model", async () => { + defaults.agentRun = null; + const dialog = await openCreate(); + const create = () => [...dialog.querySelectorAll("button")].find((button) => button.textContent === "Create")!; + expect(dialog.textContent).toContain("Universe default: Not selected"); + expect(create().disabled).toBe(true); + await act(async () => { + const select = dialog.querySelector("select")!; + select.value = "custom"; + select.dispatchEvent(new Event("change", { bubbles: true })); + }); + await act(async () => { await vi.advanceTimersByTimeAsync(10); }); + expect(dialog.textContent).toContain("Profile model: Anthropic · profile-model"); + expect(create().disabled).toBe(false); + await act(async () => create().click()); + expect(mocks.api.mock.calls.find(([method]) => method === "POST")?.[2]).toEqual({ profile: { kind: "named", profileId: "custom" } }); +}); diff --git a/platform/web/src/pages/SessionsPage.tsx b/platform/web/src/pages/SessionsPage.tsx index 5449ea3ab..453ab7618 100644 --- a/platform/web/src/pages/SessionsPage.tsx +++ b/platform/web/src/pages/SessionsPage.tsx @@ -113,6 +113,7 @@ import { setupResourceFeatureError, } from "@/lib/sessions/resource-features"; import { ProviderReadinessBanner } from "@/components/provider-readiness-banner"; +import { modelFromConfig, modelLabel, resolveCreationModel, useModelDefaults } from "@/lib/model-defaults"; import { useActionPermissions } from "@/lib/permissions"; import { useActiveUniverse, useFeature } from "@/lib/universes"; import { cn } from "@/lib/utils"; @@ -158,7 +159,7 @@ export function SessionsPage({ admin }: { admin: boolean }) {
- + {!sessionId && } {sessionId ? ( +

{effectiveModel.source}: {modelPending ? "Loading…" : effectiveModel.model ? modelLabel(effectiveModel.model) : defaults.error ? "Could not load default" : "Not selected"}

+ +
+ ); const create = useMutation({ mutationFn: () => api("POST", `/api/v1/universes/${universeId}/sessions`, { ...(displayName.trim() ? { displayName: displayName.trim() } : {}), - profile: profileForCreate(profileId, inlineProfile, selectedProfile.data), + profile: creationProfile, }), onSuccess: async (session) => { await queryClient.invalidateQueries({ queryKey: ["sessions", universeId] }); @@ -969,6 +986,7 @@ function NewSessionDialog({ const resourceError = setupResourceFeatureError( inlineProfile ?? selectedProfile.data ?? {}, ); + if (modelPending || modelMissing || (creationProfile.kind === "named" && selectedProfile.error)) return; if (configError || retentionError || resourceError) { setError(configError ? `Config: ${configError}` : retentionError ? `Retention: ${retentionError}` : resourceError); return; @@ -1049,12 +1067,12 @@ function NewSessionDialog({ {(value: string) => value ? (profiles.data?.find((p) => p.profileId === value)?.displayName ?? value) - : "No profile (engine defaults)" + : "No profile (universe default)" } - No profile (engine defaults) + No profile (universe default) {(profiles.data ?? []).map((profile) => ( {profile.displayName ?? profile.profileId} @@ -1066,6 +1084,7 @@ function NewSessionDialog({ The profile is resolved at creation; later profile edits do not change this session. + {modelSummary} - @@ -1101,7 +1120,8 @@ function NewSessionDialog({ : "Inline setup for this session."} -
+
+ {modelSummary} create.mutate()} > {create.isPending ? "Creating…" : "Create session"} @@ -1184,7 +1204,7 @@ function InlineSetupEditor({ + {!embedded && ( <>
diff --git a/release/metadata.env b/release/metadata.env index 7d5f2d1e3..86069c8f4 100644 --- a/release/metadata.env +++ b/release/metadata.env @@ -11,7 +11,7 @@ LIGHTSPEED_API_PROTOCOL_VERSION=lightspeed.agent.api.v1 # The environment protocol number the gateway and envd must agree on exactly; # a change here is the release event that stops older daemons from registering. LIGHTSPEED_ENVIRONMENT_PROTOCOL_VERSION=2 -LIGHTSPEED_SCHEMA_REVISION=9 +LIGHTSPEED_SCHEMA_REVISION=10 # Drizzle journal length and the oldest platform migration boundary accepted by # the current release's automated upgrade gate. LIGHTSPEED_PLATFORM_SCHEMA_REVISION=3 diff --git a/scripts/dev/cli-connection.mjs b/scripts/dev/cli-connection.mjs index cc4902cf2..561015e4a 100644 --- a/scripts/dev/cli-connection.mjs +++ b/scripts/dev/cli-connection.mjs @@ -77,3 +77,44 @@ export async function prepareCliConnection({ root, env, full, noBootstrap, run } } return { handoff: cli ? handoff : null, platformSecret }; } + +// This is development fixture policy, never a server-side model fallback. +// Revision zero distinguishes untouched universes from deliberately cleared +// defaults. Seed only after runtime (and, for Test, Platform) readiness. +export async function seedDevelopmentModelDefaults({ env, handoff, full, fetch: request = globalThis.fetch }) { + const single = env.LIGHTSPEED_AUTH_MODE === "single"; + const connection = handoff ? JSON.parse(readPrivate(handoff)) : null; + const secret = single ? null : connection?.credentialFile + ? readPrivate(connection.credentialFile) : env.LIGHTSPEED_PLATFORM_API_KEY; + if (!single && !secret) return false; + const universes = [env.LIGHTSPEED_PG_UNIVERSE_ID]; + if (full && env.LIGHTSPEED_PLATFORM_DEV_SEED === "true") universes.push("6c696768-7473-4065-8064-000000000010"); + for (const universe of new Set(universes)) { + const headers = { "content-type": "application/json" }; + if (!single) { + headers.authorization = `Bearer ${secret}`; + headers["x-lightspeed-universe"] = universe; + } + async function call(method, params) { + const response = await request(env.LIGHTSPEED_API_URL, { + method: "POST", headers, redirect: "error", signal: AbortSignal.timeout(10_000), + body: JSON.stringify({ jsonrpc: "2.0", id: "development-model-defaults", method, params }), + }); + if (!response.ok) throw new DevError(`Development model setup failed (HTTP ${response.status}).`); + const rpc = await response.json(); + if (rpc.error) { + // Another setup or user won the revision race. Their choice wins. + if (method === "models/defaults/put" && rpc.error.data?.kind === "conflict") return; + throw new DevError(`Development model setup failed for ${method}.`, { hint: "Check that the development key has models access, then configure defaults with the CLI." }); + } + return rpc.result.result; + } + const { defaults } = await call("models/defaults/read", {}); + if (defaults.revision !== 0) continue; + await call("models/defaults/put", { + slot: "agentRun", expectedRevision: 0, + model: { providerId: "openai", apiKind: "openai:responses", model: "gpt-6-sol" }, + }); + } + return true; +} diff --git a/scripts/dev/cli-connection.test.mjs b/scripts/dev/cli-connection.test.mjs index db47751a4..87b04bf73 100644 --- a/scripts/dev/cli-connection.test.mjs +++ b/scripts/dev/cli-connection.test.mjs @@ -71,3 +71,44 @@ test("supplied bootstrap input is registered once and missing local credentials rmSync(handoff.credentialFile); await assert.rejects(prepareCliConnection({ ...f, full: false }), /missing/); }); + +test("model seeding selects universes explicitly and preserves changed or cleared defaults", async () => { + const { seedDevelopmentModelDefaults } = await import("./cli-connection.mjs"); + const calls = []; + const env = { LIGHTSPEED_AUTH_MODE: "authenticated", LIGHTSPEED_PLATFORM_API_KEY: "lsk_fixture", LIGHTSPEED_API_URL: "http://localhost/rpc", LIGHTSPEED_PG_UNIVERSE_ID: "development", LIGHTSPEED_PLATFORM_DEV_SEED: "true" }; + await seedDevelopmentModelDefaults({ env, full: true, fetch: async (_url, init) => { + const rpc = JSON.parse(init.body); + const universe = init.headers["x-lightspeed-universe"]; + calls.push({ universe, ...rpc }); + assert.equal(init.headers.authorization, "Bearer lsk_fixture"); + assert.equal(init.redirect, "error"); + const defaults = { revision: universe === "development" ? 3 : 0, agentRun: null, speechToText: null }; + return Response.json({ result: { result: { defaults } } }); + } }); + assert.deepEqual(calls.map(c => [c.universe, c.method]), [ + ["development", "models/defaults/read"], + ["6c696768-7473-4065-8064-000000000010", "models/defaults/read"], + ["6c696768-7473-4065-8064-000000000010", "models/defaults/put"], + ]); + assert.equal(calls[2].params.expectedRevision, 0); + assert.equal(calls[2].params.slot, "agentRun"); +}); + +test("single-mode model seeding sends no authority headers and tolerates a concurrent update", async () => { + const { seedDevelopmentModelDefaults } = await import("./cli-connection.mjs"); + const methods = []; + await seedDevelopmentModelDefaults({ + env: { LIGHTSPEED_AUTH_MODE: "single", LIGHTSPEED_API_URL: "http://localhost/rpc", LIGHTSPEED_PG_UNIVERSE_ID: "development" }, + full: false, + fetch: async (_url, init) => { + assert.equal(init.headers.authorization, undefined); + assert.equal(init.headers["x-lightspeed-universe"], undefined); + const rpc = JSON.parse(init.body); + methods.push(rpc.method); + return Response.json(rpc.method === "models/defaults/read" + ? { result: { result: { defaults: { revision: 0 } } } } + : { error: { code: -32009, data: { kind: "conflict" } } }); + }, + }); + assert.deepEqual(methods, ["models/defaults/read", "models/defaults/put"]); +}); diff --git a/scripts/dev/stack.mjs b/scripts/dev/stack.mjs index f129c10b7..6289a62c3 100644 --- a/scripts/dev/stack.mjs +++ b/scripts/dev/stack.mjs @@ -4,7 +4,7 @@ // // Stateful dependencies run in Docker Compose. Rust and TypeScript processes // run from the checkout so cargo, tsx, and Vite retain their normal edit loops. -import { prepareCliConnection } from "./cli-connection.mjs"; +import { prepareCliConnection, seedDevelopmentModelDefaults } from "./cli-connection.mjs"; import { spawnSync } from "node:child_process"; import { closeSync, @@ -90,6 +90,7 @@ async function main() { phase = "preparing the development CLI connection"; const prepared = await prepareCliConnection({ root: repoRoot, env: plan.env, full: plan.profile === "full", noBootstrap: cli.noApiKeyBootstrap, run: runChecked }); if (prepared.platformSecret) { + plan.env.LIGHTSPEED_PLATFORM_API_KEY = prepared.platformSecret; for (const processPlan of plan.processes) processPlan.env.LIGHTSPEED_PLATFORM_API_KEY = prepared.platformSecret; } plan.cliHandoff = prepared.handoff; @@ -99,6 +100,11 @@ async function main() { if (stopping) return; phase = `waiting for ${plan.profile} services to become ready`; await waitForReadiness(plan); + if (!stopping && (plan.profile === "full" || plan.profile === "runtime")) { + phase = "seeding development model defaults"; + const seeded = await seedDevelopmentModelDefaults({ env: plan.env, handoff: plan.cliHandoff, full: plan.profile === "full" }); + if (!seeded) console.log("[models] No development credential available; configure universe defaults with lightspeed model defaults set."); + } if (!stopping) { printRunning(plan); if (plan.cliHandoff) console.log("CLI connection ready: lightspeed connect dev"); From f7a0a3f33505177b6a88cd201e710e9e7fd1abcb Mon Sep 17 00:00:00 2001 From: lb <542828+lukebuehler@users.noreply.github.com> Date: Tue, 29 Sep 2026 00:11:11 +0200 Subject: [PATCH 02/12] transcription workflow --- clients/typescript/schema/api.schema.json | 237 +++- clients/typescript/schema/workflow.json | 5 +- .../typescript/schema/workflow.schema.json | 329 +++++ clients/typescript/src/generated/methods.ts | 78 +- clients/typescript/src/generated/types.ts | 134 +- .../src/generated/workflow-manifest.ts | 5 +- .../src/generated/workflow-types.ts | 132 ++ crates/api-projection/src/content.rs | 21 +- crates/api-projection/src/lib.rs | 6 + crates/api/contract/api-reference.md | 41 +- crates/api/contract/api.schema.json | 237 +++- crates/api/contract/methods.json | 77 +- crates/api/contract/openrpc.json | 323 ++++- crates/api/src/access.rs | 8 +- crates/api/src/bots.rs | 3 + crates/api/src/constants.rs | 6 + crates/api/src/lib.rs | 2 + crates/api/src/rpc.rs | 43 +- crates/api/src/schema_export.rs | 2 +- crates/api/src/service.rs | 19 + crates/api/src/sessions.rs | 6 - crates/api/src/tests.rs | 1 + crates/api/src/transcriptions.rs | 123 ++ crates/api/src/views.rs | 8 + crates/api/tests/schema_artifacts.rs | 2 + crates/auth/src/providers.rs | 12 +- crates/bots/src/views.rs | 57 +- crates/channels/src/media.rs | 1 + crates/cli/src/chat/driver.rs | 12 +- crates/cli/src/skills_cli.rs | 6 +- crates/llm-clients/src/content.rs | 44 - crates/llm-clients/src/openai/audio.rs | 119 +- crates/llm-runtime/src/anthropic_messages.rs | 2 +- crates/llm-runtime/src/blob_io.rs | 20 - crates/llm-runtime/src/lib.rs | 2 +- crates/llm-runtime/src/openai_completions.rs | 2 +- crates/llm-runtime/src/openai_responses.rs | 2 +- crates/llm-runtime/src/provider_keys.rs | 110 +- .../tests/content_materialization.rs | 30 +- crates/store-pg/src/bots.rs | 6 + crates/store-pg/tests/bots_pg.rs | 16 +- crates/temporal-server/src/bots/sessions.rs | 1 + .../src/channels/activities.rs | 30 +- .../src/gateway/service/errors.rs | 21 - .../src/gateway/service/input.rs | 100 +- .../src/gateway/service/mod.rs | 65 +- .../src/gateway/service/tests.rs | 316 ++--- .../src/gateway/service/transcriptions.rs | 268 ++++ crates/temporal-server/src/subagents.rs | 2 + .../src/worker/activities/audio.rs | 803 +++++++++++ .../src/worker/activities/mod.rs | 58 +- .../src/worker/activities/preprocess.rs | 1189 ----------------- .../src/worker/activities/state.rs | 17 +- .../src/worker/activities/transcriptions.rs | 156 +++ crates/temporal-server/src/worker/channels.rs | 45 + crates/temporal-server/src/worker/mod.rs | 35 +- crates/temporal-server/tests/channels_live.rs | 136 +- crates/temporal-server/tests/mcp_live.rs | 4 + crates/temporal-server/tests/profiles_live.rs | 1 + crates/temporal-server/tests/runs_live.rs | 9 + .../temporal-server/tests/runs_live_slow.rs | 1 + crates/temporal-server/tests/sessions_live.rs | 19 +- .../temporal-server/tests/subagents_live.rs | 2 + crates/temporal-server/tests/support/live.rs | 5 +- ...process_live.rs => transcriptions_live.rs} | 266 +++- .../tests/vfs_transfer_live.rs | 1 + .../tests/workflow_tool_plugins_live.rs | 9 +- .../contract/workflow-contract.md | 4 +- .../temporal-workflow/contract/workflow.json | 5 +- .../contract/workflow.schema.json | 329 +++++ crates/temporal-workflow/src/activities.rs | 20 +- crates/temporal-workflow/src/lib.rs | 17 +- crates/temporal-workflow/src/types.rs | 52 - .../src/workflow_contract.rs | 8 +- .../src/workflows/channels/activities.rs | 9 + .../src/workflows/channels/conversation.rs | 128 +- .../src/workflows/channels/types.rs | 10 + crates/temporal-workflow/src/workflows/mod.rs | 6 + .../src/workflows/session/admissions.rs | 219 +-- .../src/workflows/session/mod.rs | 8 +- .../src/workflows/session/tests.rs | 48 - .../src/workflows/transcriptions.rs | 181 +++ ...iverse-model-defaults-and-transcription.md | 203 ++- package-lock.json | 192 +++ .../configurator-mcp/src/generated/tools.ts | 187 ++- platform/server/src/routes/gateway.ts | 2 +- platform/server/src/routes/method-roles.ts | 3 + platform/web/package.json | 1 + platform/web/src/api.ts | 2 +- .../src/components/models/model-api-key.tsx | 7 + platform/web/src/lib/method-groups.ts | 1 + 91 files changed, 5191 insertions(+), 2299 deletions(-) create mode 100644 crates/api/src/transcriptions.rs create mode 100644 crates/temporal-server/src/gateway/service/transcriptions.rs create mode 100644 crates/temporal-server/src/worker/activities/audio.rs delete mode 100644 crates/temporal-server/src/worker/activities/preprocess.rs create mode 100644 crates/temporal-server/src/worker/activities/transcriptions.rs rename crates/temporal-server/tests/{preprocess_live.rs => transcriptions_live.rs} (55%) create mode 100644 crates/temporal-workflow/src/workflows/transcriptions.rs diff --git a/clients/typescript/schema/api.schema.json b/clients/typescript/schema/api.schema.json index 11f14af3b..0faeffd8a 100644 --- a/clients/typescript/schema/api.schema.json +++ b/clients/typescript/schema/api.schema.json @@ -107,12 +107,7 @@ "invalid_request", "not_found", "conflict", - "unsupported_audio_mime", "audio_blob_too_large", - "audio_duration_too_long", - "transcoder_unavailable", - "transcode_failure", - "transcription_failure", "internal" ], "type": "string" @@ -2211,6 +2206,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfTranscriptionResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/definitions/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/definitions/TranscriptionResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfVfsSnapshotCommitResponse": { "properties": { "notifications": { @@ -4575,6 +4587,13 @@ "string", "null" ] + }, + "textRef": { + "description": "Optional prepared UTF-8 text; the source attachment remains in blobRef.", + "type": [ + "string", + "null" + ] } }, "required": [ @@ -10625,13 +10644,7 @@ "InputAdmissionFailureKind": { "enum": [ "unsupportedMedia", - "unsupportedAudioMime", "blobMissing", - "blobTooLarge", - "audioDurationTooLong", - "transcoderUnavailable", - "transcodeFailure", - "transcriptionFailure", "admissionRejected" ], "type": "string" @@ -10662,6 +10675,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -10688,6 +10708,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -11734,6 +11761,7 @@ "enum": [ "vfs", "profiles", + "transcriptions", "models", "mcp", "bots", @@ -17102,6 +17130,193 @@ ], "type": "object" }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionCancelParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/definitions/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionReadParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionResponse": { + "properties": { + "transcription": { + "$ref": "#/definitions/TranscriptionView" + } + }, + "required": [ + "transcription" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/definitions/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "status": { + "$ref": "#/definitions/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, "UniverseAction": { "description": "What a public method does, independent of its RPC spelling. Requests are\ngated by their key's groups; this classifies what the runtime's own work\nmay do and which role a gate built on the contract should require.", "oneOf": [ diff --git a/clients/typescript/schema/workflow.json b/clients/typescript/schema/workflow.json index 47ea11762..c9baa301b 100644 --- a/clients/typescript/schema/workflow.json +++ b/clients/typescript/schema/workflow.json @@ -65,7 +65,10 @@ "ChannelDeliveryCommand", "ChannelDeliveryResult", "PrepareChannelMediaInput", - "PrepareChannelMediaResult" + "PrepareChannelMediaResult", + "TranscriptionWorkflowArgs", + "TranscriptionSnapshot", + "TranscriptionActivityResult" ], "signals": { "deliverEmission": "deliver_emission" diff --git a/clients/typescript/schema/workflow.schema.json b/clients/typescript/schema/workflow.schema.json index 624554450..214d658c8 100644 --- a/clients/typescript/schema/workflow.schema.json +++ b/clients/typescript/schema/workflow.schema.json @@ -1,6 +1,79 @@ { "$schema": "http://json-schema.org/draft-07/schema#", "definitions": { + "Attribution": { + "description": "Who created a resource or authored bytes. An actor is whatever a key\nallowed to assert one said; core compares it and never resolves it.", + "oneOf": [ + { + "description": "The actor a key asserted for its request.", + "properties": { + "id": { + "type": "string" + }, + "kind": { + "const": "actor", + "type": "string" + } + }, + "required": [ + "kind", + "id" + ], + "type": "object" + }, + { + "description": "A key acting for itself, named by its display prefix.", + "properties": { + "kind": { + "const": "key", + "type": "string" + }, + "prefix": { + "type": "string" + } + }, + "required": [ + "kind", + "prefix" + ], + "type": "object" + }, + { + "description": "An unauthenticated local development request, or an in-process call.", + "properties": { + "kind": { + "const": "local", + "type": "string" + } + }, + "required": [ + "kind" + ], + "type": "object" + }, + { + "description": "The runtime's own work: a bot, a delegated session, a registration,\nor host administration through the server CLI.", + "properties": { + "cause": { + "type": "string" + }, + "component": { + "type": "string" + }, + "kind": { + "const": "internal", + "type": "string" + } + }, + "required": [ + "kind", + "component", + "cause" + ], + "type": "object" + } + ] + }, "BotId": { "type": "string" }, @@ -614,6 +687,25 @@ ], "type": "object" }, + "ModelConfig": { + "properties": { + "apiKind": { + "type": "string" + }, + "model": { + "type": "string" + }, + "providerId": { + "type": "string" + } + }, + "required": [ + "providerId", + "apiKind", + "model" + ], + "type": "object" + }, "PrepareChannelMediaInput": { "description": "`prepare_channel_media`: the connector downloads the provider file and\nstores it in the universe's CAS.", "properties": { @@ -761,6 +853,243 @@ "ToolCallId": { "type": "string" }, + "TranscriptionActivityResult": { + "oneOf": [ + { + "properties": { + "kind": { + "const": "succeeded", + "type": "string" + }, + "transcript_ref": { + "type": "string" + } + }, + "required": [ + "kind", + "transcript_ref" + ], + "type": "object" + }, + { + "properties": { + "failure": { + "$ref": "#/definitions/TranscriptionFailure" + }, + "kind": { + "const": "failed", + "type": "string" + } + }, + "required": [ + "kind", + "failure" + ], + "type": "object" + } + ] + }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/definitions/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionSnapshot": { + "properties": { + "request": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "view": { + "$ref": "#/definitions/TranscriptionView" + } + }, + "required": [ + "request", + "view" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/definitions/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "status": { + "$ref": "#/definitions/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, + "TranscriptionWorkflowArgs": { + "properties": { + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "request": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "transcriptionId": { + "type": "string" + }, + "universeId": { + "format": "uuid", + "type": "string" + } + }, + "required": [ + "universeId", + "transcriptionId", + "createdBy", + "request", + "model", + "createdAtMs" + ], + "type": "object" + }, "WorkflowToolId": { "type": "string" }, diff --git a/clients/typescript/src/generated/methods.ts b/clients/typescript/src/generated/methods.ts index 98b6acc60..1008cebd1 100644 --- a/clients/typescript/src/generated/methods.ts +++ b/clients/typescript/src/generated/methods.ts @@ -54,6 +54,9 @@ export const METHODS = [ "environments/registration-keys/read", "environments/registration-keys/list", "environments/registration-keys/revoke", + "transcriptions/start", + "transcriptions/read", + "transcriptions/cancel", "models/defaults/read", "models/defaults/put", "models/list", @@ -229,7 +232,7 @@ export const METHOD_INFO = { scope: "universe", access: {"action":"control_session","kind":"universe"}, summary: "Append keyed session context", - description: "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries.", + description: "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries.", }, "session/context/remove": { scope: "universe", @@ -435,6 +438,24 @@ export const METHOD_INFO = { summary: "Revoke an environment registration key", description: "Stops the key from admitting new daemon identities; already registered daemons keep reconnecting. With closeEnvironments, also closes every non-closed environment the key admitted. Idempotent.", }, + "transcriptions/start": { + scope: "universe", + access: {"action":"use_resource","kind":"universe"}, + summary: "Start transcription", + description: "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created.", + }, + "transcriptions/read": { + scope: "universe", + access: {"action":"read","kind":"universe"}, + summary: "Read transcription", + description: "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire.", + }, + "transcriptions/cancel": { + scope: "universe", + access: {"action":"use_resource","kind":"universe"}, + summary: "Cancel transcription", + description: "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority.", + }, "models/defaults/read": { scope: "universe", access: {"action":"read","kind":"universe"}, @@ -1083,7 +1104,7 @@ export interface MethodMap { /** * Append keyed session context * - * Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries. + * Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries. */ "session/context/append": { params: Api.ContextAppendParams; @@ -1395,6 +1416,33 @@ export interface MethodMap { params: Api.EnvironmentRegistrationKeyRevokeParams; result: Api.AgentApiOutcomeOfEnvironmentRegistrationKeyRevokeResponse; }; + /** + * Start transcription + * + * Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created. + */ + "transcriptions/start": { + params: Api.TranscriptionStartParams; + result: Api.AgentApiOutcomeOfTranscriptionResponse; + }; + /** + * Read transcription + * + * Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire. + */ + "transcriptions/read": { + params: Api.TranscriptionReadParams; + result: Api.AgentApiOutcomeOfTranscriptionResponse; + }; + /** + * Cancel transcription + * + * Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority. + */ + "transcriptions/cancel": { + params: Api.TranscriptionCancelParams; + result: Api.AgentApiOutcomeOfTranscriptionResponse; + }; /** * Read universe model defaults * @@ -2276,7 +2324,7 @@ export const rpc = { /** * Append keyed session context * - * Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries. + * Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries. */ sessionContextAppend(client: RpcCaller, params: Api.ContextAppendParams): Promise { return client.call("session/context/append", params); @@ -2553,6 +2601,30 @@ export const rpc = { environmentsRegistrationKeysRevoke(client: RpcCaller, params: Api.EnvironmentRegistrationKeyRevokeParams): Promise { return client.call("environments/registration-keys/revoke", params); }, + /** + * Start transcription + * + * Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created. + */ + transcriptionsStart(client: RpcCaller, params: Api.TranscriptionStartParams): Promise { + return client.call("transcriptions/start", params); + }, + /** + * Read transcription + * + * Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire. + */ + transcriptionsRead(client: RpcCaller, params: Api.TranscriptionReadParams): Promise { + return client.call("transcriptions/read", params); + }, + /** + * Cancel transcription + * + * Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority. + */ + transcriptionsCancel(client: RpcCaller, params: Api.TranscriptionCancelParams): Promise { + return client.call("transcriptions/cancel", params); + }, /** * Read universe model defaults * diff --git a/clients/typescript/src/generated/types.ts b/clients/typescript/src/generated/types.ts index 67f4ffa6a..4ff6f81db 100644 --- a/clients/typescript/src/generated/types.ts +++ b/clients/typescript/src/generated/types.ts @@ -80,18 +80,7 @@ export type ToolParallelismView = "exclusive" | "parallelSafe"; * via the `definition` "AgentApiErrorKind". */ export type AgentApiErrorKind = - | ( - | "invalid_request" - | "not_found" - | "conflict" - | "unsupported_audio_mime" - | "audio_blob_too_large" - | "audio_duration_too_long" - | "transcoder_unavailable" - | "transcode_failure" - | "transcription_failure" - | "internal" - ) + | ("invalid_request" | "not_found" | "conflict" | "audio_blob_too_large" | "internal") | "rejected" | "unauthenticated" | "forbidden" @@ -849,6 +838,11 @@ export type InputItem = * This metadata is not an authorization identity or model input text. */ origin?: string | null; + /** + * Optional source blob in this universe, retained with the session. + * Provenance is metadata, not model input or an authorization identity. + */ + provenanceRef?: string | null; text: string; type: "text"; } @@ -861,6 +855,11 @@ export type InputItem = * This metadata is not an authorization identity or model input text. */ origin?: string | null; + /** + * Optional source blob in this universe, retained with the session. + * Provenance is metadata, not model input or an authorization identity. + */ + provenanceRef?: string | null; type: "textRef"; } | { @@ -1287,16 +1286,7 @@ export type ChannelPairedVia = "open" | "code"; * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "InputAdmissionFailureKind". */ -export type InputAdmissionFailureKind = - | "unsupportedMedia" - | "unsupportedAudioMime" - | "blobMissing" - | "blobTooLarge" - | "audioDurationTooLong" - | "transcoderUnavailable" - | "transcodeFailure" - | "transcriptionFailure" - | "admissionRejected"; +export type InputAdmissionFailureKind = "unsupportedMedia" | "blobMissing" | "admissionRejected"; /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "ContextAppendStatus". @@ -1319,6 +1309,7 @@ export type MethodGroup = | ( | "vfs" | "profiles" + | "transcriptions" | "models" | "mcp" | "bots" @@ -1620,6 +1611,18 @@ export type SkillCatalogSource = environmentId: string; type: "environment"; }; +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionFailureKind". + */ +export type TranscriptionFailureKind = + "invalidAudio" | "configuration" | "provider" | "timeout" | "internal"; +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionStatus". + */ +export type TranscriptionStatus = + "pending" | "running" | "succeeded" | "failed" | "cancelled" | "expired"; /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "AuthProviderConfigInput". @@ -3575,6 +3578,10 @@ export interface BotEventMedia { kind: BotEventMediaKind; mime: string; name?: string | null; + /** + * Optional prepared UTF-8 text; the source attachment remains in blobRef. + */ + textRef?: string | null; } /** * The routed session an event was admitted to. @@ -6186,6 +6193,59 @@ export interface SkillLocationView { skillDirPath: string; skillDocPath: string; } +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "AgentApiOutcomeOfTranscriptionResponse". + */ +export interface AgentApiOutcomeOfTranscriptionResponse { + notifications?: AgentNotification[]; + result: TranscriptionResponse; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionResponse". + */ +export interface TranscriptionResponse { + transcription: TranscriptionView; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionView". + */ +export interface TranscriptionView { + audio: TranscriptionAudio; + createdAtMs: number; + createdBy: Attribution; + failure?: TranscriptionFailure | null; + model: ModelConfig; + status: TranscriptionStatus; + text?: string | null; + /** + * Plain UTF-8 transcript blob, usable as ordinary textRef input. + * Unsubmitted content can be swept after the ordinary CAS grace period. + */ + transcriptRef?: string | null; + transcriptionId: string; +} +/** + * Immutable audio input in this universe's content store. + * + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionAudio". + */ +export interface TranscriptionAudio { + blobRef: string; + mime: string; + name: string; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionFailure". + */ +export interface TranscriptionFailure { + kind: TranscriptionFailureKind; + message: string; +} /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "AgentApiOutcomeOfVfsSnapshotCommitResponse". @@ -8017,6 +8077,36 @@ export interface SessionStartParams { export interface SkillListParams { sessionId: string; } +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionCancelParams". + */ +export interface TranscriptionCancelParams { + transcriptionId: string; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionReadParams". + */ +export interface TranscriptionReadParams { + transcriptionId: string; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionStartParams". + */ +export interface TranscriptionStartParams { + audio: TranscriptionAudio; + /** + * Scoped to the requester. Matching retries rejoin the original job, + * including after defaults change; changed requests conflict. Identity is + * retained for the Temporal namespace's workflow-history retention period. + */ + idempotencyKey: string; + language?: string | null; + model?: ModelConfig | null; + prompt?: string | null; +} /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "VfsSnapshotCommitParams". diff --git a/clients/typescript/src/generated/workflow-manifest.ts b/clients/typescript/src/generated/workflow-manifest.ts index 5d10a5448..b0c1da9ee 100644 --- a/clients/typescript/src/generated/workflow-manifest.ts +++ b/clients/typescript/src/generated/workflow-manifest.ts @@ -70,7 +70,10 @@ export const WORKFLOW_CONTRACT_MANIFEST = "ChannelDeliveryCommand", "ChannelDeliveryResult", "PrepareChannelMediaInput", - "PrepareChannelMediaResult" + "PrepareChannelMediaResult", + "TranscriptionWorkflowArgs", + "TranscriptionSnapshot", + "TranscriptionActivityResult" ], "signals": { "deliverEmission": "deliver_emission" diff --git a/clients/typescript/src/generated/workflow-types.ts b/clients/typescript/src/generated/workflow-types.ts index 9d847e8c4..2f2607331 100644 --- a/clients/typescript/src/generated/workflow-types.ts +++ b/clients/typescript/src/generated/workflow-types.ts @@ -3,6 +3,30 @@ * Do not edit by hand. */ +/** + * Who created a resource or authored bytes. An actor is whatever a key + * allowed to assert one said; core compares it and never resolves it. + * + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "Attribution". + */ +export type Attribution = + | { + id: string; + kind: "actor"; + } + | { + kind: "key"; + prefix: string; + } + | { + kind: "local"; + } + | { + cause: string; + component: string; + kind: "internal"; + }; /** * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema * via the `definition` "BotId". @@ -197,6 +221,31 @@ export type EmissionProducer = universe_id: string; workflow_id: string; }; +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionActivityResult". + */ +export type TranscriptionActivityResult = + | { + kind: "succeeded"; + transcript_ref: string; + } + | { + failure: TranscriptionFailure; + kind: "failed"; + }; +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionFailureKind". + */ +export type TranscriptionFailureKind = + "invalidAudio" | "configuration" | "provider" | "timeout" | "internal"; +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionStatus". + */ +export type TranscriptionStatus = + "pending" | "running" | "succeeded" | "failed" | "cancelled" | "expired"; /** * Envelope and start-on-call types of the fixed deliver_emission transport between sessions and receiver workflows. @@ -413,6 +462,15 @@ export interface EmissionEnvelope { export interface MaintainChannelTypingInput { route: ChannelRoute; } +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "ModelConfig". + */ +export interface ModelConfig { + apiKind: string; + model: string; + providerId: string; +} /** * `prepare_channel_media`: the connector downloads the provider file and * stores it in the universe's CAS. @@ -444,6 +502,80 @@ export interface PreparedMediaItem { mime: string; name?: string | null; } +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionFailure". + */ +export interface TranscriptionFailure { + kind: TranscriptionFailureKind; + message: string; +} +/** + * Immutable audio input in this universe's content store. + * + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionAudio". + */ +export interface TranscriptionAudio { + blobRef: string; + mime: string; + name: string; +} +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionSnapshot". + */ +export interface TranscriptionSnapshot { + request: TranscriptionStartParams; + view: TranscriptionView; +} +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionStartParams". + */ +export interface TranscriptionStartParams { + audio: TranscriptionAudio; + /** + * Scoped to the requester. Matching retries rejoin the original job, + * including after defaults change; changed requests conflict. Identity is + * retained for the Temporal namespace's workflow-history retention period. + */ + idempotencyKey: string; + language?: string | null; + model?: ModelConfig | null; + prompt?: string | null; +} +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionView". + */ +export interface TranscriptionView { + audio: TranscriptionAudio; + createdAtMs: number; + createdBy: Attribution; + failure?: TranscriptionFailure | null; + model: ModelConfig; + status: TranscriptionStatus; + text?: string | null; + /** + * Plain UTF-8 transcript blob, usable as ordinary textRef input. + * Unsubmitted content can be swept after the ordinary CAS grace period. + */ + transcriptRef?: string | null; + transcriptionId: string; +} +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionWorkflowArgs". + */ +export interface TranscriptionWorkflowArgs { + createdAtMs: number; + createdBy: Attribution; + model: ModelConfig; + request: TranscriptionStartParams; + transcriptionId: string; + universeId: string; +} /** * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema * via the `definition` "WorkflowToolRecipeV1". diff --git a/crates/api-projection/src/content.rs b/crates/api-projection/src/content.rs index 238969e55..a62abcd50 100644 --- a/crates/api-projection/src/content.rs +++ b/crates/api-projection/src/content.rs @@ -48,9 +48,6 @@ pub async fn project_content_text( Some(llm_clients::content::ANTHROPIC_THINKING_PROVIDER_KIND) => { llm_clients::content::anthropic_thinking } - Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) => { - llm_clients::content::audio_transcript - } Some(_) => { return Err(AgentApiError::invalid_request( "unsupported native content text projection", @@ -92,7 +89,7 @@ mod tests { } #[tokio::test(flavor = "current_thread")] - async fn audio_run_summary_previews_project_text_before_applying_the_limit() { + async fn run_summary_previews_preserve_unicode_at_the_limit() { let blobs = InMemoryBlobStore::new(); let projector = crate::CoreAgentProjector::new(&blobs); let boundary = format!("{}é", "🦀".repeat(127)); @@ -108,11 +105,7 @@ mod tests { true, ), ] { - let bytes = serde_json::to_vec(&AudioTranscript { - filename: "voice.ogg".to_owned(), - text, - }) - .unwrap(); + let bytes = text.into_bytes(); let reference = blobs.put_bytes(bytes.clone()).await.unwrap(); let source = engine::RunSource::Input { input: vec![engine::ContextEntryInput { @@ -121,10 +114,10 @@ mod tests { }, content: ContentRef { content_ref: reference.clone(), - media_type: Some("application/json".into()), - provider_kind: Some(AUDIO_TRANSCRIPT_PROVIDER_KIND.into()), + media_type: Some("text/plain".into()), + provider_kind: None, }, - preview: Some("short preprocessing preview".into()), + preview: None, origin: None, provenance_ref: None, token_estimate: None, @@ -158,8 +151,6 @@ mod tests { serde_json::to_vec(&json!({"type":"message","role":"assistant","content":[{"type":"output_text","text":full,"annotations":[]}]})).unwrap()), (Some(OPENAI_COMPLETIONS_MESSAGE_PROVIDER_KIND), serde_json::to_vec(&json!({"role":"assistant","content":full,"annotations":[]})).unwrap()), - (Some(AUDIO_TRANSCRIPT_PROVIDER_KIND), - serde_json::to_vec(&json!({"filename":"note.ogg","text":full})).unwrap()), ] { let content = ContentRef { content_ref: blobs.put_bytes(bytes.clone()).await.unwrap(), @@ -180,7 +171,7 @@ mod tests { assert_eq!(view.text.as_deref(), Some(full.as_str())); assert!(!view.text_truncated); assert_eq!(projector.project_input_entries(&[input]).await.unwrap(), - vec![api::InputItem::Text { origin: None, text: full.clone() }]); + vec![api::InputItem::Text { provenance_ref: None, origin: None, text: full.clone() }]); // The retained terminal output still resolves after its context is gone. let source = engine::RunSource::Input { input: Vec::new() }; diff --git a/crates/api-projection/src/lib.rs b/crates/api-projection/src/lib.rs index 4410178bb..cfb5077b4 100644 --- a/crates/api-projection/src/lib.rs +++ b/crates/api-projection/src/lib.rs @@ -611,6 +611,7 @@ impl<'a> CoreAgentProjector<'a> { // the blob as UTF-8 text would fail. if is_text_message_media_type(entry.content.media_type.as_deref()) { InputItem::Text { + provenance_ref: entry.provenance_ref.as_ref().map(ToString::to_string), origin: entry.origin.clone(), text: project_content_text(self.blobs, &entry.content) .await? @@ -628,6 +629,7 @@ impl<'a> CoreAgentProjector<'a> { } } _ => InputItem::TextRef { + provenance_ref: entry.provenance_ref.as_ref().map(ToString::to_string), origin: entry.origin.clone(), blob_ref: entry.content.content_ref.as_str().to_owned(), }, @@ -4562,14 +4564,17 @@ mod tests { fn input_text_joins_non_empty_text_items() { let text = input_text(&[ InputItem::Text { + provenance_ref: None, origin: None, text: " first ".to_owned(), }, InputItem::Text { + provenance_ref: None, origin: None, text: "".to_owned(), }, InputItem::Text { + provenance_ref: None, origin: None, text: "second".to_owned(), }, @@ -4582,6 +4587,7 @@ mod tests { #[test] fn input_text_rejects_unresolved_text_refs() { let error = input_text(&[InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: BlobRef::from_bytes(b"hello").as_str().to_owned(), }]) diff --git a/crates/api/contract/api-reference.md b/crates/api/contract/api-reference.md index 330c2e834..4a8806a27 100644 --- a/crates/api/contract/api-reference.md +++ b/crates/api/contract/api-reference.md @@ -198,7 +198,7 @@ Returns chronological events. Forward (default) follows after and supports long- **Append keyed session context** -Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries. +Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries. - Access: `{"kind":"universe","action":"control_session"}` - Group: `session` @@ -649,6 +649,45 @@ Stops the key from admitting new daemon identities; already registered daemons k - Params: `EnvironmentRegistrationKeyRevokeParams` - Result: `AgentApiOutcome` +### `transcriptions/start` + +**Start transcription** + +Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created. + +- Access: `{"kind":"universe","action":"use_resource"}` +- Group: `transcriptions` +- Role: `contributor` +- Target: `none` +- Params: `TranscriptionStartParams` +- Result: `AgentApiOutcome` + +### `transcriptions/read` + +**Read transcription** + +Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire. + +- Access: `{"kind":"universe","action":"read"}` +- Group: `transcriptions` +- Role: `viewer` +- Target: `none` +- Params: `TranscriptionReadParams` +- Result: `AgentApiOutcome` + +### `transcriptions/cancel` + +**Cancel transcription** + +Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority. + +- Access: `{"kind":"universe","action":"use_resource"}` +- Group: `transcriptions` +- Role: `contributor` +- Target: `none` +- Params: `TranscriptionCancelParams` +- Result: `AgentApiOutcome` + ### `models/defaults/read` **Read universe model defaults** diff --git a/crates/api/contract/api.schema.json b/crates/api/contract/api.schema.json index 11f14af3b..0faeffd8a 100644 --- a/crates/api/contract/api.schema.json +++ b/crates/api/contract/api.schema.json @@ -107,12 +107,7 @@ "invalid_request", "not_found", "conflict", - "unsupported_audio_mime", "audio_blob_too_large", - "audio_duration_too_long", - "transcoder_unavailable", - "transcode_failure", - "transcription_failure", "internal" ], "type": "string" @@ -2211,6 +2206,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfTranscriptionResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/definitions/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/definitions/TranscriptionResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfVfsSnapshotCommitResponse": { "properties": { "notifications": { @@ -4575,6 +4587,13 @@ "string", "null" ] + }, + "textRef": { + "description": "Optional prepared UTF-8 text; the source attachment remains in blobRef.", + "type": [ + "string", + "null" + ] } }, "required": [ @@ -10625,13 +10644,7 @@ "InputAdmissionFailureKind": { "enum": [ "unsupportedMedia", - "unsupportedAudioMime", "blobMissing", - "blobTooLarge", - "audioDurationTooLong", - "transcoderUnavailable", - "transcodeFailure", - "transcriptionFailure", "admissionRejected" ], "type": "string" @@ -10662,6 +10675,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -10688,6 +10708,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -11734,6 +11761,7 @@ "enum": [ "vfs", "profiles", + "transcriptions", "models", "mcp", "bots", @@ -17102,6 +17130,193 @@ ], "type": "object" }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionCancelParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/definitions/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionReadParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionResponse": { + "properties": { + "transcription": { + "$ref": "#/definitions/TranscriptionView" + } + }, + "required": [ + "transcription" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/definitions/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "status": { + "$ref": "#/definitions/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, "UniverseAction": { "description": "What a public method does, independent of its RPC spelling. Requests are\ngated by their key's groups; this classifies what the runtime's own work\nmay do and which role a gate built on the contract should require.", "oneOf": [ diff --git a/crates/api/contract/methods.json b/crates/api/contract/methods.json index f9c1d80b6..3fe0b7584 100644 --- a/crates/api/contract/methods.json +++ b/crates/api/contract/methods.json @@ -354,7 +354,7 @@ "action": "control_session", "kind": "universe" }, - "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries.", + "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries.", "group": "session", "method": "session/context/append", "params": { @@ -1224,6 +1224,81 @@ "summary": "Revoke an environment registration key", "target": null }, + { + "access": { + "action": "use_resource", + "kind": "universe" + }, + "description": "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created.", + "group": "transcriptions", + "method": "transcriptions/start", + "params": { + "schema": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "type": "TranscriptionStartParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfTranscriptionResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "contributor", + "scope": "universe", + "summary": "Start transcription", + "target": null + }, + { + "access": { + "action": "read", + "kind": "universe" + }, + "description": "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire.", + "group": "transcriptions", + "method": "transcriptions/read", + "params": { + "schema": { + "$ref": "#/definitions/TranscriptionReadParams" + }, + "type": "TranscriptionReadParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfTranscriptionResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "viewer", + "scope": "universe", + "summary": "Read transcription", + "target": null + }, + { + "access": { + "action": "use_resource", + "kind": "universe" + }, + "description": "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority.", + "group": "transcriptions", + "method": "transcriptions/cancel", + "params": { + "schema": { + "$ref": "#/definitions/TranscriptionCancelParams" + }, + "type": "TranscriptionCancelParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfTranscriptionResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "contributor", + "scope": "universe", + "summary": "Cancel transcription", + "target": null + }, { "access": { "action": "read", diff --git a/crates/api/contract/openrpc.json b/crates/api/contract/openrpc.json index 2d8bcc43e..7a4ac4cf9 100644 --- a/crates/api/contract/openrpc.json +++ b/crates/api/contract/openrpc.json @@ -107,12 +107,7 @@ "invalid_request", "not_found", "conflict", - "unsupported_audio_mime", "audio_blob_too_large", - "audio_duration_too_long", - "transcoder_unavailable", - "transcode_failure", - "transcription_failure", "internal" ], "type": "string" @@ -2211,6 +2206,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfTranscriptionResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/components/schemas/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/components/schemas/TranscriptionResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfVfsSnapshotCommitResponse": { "properties": { "notifications": { @@ -4575,6 +4587,13 @@ "string", "null" ] + }, + "textRef": { + "description": "Optional prepared UTF-8 text; the source attachment remains in blobRef.", + "type": [ + "string", + "null" + ] } }, "required": [ @@ -10625,13 +10644,7 @@ "InputAdmissionFailureKind": { "enum": [ "unsupportedMedia", - "unsupportedAudioMime", "blobMissing", - "blobTooLarge", - "audioDurationTooLong", - "transcoderUnavailable", - "transcodeFailure", - "transcriptionFailure", "admissionRejected" ], "type": "string" @@ -10662,6 +10675,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -10688,6 +10708,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -11734,6 +11761,7 @@ "enum": [ "vfs", "profiles", + "transcriptions", "models", "mcp", "bots", @@ -17102,6 +17130,193 @@ ], "type": "object" }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionCancelParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/components/schemas/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionReadParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionResponse": { + "properties": { + "transcription": { + "$ref": "#/components/schemas/TranscriptionView" + } + }, + "required": [ + "transcription" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/components/schemas/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/components/schemas/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/components/schemas/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/components/schemas/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/components/schemas/ModelConfig" + }, + "status": { + "$ref": "#/components/schemas/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, "UniverseAction": { "description": "What a public method does, independent of its RPC spelling. Requests are\ngated by their key's groups; this classifies what the runtime's own work\nmay do and which role a gate built on the contract should require.", "oneOf": [ @@ -18420,7 +18635,7 @@ "x-lightspeed-target": "sessionId" }, { - "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries.", + "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries.", "name": "session/context/append", "paramStructure": "by-name", "params": [ @@ -19399,6 +19614,90 @@ "x-lightspeed-role": "operator", "x-lightspeed-target": null }, + { + "description": "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created.", + "name": "transcriptions/start", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/TranscriptionStartParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfTranscriptionResponse" + } + }, + "summary": "Start transcription", + "x-lightspeed-access": { + "action": "use_resource", + "kind": "universe" + }, + "x-lightspeed-group": "transcriptions", + "x-lightspeed-role": "contributor", + "x-lightspeed-target": null + }, + { + "description": "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire.", + "name": "transcriptions/read", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/TranscriptionReadParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfTranscriptionResponse" + } + }, + "summary": "Read transcription", + "x-lightspeed-access": { + "action": "read", + "kind": "universe" + }, + "x-lightspeed-group": "transcriptions", + "x-lightspeed-role": "viewer", + "x-lightspeed-target": null + }, + { + "description": "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority.", + "name": "transcriptions/cancel", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/TranscriptionCancelParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfTranscriptionResponse" + } + }, + "summary": "Cancel transcription", + "x-lightspeed-access": { + "action": "use_resource", + "kind": "universe" + }, + "x-lightspeed-group": "transcriptions", + "x-lightspeed-role": "contributor", + "x-lightspeed-target": null + }, { "description": "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers.", "name": "models/defaults/read", diff --git a/crates/api/src/access.rs b/crates/api/src/access.rs index 58d6231f8..538310352 100644 --- a/crates/api/src/access.rs +++ b/crates/api/src/access.rs @@ -55,6 +55,8 @@ pub enum MethodGroup { Vfs, #[serde(rename = "profiles")] Profiles, + #[serde(rename = "transcriptions")] + Transcriptions, #[serde(rename = "models")] Models, #[serde(rename = "mcp")] @@ -89,12 +91,13 @@ pub enum MethodGroup { } impl MethodGroup { - pub const ALL: [MethodGroup; 16] = [ + pub const ALL: [MethodGroup; 17] = [ Self::Session, Self::BlobsPut, Self::Vfs, Self::Profiles, Self::Models, + Self::Transcriptions, Self::Mcp, Self::Environments, Self::Bots, @@ -116,6 +119,7 @@ impl MethodGroup { Self::Vfs => "vfs", Self::Profiles => "profiles", Self::Models => "models", + Self::Transcriptions => "transcriptions", Self::Mcp => "mcp", Self::Environments => "environments", Self::Bots => "bots", @@ -164,6 +168,8 @@ impl MethodGroup { Self::BlobsPut } else if group("session/") || group("blobs/") { Self::Session + } else if group("transcriptions/") { + Self::Transcriptions } else if group("vfs/") { Self::Vfs } else if group("profiles/") { diff --git a/crates/api/src/bots.rs b/crates/api/src/bots.rs index 6cef07f7e..1a471bdb7 100644 --- a/crates/api/src/bots.rs +++ b/crates/api/src/bots.rs @@ -699,6 +699,9 @@ pub enum BotEventMediaKind { #[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] #[serde(rename_all = "camelCase")] pub struct BotEventMedia { + /// Optional prepared UTF-8 text; the source attachment remains in blobRef. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub text_ref: Option, pub blob_ref: String, pub kind: BotEventMediaKind, pub mime: String, diff --git a/crates/api/src/constants.rs b/crates/api/src/constants.rs index dc6a1482e..32d7ddaa4 100644 --- a/crates/api/src/constants.rs +++ b/crates/api/src/constants.rs @@ -177,3 +177,9 @@ pub const NOTIFY_SESSION_EVENT: &str = "session/event"; pub const NOTIFY_SESSION_RUNS_STARTED: &str = "session/runs/started"; pub const NOTIFY_SESSION_RUNS_COMPLETED: &str = "session/runs/completed"; pub const NOTIFY_ERROR: &str = "error"; + +// ── Transcriptions ─────────────────────────────────────────────────────────── + +pub const METHOD_TRANSCRIPTIONS_START: &str = "transcriptions/start"; +pub const METHOD_TRANSCRIPTIONS_READ: &str = "transcriptions/read"; +pub const METHOD_TRANSCRIPTIONS_CANCEL: &str = "transcriptions/cancel"; diff --git a/crates/api/src/lib.rs b/crates/api/src/lib.rs index ebfd0c89d..f7944e68e 100644 --- a/crates/api/src/lib.rs +++ b/crates/api/src/lib.rs @@ -38,6 +38,7 @@ mod service; mod sessions; mod skills; mod storage; +mod transcriptions; mod views; pub use access::*; @@ -62,6 +63,7 @@ pub use service::*; pub use sessions::*; pub use skills::*; pub use storage::*; +pub use transcriptions::*; pub use views::*; #[cfg(test)] diff --git a/crates/api/src/rpc.rs b/crates/api/src/rpc.rs index 93fa40df7..8c5839d70 100644 --- a/crates/api/src/rpc.rs +++ b/crates/api/src/rpc.rs @@ -15,12 +15,7 @@ pub enum AgentApiErrorKind { Forbidden, /// No model was supplied and this universe has no default for the requested use. ModelDefaultUnset, - UnsupportedAudioMime, AudioBlobTooLarge, - AudioDurationTooLong, - TranscoderUnavailable, - TranscodeFailure, - TranscriptionFailure, /// The session's agent workflow exists but failed during bootstrap /// (rehydration) and cannot serve runs. Distinct from `NotFound` (no /// workflow) so clients/bridges treat it as a session recovery problem @@ -96,30 +91,10 @@ impl AgentApiError { Self::new(AgentApiErrorKind::Forbidden, "request is not authorized") } - pub fn unsupported_audio_mime(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::UnsupportedAudioMime, message) - } - pub fn audio_blob_too_large(message: impl Into) -> Self { Self::new(AgentApiErrorKind::AudioBlobTooLarge, message) } - pub fn audio_duration_too_long(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::AudioDurationTooLong, message) - } - - pub fn transcoder_unavailable(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::TranscoderUnavailable, message) - } - - pub fn transcode_failure(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::TranscodeFailure, message) - } - - pub fn transcription_failure(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::TranscriptionFailure, message) - } - pub fn session_bootstrap_failed(message: impl Into) -> Self { Self::new(AgentApiErrorKind::SessionBootstrapFailed, message) } @@ -138,18 +113,12 @@ impl AgentApiError { pub fn json_rpc_code(&self) -> i64 { match self.kind { - AgentApiErrorKind::InvalidRequest - | AgentApiErrorKind::UnsupportedAudioMime - | AgentApiErrorKind::AudioBlobTooLarge - | AgentApiErrorKind::AudioDurationTooLong - | AgentApiErrorKind::TranscoderUnavailable => -32602, + AgentApiErrorKind::InvalidRequest | AgentApiErrorKind::AudioBlobTooLarge => -32602, AgentApiErrorKind::Unauthenticated => -32001, AgentApiErrorKind::Forbidden => -32003, AgentApiErrorKind::NotFound => -32004, AgentApiErrorKind::Conflict => -32009, - AgentApiErrorKind::Rejected - | AgentApiErrorKind::TranscodeFailure - | AgentApiErrorKind::TranscriptionFailure => -32010, + AgentApiErrorKind::Rejected => -32010, AgentApiErrorKind::SessionBootstrapFailed => -32011, AgentApiErrorKind::EnvironmentNotReady => -32012, AgentApiErrorKind::ResponseTooLarge => -32013, @@ -392,7 +361,7 @@ api_methods! { METHOD_SESSION_EVENTS_READ => read_session_events(SessionEventsReadParams) -> SessionEventsReadResponse => ["Read the session event stream", "Returns chronological events. Forward (default) follows after and supports long-polling. Backward reads the latest window below before (or the head); pass nextCursor as before until complete. Follow live events after the initial backward headCursor. Windows may split runs/tool batches; keep historical reconstruction separate from live controls."], access: MethodAccess::Universe(UniverseAction::Read), METHOD_SESSION_CONTEXT_APPEND => append_context(ContextAppendParams) -> ContextAppendResponse => - ["Append keyed session context", "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries."], access: MethodAccess::Universe(UniverseAction::ControlSession), + ["Append keyed session context", "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries."], access: MethodAccess::Universe(UniverseAction::ControlSession), METHOD_SESSION_CONTEXT_REMOVE => remove_context(ContextRemoveParams) -> ContextRemoveResponse => ["Remove keyed session context", "Removes active entries by stable key with per-key results. Missing keys are idempotent no-ops; runtime-reserved run keys cannot be removed."], access: MethodAccess::Universe(UniverseAction::ControlSession), METHOD_SESSION_CONTEXT_COMPACT => compact_context(ContextCompactParams) -> ContextCompactResponse => @@ -461,6 +430,12 @@ api_methods! { ["List environment registration keys", "Lists this universe's registration keys with policy, status, and derived counts. Each key is the group of the environments it admitted."], access: MethodAccess::Universe(UniverseAction::ConfigureResource), METHOD_ENVIRONMENTS_REGISTRATION_KEYS_REVOKE => revoke_environment_registration_key(EnvironmentRegistrationKeyRevokeParams) -> EnvironmentRegistrationKeyRevokeResponse => ["Revoke an environment registration key", "Stops the key from admitting new daemon identities; already registered daemons keep reconnecting. With closeEnvironments, also closes every non-closed environment the key admitted. Idempotent."], access: MethodAccess::Universe(UniverseAction::ConfigureResource), + METHOD_TRANSCRIPTIONS_START => start_transcription(TranscriptionStartParams) -> TranscriptionResponse => + ["Start transcription", "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created."], access: MethodAccess::Universe(UniverseAction::UseResource), + METHOD_TRANSCRIPTIONS_READ => read_transcription(TranscriptionReadParams) -> TranscriptionResponse => + ["Read transcription", "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire."], access: MethodAccess::Universe(UniverseAction::Read), + METHOD_TRANSCRIPTIONS_CANCEL => cancel_transcription(TranscriptionCancelParams) -> TranscriptionResponse => + ["Cancel transcription", "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority."], access: MethodAccess::Universe(UniverseAction::UseResource), METHOD_MODELS_DEFAULTS_READ => read_model_defaults(ModelDefaultsReadParams) -> ModelDefaultsResponse => ["Read universe model defaults", "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers."], access: MethodAccess::Universe(UniverseAction::Read), METHOD_MODELS_DEFAULTS_PUT => put_model_defaults(ModelDefaultsPutParams) -> ModelDefaultsResponse => diff --git a/crates/api/src/schema_export.rs b/crates/api/src/schema_export.rs index 49408dd9e..68b1ba1e1 100644 --- a/crates/api/src/schema_export.rs +++ b/crates/api/src/schema_export.rs @@ -205,7 +205,7 @@ mod tests { methods.sort_unstable(); methods.dedup(); assert_eq!(methods.len(), total, "duplicate method in manifest"); - assert_eq!(total, 133); + assert_eq!(total, 136); assert_eq!( manifest .iter() diff --git a/crates/api/src/service.rs b/crates/api/src/service.rs index 16a37cac2..6d745b78f 100644 --- a/crates/api/src/service.rs +++ b/crates/api/src/service.rs @@ -2,6 +2,25 @@ use super::*; #[async_trait] pub trait AgentApiService: Send + Sync { + async fn start_transcription( + &self, + _params: TranscriptionStartParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("transcription is unavailable")) + } + async fn read_transcription( + &self, + _params: TranscriptionReadParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("transcription is unavailable")) + } + async fn cancel_transcription( + &self, + _params: TranscriptionCancelParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("transcription is unavailable")) + } + async fn read_vfs_workspace_file( &self, _params: VfsWorkspaceFileReadParams, diff --git a/crates/api/src/sessions.rs b/crates/api/src/sessions.rs index 5e47da7e0..0380ddbfe 100644 --- a/crates/api/src/sessions.rs +++ b/crates/api/src/sessions.rs @@ -862,13 +862,7 @@ pub struct InputAdmissionFailureView { #[serde(rename_all = "camelCase")] pub enum InputAdmissionFailureKind { UnsupportedMedia, - UnsupportedAudioMime, BlobMissing, - BlobTooLarge, - AudioDurationTooLong, - TranscoderUnavailable, - TranscodeFailure, - TranscriptionFailure, AdmissionRejected, } diff --git a/crates/api/src/tests.rs b/crates/api/src/tests.rs index ca1de915a..d9a71b616 100644 --- a/crates/api/src/tests.rs +++ b/crates/api/src/tests.rs @@ -59,6 +59,7 @@ fn notification_serializes_as_json_rpc_lite_shape() { completed_at_ms: Some(20), source: RunViewSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "hello".to_owned(), }], diff --git a/crates/api/src/transcriptions.rs b/crates/api/src/transcriptions.rs new file mode 100644 index 000000000..7eebb0d04 --- /dev/null +++ b/crates/api/src/transcriptions.rs @@ -0,0 +1,123 @@ +use super::*; + +/// Immutable audio input in this universe's content store. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct TranscriptionAudio { + pub blob_ref: String, + pub mime: String, + pub name: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct TranscriptionStartParams { + /// Scoped to the requester. Matching retries rejoin the original job, + /// including after defaults change; changed requests conflict. Identity is + /// retained for the Temporal namespace's workflow-history retention period. + pub idempotency_key: String, + pub audio: TranscriptionAudio, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub language: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub prompt: Option, +} + +impl TranscriptionStartParams { + pub fn validate(&self) -> Result<(), AgentApiError> { + for (name, value, limit) in [ + ("idempotencyKey", self.idempotency_key.as_str(), 200), + ("audio.mime", self.audio.mime.as_str(), 128), + ("audio.name", self.audio.name.as_str(), 256), + ] { + if value.trim().is_empty() || value.len() > limit { + return Err(AgentApiError::invalid_request(format!( + "{name} must contain 1–{limit} bytes" + ))); + } + } + if self + .language + .as_ref() + .is_some_and(|v| v.is_empty() || v.len() > 32) + || self.prompt.as_ref().is_some_and(|v| v.len() > 8192) + { + return Err(AgentApiError::invalid_request( + "language or prompt exceeds the transcription limit", + )); + } + if let Some(model) = &self.model { + ModelDefaultSlot::SpeechToText.validate_model(model)?; + } + Ok(()) + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct TranscriptionReadParams { + pub transcription_id: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct TranscriptionCancelParams { + pub transcription_id: String, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub enum TranscriptionStatus { + Pending, + Running, + Succeeded, + Failed, + Cancelled, + Expired, +} +impl TranscriptionStatus { + pub fn is_terminal(self) -> bool { + !matches!(self, Self::Pending | Self::Running) + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub enum TranscriptionFailureKind { + InvalidAudio, + Configuration, + Provider, + Timeout, + Internal, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionFailure { + pub kind: TranscriptionFailureKind, + pub message: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionView { + pub transcription_id: String, + pub created_by: Attribution, + pub audio: TranscriptionAudio, + pub model: ModelConfig, + pub status: TranscriptionStatus, + pub created_at_ms: u64, + /// Plain UTF-8 transcript blob, usable as ordinary textRef input. + /// Unsubmitted content can be swept after the ordinary CAS grace period. + pub transcript_ref: Option, + pub text: Option, + pub failure: Option, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionResponse { + pub transcription: TranscriptionView, +} diff --git a/crates/api/src/views.rs b/crates/api/src/views.rs index 766ffb7df..9d315566d 100644 --- a/crates/api/src/views.rs +++ b/crates/api/src/views.rs @@ -364,6 +364,10 @@ pub enum InputItem { /// bot deliveries; other values are allowed. Omitted means unknown. /// This metadata is not an authorization identity or model input text. origin: Option, + /// Optional source blob in this universe, retained with the session. + /// Provenance is metadata, not model input or an authorization identity. + #[serde(default, skip_serializing_if = "Option::is_none")] + provenance_ref: Option, text: String, }, TextRef { @@ -373,6 +377,10 @@ pub enum InputItem { /// bot deliveries; other values are allowed. Omitted means unknown. /// This metadata is not an authorization identity or model input text. origin: Option, + /// Optional source blob in this universe, retained with the session. + /// Provenance is metadata, not model input or an authorization identity. + #[serde(default, skip_serializing_if = "Option::is_none")] + provenance_ref: Option, blob_ref: String, }, Media { diff --git a/crates/api/tests/schema_artifacts.rs b/crates/api/tests/schema_artifacts.rs index 147f9d5ac..f470bbcd0 100644 --- a/crates/api/tests/schema_artifacts.rs +++ b/crates/api/tests/schema_artifacts.rs @@ -77,6 +77,7 @@ fn serialized_fixtures_validate_against_exported_schemas() { session_id: "session_1".to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "hello".to_owned(), }], @@ -96,6 +97,7 @@ fn serialized_fixtures_validate_against_exported_schemas() { completed_at_ms: Some(20), source: RunViewSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "hello".to_owned(), }], diff --git a/crates/auth/src/providers.rs b/crates/auth/src/providers.rs index 72da1a26e..e53499f62 100644 --- a/crates/auth/src/providers.rs +++ b/crates/auth/src/providers.rs @@ -216,7 +216,10 @@ fn validate_model_endpoint(endpoint: &ModelEndpointConfig) -> Result<(), AuthReg } let mut kinds = BTreeSet::new(); for kind in &endpoint.api_kinds { - if !matches!(kind.as_str(), "openai:responses" | "openai:completions") { + if !matches!( + kind.as_str(), + "openai:responses" | "openai:completions" | "openai:audio-transcriptions" + ) { return Err(AuthRegistryError::InvalidInput { message: format!("model endpoint API kind {kind:?} is not supported"), }); @@ -613,6 +616,13 @@ mod tests { assert!(validate_model_endpoint(&reserved).is_err()); } + #[test] + fn speech_endpoints_declare_the_audio_protocol() { + let mut speech = endpoint("http://localhost:9000/v1"); + speech.api_kinds = vec!["openai:audio-transcriptions".into()]; + validate_model_endpoint(&speech).expect("speech endpoint"); + } + #[test] fn credentialless_model_endpoint_validates_and_round_trips() { let config = AuthProviderConfig::ModelEndpoint(ModelEndpointOnlyConfig { diff --git a/crates/bots/src/views.rs b/crates/bots/src/views.rs index dfa315d3d..fcf6fdd20 100644 --- a/crates/bots/src/views.rs +++ b/crates/bots/src/views.rs @@ -605,18 +605,29 @@ fn media_kind(kind: BotEventMediaKind) -> MediaKind { /// and its media. fn event_input_items(event: &BotEvent) -> impl Iterator + '_ { std::iter::once(InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: event .prompt_ref .clone() .unwrap_or_else(|| event.document_ref.clone()), }) - .chain(event.media.iter().map(|item| InputItem::Media { - origin: Some("event".to_owned()), - blob_ref: item.blob_ref.clone(), - mime: item.mime.clone(), - kind: media_kind(item.kind), - name: item.name.clone(), + .chain(event.media.iter().map(|item| { + if let Some(text_ref) = &item.text_ref { + InputItem::TextRef { + provenance_ref: Some(item.blob_ref.clone()), + origin: Some("event".into()), + blob_ref: text_ref.clone(), + } + } else { + InputItem::Media { + origin: Some("event".to_owned()), + blob_ref: item.blob_ref.clone(), + mime: item.mime.clone(), + kind: media_kind(item.kind), + name: item.name.clone(), + } + } })) } @@ -627,7 +638,7 @@ fn event_input_items(event: &BotEvent) -> impl Iterator + '_ { pub fn delivery_input_items(events: &[BotEvent]) -> Vec { let mut items = Vec::new(); if events.len() > 1 { - items.push(InputItem::Text { origin: Some("event".to_owned()), + items.push(InputItem::Text { provenance_ref: None, origin: Some("event".to_owned()), text: format!( "{} events delivered as one batch — handle them together and resolve the delivery once.", events.len() @@ -641,6 +652,7 @@ pub fn delivery_input_items(events: &[BotEvent]) -> Vec { /// Steering input for events folded into a running run. pub fn steer_input_items(events: &[BotEvent]) -> Vec { let mut items = vec![InputItem::Text { + provenance_ref: None, origin: Some("event".to_owned()), text: format!( "{} more event(s) arrived while you were working — fold them into your current work where relevant.", @@ -1471,6 +1483,7 @@ mod tests { assert_eq!( delivery_input_items(&[signal_event(&document_ref, Some(&prompt_ref))]), vec![InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: prompt_ref.clone(), }] @@ -1479,6 +1492,7 @@ mod tests { assert_eq!( delivery_input_items(&[signal_event(&document_ref, None)]), vec![InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: document_ref.clone(), }] @@ -1486,6 +1500,7 @@ mod tests { // Media follows its event's rendering. let mut with_media = signal_event(&document_ref, Some(&prompt_ref)); with_media.media = vec![BotEventMedia { + text_ref: None, blob_ref: format!("sha256:{}", "c".repeat(64)), kind: BotEventMediaKind::Image, mime: "image/png".to_owned(), @@ -1505,13 +1520,34 @@ mod tests { ); } + #[test] + fn prepared_attachment_delivery_uses_text_with_source_provenance() { + let reference = format!("sha256:{}", "d".repeat(64)); + let mut event = signal_event(&format!("sha256:{}", "a".repeat(64)), None); + event.media.push(BotEventMedia { + text_ref: Some(reference.clone()), + blob_ref: format!("sha256:{}", "c".repeat(64)), + kind: BotEventMediaKind::Audio, + mime: "audio/ogg".into(), + name: Some("voice.ogg".into()), + }); + assert_eq!( + delivery_input_items(&[event])[1], + InputItem::TextRef { + provenance_ref: Some(format!("sha256:{}", "c".repeat(64))), + origin: Some("event".into()), + blob_ref: reference + } + ); + } + #[test] fn frames_a_batch_with_one_header_line_binding_it_to_one_decision() { let a = format!("sha256:{}", "a".repeat(64)); let b = format!("sha256:{}", "b".repeat(64)); let items = delivery_input_items(&[signal_event(&a, Some(&b)), signal_event(&b, Some(&a))]); assert_eq!(items.len(), 3); - let InputItem::Text { text, origin } = &items[0] else { + let InputItem::Text { text, origin, .. } = &items[0] else { panic!("expected a text header, got {:?}", items[0]); }; assert_eq!(origin.as_deref(), Some("event")); @@ -1520,6 +1556,7 @@ mod tests { assert_eq!( items[1], InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: b.clone() } @@ -1527,6 +1564,7 @@ mod tests { assert_eq!( items[2], InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: a.clone() } @@ -1539,7 +1577,7 @@ mod tests { let b = format!("sha256:{}", "b".repeat(64)); let items = steer_input_items(&[signal_event(&a, Some(&b))]); assert_eq!(items.len(), 2); - let InputItem::Text { text, origin } = &items[0] else { + let InputItem::Text { text, origin, .. } = &items[0] else { panic!("expected a text header, got {:?}", items[0]); }; assert_eq!(origin.as_deref(), Some("event")); @@ -1548,6 +1586,7 @@ mod tests { assert_eq!( items[1], InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: b } diff --git a/crates/channels/src/media.rs b/crates/channels/src/media.rs index 1e2d12a80..57f772775 100644 --- a/crates/channels/src/media.rs +++ b/crates/channels/src/media.rs @@ -248,6 +248,7 @@ impl From for BotEventMedia { Self { blob_ref: item.blob_ref, kind: bot_event_media_kind(item.kind), + text_ref: None, mime: item.mime, name: item.name, } diff --git a/crates/cli/src/chat/driver.rs b/crates/cli/src/chat/driver.rs index 5d44408e7..e88688d15 100644 --- a/crates/cli/src/chat/driver.rs +++ b/crates/cli/src/chat/driver.rs @@ -581,7 +581,11 @@ impl ChatSessionDriver { notify_on_terminal: None, session_id, source: RunStartSource::Input { - items: vec![InputItem::Text { origin: None, text }], + items: vec![InputItem::Text { + provenance_ref: None, + origin: None, + text, + }], }, submission_id: Some(new_submission_id()), config: Some(config), @@ -648,7 +652,11 @@ impl ChatSessionDriver { .steer_run(api::RunSteerParams { session_id: self.session_id.clone(), run_id, - items: vec![InputItem::Text { origin: None, text }], + items: vec![InputItem::Text { + provenance_ref: None, + origin: None, + text, + }], }) .await .map_err(api_error)? diff --git a/crates/cli/src/skills_cli.rs b/crates/cli/src/skills_cli.rs index dcda7d99c..6698dbf94 100644 --- a/crates/cli/src/skills_cli.rs +++ b/crates/cli/src/skills_cli.rs @@ -128,7 +128,11 @@ async fn use_skill(args: SkillsUseArgs) -> Result<()> { .map_err(api_error)? .result .session; - let items = vec![api::InputItem::Text { origin: None, text }]; + let items = vec![api::InputItem::Text { + provenance_ref: None, + origin: None, + text, + }]; if let Some(run) = session.active_run { let response = api .steer_run(api::RunSteerParams { diff --git a/crates/llm-clients/src/content.rs b/crates/llm-clients/src/content.rs index f55224fe7..b21746daf 100644 --- a/crates/llm-clients/src/content.rs +++ b/crates/llm-clients/src/content.rs @@ -35,32 +35,6 @@ pub const OPENAI_COMPLETIONS_MESSAGE_PROVIDER_KIND: &str = "openai.completions.m pub const OPENAI_COMPLETIONS_REASONING_PROVIDER_KIND: &str = "openai.completions.reasoning_state"; pub const OPENAI_RESPONSES_REASONING_PROVIDER_KIND: &str = "openai.responses.reasoning"; pub const ANTHROPIC_THINKING_PROVIDER_KIND: &str = "anthropic.messages.thinking"; -pub const AUDIO_TRANSCRIPT_PROVIDER_KIND: &str = "lightspeed.audio.transcript"; - -/// Transcript content between audio preprocessing and model message lowering. -/// The source audio is recorded separately as the context entry's provenance. -#[derive(Clone, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub struct AudioTranscript { - pub filename: String, - pub text: String, -} - -impl AudioTranscript { - pub fn header(&self) -> String { - format!("[audio transcript: {}]", self.filename) - } - - pub fn model_text(&self) -> String { - format!("{}\n{}", self.header(), self.text) - } -} - -pub fn audio_transcript(raw: &Value) -> Option { - serde_json::from_value::(raw.clone()) - .ok() - .map(|transcript| transcript.text) -} - /// Only provider-exposed text participates in display. Signatures and encrypted /// continuation state remain in the payload for replay. pub fn anthropic_thinking(raw: &Value) -> Option { @@ -194,22 +168,4 @@ mod tests { Some("visible") ); } - - #[test] - fn transcript_labels_are_rendered_without_parsing_body_text() { - let transcript = AudioTranscript { - filename: "voice.ogg".into(), - text: "[audio transcript: quoted]\nkeep this header as speech".into(), - }; - let raw = serde_json::to_value(&transcript).unwrap(); - assert_eq!( - audio_transcript(&raw).as_deref(), - Some(transcript.text.as_str()) - ); - assert_eq!( - transcript.model_text(), - "[audio transcript: voice.ogg]\n[audio transcript: quoted]\nkeep this header as speech" - ); - assert!(audio_transcript(&json!({"text":"missing filename"})).is_none()); - } } diff --git a/crates/llm-clients/src/openai/audio.rs b/crates/llm-clients/src/openai/audio.rs index 23587e3b7..28a056a34 100644 --- a/crates/llm-clients/src/openai/audio.rs +++ b/crates/llm-clients/src/openai/audio.rs @@ -7,7 +7,9 @@ use crate::error::{ ConfigurationError, DecodeError, LlmApiError, ProviderHttpError, TransportError, }; use crate::transport::http::{join_url, normalize_base_url}; -use crate::transport::{ApiResponse, HeaderSnapshot, HttpClient, HttpClientConfig}; +use crate::transport::{ + ApiResponse, EndpointOverride, HeaderSnapshot, HttpClient, HttpClientConfig, +}; use reqwest::header::{AUTHORIZATION, HeaderValue}; use reqwest::{Method, StatusCode, Url}; use serde::de::DeserializeOwned; @@ -141,7 +143,20 @@ impl Client { request: CreateTranscriptionRequest, auth: Option>, ) -> Result, LlmApiError> { - let auth = self.auth_header(auth)?; + self.create_transcription_with_transport(request, auth, None) + .await + } + + pub async fn create_transcription_with_transport( + &self, + request: CreateTranscriptionRequest, + auth: Option>, + endpoint: Option<&EndpointOverride>, + ) -> Result, LlmApiError> { + let auth = match auth { + Some(crate::RequestAuth::None) if endpoint.is_some() => None, + other => Some(self.auth_header(other)?), + }; let file_part = reqwest::multipart::Part::bytes(request.file.bytes) .file_name(request.file.filename) .mime_str(&request.file.mime) @@ -159,10 +174,16 @@ impl Client { form = form.text("prompt", prompt); } - let response = self - .http - .request(Method::POST, self.transcriptions_url.clone()) - .header(AUTHORIZATION, auth) + let mut request_builder = self.http.request_with_endpoint( + Method::POST, + self.transcriptions_url.clone(), + "audio/transcriptions", + endpoint, + )?; + if let Some(auth) = auth { + request_builder = request_builder.header(AUTHORIZATION, auth); + } + let mut response = request_builder .multipart(form) .send() .await @@ -170,10 +191,21 @@ impl Client { let status = response.status(); let headers = HeaderSnapshot::from_headermap(response.headers()); - let body = response - .text() + let mut bytes = Vec::new(); + while let Some(chunk) = response + .chunk() .await - .map_err(|err| map_reqwest_error(err, self.http.config().request_timeout))?; + .map_err(|err| map_reqwest_error(err, self.http.config().request_timeout))? + { + if bytes.len().saturating_add(chunk.len()) > 2 * 1024 * 1024 { + return Err( + DecodeError::new("transcription response exceeds the 2 MiB limit").into(), + ); + } + bytes.extend_from_slice(&chunk); + } + let body = String::from_utf8(bytes) + .map_err(|_| DecodeError::new("transcription response is not UTF-8"))?; parse_json_response(status, headers, body, "OpenAI audio transcription") } } @@ -334,4 +366,73 @@ mod tests { assert!(matches!(error, LlmApiError::Configuration(_))); } + #[tokio::test(flavor = "current_thread")] + async fn custom_transport_isolated_from_deployment_headers_and_key() { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + for auth in [ + crate::RequestAuth::None, + crate::RequestAuth::Bearer("route-key"), + ] { + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = EndpointOverride::from_parts( + &format!("http://{}/custom/v1", listener.local_addr().unwrap()), + &BTreeMap::from([("x-route".into(), "speech".into())]), + ) + .unwrap(); + let server = tokio::spawn(async move { + let (mut stream, _) = listener.accept().await.unwrap(); + let mut bytes = Vec::new(); + loop { + let mut buffer = [0; 4096]; + let count = stream.read(&mut buffer).await.unwrap(); + assert!(count > 0); + bytes.extend_from_slice(&buffer[..count]); + if let Some(end) = bytes.windows(4).position(|v| v == b"\r\n\r\n") { + let headers = String::from_utf8_lossy(&bytes[..end]).to_ascii_lowercase(); + let size: usize = headers + .lines() + .find_map(|l| l.strip_prefix("content-length: ")) + .unwrap() + .parse() + .unwrap(); + if bytes.len() >= end + 4 + size { + break; + } + } + } + let body = r#"{"text":"hello"}"#; + stream.write_all(format!("HTTP/1.1 200 OK\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len()).as_bytes()).await.unwrap(); + String::from_utf8(bytes).unwrap() + }); + let mut config = Config::new("deployment-key"); + config.organization = Some("deployment-org".into()); + config.project = Some("deployment-project".into()); + let client = Client::new(config).unwrap(); + let mut request = CreateTranscriptionRequest::new(AudioFile { + bytes: b"voice".to_vec(), + filename: "voice.ogg".into(), + mime: "audio/ogg".into(), + }); + request.model = "custom-speech-model".into(); + request.language = Some("de".into()); + request.prompt = Some("Names".into()); + let result = client + .create_transcription_with_transport(request, Some(auth), Some(&endpoint)) + .await + .unwrap(); + assert_eq!(result.parsed.text, "hello"); + let request = server.await.unwrap(); + assert!(request.starts_with("POST /custom/v1/audio/transcriptions ")); + assert!(request.contains("x-route: speech")); + assert!(request.contains("custom-speech-model")); + assert!(request.contains("Names")); + assert!(!request.contains("deployment-")); + match auth { + crate::RequestAuth::None => { + assert!(!request.to_ascii_lowercase().contains("authorization:")) + } + _ => assert!(request.contains("Bearer route-key")), + } + } + } } diff --git a/crates/llm-runtime/src/anthropic_messages.rs b/crates/llm-runtime/src/anthropic_messages.rs index 0aa493b67..b7869368d 100644 --- a/crates/llm-runtime/src/anthropic_messages.rs +++ b/crates/llm-runtime/src/anthropic_messages.rs @@ -798,7 +798,7 @@ async fn materialize_block( }; return Ok((role, blocks)); } - let text = crate::blob_io::read_message_text(blobs, &entry.content).await?; + let text = crate::blob_io::read_text(blobs, &entry.content.content_ref).await?; Ok((role, vec![am::ContentBlockParam::text(text)])) } ContextEntryKind::ToolResult { call_id, is_error } => { diff --git a/crates/llm-runtime/src/blob_io.rs b/crates/llm-runtime/src/blob_io.rs index 1a6e01982..0cadb754f 100644 --- a/crates/llm-runtime/src/blob_io.rs +++ b/crates/llm-runtime/src/blob_io.rs @@ -71,26 +71,6 @@ pub fn document_entry(media_type: Option<&str>, preview: Option<&str>) -> Option }) } -/// Render a preprocessed transcript at the provider boundary. Ordinary authored -/// text stays byte-for-byte text, even when it resembles JSON or a transcript. -pub async fn read_message_text( - blobs: &dyn BlobStore, - content: &engine::ContentRef, -) -> LlmAdapterResult { - if content.provider_kind.as_deref() - == Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) - { - let raw = read_json(blobs, &content.content_ref).await?; - let transcript: llm_clients::content::AudioTranscript = serde_json::from_value(raw) - .map_err(|error| LlmAdapterError::InvalidJson { - blob_ref: content.content_ref.clone(), - message: error.to_string(), - })?; - return Ok(transcript.model_text()); - } - read_text(blobs, &content.content_ref).await -} - pub async fn read_base64(blobs: &dyn BlobStore, blob_ref: &BlobRef) -> LlmAdapterResult { use base64::Engine as _; let bytes = blobs.read_bytes(blob_ref).await?; diff --git a/crates/llm-runtime/src/lib.rs b/crates/llm-runtime/src/lib.rs index 93e22bf6f..ed8a2a5f2 100644 --- a/crates/llm-runtime/src/lib.rs +++ b/crates/llm-runtime/src/lib.rs @@ -36,7 +36,7 @@ pub use params::{ pub use provider_keys::{ ModelProviderResolver, NoStoredModelProviders, NoStoredProviderKeys, ProviderAuthScheme, ProviderKeyError, ResolvedEndpoint, ResolvedModelProvider, ResolvedProviderAuth, - StaticModelProviders, StaticProviderKeys, + StaticModelProviders, StaticProviderKeys, resolve_provider_route, }; pub use result::{LlmDebugDumps, LlmGenerationExecution, failed_generation_result}; pub use secrets::{ diff --git a/crates/llm-runtime/src/openai_completions.rs b/crates/llm-runtime/src/openai_completions.rs index 73fb69f3a..984e24ff0 100644 --- a/crates/llm-runtime/src/openai_completions.rs +++ b/crates/llm-runtime/src/openai_completions.rs @@ -688,7 +688,7 @@ async fn materialize_message( } } else { oai_c::CompletionMessageContent::Text( - crate::blob_io::read_message_text(blobs, &entry.content).await?, + crate::blob_io::read_text(blobs, &entry.content.content_ref).await?, ) }; Ok(oai_c::CompletionMessage { diff --git a/crates/llm-runtime/src/openai_responses.rs b/crates/llm-runtime/src/openai_responses.rs index 6b14eda2e..0122bca78 100644 --- a/crates/llm-runtime/src/openai_responses.rs +++ b/crates/llm-runtime/src/openai_responses.rs @@ -514,7 +514,7 @@ async fn materialize_input_item( extra: Default::default(), })); } - let text = crate::blob_io::read_message_text(blobs, &item.content).await?; + let text = crate::blob_io::read_text(blobs, &item.content.content_ref).await?; Ok(oai::ResponseInputItem::Message(oai::InputMessage { role, content: oai::InputMessageContent::Text(text), diff --git a/crates/llm-runtime/src/provider_keys.rs b/crates/llm-runtime/src/provider_keys.rs index c0a69f1b4..5084760fd 100644 --- a/crates/llm-runtime/src/provider_keys.rs +++ b/crates/llm-runtime/src/provider_keys.rs @@ -72,9 +72,8 @@ impl ResolvedEndpoint { }) } - fn supports(&self, api_kind: &ProviderApiKind) -> bool { - self.api_kinds - .contains_key(provider_api_kind_name(api_kind)) + pub fn supports(&self, api_kind: &str) -> bool { + self.api_kinds.contains_key(api_kind) } } @@ -133,42 +132,57 @@ pub(crate) async fn resolve_model_provider( resolver: &dyn ModelProviderResolver, model: &ModelSelection, ) -> Result, LlmAdapterError> { - let resolved = resolver - .resolve_model_provider(&model.provider_id) - .await - .map_err(|error| LlmAdapterError::ProviderKeyResolution { - message: error.to_string(), - })?; - if resolved.is_none() && !is_builtin_provider(&model.provider_id) { - return Err(LlmAdapterError::ProviderKeyResolution { + resolve_provider_route( + resolver, + &model.provider_id, + provider_api_kind_name(&model.api_kind), + ) + .await + .map_err(|error| LlmAdapterError::ProviderKeyResolution { + message: error.to_string(), + }) +} + +/// Resolve a protocol route without extending the engine's generation model enum. +/// Only built-in providers may use deployment transport defaults. +pub async fn resolve_provider_route( + resolver: &dyn ModelProviderResolver, + provider_id: &str, + api_kind: &str, +) -> Result, ProviderKeyError> { + let resolved = resolver.resolve_model_provider(provider_id).await?; + if resolved.is_none() && !is_builtin_provider(provider_id) { + return Err(ProviderKeyError::NotUsable { + provider_id: provider_id.into(), message: format!( "custom model provider {} has no universe model-provider record", - model.provider_id + provider_id ), }); } - if !is_builtin_provider(&model.provider_id) + if !is_builtin_provider(provider_id) && resolved .as_ref() .is_some_and(|provider| provider.endpoint.is_none()) { - return Err(LlmAdapterError::ProviderKeyResolution { + return Err(ProviderKeyError::NotUsable { + provider_id: provider_id.into(), message: format!( "custom model provider {} has no endpoint configuration", - model.provider_id + provider_id ), }); } if let Some(endpoint) = resolved .as_ref() .and_then(|provider| provider.endpoint.as_ref()) - && !endpoint.supports(&model.api_kind) + && !endpoint.supports(api_kind) { - return Err(LlmAdapterError::ProviderKeyResolution { + return Err(ProviderKeyError::NotUsable { + provider_id: provider_id.into(), message: format!( "model provider {} endpoint does not admit API kind {}", - model.provider_id, - provider_api_kind_name(&model.api_kind) + provider_id, api_kind ), }); } @@ -380,3 +394,61 @@ mod tests { )); } } + +#[cfg(test)] +mod route_tests { + use super::*; + #[tokio::test(flavor = "current_thread")] + async fn transcription_routes_require_matching_explicit_custom_endpoints() { + assert!( + resolve_provider_route( + &NoStoredModelProviders, + "openai", + "openai:audio-transcriptions" + ) + .await + .unwrap() + .is_none() + ); + assert!(matches!( + resolve_provider_route( + &NoStoredModelProviders, + "custom", + "openai:audio-transcriptions" + ) + .await, + Err(ProviderKeyError::NotUsable { .. }) + )); + let route = ResolvedModelProvider { + auth: None, + endpoint: Some( + ResolvedEndpoint::new( + "http://localhost:9999/v1", + &BTreeMap::new(), + ["openai:audio-transcriptions".into()], + ) + .unwrap(), + ), + }; + struct Resolver(ResolvedModelProvider); + #[async_trait] + impl ModelProviderResolver for Resolver { + async fn resolve_model_provider( + &self, + _: &str, + ) -> Result, ProviderKeyError> { + Ok(Some(self.0.clone())) + } + } + let resolver = Resolver(route); + let resolved = resolve_provider_route(&resolver, "custom", "openai:audio-transcriptions") + .await + .unwrap() + .unwrap(); + assert!(resolved.auth.is_none()); + assert!(matches!( + resolve_provider_route(&resolver, "custom", "openai:responses").await, + Err(ProviderKeyError::NotUsable { .. }) + )); + } +} diff --git a/crates/llm-runtime/tests/content_materialization.rs b/crates/llm-runtime/tests/content_materialization.rs index f63a622cb..43754cc9a 100644 --- a/crates/llm-runtime/tests/content_materialization.rs +++ b/crates/llm-runtime/tests/content_materialization.rs @@ -1,27 +1,20 @@ use engine::{ - BlobRef, ContentRef, ContextEntry, ContextEntryId, ContextEntryKind, ContextEntrySource, + ContentRef, ContextEntry, ContextEntryId, ContextEntryKind, ContextEntrySource, ContextMessageRole, ContextSnapshot, LlmRequest, ModelSelection, ProviderApiKind, storage::{BlobStore, InMemoryBlobStore}, }; -use llm_clients::content::{AUDIO_TRANSCRIPT_PROVIDER_KIND, AudioTranscript}; use serde_json::{Value, json}; #[tokio::test(flavor = "current_thread")] -async fn structured_transcripts_and_authored_json_lower_across_all_provider_apis() { +async fn text_with_provenance_is_unchanged_across_all_provider_apis() { let blobs = InMemoryBlobStore::new(); - let transcript = AudioTranscript { - filename: "voice.ogg".into(), - text: "[audio transcript: quoted]\nKeep these spoken words.".into(), - }; - let bytes = serde_json::to_vec(&transcript).unwrap(); - let reference = blobs.put_bytes(bytes.clone()).await.unwrap(); let source = blobs.put_bytes(b"source audio".to_vec()).await.unwrap(); - for structured in [true, false] { - let expected = if structured { - transcript.model_text() - } else { - String::from_utf8(bytes.clone()).unwrap() - }; + for expected in [ + "[audio transcript: quoted]\nKeep these spoken words.", + r#"{"filename":"voice.ogg","text":"ordinary authored JSON"}"#, + ] { + let bytes = expected.as_bytes().to_vec(); + let reference = blobs.put_bytes(bytes.clone()).await.unwrap(); for api_kind in [ ProviderApiKind::OpenAiResponses, ProviderApiKind::OpenAiCompletions, @@ -36,10 +29,10 @@ async fn structured_transcripts_and_authored_json_lower_across_all_provider_apis source: ContextEntrySource::ContextEdit, content: ContentRef { content_ref: reference.clone(), - media_type: Some("application/json".into()), - provider_kind: structured.then(|| AUDIO_TRANSCRIPT_PROVIDER_KIND.to_owned()), + media_type: Some("text/plain".into()), + provider_kind: None, }, - preview: structured.then(|| transcript.header()), + preview: None, origin: None, provenance_ref: Some(source.clone()), token_estimate: None, @@ -103,5 +96,4 @@ async fn structured_transcripts_and_authored_json_lower_across_all_provider_apis assert_eq!(blobs.read_bytes(&reference).await.unwrap(), bytes); } } - assert_eq!(reference, BlobRef::from_bytes(&bytes)); } diff --git a/crates/store-pg/src/bots.rs b/crates/store-pg/src/bots.rs index 7dcf5bd50..1ea875d51 100644 --- a/crates/store-pg/src/bots.rs +++ b/crates/store-pg/src/bots.rs @@ -981,6 +981,12 @@ impl BotEventStore for PgStore { let mut refs = std::collections::BTreeSet::from([record.document_ref.clone()]); refs.extend(record.prompt_ref.iter().cloned()); refs.extend(record.media.iter().map(|media| media.blob_ref.clone())); + refs.extend( + record + .media + .iter() + .filter_map(|media| media.text_ref.clone()), + ); refs.extend( record .receiver diff --git a/crates/store-pg/tests/bots_pg.rs b/crates/store-pg/tests/bots_pg.rs index 97361b5e9..0d45c2826 100644 --- a/crates/store-pg/tests/bots_pg.rs +++ b/crates/store-pg/tests/bots_pg.rs @@ -1253,6 +1253,10 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs .put_bytes(b"media attachment".to_vec()) .await .expect("media"); + let transcript = store + .put_bytes(b"prepared transcript".to_vec()) + .await + .expect("transcript"); let mut record = event(&bot_id, "with-tools", 1, 2, None, None); record.receiver = Some(bots::EventReceiver::Workflow { workflow_id: "conversation".to_owned(), @@ -1261,9 +1265,10 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs tools_ref: Some(tools.to_string()), }); record.media = vec![api::BotEventMedia { + text_ref: Some(transcript.to_string()), blob_ref: media.to_string(), - kind: api::BotEventMediaKind::Image, - mime: "image/png".to_owned(), + kind: api::BotEventMediaKind::Audio, + mime: "audio/ogg".to_owned(), name: None, }]; store @@ -1279,8 +1284,8 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs .await .expect("roots"); assert_eq!( - roots, 4, - "document, prompt, media, and receiver tools are retained" + roots, 5, + "document, prompt, source audio, transcript, and receiver tools are retained" ); sqlx::query("UPDATE cas_blobs SET created_at_ms = 1, touched_at_ms = 1 WHERE universe_id = $1") .bind(store.config().universe_id) @@ -1291,6 +1296,7 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs &record.document_ref, record.prompt_ref.as_ref().unwrap(), &media.to_string(), + &transcript.to_string(), &tools.to_string(), ] .into_iter() @@ -1342,7 +1348,7 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs .await .expect("sweep released roots") .len(), - 4 + 5 ); drop_universe(&store).await; } diff --git a/crates/temporal-server/src/bots/sessions.rs b/crates/temporal-server/src/bots/sessions.rs index 1509f0a1f..b8f62199a 100644 --- a/crates/temporal-server/src/bots/sessions.rs +++ b/crates/temporal-server/src/bots/sessions.rs @@ -672,6 +672,7 @@ pub async fn append_context( .map(|event| ContextAppendEntry { key: appended_event_context_key(&event.id), item: InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: event .prompt_ref diff --git a/crates/temporal-server/src/channels/activities.rs b/crates/temporal-server/src/channels/activities.rs index 8b850d6b6..6b51800d0 100644 --- a/crates/temporal-server/src/channels/activities.rs +++ b/crates/temporal-server/src/channels/activities.rs @@ -664,9 +664,34 @@ pub async fn emit_chat_event( &request.conversation.key(), &request.message.message_id, ); - let mut input = StoreBotEventInput::new(event_id.clone(), chat_message_document(&request)); + let original_summary = chat_message_document(&request).summary; + let original_text = request.message.text.clone(); + let mut prepared_media = Vec::with_capacity(request.media.len()); + for item in &request.media { + let mut prepared = BotEventMedia::from(item.clone()); + if let Some(reference) = request.transcript_refs.get(&item.blob_ref) { + let text = store + .read_text(&parse_blob_ref(reference)?) + .await + .map_err(|e| blob_error("read transcript", e))?; + if !request.message.text.is_empty() { + request.message.text.push_str("\n\n"); + } + request.message.text.push_str(&text); + prepared.text_ref = Some(reference.clone()); + } + prepared_media.push(prepared); + } + let mut document = chat_message_document(&request); + document.summary = original_summary; + if !request.transcript_refs.is_empty() { + if let Some(data) = document.data.as_mut() { + data["message"]["originalText"] = serde_json::Value::String(original_text); + } + } + let mut input = StoreBotEventInput::new(event_id.clone(), document); input.prompt_data = Some(chat_prompt_data(&request.media)); - input.media = request.media.into_iter().map(BotEventMedia::from).collect(); + input.media = prepared_media; input.receiver = Some(EventReceiver::Workflow { workflow_id: request.notify.workflow_id, workflow_kind: request.notify.workflow_kind, @@ -913,6 +938,7 @@ mod tests { fn emit_request(text: &str, media: Vec) -> ChatEmitEventRequest { ChatEmitEventRequest { + transcript_refs: Default::default(), universe_id: Uuid::nil(), bot_id: BotId::new("triage"), trigger_id: BotTriggerId::new("tg"), diff --git a/crates/temporal-server/src/gateway/service/errors.rs b/crates/temporal-server/src/gateway/service/errors.rs index 265f812ce..5bd2800e6 100644 --- a/crates/temporal-server/src/gateway/service/errors.rs +++ b/crates/temporal-server/src/gateway/service/errors.rs @@ -19,27 +19,6 @@ pub(super) fn map_admission_failure_to_api_error(failure: &AgentAdmissionFailure AgentAdmissionFailureKind::RejectedCommand => { AgentApiError::rejected(failure.message.clone()) } - AgentAdmissionFailureKind::UnsupportedAudioMime => { - AgentApiError::unsupported_audio_mime(failure.message.clone()) - } - AgentAdmissionFailureKind::AudioBlobMissing => { - AgentApiError::invalid_request(failure.message.clone()) - } - AgentAdmissionFailureKind::AudioBlobTooLarge => { - AgentApiError::audio_blob_too_large(failure.message.clone()) - } - AgentAdmissionFailureKind::AudioDurationTooLong => { - AgentApiError::audio_duration_too_long(failure.message.clone()) - } - AgentAdmissionFailureKind::TranscoderUnavailable => { - AgentApiError::transcoder_unavailable(failure.message.clone()) - } - AgentAdmissionFailureKind::TranscodeFailure => { - AgentApiError::transcode_failure(failure.message.clone()) - } - AgentAdmissionFailureKind::TranscriptionFailure => { - AgentApiError::transcription_failure(failure.message.clone()) - } } } diff --git a/crates/temporal-server/src/gateway/service/input.rs b/crates/temporal-server/src/gateway/service/input.rs index 69b4e19c8..5045abad5 100644 --- a/crates/temporal-server/src/gateway/service/input.rs +++ b/crates/temporal-server/src/gateway/service/input.rs @@ -2,19 +2,6 @@ use super::*; /// Images and documents, bounded per run. const ALLOWED_IMAGE_MIMES: &[&str] = &["image/jpeg", "image/png", "image/webp", "image/gif"]; -/// Bounded audio blobs are accepted at admission, then rewritten by -/// workflow preprocessing before core planning. -const ALLOWED_AUDIO_MIMES: &[&str] = &[ - "audio/mpeg", - "audio/mp4", - "audio/wav", - "audio/webm", - "audio/ogg", - "audio/aac", - "audio/amr", - "audio/3gpp", - "audio/3gpp2", -]; /// PDF is the only document type both providers accept natively; the text /// MIMEs are inlined as text by the llm-runtime adapters. const PDF_MIME: &str = "application/pdf"; @@ -25,7 +12,6 @@ const TEXT_DOCUMENT_MIMES: &[&str] = &[ "application/json", ]; const MAX_IMAGE_BYTES: u64 = 10 * 1024 * 1024; -const MAX_AUDIO_BYTES: u64 = 25 * 1024 * 1024; const MAX_PDF_BYTES: u64 = 10 * 1024 * 1024; /// Text documents land in model context verbatim; keep them small. const MAX_TEXT_DOCUMENT_BYTES: u64 = 1024 * 1024; @@ -39,6 +25,7 @@ pub(super) async fn run_input_from_api( let mut media_items = 0usize; for item in input { let origin = input_origin_from_api(item)?; + let provenance_ref = input_provenance_from_api(store, item).await?; let first = entries.len(); match item { InputItem::Text { text, .. } => { @@ -87,6 +74,7 @@ pub(super) async fn run_input_from_api( } for entry in &mut entries[first..] { entry.origin = origin.clone(); + entry.provenance_ref = provenance_ref.clone(); } } @@ -103,17 +91,12 @@ async fn media_message_input( kind: MediaKind, name: Option<&str>, ) -> Result { - let raw_mime = mime.trim().to_ascii_lowercase(); - let mime = if matches!(kind, MediaKind::Audio) { - normalize_audio_mime(&raw_mime) - } else { - raw_mime - .split(';') - .next() - .unwrap_or_default() - .trim() - .to_owned() - }; + let mime = mime + .split(';') + .next() + .unwrap_or_default() + .trim() + .to_ascii_lowercase(); let (label, max_bytes) = match kind { MediaKind::Image => { if !ALLOWED_IMAGE_MIMES.contains(&mime.as_str()) { @@ -125,13 +108,9 @@ async fn media_message_input( ("image", MAX_IMAGE_BYTES) } MediaKind::Audio => { - if !ALLOWED_AUDIO_MIMES.contains(&mime.as_str()) { - return Err(AgentApiError::unsupported_audio_mime(format!( - "unsupported audio mime type {mime}; allowed: {}", - ALLOWED_AUDIO_MIMES.join(", ") - ))); - } - ("audio", MAX_AUDIO_BYTES) + return Err(AgentApiError::invalid_request( + "audio must be transcribed with transcriptions/start before session admission; submit text or a text reference", + )); } MediaKind::Document if mime == PDF_MIME => ("document", MAX_PDF_BYTES), MediaKind::Document if TEXT_DOCUMENT_MIMES.contains(&mime.as_str()) => { @@ -150,17 +129,10 @@ async fn media_message_input( .await .map_err(map_input_blob_store_error)?; if info.byte_len > max_bytes { - return if matches!(kind, MediaKind::Audio) { - Err(AgentApiError::audio_blob_too_large(format!( - "{label} blob is {} bytes; the limit is {max_bytes} bytes", - info.byte_len - ))) - } else { - Err(AgentApiError::invalid_request(format!( - "{label} blob is {} bytes; the limit is {max_bytes} bytes", - info.byte_len - ))) - }; + return Err(AgentApiError::invalid_request(format!( + "{label} blob is {} bytes; the limit is {max_bytes} bytes", + info.byte_len + ))); } if matches!(kind, MediaKind::Document) && mime != PDF_MIME { // Text documents reach the model as text; reject undecodable bytes @@ -244,6 +216,7 @@ pub(super) async fn context_entry_input_from_api( } }?; entry.origin = origin; + entry.provenance_ref = input_provenance_from_api(store, item).await?; Ok(entry) } @@ -288,26 +261,6 @@ pub(super) fn empty_run_input_error() -> AgentApiError { ) } -fn normalize_audio_mime(mime: &str) -> String { - let mime = mime - .split(';') - .next() - .unwrap_or_default() - .trim() - .to_ascii_lowercase(); - match mime.as_str() { - "audio/mp3" => "audio/mpeg", - "audio/x-m4a" | "audio/m4a" => "audio/mp4", - "audio/x-wav" | "audio/wave" | "audio/vnd.wave" => "audio/wav", - "audio/oga" | "audio/opus" => "audio/ogg", - "audio/x-aac" => "audio/aac", - "audio/3gp" => "audio/3gpp", - "audio/3gpp2" | "audio/3g2" => "audio/3gpp2", - other => other, - } - .to_owned() -} - fn input_origin_from_api(item: &InputItem) -> Result, AgentApiError> { let origin = match item { InputItem::Text { origin, .. } @@ -324,3 +277,24 @@ fn input_origin_from_api(item: &InputItem) -> Result, AgentApiErr } Ok(origin.clone()) } + +async fn input_provenance_from_api( + store: &dyn BlobStore, + item: &InputItem, +) -> Result, AgentApiError> { + let reference = match item { + InputItem::Text { provenance_ref, .. } | InputItem::TextRef { provenance_ref, .. } => { + provenance_ref + } + _ => return Ok(None), + }; + let Some(reference) = reference else { + return Ok(None); + }; + let reference = parse_blob_ref(reference)?; + store + .stat_blob(&reference) + .await + .map_err(map_input_blob_store_error)?; + Ok(Some(reference)) +} diff --git a/crates/temporal-server/src/gateway/service/mod.rs b/crates/temporal-server/src/gateway/service/mod.rs index 630adc7eb..2fd14614d 100644 --- a/crates/temporal-server/src/gateway/service/mod.rs +++ b/crates/temporal-server/src/gateway/service/mod.rs @@ -24,6 +24,7 @@ mod models_api; mod oauth_api; mod parse; mod profiles; +mod transcriptions; pub(crate) use crate::environments::provider_controllers; mod session_jobs; mod session_lifecycle; @@ -394,9 +395,7 @@ async fn context_append_result( let entry = project_context_entry_inputs(std::slice::from_ref(input)) .into_iter() .next(); - let activation_text = if is_audio_transcript_entry(input) { - api_projection::project_content_text(store, &input.content).await? - } else if context_append_entry_has_activation_text(input) { + let activation_text = if context_append_entry_has_activation_text(input) { // The submitted text is reused when it produced this exact entry so // plain-text appends do not pay a blob read per response entry. match submitted_text { @@ -483,35 +482,11 @@ fn active_entry_input(entry: &ContextEntry) -> ContextEntryInput { } fn active_context_entry_matches_input(active: &ContextEntry, input: &ContextEntryInput) -> bool { - let active_input = active_entry_input(active); - active_input == *input || audio_input_matches_transcript(input, &active_input) -} - -fn audio_input_matches_transcript(input: &ContextEntryInput, active: &ContextEntryInput) -> bool { - input - .content - .media_type - .as_deref() - .is_some_and(|mime| mime.trim().to_ascii_lowercase().starts_with("audio/")) - && is_audio_transcript_entry(active) - && active.provenance_ref.as_ref() == Some(&input.content.content_ref) -} - -fn is_audio_transcript_entry(input: &ContextEntryInput) -> bool { - input.content.provider_kind.as_deref() - == Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) + active_entry_input(active) == *input } fn input_admission_failure_from_api_error(error: AgentApiError) -> InputAdmissionFailureView { let kind = match error.kind { - AgentApiErrorKind::UnsupportedAudioMime => InputAdmissionFailureKind::UnsupportedAudioMime, - AgentApiErrorKind::AudioBlobTooLarge => InputAdmissionFailureKind::BlobTooLarge, - AgentApiErrorKind::AudioDurationTooLong => InputAdmissionFailureKind::AudioDurationTooLong, - AgentApiErrorKind::TranscoderUnavailable => { - InputAdmissionFailureKind::TranscoderUnavailable - } - AgentApiErrorKind::TranscodeFailure => InputAdmissionFailureKind::TranscodeFailure, - AgentApiErrorKind::TranscriptionFailure => InputAdmissionFailureKind::TranscriptionFailure, AgentApiErrorKind::NotFound => InputAdmissionFailureKind::BlobMissing, _ => InputAdmissionFailureKind::UnsupportedMedia, }; @@ -525,21 +500,6 @@ fn input_admission_failure_from_workflow( failure: &AgentAdmissionFailure, ) -> InputAdmissionFailureView { let kind = match failure.kind { - AgentAdmissionFailureKind::UnsupportedAudioMime => { - InputAdmissionFailureKind::UnsupportedAudioMime - } - AgentAdmissionFailureKind::AudioBlobMissing => InputAdmissionFailureKind::BlobMissing, - AgentAdmissionFailureKind::AudioBlobTooLarge => InputAdmissionFailureKind::BlobTooLarge, - AgentAdmissionFailureKind::AudioDurationTooLong => { - InputAdmissionFailureKind::AudioDurationTooLong - } - AgentAdmissionFailureKind::TranscoderUnavailable => { - InputAdmissionFailureKind::TranscoderUnavailable - } - AgentAdmissionFailureKind::TranscodeFailure => InputAdmissionFailureKind::TranscodeFailure, - AgentAdmissionFailureKind::TranscriptionFailure => { - InputAdmissionFailureKind::TranscriptionFailure - } AgentAdmissionFailureKind::RejectedCommand => InputAdmissionFailureKind::AdmissionRejected, }; InputAdmissionFailureView { @@ -2014,6 +1974,25 @@ impl AgentApiService for GatewayAgentApi { .map(AgentApiOutcome::new) } + async fn start_transcription( + &self, + params: TranscriptionStartParams, + ) -> Result, AgentApiError> { + self.start_transcription_impl(params).await + } + async fn read_transcription( + &self, + params: TranscriptionReadParams, + ) -> Result, AgentApiError> { + self.read_transcription_impl(params).await + } + async fn cancel_transcription( + &self, + params: TranscriptionCancelParams, + ) -> Result, AgentApiError> { + self.cancel_transcription_impl(params).await + } + async fn read_model_defaults( &self, _params: api::ModelDefaultsReadParams, diff --git a/crates/temporal-server/src/gateway/service/tests.rs b/crates/temporal-server/src/gateway/service/tests.rs index 3997e5954..aec1ca861 100644 --- a/crates/temporal-server/src/gateway/service/tests.rs +++ b/crates/temporal-server/src/gateway/service/tests.rs @@ -18,30 +18,6 @@ fn admission_failure_mapping_uses_gateway_error_kinds() { map_admission_failure_to_api_error(&revision_conflict).kind, AgentApiErrorKind::Conflict ); - assert_eq!( - map_admission_failure_to_api_error(&failure( - AgentAdmissionFailureKind::UnsupportedAudioMime - )) - .kind, - AgentApiErrorKind::UnsupportedAudioMime - ); - assert_eq!( - map_admission_failure_to_api_error(&failure(AgentAdmissionFailureKind::AudioBlobMissing)) - .kind, - AgentApiErrorKind::InvalidRequest - ); - assert_eq!( - map_admission_failure_to_api_error(&failure( - AgentAdmissionFailureKind::TranscriptionFailure - )) - .kind, - AgentApiErrorKind::TranscriptionFailure - ); - assert_eq!( - map_admission_failure_to_api_error(&failure(AgentAdmissionFailureKind::TranscodeFailure)) - .kind, - AgentApiErrorKind::TranscodeFailure - ); } #[test] @@ -1531,6 +1507,7 @@ async fn context_entry_input_from_api_stores_text_as_user_message() { let entry = context_entry_input_from_api( &store, &InputItem::Text { + provenance_ref: None, origin: None, text: " [telegram] Alice (12:01): hi ".to_owned(), }, @@ -1561,6 +1538,7 @@ async fn context_entry_input_from_api_rejects_empty_text() { let error = context_entry_input_from_api( &store, &InputItem::Text { + provenance_ref: None, origin: None, text: " ".to_owned(), }, @@ -1637,6 +1615,7 @@ async fn context_entry_input_from_api_preserves_text_ref() { let entry = context_entry_input_from_api( &store, &InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: blob_ref.as_str().to_owned(), }, @@ -1659,6 +1638,7 @@ async fn run_input_from_api_maps_image_media_to_user_message_entry() { &store, &[ InputItem::Text { + provenance_ref: None, origin: None, text: "what is this?".to_owned(), }, @@ -1775,30 +1755,28 @@ async fn run_input_from_api_rejects_unsupported_document_media() { } #[tokio::test(flavor = "current_thread")] -async fn run_input_from_api_maps_audio_media_to_user_message_entry() { +async fn run_input_from_api_rejects_unprepared_audio() { let store = engine::storage::InMemoryBlobStore::new(); let blob_ref = store .put_bytes(b"OggS fake voice note".to_vec()) .await .expect("store audio"); - let input = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: blob_ref.as_str().to_owned(), - mime: "audio/ogg".to_owned(), - kind: api::MediaKind::Audio, - name: Some("voice.ogg".to_owned()), - }], - ) - .await - .expect("input"); - - assert_eq!(input.len(), 1); - assert_eq!(input[0].content.content_ref, blob_ref); - assert_eq!(input[0].content.media_type.as_deref(), Some("audio/ogg")); - assert_eq!(input[0].preview.as_deref(), Some("[audio: voice.ogg]")); + let item = InputItem::Media { + origin: None, + blob_ref: blob_ref.to_string(), + mime: "audio/ogg".into(), + kind: api::MediaKind::Audio, + name: Some("voice.ogg".into()), + }; + let run_error = run_input_from_api(&store, std::slice::from_ref(&item)) + .await + .expect_err("audio requires transcription"); + let append_error = context_entry_input_from_api(&store, &item) + .await + .expect_err("context audio requires transcription"); + assert_eq!(run_error.kind, AgentApiErrorKind::InvalidRequest); + assert_eq!(append_error, run_error); } #[tokio::test(flavor = "current_thread")] @@ -1806,20 +1784,6 @@ async fn run_input_from_api_rejects_unsupported_media() { let store = engine::storage::InMemoryBlobStore::new(); let blob_ref = store.put_bytes(vec![1, 2, 3]).await.expect("store blob"); - let audio = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: blob_ref.as_str().to_owned(), - mime: "audio/flac".to_owned(), - kind: api::MediaKind::Audio, - name: None, - }], - ) - .await - .expect_err("unsupported audio mime must be rejected"); - assert_eq!(audio.kind, AgentApiErrorKind::UnsupportedAudioMime); - let bad_mime = run_input_from_api( &store, &[InputItem::Media { @@ -1835,73 +1799,6 @@ async fn run_input_from_api_rejects_unsupported_media() { assert_eq!(bad_mime.kind, AgentApiErrorKind::InvalidRequest); } -#[tokio::test(flavor = "current_thread")] -async fn run_input_from_api_accepts_transcodable_audio_media() { - let store = engine::storage::InMemoryBlobStore::new(); - let blob_ref = store.put_bytes(vec![1, 2, 3]).await.expect("store blob"); - - let input = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: blob_ref.as_str().to_owned(), - mime: "audio/x-aac".to_owned(), - kind: api::MediaKind::Audio, - name: Some("clip.aac".to_owned()), - }], - ) - .await - .expect("transcodable audio should be admitted"); - - assert_eq!(input[0].content.content_ref, blob_ref); - assert_eq!(input[0].content.media_type.as_deref(), Some("audio/aac")); - assert_eq!(input[0].preview.as_deref(), Some("[audio: clip.aac]")); -} - -#[tokio::test(flavor = "current_thread")] -async fn run_input_from_api_rejects_audio_over_byte_cap() { - let store = engine::storage::InMemoryBlobStore::new(); - let blob_ref = store - .put_bytes(vec![0; 25 * 1024 * 1024 + 1]) - .await - .expect("store large audio"); - - let error = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: blob_ref.as_str().to_owned(), - mime: "audio/ogg".to_owned(), - kind: api::MediaKind::Audio, - name: None, - }], - ) - .await - .expect_err("oversized audio must be rejected"); - - assert_eq!(error.kind, AgentApiErrorKind::AudioBlobTooLarge); -} - -#[tokio::test(flavor = "current_thread")] -async fn run_input_from_api_rejects_missing_audio_blob() { - let store = engine::storage::InMemoryBlobStore::new(); - - let error = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: BlobRef::from_bytes(b"missing-audio").as_str().to_owned(), - mime: "audio/ogg".to_owned(), - kind: api::MediaKind::Audio, - name: None, - }], - ) - .await - .expect_err("missing audio blob must be rejected"); - - assert_eq!(error.kind, AgentApiErrorKind::InvalidRequest); -} - #[tokio::test(flavor = "current_thread")] async fn context_entry_input_from_api_accepts_media() { let store = engine::storage::InMemoryBlobStore::new(); @@ -1939,6 +1836,7 @@ async fn run_input_from_api_preserves_single_text_ref() { let input = run_input_from_api( &store, &[InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: blob_ref.as_str().to_owned(), }], @@ -1965,10 +1863,12 @@ async fn run_input_from_api_stores_text_and_preserves_refs() { &store, &[ InputItem::Text { + provenance_ref: None, origin: None, text: " first ".to_owned(), }, InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: blob_ref.as_str().to_owned(), }, @@ -2366,86 +2266,28 @@ fn auth_flow_views_carry_derived_status() { assert_eq!(expired.status, api::AuthFlowStatus::Expired); } -#[tokio::test(flavor = "current_thread")] -async fn structured_transcript_append_keeps_spoken_headers_and_source_idempotency() { - use llm_clients::content::{AUDIO_TRANSCRIPT_PROVIDER_KIND, AudioTranscript}; - let blobs = engine::storage::InMemoryBlobStore::new(); - let audio_ref = blobs.put_bytes(b"source audio".to_vec()).await.unwrap(); - let original = ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/ogg".into()), - provider_kind: None, - }, - preview: Some("[audio: voice.ogg]".into()), - origin: None, - provenance_ref: None, - token_estimate: None, - }; - let transcript = AudioTranscript { - filename: "voice.ogg".into(), - text: "[audio transcript: spoken words]\nDo not strip this line.".into(), - }; - let rewritten = ContextEntryInput { - kind: original.kind.clone(), - content: engine::ContentRef { - content_ref: blobs - .put_bytes(serde_json::to_vec(&transcript).unwrap()) - .await - .unwrap(), - media_type: Some("application/json".into()), - provider_kind: Some(AUDIO_TRANSCRIPT_PROVIDER_KIND.into()), - }, - preview: Some(transcript.header()), - origin: None, - provenance_ref: Some(audio_ref.clone()), - token_estimate: None, - }; - assert!(audio_input_matches_transcript(&original, &rewritten)); - let mut different = original; - different.content.content_ref = BlobRef::from_bytes(b"different audio"); - assert!(!audio_input_matches_transcript(&different, &rewritten)); - let result = context_append_result( - &blobs, - "audio-note".into(), - ContextAppendStatus::Applied, - &rewritten, - None, - ) - .await - .unwrap(); - assert_eq!( - result.activation_text.as_deref(), - Some(transcript.text.as_str()) - ); - assert!(!result.activation_text_truncated); - assert_eq!( - result.entry.unwrap().provenance_ref.as_deref(), - Some(audio_ref.as_str()) - ); -} - #[tokio::test(flavor = "current_thread")] async fn input_origin_survives_admission_and_projection_without_changing_model_content() { let store = engine::storage::InMemoryBlobStore::new(); let body = store.put_bytes(b"event body".to_vec()).await.unwrap(); let items = vec![ InputItem::Text { + provenance_ref: None, text: "human body".into(), origin: Some("user:operator".into()), }, InputItem::TextRef { + provenance_ref: None, blob_ref: body.as_str().into(), origin: Some("event".into()), }, InputItem::Text { + provenance_ref: None, text: "custom body".into(), origin: Some("integration:example".into()), }, InputItem::Text { + provenance_ref: None, text: "unknown body".into(), origin: None, }, @@ -2465,7 +2307,7 @@ async fn input_origin_survives_admission_and_projection_without_changing_model_c { assert_eq!(entries[i].origin.as_deref(), expected); assert_eq!(accepted[i].origin.as_deref(), expected); - let InputItem::Text { origin, text } = &inputs[i] else { + let InputItem::Text { origin, text, .. } = &inputs[i] else { panic!("expected text input"); }; assert_eq!(origin.as_deref(), expected); @@ -2516,6 +2358,7 @@ async fn input_origin_rejects_blank_and_oversized_values() { let store = engine::storage::InMemoryBlobStore::new(); for origin in ["".to_owned(), " ".to_owned(), "x".repeat(201)] { let item = InputItem::Text { + provenance_ref: None, text: "hello".into(), origin: Some(origin), }; @@ -2783,3 +2626,104 @@ fn a_missing_resource_names_only_its_kind_and_id() { "environment not found: prod-1" ); } + +#[tokio::test(flavor = "current_thread")] +async fn text_inputs_preserve_generic_provenance_at_all_input_boundaries() { + let store = engine::storage::InMemoryBlobStore::new(); + let source = store + .put_bytes(b"source document or recording".to_vec()) + .await + .unwrap(); + let text = "[audio transcript: quoted]\nReviewed text"; + let reference = store.put_bytes(text.as_bytes().to_vec()).await.unwrap(); + for item in [ + InputItem::Text { + origin: Some("user:test".into()), + text: text.into(), + provenance_ref: Some(source.to_string()), + }, + InputItem::TextRef { + origin: Some("user:test".into()), + blob_ref: reference.to_string(), + provenance_ref: Some(source.to_string()), + }, + ] { + let run = run_input_from_api(&store, std::slice::from_ref(&item)) + .await + .unwrap(); + let append = context_entry_input_from_api(&store, &item).await.unwrap(); + assert_eq!(run, vec![append.clone()]); + assert_eq!(append.content, engine::ContentRef::text(reference.clone())); + assert_eq!(append.provenance_ref.as_ref(), Some(&source)); + assert_eq!(append.origin.as_deref(), Some("user:test")); + let projected = api_projection::CoreAgentProjector::new(&store) + .project_input_entries(&run) + .await + .unwrap(); + assert_eq!( + projected, + vec![InputItem::Text { + origin: Some("user:test".into()), + text: text.into(), + provenance_ref: Some(source.to_string()) + }] + ); + let result = context_append_result( + &store, + "note".into(), + ContextAppendStatus::Applied, + &append, + None, + ) + .await + .unwrap(); + assert_eq!(result.activation_text.as_deref(), Some(text)); + assert_eq!( + result.entry.unwrap().provenance_ref, + Some(source.to_string()) + ); + } + let plain = context_entry_input_from_api( + &store, + &InputItem::Text { + origin: None, + text: "edited dictation".into(), + provenance_ref: None, + }, + ) + .await + .unwrap(); + assert!(plain.provenance_ref.is_none()); +} + +#[tokio::test(flavor = "current_thread")] +async fn text_input_provenance_requires_an_existing_source_blob() { + let store = engine::storage::InMemoryBlobStore::new(); + let text_ref = store.put_bytes(b"text".to_vec()).await.unwrap(); + for reference in [ + "not-a-blob".into(), + BlobRef::from_bytes(b"missing source").to_string(), + ] { + for item in [ + InputItem::Text { + origin: None, + text: "text".into(), + provenance_ref: Some(reference.clone()), + }, + InputItem::TextRef { + origin: None, + blob_ref: text_ref.to_string(), + provenance_ref: Some(reference.clone()), + }, + ] { + let run_error = run_input_from_api(&store, std::slice::from_ref(&item)) + .await + .unwrap_err(); + let append_error = context_entry_input_from_api(&store, &item) + .await + .unwrap_err(); + assert_eq!(run_error.kind, AgentApiErrorKind::InvalidRequest); + assert_eq!(append_error.kind, run_error.kind); + } + } +} diff --git a/crates/temporal-server/src/gateway/service/transcriptions.rs b/crates/temporal-server/src/gateway/service/transcriptions.rs new file mode 100644 index 000000000..9ccfb3ed5 --- /dev/null +++ b/crates/temporal-server/src/gateway/service/transcriptions.rs @@ -0,0 +1,268 @@ +use super::*; +use temporal_workflow::{TranscriptionSnapshot, TranscriptionWorkflow, TranscriptionWorkflowArgs}; +use temporalio_common::protos::temporal::api::enums::v1::WorkflowIdReusePolicy; + +fn validate_id(id: &str) -> Result<(), AgentApiError> { + if id + .strip_prefix("transcription_") + .is_some_and(|s| s.len() == 64 && s.bytes().all(|b| b.is_ascii_hexdigit())) + { + Ok(()) + } else { + Err(AgentApiError::invalid_request("invalid transcription id")) + } +} + +impl GatewayAgentApi { + pub(crate) async fn transcription_snapshot( + &self, + id: &str, + ) -> Result, AgentApiError> { + validate_id(id)?; + let handle = self + .client + .get_workflow_handle::(format!("{}/{id}", self.universe_id())); + match handle + .query( + TranscriptionWorkflow::snapshot, + (), + WorkflowQueryOptions::default(), + ) + .await + { + Ok(Some(mut snapshot)) => { + if !snapshot.view.status.is_terminal() { + let description = handle + .describe(WorkflowDescribeOptions::default()) + .await + .map_err(map_workflow_interaction_error)?; + match description.status() { + WorkflowExecutionStatus::TimedOut => { + snapshot.view.status = TranscriptionStatus::Failed; + snapshot.view.failure = Some(TranscriptionFailure { + kind: TranscriptionFailureKind::Timeout, + message: "Transcription exceeded its workflow deadline.".into(), + }); + } + WorkflowExecutionStatus::Canceled | WorkflowExecutionStatus::Terminated => { + snapshot.view.status = TranscriptionStatus::Cancelled + } + WorkflowExecutionStatus::Failed => { + snapshot.view.status = TranscriptionStatus::Failed; + snapshot.view.failure = Some(TranscriptionFailure { + kind: TranscriptionFailureKind::Internal, + message: "Transcription workflow failed.".into(), + }); + } + _ => {} + } + } + Ok(Some(snapshot)) + } + Ok(None) => Err(AgentApiError::conflict( + "transcription is starting; retry shortly", + )), + Err(WorkflowQueryError::NotFound(_)) => Ok(None), + Err(error) => Err(map_workflow_query_error(error)), + } + } + + /// Trusted internal callers supply their controller attribution explicitly. + pub(crate) async fn admit_transcription( + &self, + request: TranscriptionStartParams, + owner: Attribution, + ) -> Result { + request.validate()?; + let source = parse_blob_ref(&request.audio.blob_ref)?; + let id = temporal_workflow::transcription_id(&owner, &request.idempotency_key); + if let Some(snapshot) = self.transcription_snapshot(&id).await? { + return matching_request(snapshot, &request); + } + let model = match &request.model { + Some(model) => model.clone(), + None => self + .store + .read_model_defaults() + .await + .map_err(model_defaults::map_store_error)? + .speech_to_text + .ok_or_else(|| { + AgentApiError::model_default_unset(ModelDefaultSlot::SpeechToText) + })?, + }; + ModelDefaultSlot::SpeechToText.validate_model(&model)?; + let info = self + .store + .stat_blob(&source) + .await + .map_err(map_input_blob_store_error)?; + if info.byte_len > 25 * 1024 * 1024 { + return Err(AgentApiError::audio_blob_too_large( + "audio exceeds the 25 MiB limit", + )); + } + self.store + .touch_blob_refs(&[source]) + .await + .map_err(map_input_blob_store_error)?; + let args = TranscriptionWorkflowArgs { + universe_id: self.universe_id(), + transcription_id: id.clone(), + created_by: owner, + request: request.clone(), + model, + created_at_ms: now_ms()? as u64, + }; + let pending = args.pending(); + match self + .client + .start_workflow( + TranscriptionWorkflow::run, + args, + WorkflowStartOptions::new( + self.task_queue.clone(), + format!("{}/{id}", self.universe_id()), + ) + .id_reuse_policy(WorkflowIdReusePolicy::RejectDuplicate) + .execution_timeout(Duration::from_secs(960)) + .build(), + ) + .await + { + Ok(_) => Ok(pending), + Err(WorkflowStartError::AlreadyStarted { .. }) => { + // Another admission won the race. Its pinned model is authoritative. + let snapshot = self.transcription_snapshot(&id).await?.ok_or_else(|| { + AgentApiError::conflict("transcription is starting; retry the same request") + })?; + matching_request(snapshot, &request) + } + Err(error) => Err(map_workflow_start_error(error)), + } + } + + pub(crate) async fn transcription_content( + &self, + mut view: TranscriptionView, + ) -> Result { + if let Some(reference) = &view.transcript_ref { + let reference = parse_blob_ref(reference)?; + let content = async { + // Check metadata before the byte cache: a swept result is no + // longer admissible even while this process has cached bytes. + self.store.stat_blob(&reference).await?; + self.store.read_bytes(&reference).await + } + .await; + match content { + Ok(bytes) => { + view.text = Some(String::from_utf8(bytes).map_err(|_| { + AgentApiError::internal("transcription result is not UTF-8") + })?); + } + Err(BlobStoreError::NotFound { .. }) => { + view.status = TranscriptionStatus::Expired; + } + Err(error) => return Err(map_blob_store_error(error)), + } + } + Ok(view) + } + + fn authorize_transcription_owner(&self, view: &TranscriptionView) -> Result<(), AgentApiError> { + // Person callers can only access their own drafts. Direct universe keys + // retain their method-group authority; core does not resolve person roles. + if let Some(actor) = self.caller()?.actor + && view.created_by != (Attribution::Actor { id: actor }) + { + return Err(AgentApiError::forbidden()); + } + Ok(()) + } + + pub(super) async fn start_transcription_impl( + &self, + params: TranscriptionStartParams, + ) -> Result, AgentApiError> { + self.authorize_method(METHOD_TRANSCRIPTIONS_START, None) + .await?; + let view = self + .admit_transcription(params, self.attribution()?) + .await?; + Ok(AgentApiOutcome::new(TranscriptionResponse { + transcription: self.transcription_content(view).await?, + })) + } + + pub(super) async fn read_transcription_impl( + &self, + params: TranscriptionReadParams, + ) -> Result, AgentApiError> { + self.authorize_method(METHOD_TRANSCRIPTIONS_READ, None) + .await?; + let view = self + .transcription_snapshot(¶ms.transcription_id) + .await? + .ok_or_else(|| AgentApiError::not_found("transcription not found"))? + .view; + self.authorize_transcription_owner(&view)?; + Ok(AgentApiOutcome::new(TranscriptionResponse { + transcription: self.transcription_content(view).await?, + })) + } + + pub(super) async fn cancel_transcription_impl( + &self, + params: TranscriptionCancelParams, + ) -> Result, AgentApiError> { + self.authorize_method(METHOD_TRANSCRIPTIONS_CANCEL, None) + .await?; + let view = self + .transcription_snapshot(¶ms.transcription_id) + .await? + .ok_or_else(|| AgentApiError::not_found("transcription not found"))? + .view; + self.authorize_transcription_owner(&view)?; + if !view.status.is_terminal() { + let handle = self + .client + .get_workflow_handle::(format!( + "{}/{}", + self.universe_id(), + params.transcription_id + )); + match handle + .signal( + TranscriptionWorkflow::cancel, + (), + WorkflowSignalOptions::default(), + ) + .await + { + Ok(()) | Err(WorkflowInteractionError::NotFound(_)) => {} + Err(error) => return Err(map_workflow_interaction_error(error)), + } + } + let view = self + .transcription_snapshot(¶ms.transcription_id) + .await? + .map(|s| s.view) + .unwrap_or(view); + Ok(AgentApiOutcome::new(TranscriptionResponse { + transcription: self.transcription_content(view).await?, + })) + } +} + +fn matching_request( + snapshot: TranscriptionSnapshot, + request: &TranscriptionStartParams, +) -> Result { + if snapshot.request != *request { + return Err(AgentApiError::conflict( + "transcription idempotency key was used with different input or options", + )); + } + Ok(snapshot.view) +} diff --git a/crates/temporal-server/src/subagents.rs b/crates/temporal-server/src/subagents.rs index 8ca05d096..688f1835f 100644 --- a/crates/temporal-server/src/subagents.rs +++ b/crates/temporal-server/src/subagents.rs @@ -403,6 +403,7 @@ impl SubagentService { .start_run( &child_session_id, vec![InputItem::Text { + provenance_ref: None, origin: None, text: args.input, }], @@ -1122,6 +1123,7 @@ mod tests { assert_eq!( runs[0].1, vec![InputItem::Text { + provenance_ref: None, origin: None, text: "review the change".to_owned() }] diff --git a/crates/temporal-server/src/worker/activities/audio.rs b/crates/temporal-server/src/worker/activities/audio.rs new file mode 100644 index 000000000..3ac2d64be --- /dev/null +++ b/crates/temporal-server/src/worker/activities/audio.rs @@ -0,0 +1,803 @@ +use std::{ + env, + ffi::{OsStr, OsString}, + path::{Path, PathBuf}, + process::Stdio, + sync::Arc, + time::Duration, +}; + +use async_trait::async_trait; +use llm_clients::{LlmApiError, openai::audio as oai}; +use llm_runtime::ModelProviderResolver; +pub(super) const MAX_AUDIO_BYTES: u64 = 25 * 1024 * 1024; +pub(super) const MAX_AUDIO_DURATION_MS: u64 = 10 * 60 * 1000; +const OPENAI_PROVIDER_ID: &str = "openai"; +pub(super) const PROVIDER_ACCEPTED_AUDIO_MIMES: &[&str] = &[ + "audio/mpeg", + "audio/mp4", + "audio/wav", + "audio/webm", + "audio/ogg", +]; +pub(super) const TRANSCODABLE_AUDIO_MIMES: &[&str] = + &["audio/aac", "audio/amr", "audio/3gpp", "audio/3gpp2"]; +const TRANSCODED_AUDIO_MIME: &str = "audio/wav"; +const DEFAULT_FFMPEG_PATH: &str = "ffmpeg"; +const DEFAULT_TRANSCODE_TIMEOUT: Duration = Duration::from_secs(30); + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AudioTranscriptionRequest { + pub bytes: Vec, + pub mime: String, + pub name: String, + pub model: api::ModelConfig, + pub language: Option, + pub prompt: Option, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AudioTranscription { + pub text: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +#[error("audio transcription failed: {message}")] +pub struct AudioTranscriptionError { + pub message: String, + pub retryable: bool, + pub configuration: bool, +} + +#[async_trait] +pub trait AudioTranscriber: Send + Sync { + async fn transcribe( + &self, + request: AudioTranscriptionRequest, + ) -> Result; +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AudioTranscodeRequest { + pub bytes: Vec, + pub mime: String, + pub name: String, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AudioTranscodeOutput { + pub bytes: Vec, + pub mime: String, + pub name: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +#[error("audio transcode failed: {message}")] +pub struct AudioTranscodeError { + pub message: String, +} + +#[async_trait] +pub trait AudioTranscoder: Send + Sync { + async fn transcode( + &self, + request: AudioTranscodeRequest, + ) -> Result; +} + +pub struct UnavailableAudioTranscriber; + +#[async_trait] +impl AudioTranscriber for UnavailableAudioTranscriber { + async fn transcribe( + &self, + _request: AudioTranscriptionRequest, + ) -> Result { + Err(AudioTranscriptionError { + message: "audio transcriber is not configured".to_owned(), + retryable: false, + configuration: true, + }) + } +} + +#[derive(Clone, Debug)] +pub struct FfmpegAudioTranscoder { + ffmpeg_path: PathBuf, + timeout: Duration, + max_output_bytes: u64, +} + +impl FfmpegAudioTranscoder { + pub fn new(ffmpeg_path: impl Into) -> Self { + Self { + ffmpeg_path: ffmpeg_path.into(), + timeout: DEFAULT_TRANSCODE_TIMEOUT, + max_output_bytes: MAX_AUDIO_BYTES, + } + } + + pub fn with_timeout(mut self, timeout: Duration) -> Self { + self.timeout = timeout; + self + } + + pub fn from_env() -> Self { + let ffmpeg_path = env::var("LIGHTSPEED_FFMPEG_PATH") + .ok() + .filter(|value| !value.trim().is_empty()) + .unwrap_or_else(|| DEFAULT_FFMPEG_PATH.to_owned()); + let timeout = env::var("LIGHTSPEED_AUDIO_TRANSCODE_TIMEOUT_MS") + .ok() + .and_then(|value| value.parse::().ok()) + .filter(|millis| *millis > 0) + .map(Duration::from_millis) + .unwrap_or(DEFAULT_TRANSCODE_TIMEOUT); + Self::new(ffmpeg_path).with_timeout(timeout) + } +} + +#[async_trait] +impl AudioTranscoder for FfmpegAudioTranscoder { + async fn transcode( + &self, + request: AudioTranscodeRequest, + ) -> Result { + let temp_dir = tempfile::Builder::new() + .prefix("lightspeed-audio-transcode-") + .tempdir() + .map_err(|error| AudioTranscodeError { + message: format!("create transcode temp dir: {error}"), + })?; + let input_path = temp_dir + .path() + .join(format!("input.{}", extension_for_mime(&request.mime))); + let output_path = temp_dir.path().join("output.wav"); + tokio::fs::write(&input_path, &request.bytes) + .await + .map_err(|error| AudioTranscodeError { + message: format!("write transcode input: {error}"), + })?; + + let mut command = tokio::process::Command::new(&self.ffmpeg_path); + command + .args(ffmpeg_args(&input_path, &output_path)) + .kill_on_drop(true) + .stdin(Stdio::null()) + .stdout(Stdio::null()) + .stderr(Stdio::piped()); + let output = tokio::time::timeout(self.timeout, command.output()) + .await + .map_err(|_| AudioTranscodeError { + message: format!( + "ffmpeg did not finish within {}ms", + self.timeout.as_millis() + ), + })? + .map_err(|error| AudioTranscodeError { + message: format!("run ffmpeg: {error}"), + })?; + if !output.status.success() { + return Err(AudioTranscodeError { + message: ffmpeg_failure_message(&output.stderr), + }); + } + let metadata = + tokio::fs::metadata(&output_path) + .await + .map_err(|error| AudioTranscodeError { + message: format!("read ffmpeg output metadata: {error}"), + })?; + if metadata.len() > self.max_output_bytes { + return Err(AudioTranscodeError { + message: format!( + "transcoded audio {} is {} bytes; the limit is {} bytes", + output_path.display(), + metadata.len(), + self.max_output_bytes + ), + }); + } + let bytes = tokio::fs::read(&output_path) + .await + .map_err(|error| AudioTranscodeError { + message: format!("read ffmpeg output: {error}"), + })?; + Ok(AudioTranscodeOutput { + bytes, + mime: TRANSCODED_AUDIO_MIME.to_owned(), + name: transcoded_audio_filename(&request.name), + }) + } +} + +pub struct OpenAiAudioTranscriber { + client: Arc, + provider_keys: Arc, +} + +impl OpenAiAudioTranscriber { + pub fn new(client: Arc, provider_keys: Arc) -> Self { + Self { + client, + provider_keys, + } + } +} + +#[async_trait] +impl AudioTranscriber for OpenAiAudioTranscriber { + async fn transcribe( + &self, + request: AudioTranscriptionRequest, + ) -> Result { + api::ModelDefaultSlot::SpeechToText + .validate_model(&request.model) + .map_err(|error| AudioTranscriptionError { + message: error.to_string(), + retryable: false, + configuration: true, + })?; + let stored = llm_runtime::resolve_provider_route( + self.provider_keys.as_ref(), + &request.model.provider_id, + &request.model.api_kind, + ) + .await + .map_err(|error| { + let retryable = matches!(error, llm_runtime::ProviderKeyError::Backend { .. }); + AudioTranscriptionError { + message: "Transcription provider configuration is missing or unusable.".into(), + retryable, + configuration: !retryable, + } + })?; + if request.model.provider_id != OPENAI_PROVIDER_ID + && stored.as_ref().and_then(|p| p.endpoint.as_ref()).is_none() + { + return Err(AudioTranscriptionError { + message: "Transcription provider requires an explicit endpoint.".into(), + retryable: false, + configuration: true, + }); + } + let mut input = oai::CreateTranscriptionRequest::new(oai::AudioFile { + bytes: request.bytes, + filename: request.name, + mime: request.mime, + }); + input.model = request.model.model; + input.language = request.language; + input.prompt = request.prompt; + let response = self + .client + .create_transcription_with_transport( + input, + stored.as_ref().map(|provider| provider.as_request_auth()), + stored + .as_ref() + .and_then(|provider| provider.endpoint.as_ref()) + .map(|endpoint| &endpoint.transport), + ) + .await + .map_err(map_openai_error)?; + Ok(AudioTranscription { + text: response.parsed.text, + }) + } +} + +#[derive(Debug, thiserror::Error)] +pub(super) enum AudioPreparationError { + #[error("Audio requires a transcoder, but none is configured.")] + TranscoderUnavailable, + #[error(transparent)] + Transcode(#[from] AudioTranscodeError), + #[error("Transcoded audio exceeds the 25 MiB limit.")] + OutputTooLarge, + #[error("Audio transcoder returned unsupported MIME type {0}.")] + UnsupportedOutputMime(String), +} + +#[derive(Debug)] +pub(super) struct PreparedAudio { + pub bytes: Vec, + pub mime: String, + pub name: String, +} + +pub(super) async fn prepare_audio_for_transcription( + bytes: Vec, + mime: &str, + name: &str, + transcoder: Option<&dyn AudioTranscoder>, +) -> Result { + if PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&mime) { + return Ok(PreparedAudio { + bytes, + mime: mime.to_owned(), + name: audio_filename(name), + }); + } + let Some(transcoder) = transcoder else { + return Err(AudioPreparationError::TranscoderUnavailable); + }; + let output = transcoder + .transcode(AudioTranscodeRequest { + bytes, + mime: mime.to_owned(), + name: audio_filename(name), + }) + .await + .map_err(AudioPreparationError::Transcode)?; + if output.bytes.len() as u64 > MAX_AUDIO_BYTES { + return Err(AudioPreparationError::OutputTooLarge); + } + if !PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&output.mime.as_str()) { + return Err(AudioPreparationError::UnsupportedOutputMime(output.mime)); + } + Ok(PreparedAudio { + bytes: output.bytes, + mime: output.mime, + name: output.name, + }) +} + +pub(super) fn normalized_mime(mime: Option<&str>) -> String { + let mime = mime + .unwrap_or_default() + .split(';') + .next() + .unwrap_or_default() + .trim() + .to_ascii_lowercase(); + match mime.as_str() { + "audio/mp3" => "audio/mpeg", + "audio/x-m4a" | "audio/m4a" => "audio/mp4", + "audio/x-wav" | "audio/wave" | "audio/vnd.wave" => "audio/wav", + "audio/oga" | "audio/opus" => "audio/ogg", + "audio/x-aac" => "audio/aac", + "audio/3gp" => "audio/3gpp", + "audio/3g2" => "audio/3gpp2", + other => other, + } + .to_owned() +} + +fn audio_filename(label: &str) -> String { + let trimmed = label.trim(); + if trimmed.is_empty() || trimmed == "audio" { + "audio.ogg".to_owned() + } else { + trimmed.to_owned() + } +} + +fn transcoded_audio_filename(label: &str) -> String { + let filename = audio_filename(label); + let stem = Path::new(&filename) + .file_stem() + .and_then(OsStr::to_str) + .filter(|value| !value.trim().is_empty()) + .unwrap_or("audio"); + format!("{stem}.wav") +} + +fn extension_for_mime(mime: &str) -> &'static str { + match mime { + "audio/mpeg" => "mp3", + "audio/mp4" => "m4a", + "audio/wav" => "wav", + "audio/webm" => "webm", + "audio/ogg" => "ogg", + "audio/aac" => "aac", + "audio/amr" => "amr", + "audio/3gpp" => "3gp", + "audio/3gpp2" => "3g2", + _ => "bin", + } +} + +fn ffmpeg_args(input_path: &Path, output_path: &Path) -> Vec { + [ + OsString::from("-nostdin"), + OsString::from("-hide_banner"), + OsString::from("-loglevel"), + OsString::from("error"), + OsString::from("-y"), + OsString::from("-i"), + input_path.as_os_str().to_owned(), + OsString::from("-vn"), + OsString::from("-ac"), + OsString::from("1"), + OsString::from("-ar"), + OsString::from("16000"), + OsString::from("-acodec"), + OsString::from("pcm_s16le"), + OsString::from("-f"), + OsString::from("wav"), + output_path.as_os_str().to_owned(), + ] + .into_iter() + .collect() +} + +fn ffmpeg_failure_message(stderr: &[u8]) -> String { + let stderr = String::from_utf8_lossy(stderr); + let stderr = stderr.trim(); + if stderr.is_empty() { + "ffmpeg exited with a non-zero status".to_owned() + } else { + format!("ffmpeg exited with a non-zero status: {stderr}") + } +} + +pub(super) fn audio_duration_ms(mime: &str, bytes: &[u8]) -> Option { + // Cheap duration enforcement is intentionally narrow for the first cut: + // OGG/Opus and WAV are covered; MP3/M4A/WebM rely on the byte cap unless + // a non-provider container is transcoded to WAV. + match mime { + "audio/ogg" => ogg_opus_duration_ms(bytes), + "audio/wav" => wav_duration_ms(bytes), + _ => None, + } +} + +fn ogg_opus_duration_ms(bytes: &[u8]) -> Option { + let mut offset = 0usize; + let mut last_granule = None; + while offset + 27 <= bytes.len() { + if &bytes[offset..offset + 4] != b"OggS" { + return None; + } + let granule = u64::from_le_bytes(bytes[offset + 6..offset + 14].try_into().ok()?); + if granule != u64::MAX { + last_granule = Some(granule); + } + let segments = bytes[offset + 26] as usize; + let lacing_start = offset + 27; + let lacing_end = lacing_start.checked_add(segments)?; + if lacing_end > bytes.len() { + return None; + } + let mut body_len = 0usize; + for segment in &bytes[lacing_start..lacing_end] { + body_len = body_len.checked_add(*segment as usize)?; + } + offset = lacing_end.checked_add(body_len)?; + } + last_granule.map(|samples| samples.saturating_mul(1000) / 48_000) +} + +fn wav_duration_ms(bytes: &[u8]) -> Option { + if bytes.len() < 12 || &bytes[0..4] != b"RIFF" || &bytes[8..12] != b"WAVE" { + return None; + } + let mut offset = 12usize; + let mut byte_rate = None; + let mut data_len = None; + while offset + 8 <= bytes.len() { + let chunk_id = &bytes[offset..offset + 4]; + let chunk_len = u32::from_le_bytes(bytes[offset + 4..offset + 8].try_into().ok()?) as usize; + let data_start = offset + 8; + let data_end = data_start.checked_add(chunk_len)?; + if data_end > bytes.len() { + return None; + } + if chunk_id == b"fmt " && chunk_len >= 16 { + byte_rate = Some(u32::from_le_bytes( + bytes[data_start + 8..data_start + 12].try_into().ok()?, + ) as u64); + } else if chunk_id == b"data" { + data_len = Some(chunk_len as u64); + } + if byte_rate.is_some() && data_len.is_some() { + break; + } + offset = data_end.checked_add(chunk_len % 2)?; + } + let byte_rate = byte_rate?; + if byte_rate == 0 { + return None; + } + Some(data_len?.saturating_mul(1000) / byte_rate) +} + +fn map_openai_error(error: LlmApiError) -> AudioTranscriptionError { + let retryable = error.retryable(); + let configuration = matches!(error, LlmApiError::Configuration(_)); + AudioTranscriptionError { + message: if configuration { + "Transcription provider is not configured." + } else { + "Transcription provider request failed." + } + .into(), + retryable, + configuration, + } +} + +pub(super) fn default_openai_audio_transcriber( + provider_keys: Arc, +) -> Result, anyhow::Error> { + let client = oai::Client::new(oai::Config::from_env_allow_missing_key()) + .map_err(|error| anyhow::anyhow!("construct OpenAI audio client: {error}"))?; + Ok(Arc::new(OpenAiAudioTranscriber::new( + Arc::new(client), + provider_keys, + ))) +} + +pub fn default_audio_transcoder_from_env() -> anyhow::Result>> { + let Some(kind) = env::var("LIGHTSPEED_AUDIO_TRANSCODER") + .ok() + .map(|value| value.trim().to_ascii_lowercase()) + .filter(|value| !value.is_empty() && value != "none") + else { + return Ok(None); + }; + match kind.as_str() { + "ffmpeg" => Ok(Some(Arc::new(FfmpegAudioTranscoder::from_env()))), + other => anyhow::bail!( + "unsupported LIGHTSPEED_AUDIO_TRANSCODER value {other:?}; expected \"ffmpeg\" or \"none\"" + ), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::sync::Mutex; + + struct StaticTranscoder { + result: Result, + requests: Mutex>, + } + + impl StaticTranscoder { + fn new(result: Result) -> Self { + Self { + result, + requests: Mutex::new(Vec::new()), + } + } + + fn requests(&self) -> Vec { + self.requests + .lock() + .expect("static transcoder requests") + .clone() + } + } + + #[async_trait] + impl AudioTranscoder for StaticTranscoder { + async fn transcode( + &self, + request: AudioTranscodeRequest, + ) -> Result { + self.requests + .lock() + .expect("static transcoder requests") + .push(request); + self.result.clone() + } + } + + #[tokio::test(flavor = "current_thread")] + async fn accepted_audio_bypasses_transcoding() { + let transcoder = StaticTranscoder::new(Err(AudioTranscodeError { + message: "must not run".into(), + })); + let audio = prepare_audio_for_transcription( + b"audio".to_vec(), + "audio/ogg", + "voice.ogg", + Some(&transcoder), + ) + .await + .unwrap(); + assert_eq!(audio.bytes, b"audio"); + assert_eq!(audio.mime, "audio/ogg"); + assert_eq!(audio.name, "voice.ogg"); + assert!(transcoder.requests().is_empty()); + } + + #[tokio::test(flavor = "current_thread")] + async fn transcodable_audio_requires_transcoder() { + let error = prepare_audio_for_transcription(vec![], "audio/aac", "voice.aac", None) + .await + .unwrap_err(); + assert!(matches!( + error, + AudioPreparationError::TranscoderUnavailable + )); + } + + #[tokio::test(flavor = "current_thread")] + async fn transcodable_audio_is_prepared_for_provider() { + let transcoder = StaticTranscoder::new(Ok(AudioTranscodeOutput { + bytes: tiny_wav_bytes(), + mime: "audio/wav".into(), + name: "voice.wav".into(), + })); + let audio = prepare_audio_for_transcription( + b"aac fake".to_vec(), + "audio/aac", + "voice.aac", + Some(&transcoder), + ) + .await + .unwrap(); + assert_eq!(audio.bytes, tiny_wav_bytes()); + assert_eq!(audio.mime, "audio/wav"); + assert_eq!(audio.name, "voice.wav"); + assert_eq!( + transcoder.requests(), + vec![AudioTranscodeRequest { + bytes: b"aac fake".to_vec(), + mime: "audio/aac".into(), + name: "voice.aac".into(), + }] + ); + } + + #[tokio::test(flavor = "current_thread")] + async fn invalid_transcoder_output_is_rejected() { + for (output, too_large) in [ + ( + AudioTranscodeOutput { + bytes: vec![0; MAX_AUDIO_BYTES as usize + 1], + mime: "audio/wav".into(), + name: "voice.wav".into(), + }, + true, + ), + ( + AudioTranscodeOutput { + bytes: vec![], + mime: "audio/aac".into(), + name: "voice.aac".into(), + }, + false, + ), + ] { + let transcoder = StaticTranscoder::new(Ok(output)); + let error = prepare_audio_for_transcription( + vec![], + "audio/aac", + "voice.aac", + Some(&transcoder), + ) + .await + .unwrap_err(); + if too_large { + assert!(matches!(error, AudioPreparationError::OutputTooLarge)); + } else { + assert!(matches!( + error, + AudioPreparationError::UnsupportedOutputMime(_) + )); + } + } + } + + #[tokio::test(flavor = "current_thread")] + async fn transcode_failure_is_preserved() { + let transcoder = StaticTranscoder::new(Err(AudioTranscodeError { + message: "unsupported codec".into(), + })); + let error = + prepare_audio_for_transcription(vec![], "audio/aac", "voice.aac", Some(&transcoder)) + .await + .unwrap_err(); + assert!(matches!(error, AudioPreparationError::Transcode(_))); + } + + #[test] + fn duration_readers_measure_audio() { + assert_eq!( + audio_duration_ms("audio/wav", &tiny_wav_bytes()), + Some(1000) + ); + assert_eq!( + audio_duration_ms("audio/ogg", &ogg_page(601 * 48_000)), + Some(601_000) + ); + } + + #[test] + fn ffmpeg_command_args_normalize_audio_to_mono_wav() { + let args = ffmpeg_args(Path::new("/tmp/in.aac"), Path::new("/tmp/out.wav")); + let args: Vec = args + .iter() + .map(|arg| arg.to_string_lossy().into_owned()) + .collect(); + + assert_eq!( + args, + vec![ + "-nostdin", + "-hide_banner", + "-loglevel", + "error", + "-y", + "-i", + "/tmp/in.aac", + "-vn", + "-ac", + "1", + "-ar", + "16000", + "-acodec", + "pcm_s16le", + "-f", + "wav", + "/tmp/out.wav" + ] + ); + } + + #[tokio::test(flavor = "current_thread")] + #[ignore = "requires ffmpeg on PATH or LIGHTSPEED_FFMPEG_PATH"] + async fn ffmpeg_audio_transcoder_smoke_test() { + let transcoder = FfmpegAudioTranscoder::from_env(); + + let output = transcoder + .transcode(AudioTranscodeRequest { + bytes: tiny_wav_bytes(), + mime: "audio/wav".to_owned(), + name: "voice.wav".to_owned(), + }) + .await + .expect("ffmpeg should transcode tiny wav input"); + + assert_eq!(output.mime, "audio/wav"); + assert_eq!(output.name, "voice.wav"); + assert!(output.bytes.starts_with(b"RIFF")); + assert_eq!(audio_duration_ms(&output.mime, &output.bytes), Some(1000)); + } + + fn tiny_wav_bytes() -> Vec { + let sample_rate = 8_000u32; + let channels = 1u16; + let bits_per_sample = 16u16; + let sample_count = sample_rate as usize; + let byte_rate = sample_rate * channels as u32 * bits_per_sample as u32 / 8; + let block_align = channels * bits_per_sample / 8; + let data_len = sample_count * block_align as usize; + + let mut bytes = Vec::with_capacity(44 + data_len); + bytes.extend_from_slice(b"RIFF"); + bytes.extend_from_slice(&(36 + data_len as u32).to_le_bytes()); + bytes.extend_from_slice(b"WAVE"); + bytes.extend_from_slice(b"fmt "); + bytes.extend_from_slice(&16u32.to_le_bytes()); + bytes.extend_from_slice(&1u16.to_le_bytes()); + bytes.extend_from_slice(&channels.to_le_bytes()); + bytes.extend_from_slice(&sample_rate.to_le_bytes()); + bytes.extend_from_slice(&byte_rate.to_le_bytes()); + bytes.extend_from_slice(&block_align.to_le_bytes()); + bytes.extend_from_slice(&bits_per_sample.to_le_bytes()); + bytes.extend_from_slice(b"data"); + bytes.extend_from_slice(&(data_len as u32).to_le_bytes()); + bytes.resize(44 + data_len, 0); + bytes + } + + fn ogg_page(granule_position: u64) -> Vec { + let mut page = Vec::new(); + page.extend_from_slice(b"OggS"); + page.push(0); + page.push(0); + page.extend_from_slice(&granule_position.to_le_bytes()); + page.extend_from_slice(&1u32.to_le_bytes()); + page.extend_from_slice(&0u32.to_le_bytes()); + page.extend_from_slice(&0u32.to_le_bytes()); + page.push(1); + page.push(1); + page.push(0); + page + } +} diff --git a/crates/temporal-server/src/worker/activities/mod.rs b/crates/temporal-server/src/worker/activities/mod.rs index 296f753d8..9bc7a3827 100644 --- a/crates/temporal-server/src/worker/activities/mod.rs +++ b/crates/temporal-server/src/worker/activities/mod.rs @@ -16,29 +16,29 @@ use crate::worker::{ ACTIVITY_CONTEXT_COMPACT, ACTIVITY_CREATE_OR_LOAD_SESSION, ACTIVITY_ENVIRONMENT_JOB_CANCEL, ACTIVITY_ENVIRONMENT_JOB_POLL, ACTIVITY_ENVIRONMENT_JOB_PREPARE_WORKFLOW_TOOL, ACTIVITY_ENVIRONMENT_JOB_START, ACTIVITY_LLM_GENERATE, ACTIVITY_MATERIALIZE_AWAIT_RESULT, - ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PREPROCESS_RUN_INPUT, ACTIVITY_PUT_BLOB, - ACTIVITY_READ_BLOB, ACTIVITY_RUNTIME_PROJECTION_REFRESH, - ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, - ACTIVITY_SUBAGENT_RESOLVE, ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, - ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, - AppendEventsRequest, AwaitEnvironmentReadyActivityRequest, AwaitEnvironmentReadyActivityResult, + ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PUT_BLOB, ACTIVITY_READ_BLOB, + ACTIVITY_RUNTIME_PROJECTION_REFRESH, ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, + ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, ACTIVITY_SUBAGENT_RESOLVE, + ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, + ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, AppendEventsRequest, + AwaitEnvironmentReadyActivityRequest, AwaitEnvironmentReadyActivityResult, ContextCompactActivityRequest, CreateOrLoadSessionRequest, CreateOrLoadSessionResult, EnvironmentJobCancelActivityRequest, EnvironmentJobPollActivityRequest, EnvironmentJobPollActivityResult, EnvironmentJobStartActivityRequest, - EnvironmentJobStartActivityResult, LlmGenerateActivityRequest, - PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult, PutBlobRequest, - ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, + EnvironmentJobStartActivityResult, LlmGenerateActivityRequest, PutBlobRequest, ReadBlobRequest, + ReadBlobResult, RuntimeProjectionRefreshActivityRequest, RuntimeProjectionRefreshActivityResult, ToolInvokeBatchActivityRequest, ToolInvokeCallActivityRequest, ToolInvokeCallActivityResult, ToolPreparePromiseControlsActivityRequest, }; +mod audio; mod common; mod compaction; mod context_refresh; mod environment_jobs; mod llm; -mod preprocess; +mod transcriptions; pub use context_refresh::subagent_catalog_snapshot; mod state; mod storage; @@ -46,13 +46,13 @@ mod subagents; mod tools; mod workflow_tools; -pub use preprocess::{ +pub use audio::{ AudioTranscodeError, AudioTranscodeOutput, AudioTranscodeRequest, AudioTranscoder, AudioTranscriber, AudioTranscription, AudioTranscriptionError, AudioTranscriptionRequest, FfmpegAudioTranscoder, default_audio_transcoder_from_env, }; pub use state::{ - ActivityState, LlmActivityDeps, PreprocessActivityDeps, RuntimeProjectionActivityDeps, + ActivityState, AudioActivityDeps, LlmActivityDeps, RuntimeProjectionActivityDeps, StorageActivityDeps, ToolActivityDeps, }; @@ -226,8 +226,8 @@ mod tests { temporal_workflow::WorkflowActivities::llm_generate.name() ); assert_eq!( - WorkerActivities::preprocess_run_input.name(), - temporal_workflow::WorkflowActivities::preprocess_run_input.name() + WorkerActivities::execute_transcription.name(), + temporal_workflow::WorkflowActivities::execute_transcription.name() ); assert_eq!( WorkerActivities::context_compact.name(), @@ -463,6 +463,26 @@ mod tests { #[activities] impl WorkerActivities { + #[activity(name = "WorkflowActivities::execute_transcription")] + pub async fn execute_transcription( + self: Arc, + ctx: ActivityContext, + args: temporal_workflow::TranscriptionWorkflowArgs, + ) -> Result { + if ctx + .info() + .workflow_execution + .as_ref() + .map(|w| w.workflow_id.as_str()) + != Some(temporal_workflow::transcription_workflow_id(&args).as_str()) + { + return Err(common::activity_error(anyhow::anyhow!( + "transcription workflow identity mismatch" + ))); + } + let state = self.state_for(&ctx).await?; + common::cancellable(&ctx, transcriptions::execute(&state, args)).await + } #[activity(name = ACTIVITY_CREATE_OR_LOAD_SESSION)] pub async fn create_or_load_session( self: Arc, @@ -534,16 +554,6 @@ impl WorkerActivities { common::cancellable(&ctx, llm::generate(state.llm(), attempt, request)).await } - #[activity(name = ACTIVITY_PREPROCESS_RUN_INPUT)] - pub async fn preprocess_run_input( - self: Arc, - ctx: ActivityContext, - request: PreprocessRunInputActivityRequest, - ) -> Result { - let state = self.state_for(&ctx).await?; - preprocess::preprocess_run_input(state.preprocess(), request).await - } - #[activity(name = ACTIVITY_CONTEXT_COMPACT)] pub async fn context_compact( self: Arc, diff --git a/crates/temporal-server/src/worker/activities/preprocess.rs b/crates/temporal-server/src/worker/activities/preprocess.rs deleted file mode 100644 index 89a2640ee..000000000 --- a/crates/temporal-server/src/worker/activities/preprocess.rs +++ /dev/null @@ -1,1189 +0,0 @@ -use std::{ - env, - ffi::{OsStr, OsString}, - path::{Path, PathBuf}, - process::Stdio, - sync::Arc, - time::Duration, -}; - -use async_trait::async_trait; -use engine::{ - ContextEntryInput, ContextEntryKind, ContextMessageRole, - storage::{BlobStore, BlobStoreError}, -}; -use llm_clients::{LlmApiError, openai::audio as oai}; -use llm_runtime::ModelProviderResolver; -use temporalio_sdk::activities::ActivityError; - -use crate::worker::{PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult}; -use temporal_workflow::{ - PreprocessRunInputFailure, PreprocessRunInputFailureKind, PreprocessRunInputOutcome, -}; - -use super::state::PreprocessActivityDeps; -use llm_clients::content::{AUDIO_TRANSCRIPT_PROVIDER_KIND, AudioTranscript}; - -const MAX_AUDIO_BYTES: u64 = 25 * 1024 * 1024; -const MAX_AUDIO_DURATION_MS: u64 = 10 * 60 * 1000; -const OPENAI_PROVIDER_ID: &str = "openai"; -const PROVIDER_ACCEPTED_AUDIO_MIMES: &[&str] = &[ - "audio/mpeg", - "audio/mp4", - "audio/wav", - "audio/webm", - "audio/ogg", -]; -const TRANSCODABLE_AUDIO_MIMES: &[&str] = &["audio/aac", "audio/amr", "audio/3gpp", "audio/3gpp2"]; -const TRANSCODED_AUDIO_MIME: &str = "audio/wav"; -const DEFAULT_FFMPEG_PATH: &str = "ffmpeg"; -const DEFAULT_TRANSCODE_TIMEOUT: Duration = Duration::from_secs(30); - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct AudioTranscriptionRequest { - pub bytes: Vec, - pub mime: String, - pub name: String, -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct AudioTranscription { - pub text: String, -} - -#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] -#[error("audio transcription failed: {message}")] -pub struct AudioTranscriptionError { - pub message: String, -} - -#[async_trait] -pub trait AudioTranscriber: Send + Sync { - async fn transcribe( - &self, - request: AudioTranscriptionRequest, - ) -> Result; -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct AudioTranscodeRequest { - pub bytes: Vec, - pub mime: String, - pub name: String, -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct AudioTranscodeOutput { - pub bytes: Vec, - pub mime: String, - pub name: String, -} - -#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] -#[error("audio transcode failed: {message}")] -pub struct AudioTranscodeError { - pub message: String, -} - -#[async_trait] -pub trait AudioTranscoder: Send + Sync { - async fn transcode( - &self, - request: AudioTranscodeRequest, - ) -> Result; -} - -pub struct UnavailableAudioTranscriber; - -#[async_trait] -impl AudioTranscriber for UnavailableAudioTranscriber { - async fn transcribe( - &self, - _request: AudioTranscriptionRequest, - ) -> Result { - Err(AudioTranscriptionError { - message: "audio transcriber is not configured".to_owned(), - }) - } -} - -#[derive(Clone, Debug)] -pub struct FfmpegAudioTranscoder { - ffmpeg_path: PathBuf, - timeout: Duration, - max_output_bytes: u64, -} - -impl FfmpegAudioTranscoder { - pub fn new(ffmpeg_path: impl Into) -> Self { - Self { - ffmpeg_path: ffmpeg_path.into(), - timeout: DEFAULT_TRANSCODE_TIMEOUT, - max_output_bytes: MAX_AUDIO_BYTES, - } - } - - pub fn with_timeout(mut self, timeout: Duration) -> Self { - self.timeout = timeout; - self - } - - pub fn from_env() -> Self { - let ffmpeg_path = env::var("LIGHTSPEED_FFMPEG_PATH") - .ok() - .filter(|value| !value.trim().is_empty()) - .unwrap_or_else(|| DEFAULT_FFMPEG_PATH.to_owned()); - let timeout = env::var("LIGHTSPEED_AUDIO_TRANSCODE_TIMEOUT_MS") - .ok() - .and_then(|value| value.parse::().ok()) - .filter(|millis| *millis > 0) - .map(Duration::from_millis) - .unwrap_or(DEFAULT_TRANSCODE_TIMEOUT); - Self::new(ffmpeg_path).with_timeout(timeout) - } -} - -#[async_trait] -impl AudioTranscoder for FfmpegAudioTranscoder { - async fn transcode( - &self, - request: AudioTranscodeRequest, - ) -> Result { - let temp_dir = tempfile::Builder::new() - .prefix("lightspeed-audio-transcode-") - .tempdir() - .map_err(|error| AudioTranscodeError { - message: format!("create transcode temp dir: {error}"), - })?; - let input_path = temp_dir - .path() - .join(format!("input.{}", extension_for_mime(&request.mime))); - let output_path = temp_dir.path().join("output.wav"); - tokio::fs::write(&input_path, &request.bytes) - .await - .map_err(|error| AudioTranscodeError { - message: format!("write transcode input: {error}"), - })?; - - let mut command = tokio::process::Command::new(&self.ffmpeg_path); - command - .args(ffmpeg_args(&input_path, &output_path)) - .kill_on_drop(true) - .stdin(Stdio::null()) - .stdout(Stdio::null()) - .stderr(Stdio::piped()); - let output = tokio::time::timeout(self.timeout, command.output()) - .await - .map_err(|_| AudioTranscodeError { - message: format!( - "ffmpeg did not finish within {}ms", - self.timeout.as_millis() - ), - })? - .map_err(|error| AudioTranscodeError { - message: format!("run ffmpeg: {error}"), - })?; - if !output.status.success() { - return Err(AudioTranscodeError { - message: ffmpeg_failure_message(&output.stderr), - }); - } - let metadata = - tokio::fs::metadata(&output_path) - .await - .map_err(|error| AudioTranscodeError { - message: format!("read ffmpeg output metadata: {error}"), - })?; - if metadata.len() > self.max_output_bytes { - return Err(AudioTranscodeError { - message: format!( - "transcoded audio {} is {} bytes; the limit is {} bytes", - output_path.display(), - metadata.len(), - self.max_output_bytes - ), - }); - } - let bytes = tokio::fs::read(&output_path) - .await - .map_err(|error| AudioTranscodeError { - message: format!("read ffmpeg output: {error}"), - })?; - Ok(AudioTranscodeOutput { - bytes, - mime: TRANSCODED_AUDIO_MIME.to_owned(), - name: transcoded_audio_filename(&request.name), - }) - } -} - -pub struct OpenAiAudioTranscriber { - client: Arc, - provider_keys: Arc, -} - -impl OpenAiAudioTranscriber { - pub fn new(client: Arc, provider_keys: Arc) -> Self { - Self { - client, - provider_keys, - } - } -} - -#[async_trait] -impl AudioTranscriber for OpenAiAudioTranscriber { - async fn transcribe( - &self, - request: AudioTranscriptionRequest, - ) -> Result { - let stored_key = self - .provider_keys - .resolve_model_provider(OPENAI_PROVIDER_ID) - .await - .map_err(|error| AudioTranscriptionError { - message: error.to_string(), - })?; - let response = self - .client - .create_transcription_with_auth( - oai::CreateTranscriptionRequest::new(oai::AudioFile { - bytes: request.bytes, - filename: request.name, - mime: request.mime, - }), - stored_key - .as_ref() - .and_then(|provider| provider.auth.as_ref()) - .map(|auth| auth.as_request_auth()), - ) - .await - .map_err(map_openai_error)?; - Ok(AudioTranscription { - text: response.parsed.text, - }) - } -} - -pub(super) async fn preprocess_run_input( - deps: &PreprocessActivityDeps, - request: PreprocessRunInputActivityRequest, -) -> Result { - let outcome = match rewrite_run_input( - deps.blobs.as_ref(), - deps.transcriber.as_ref(), - deps.transcoder.as_deref(), - request.input, - ) - .await - { - Ok(input) => PreprocessRunInputOutcome::Succeeded { input }, - Err(failure) => PreprocessRunInputOutcome::Failed { failure }, - }; - Ok(PreprocessRunInputActivityResult { outcome }) -} - -async fn rewrite_run_input( - blobs: &dyn BlobStore, - transcriber: &dyn AudioTranscriber, - transcoder: Option<&dyn AudioTranscoder>, - input: Vec, -) -> Result, PreprocessRunInputFailure> { - let mut rewritten = Vec::with_capacity(input.len()); - for entry in input { - if !is_audio_entry(&entry) { - rewritten.push(entry); - continue; - } - rewritten.push(transcribe_entry(blobs, transcriber, transcoder, entry).await?); - } - Ok(rewritten) -} - -async fn transcribe_entry( - blobs: &dyn BlobStore, - transcriber: &dyn AudioTranscriber, - transcoder: Option<&dyn AudioTranscoder>, - entry: ContextEntryInput, -) -> Result { - let mime = normalized_mime(entry.content.media_type.as_deref()); - let name = audio_label(&entry); - if !PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&mime.as_str()) - && !TRANSCODABLE_AUDIO_MIMES.contains(&mime.as_str()) - { - return Err(failure( - PreprocessRunInputFailureKind::UnsupportedAudioMime, - format!( - "unsupported audio mime type {mime} for {name}; accepted: {}", - accepted_audio_mimes().join(", ") - ), - )); - } - - let info = blobs - .stat_blob(&entry.content.content_ref) - .await - .map_err(map_audio_blob_error)?; - if info.byte_len > MAX_AUDIO_BYTES { - return Err(failure( - PreprocessRunInputFailureKind::AudioBlobTooLarge, - format!( - "audio entry {name} blob {} is {} bytes; the limit is {MAX_AUDIO_BYTES} bytes", - entry.content.content_ref, info.byte_len - ), - )); - } - - let bytes = blobs - .read_bytes(&entry.content.content_ref) - .await - .map_err(map_audio_blob_error)?; - if let Some(duration_ms) = audio_duration_ms(&mime, &bytes) - && duration_ms > MAX_AUDIO_DURATION_MS - { - return Err(failure( - PreprocessRunInputFailureKind::AudioDurationTooLong, - format!( - "audio entry {name} duration is {}; the limit is {}", - format_duration_ms(duration_ms), - format_duration_ms(MAX_AUDIO_DURATION_MS) - ), - )); - } - let audio = prepare_audio_for_transcription(bytes, &mime, &name, transcoder).await?; - if let Some(duration_ms) = audio_duration_ms(&audio.mime, &audio.bytes) - && duration_ms > MAX_AUDIO_DURATION_MS - { - return Err(failure( - PreprocessRunInputFailureKind::AudioDurationTooLong, - format!( - "audio entry {name} duration is {}; the limit is {}", - format_duration_ms(duration_ms), - format_duration_ms(MAX_AUDIO_DURATION_MS) - ), - )); - } - let transcript = transcriber - .transcribe(AudioTranscriptionRequest { - bytes: audio.bytes, - mime: audio.mime, - name: audio.name, - }) - .await - .map_err(map_transcription_error)?; - let transcript = AudioTranscript { - filename: name, - text: transcript.text.trim().to_owned(), - }; - let transcript_bytes = serde_json::to_vec(&transcript).map_err(|error| { - failure( - PreprocessRunInputFailureKind::TranscriptionFailure, - format!("failed to encode transcript: {error}"), - ) - })?; - let transcript_ref = blobs.put_bytes(transcript_bytes).await.map_err(|error| { - failure( - PreprocessRunInputFailureKind::TranscriptionFailure, - format!("failed to store audio transcript: {error}"), - ) - })?; - - Ok(ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: transcript_ref, - media_type: Some("application/json".to_owned()), - provider_kind: Some(AUDIO_TRANSCRIPT_PROVIDER_KIND.to_owned()), - }, - preview: Some(transcript.header().chars().take(256).collect()), - origin: entry.origin.clone(), - provenance_ref: Some(entry.content.content_ref.clone()), - token_estimate: None, - }) -} - -struct PreparedAudio { - bytes: Vec, - mime: String, - name: String, -} - -async fn prepare_audio_for_transcription( - bytes: Vec, - mime: &str, - name: &str, - transcoder: Option<&dyn AudioTranscoder>, -) -> Result { - if PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&mime) { - return Ok(PreparedAudio { - bytes, - mime: mime.to_owned(), - name: audio_filename(name), - }); - } - let Some(transcoder) = transcoder else { - return Err(failure( - PreprocessRunInputFailureKind::TranscoderUnavailable, - format!( - "audio entry {name} has mime type {mime}, which requires a transcoder, but no audio transcoder is configured" - ), - )); - }; - let output = transcoder - .transcode(AudioTranscodeRequest { - bytes, - mime: mime.to_owned(), - name: audio_filename(name), - }) - .await - .map_err(map_transcode_error)?; - if output.bytes.len() as u64 > MAX_AUDIO_BYTES { - return Err(failure( - PreprocessRunInputFailureKind::TranscodeFailure, - format!( - "transcoded audio entry {name} is {} bytes; the limit is {MAX_AUDIO_BYTES} bytes", - output.bytes.len() - ), - )); - } - if !PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&output.mime.as_str()) { - return Err(failure( - PreprocessRunInputFailureKind::TranscodeFailure, - format!( - "audio transcoder returned unsupported mime type {} for {name}", - output.mime - ), - )); - } - Ok(PreparedAudio { - bytes: output.bytes, - mime: output.mime, - name: output.name, - }) -} - -fn is_audio_entry(entry: &ContextEntryInput) -> bool { - entry - .content - .media_type - .as_deref() - .map(|mime| mime.trim().to_ascii_lowercase().starts_with("audio/")) - .unwrap_or(false) -} - -fn normalized_mime(mime: Option<&str>) -> String { - let mime = mime - .unwrap_or_default() - .split(';') - .next() - .unwrap_or_default() - .trim() - .to_ascii_lowercase(); - match mime.as_str() { - "audio/mp3" => "audio/mpeg", - "audio/x-m4a" | "audio/m4a" => "audio/mp4", - "audio/x-wav" | "audio/wave" | "audio/vnd.wave" => "audio/wav", - "audio/oga" | "audio/opus" => "audio/ogg", - "audio/x-aac" => "audio/aac", - "audio/3gp" => "audio/3gpp", - "audio/3g2" => "audio/3gpp2", - other => other, - } - .to_owned() -} - -fn audio_label(entry: &ContextEntryInput) -> String { - let Some(preview) = entry.preview.as_deref().map(str::trim) else { - return "audio".to_owned(); - }; - preview - .strip_prefix("[audio: ") - .and_then(|value| value.strip_suffix(']')) - .map(str::trim) - .filter(|value| !value.is_empty()) - .unwrap_or("audio") - .to_owned() -} - -fn audio_filename(label: &str) -> String { - let trimmed = label.trim(); - if trimmed.is_empty() || trimmed == "audio" { - "audio.ogg".to_owned() - } else { - trimmed.to_owned() - } -} - -fn transcoded_audio_filename(label: &str) -> String { - let filename = audio_filename(label); - let stem = Path::new(&filename) - .file_stem() - .and_then(OsStr::to_str) - .filter(|value| !value.trim().is_empty()) - .unwrap_or("audio"); - format!("{stem}.wav") -} - -fn accepted_audio_mimes() -> Vec<&'static str> { - PROVIDER_ACCEPTED_AUDIO_MIMES - .iter() - .chain(TRANSCODABLE_AUDIO_MIMES.iter()) - .copied() - .collect() -} - -fn extension_for_mime(mime: &str) -> &'static str { - match mime { - "audio/mpeg" => "mp3", - "audio/mp4" => "m4a", - "audio/wav" => "wav", - "audio/webm" => "webm", - "audio/ogg" => "ogg", - "audio/aac" => "aac", - "audio/amr" => "amr", - "audio/3gpp" => "3gp", - "audio/3gpp2" => "3g2", - _ => "bin", - } -} - -fn ffmpeg_args(input_path: &Path, output_path: &Path) -> Vec { - [ - OsString::from("-nostdin"), - OsString::from("-hide_banner"), - OsString::from("-loglevel"), - OsString::from("error"), - OsString::from("-y"), - OsString::from("-i"), - input_path.as_os_str().to_owned(), - OsString::from("-vn"), - OsString::from("-ac"), - OsString::from("1"), - OsString::from("-ar"), - OsString::from("16000"), - OsString::from("-acodec"), - OsString::from("pcm_s16le"), - OsString::from("-f"), - OsString::from("wav"), - output_path.as_os_str().to_owned(), - ] - .into_iter() - .collect() -} - -fn ffmpeg_failure_message(stderr: &[u8]) -> String { - let stderr = String::from_utf8_lossy(stderr); - let stderr = stderr.trim(); - if stderr.is_empty() { - "ffmpeg exited with a non-zero status".to_owned() - } else { - format!("ffmpeg exited with a non-zero status: {stderr}") - } -} - -fn audio_duration_ms(mime: &str, bytes: &[u8]) -> Option { - // Cheap duration enforcement is intentionally narrow for the first cut: - // OGG/Opus and WAV are covered; MP3/M4A/WebM rely on the byte cap unless - // a non-provider container is transcoded to WAV. - match mime { - "audio/ogg" => ogg_opus_duration_ms(bytes), - "audio/wav" => wav_duration_ms(bytes), - _ => None, - } -} - -fn ogg_opus_duration_ms(bytes: &[u8]) -> Option { - let mut offset = 0usize; - let mut last_granule = None; - while offset + 27 <= bytes.len() { - if &bytes[offset..offset + 4] != b"OggS" { - return None; - } - let granule = u64::from_le_bytes(bytes[offset + 6..offset + 14].try_into().ok()?); - if granule != u64::MAX { - last_granule = Some(granule); - } - let segments = bytes[offset + 26] as usize; - let lacing_start = offset + 27; - let lacing_end = lacing_start.checked_add(segments)?; - if lacing_end > bytes.len() { - return None; - } - let mut body_len = 0usize; - for segment in &bytes[lacing_start..lacing_end] { - body_len = body_len.checked_add(*segment as usize)?; - } - offset = lacing_end.checked_add(body_len)?; - } - last_granule.map(|samples| samples.saturating_mul(1000) / 48_000) -} - -fn wav_duration_ms(bytes: &[u8]) -> Option { - if bytes.len() < 12 || &bytes[0..4] != b"RIFF" || &bytes[8..12] != b"WAVE" { - return None; - } - let mut offset = 12usize; - let mut byte_rate = None; - let mut data_len = None; - while offset + 8 <= bytes.len() { - let chunk_id = &bytes[offset..offset + 4]; - let chunk_len = u32::from_le_bytes(bytes[offset + 4..offset + 8].try_into().ok()?) as usize; - let data_start = offset + 8; - let data_end = data_start.checked_add(chunk_len)?; - if data_end > bytes.len() { - return None; - } - if chunk_id == b"fmt " && chunk_len >= 16 { - byte_rate = Some(u32::from_le_bytes( - bytes[data_start + 8..data_start + 12].try_into().ok()?, - ) as u64); - } else if chunk_id == b"data" { - data_len = Some(chunk_len as u64); - } - if byte_rate.is_some() && data_len.is_some() { - break; - } - offset = data_end.checked_add(chunk_len % 2)?; - } - let byte_rate = byte_rate?; - if byte_rate == 0 { - return None; - } - Some(data_len?.saturating_mul(1000) / byte_rate) -} - -fn format_duration_ms(duration_ms: u64) -> String { - format!("{}s", duration_ms.div_ceil(1000)) -} - -fn map_audio_blob_error(error: BlobStoreError) -> PreprocessRunInputFailure { - match error { - BlobStoreError::NotFound { blob_ref } => failure( - PreprocessRunInputFailureKind::AudioBlobMissing, - format!("audio blob not found: {blob_ref}"), - ), - BlobStoreError::Store { message } => { - failure(PreprocessRunInputFailureKind::AudioBlobMissing, message) - } - } -} - -fn map_transcription_error(error: AudioTranscriptionError) -> PreprocessRunInputFailure { - failure( - PreprocessRunInputFailureKind::TranscriptionFailure, - error.message, - ) -} - -fn map_transcode_error(error: AudioTranscodeError) -> PreprocessRunInputFailure { - failure( - PreprocessRunInputFailureKind::TranscodeFailure, - error.message, - ) -} - -fn map_openai_error(error: LlmApiError) -> AudioTranscriptionError { - AudioTranscriptionError { - message: error.to_string(), - } -} - -fn failure( - kind: PreprocessRunInputFailureKind, - message: impl Into, -) -> PreprocessRunInputFailure { - PreprocessRunInputFailure { - kind, - message: message.into(), - } -} - -pub(super) fn default_openai_audio_transcriber( - provider_keys: Arc, -) -> Result, anyhow::Error> { - let client = oai::Client::new(oai::Config::from_env_allow_missing_key()) - .map_err(|error| anyhow::anyhow!("construct OpenAI audio client: {error}"))?; - Ok(Arc::new(OpenAiAudioTranscriber::new( - Arc::new(client), - provider_keys, - ))) -} - -pub fn default_audio_transcoder_from_env() -> anyhow::Result>> { - let Some(kind) = env::var("LIGHTSPEED_AUDIO_TRANSCODER") - .ok() - .map(|value| value.trim().to_ascii_lowercase()) - .filter(|value| !value.is_empty() && value != "none") - else { - return Ok(None); - }; - match kind.as_str() { - "ffmpeg" => Ok(Some(Arc::new(FfmpegAudioTranscoder::from_env()))), - other => anyhow::bail!( - "unsupported LIGHTSPEED_AUDIO_TRANSCODER value {other:?}; expected \"ffmpeg\" or \"none\"" - ), - } -} - -#[cfg(test)] -mod tests { - use super::*; - use engine::storage::{BlobStore, InMemoryBlobStore}; - use std::sync::Mutex; - - #[derive(Clone)] - struct StaticTranscriber { - text: String, - } - - #[async_trait] - impl AudioTranscriber for StaticTranscriber { - async fn transcribe( - &self, - _request: AudioTranscriptionRequest, - ) -> Result { - Ok(AudioTranscription { - text: self.text.clone(), - }) - } - } - - struct RecordingTranscriber { - text: String, - requests: Mutex>, - } - - impl RecordingTranscriber { - fn new(text: impl Into) -> Self { - Self { - text: text.into(), - requests: Mutex::new(Vec::new()), - } - } - - fn requests(&self) -> Vec { - self.requests - .lock() - .expect("recording transcriber requests") - .clone() - } - } - - #[async_trait] - impl AudioTranscriber for RecordingTranscriber { - async fn transcribe( - &self, - request: AudioTranscriptionRequest, - ) -> Result { - self.requests - .lock() - .expect("recording transcriber requests") - .push(request); - Ok(AudioTranscription { - text: self.text.clone(), - }) - } - } - - struct StaticTranscoder { - result: Result, - requests: Mutex>, - } - - impl StaticTranscoder { - fn new(result: Result) -> Self { - Self { - result, - requests: Mutex::new(Vec::new()), - } - } - - fn requests(&self) -> Vec { - self.requests - .lock() - .expect("static transcoder requests") - .clone() - } - } - - #[async_trait] - impl AudioTranscoder for StaticTranscoder { - async fn transcode( - &self, - request: AudioTranscodeRequest, - ) -> Result { - self.requests - .lock() - .expect("static transcoder requests") - .push(request); - self.result.clone() - } - } - - #[tokio::test(flavor = "current_thread")] - async fn audio_entries_are_rewritten_to_transcript_text() { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(b"OggS fake".to_vec()) - .await - .expect("put audio"); - let input = vec![ - text_entry(blobs.put_bytes(b"before".to_vec()).await.expect("put text")), - ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/ogg".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: voice.ogg]".to_owned()), - origin: Some("user:operator".into()), - provenance_ref: None, - token_estimate: None, - }, - ]; - - let rewritten = rewrite_run_input( - &blobs, - &StaticTranscriber { - text: "please summarize this".to_owned(), - }, - None, - input, - ) - .await - .expect("rewrite"); - - assert_eq!(rewritten.len(), 2); - assert_eq!(rewritten[1].origin.as_deref(), Some("user:operator")); - assert_eq!( - rewritten[1].content.media_type.as_deref(), - Some("application/json") - ); - let raw = blobs - .read_text(&rewritten[1].content.content_ref) - .await - .expect("read transcript"); - assert_eq!( - serde_json::from_str::(&raw).unwrap(), - AudioTranscript { - filename: "voice.ogg".into(), - text: "please summarize this".into() - } - ); - assert_eq!(rewritten[1].provenance_ref.as_ref(), Some(&audio_ref)); - assert_eq!( - api_projection::project_content_text(&blobs, &rewritten[1].content) - .await - .unwrap() - .as_deref(), - Some("please summarize this") - ); - } - - #[tokio::test(flavor = "current_thread")] - async fn transcodable_audio_without_transcoder_fails_whole_group() { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(b"aac fake".to_vec()) - .await - .expect("put audio"); - let input = vec![ - text_entry(blobs.put_bytes(b"before".to_vec()).await.expect("put text")), - ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/aac".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: voice.aac]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }, - ]; - - let failure = rewrite_run_input( - &blobs, - &StaticTranscriber { - text: "unused".to_owned(), - }, - None, - input, - ) - .await - .expect_err("transcodable audio without transcoder must fail group"); - - assert_eq!( - failure.kind, - PreprocessRunInputFailureKind::TranscoderUnavailable - ); - } - - #[tokio::test(flavor = "current_thread")] - async fn transcodable_audio_is_transcoded_before_transcription() { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(b"aac fake".to_vec()) - .await - .expect("put audio"); - let input = vec![ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/aac".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: voice.aac]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }]; - let transcriber = RecordingTranscriber::new("transcoded request"); - let transcoder = StaticTranscoder::new(Ok(AudioTranscodeOutput { - bytes: b"RIFF\x24\x00\x00\x00WAVEfmt ".to_vec(), - mime: "audio/wav".to_owned(), - name: "voice.wav".to_owned(), - })); - - let rewritten = rewrite_run_input(&blobs, &transcriber, Some(&transcoder), input) - .await - .expect("rewrite"); - - let raw = blobs - .read_text(&rewritten[0].content.content_ref) - .await - .expect("read transcript"); - assert_eq!( - serde_json::from_str::(&raw).unwrap(), - AudioTranscript { - filename: "voice.aac".into(), - text: "transcoded request".into() - } - ); - assert_eq!(rewritten[0].provenance_ref.as_ref(), Some(&audio_ref)); - assert_eq!( - api_projection::project_content_text(&blobs, &rewritten[0].content) - .await - .unwrap() - .as_deref(), - Some("transcoded request") - ); - let transcode_requests = transcoder.requests(); - assert_eq!(transcode_requests.len(), 1); - assert_eq!(transcode_requests[0].mime, "audio/aac"); - assert_eq!(transcode_requests[0].name, "voice.aac"); - assert_eq!(transcode_requests[0].bytes, b"aac fake"); - let transcription_requests = transcriber.requests(); - assert_eq!(transcription_requests.len(), 1); - assert_eq!(transcription_requests[0].mime, "audio/wav"); - assert_eq!(transcription_requests[0].name, "voice.wav"); - assert_eq!( - transcription_requests[0].bytes, - b"RIFF\x24\x00\x00\x00WAVEfmt " - ); - } - - #[tokio::test(flavor = "current_thread")] - async fn transcode_failure_fails_whole_group() { - let failure = transcode_failure(AudioTranscodeError { - message: "unsupported codec".to_owned(), - }) - .await; - - assert_eq!( - failure.kind, - PreprocessRunInputFailureKind::TranscodeFailure - ); - } - - #[test] - fn ffmpeg_command_args_normalize_audio_to_mono_wav() { - let args = ffmpeg_args(Path::new("/tmp/in.aac"), Path::new("/tmp/out.wav")); - let args: Vec = args - .iter() - .map(|arg| arg.to_string_lossy().into_owned()) - .collect(); - - assert_eq!( - args, - vec![ - "-nostdin", - "-hide_banner", - "-loglevel", - "error", - "-y", - "-i", - "/tmp/in.aac", - "-vn", - "-ac", - "1", - "-ar", - "16000", - "-acodec", - "pcm_s16le", - "-f", - "wav", - "/tmp/out.wav" - ] - ); - } - - #[tokio::test(flavor = "current_thread")] - #[ignore = "requires ffmpeg on PATH or LIGHTSPEED_FFMPEG_PATH"] - async fn ffmpeg_audio_transcoder_smoke_test() { - let transcoder = FfmpegAudioTranscoder::from_env(); - - let output = transcoder - .transcode(AudioTranscodeRequest { - bytes: tiny_wav_bytes(), - mime: "audio/wav".to_owned(), - name: "voice.wav".to_owned(), - }) - .await - .expect("ffmpeg should transcode tiny wav input"); - - assert_eq!(output.mime, "audio/wav"); - assert_eq!(output.name, "voice.wav"); - assert!(output.bytes.starts_with(b"RIFF")); - assert_eq!(audio_duration_ms(&output.mime, &output.bytes), Some(1000)); - } - - #[tokio::test(flavor = "current_thread")] - async fn long_ogg_audio_fails_duration_cap() { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(ogg_page(((MAX_AUDIO_DURATION_MS / 1000) + 1) * 48_000)) - .await - .expect("put audio"); - let input = vec![ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/ogg".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: long.ogg]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }]; - - let failure = rewrite_run_input( - &blobs, - &StaticTranscriber { - text: "unused".to_owned(), - }, - None, - input, - ) - .await - .expect_err("long audio must fail group"); - - assert_eq!( - failure.kind, - PreprocessRunInputFailureKind::AudioDurationTooLong - ); - assert!(failure.message.contains("long.ogg")); - } - - fn text_entry(content_ref: engine::BlobRef) -> ContextEntryInput { - ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef::text(content_ref), - preview: None, - origin: None, - provenance_ref: None, - token_estimate: None, - } - } - - async fn transcode_failure(error: AudioTranscodeError) -> PreprocessRunInputFailure { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(b"aac fake".to_vec()) - .await - .expect("put audio"); - let input = vec![ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/aac".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: voice.aac]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }]; - let transcoder = StaticTranscoder::new(Err(error)); - - rewrite_run_input( - &blobs, - &StaticTranscriber { - text: "unused".to_owned(), - }, - Some(&transcoder), - input, - ) - .await - .expect_err("transcode failure must reject group") - } - - fn tiny_wav_bytes() -> Vec { - let sample_rate = 8_000u32; - let channels = 1u16; - let bits_per_sample = 16u16; - let sample_count = sample_rate as usize; - let byte_rate = sample_rate * channels as u32 * bits_per_sample as u32 / 8; - let block_align = channels * bits_per_sample / 8; - let data_len = sample_count * block_align as usize; - - let mut bytes = Vec::with_capacity(44 + data_len); - bytes.extend_from_slice(b"RIFF"); - bytes.extend_from_slice(&(36 + data_len as u32).to_le_bytes()); - bytes.extend_from_slice(b"WAVE"); - bytes.extend_from_slice(b"fmt "); - bytes.extend_from_slice(&16u32.to_le_bytes()); - bytes.extend_from_slice(&1u16.to_le_bytes()); - bytes.extend_from_slice(&channels.to_le_bytes()); - bytes.extend_from_slice(&sample_rate.to_le_bytes()); - bytes.extend_from_slice(&byte_rate.to_le_bytes()); - bytes.extend_from_slice(&block_align.to_le_bytes()); - bytes.extend_from_slice(&bits_per_sample.to_le_bytes()); - bytes.extend_from_slice(b"data"); - bytes.extend_from_slice(&(data_len as u32).to_le_bytes()); - bytes.resize(44 + data_len, 0); - bytes - } - - fn ogg_page(granule_position: u64) -> Vec { - let mut page = Vec::new(); - page.extend_from_slice(b"OggS"); - page.push(0); - page.push(0); - page.extend_from_slice(&granule_position.to_le_bytes()); - page.extend_from_slice(&1u32.to_le_bytes()); - page.extend_from_slice(&0u32.to_le_bytes()); - page.extend_from_slice(&0u32.to_le_bytes()); - page.push(1); - page.push(1); - page.push(0); - page - } -} diff --git a/crates/temporal-server/src/worker/activities/state.rs b/crates/temporal-server/src/worker/activities/state.rs index f3c538bfb..014e496b3 100644 --- a/crates/temporal-server/src/worker/activities/state.rs +++ b/crates/temporal-server/src/worker/activities/state.rs @@ -31,7 +31,7 @@ use crate::{ worker::{BrokerSecretResolver, SessionTools, StoredProviderKeyResolver}, }; -use super::preprocess::{ +use super::audio::{ AudioTranscoder, AudioTranscriber, OpenAiAudioTranscriber, UnavailableAudioTranscriber, default_audio_transcoder_from_env, default_openai_audio_transcriber, }; @@ -77,8 +77,9 @@ pub struct RuntimeProjectionActivityDeps { pub(super) profiles: Option>, } +/// Audio I/O dependencies for standalone transcription. #[derive(Clone)] -pub struct PreprocessActivityDeps { +pub struct AudioActivityDeps { pub(super) blobs: Arc, pub(super) transcriber: Arc, pub(super) transcoder: Option>, @@ -119,7 +120,7 @@ pub struct ActivityState { tools: ToolActivityDeps, runtime_projection: Option, pub(super) preparation_store: Option>, - preprocess: PreprocessActivityDeps, + audio: AudioActivityDeps, environment_jobs: Option, workflow_tool_executions: Option, subagents: Option, @@ -151,7 +152,7 @@ impl ActivityState { }, runtime_projection: None, preparation_store: None, - preprocess: PreprocessActivityDeps { + audio: AudioActivityDeps { blobs: blobs.clone(), transcriber: Arc::new(UnavailableAudioTranscriber), transcoder: None, @@ -185,7 +186,7 @@ impl ActivityState { } pub fn with_audio_transcriber(mut self, transcriber: Arc) -> Self { - self.preprocess.transcriber = transcriber; + self.audio.transcriber = transcriber; self } @@ -207,7 +208,7 @@ impl ActivityState { } pub fn with_audio_transcoder(mut self, transcoder: Arc) -> Self { - self.preprocess.transcoder = Some(transcoder); + self.audio.transcoder = Some(transcoder); self } @@ -443,8 +444,8 @@ impl ActivityState { self.runtime_projection.as_ref() } - pub(super) fn preprocess(&self) -> &PreprocessActivityDeps { - &self.preprocess + pub(super) fn audio(&self) -> &AudioActivityDeps { + &self.audio } pub(super) fn environment_jobs(&self) -> Option<&EnvironmentJobActivityDeps> { diff --git a/crates/temporal-server/src/worker/activities/transcriptions.rs b/crates/temporal-server/src/worker/activities/transcriptions.rs new file mode 100644 index 000000000..81b4baa86 --- /dev/null +++ b/crates/temporal-server/src/worker/activities/transcriptions.rs @@ -0,0 +1,156 @@ +use super::{audio, state::ActivityState}; +use api::{TranscriptionFailure, TranscriptionFailureKind as Kind}; +use engine::{BlobRef, storage::BlobStore}; +use temporal_workflow::{TranscriptionActivityResult, TranscriptionWorkflowArgs}; +use temporalio_sdk::activities::ActivityError; + +fn activity_error(error: impl Into) -> ActivityError { + ActivityError::from(error.into()) +} +fn failure(kind: Kind, message: impl Into) -> TranscriptionFailure { + TranscriptionFailure { + kind, + message: message.into(), + } +} + +pub(super) async fn execute( + state: &ActivityState, + args: TranscriptionWorkflowArgs, +) -> Result { + match transcribe(state, &args).await { + Ok(transcript) => Ok(TranscriptionActivityResult::Succeeded { + transcript_ref: transcript.to_string(), + }), + Err(ExecutionError::Terminal(failure)) => { + Ok(TranscriptionActivityResult::Failed { failure }) + } + Err(ExecutionError::Retry(error)) => Err(activity_error(anyhow::anyhow!(error))), + } +} + +enum ExecutionError { + Terminal(TranscriptionFailure), + Retry(String), +} +impl From for ExecutionError { + fn from(error: engine::storage::BlobStoreError) -> Self { + if matches!(error, engine::storage::BlobStoreError::NotFound { .. }) { + Self::Terminal(failure( + Kind::InvalidAudio, + "Audio is unavailable; upload it again.", + )) + } else { + Self::Retry("Transcription content storage is unavailable.".into()) + } + } +} + +async fn transcribe( + state: &ActivityState, + record: &TranscriptionWorkflowArgs, +) -> Result { + let deps = state.audio(); + let request = &record.request; + let source = BlobRef::parse(&request.audio.blob_ref).map_err(|_| { + ExecutionError::Terminal(failure(Kind::InvalidAudio, "Invalid audio reference.")) + })?; + let mime = audio::normalized_mime(Some(&request.audio.mime)); + if !audio::PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&mime.as_str()) + && !audio::TRANSCODABLE_AUDIO_MIMES.contains(&mime.as_str()) + { + return Err(ExecutionError::Terminal(failure( + Kind::InvalidAudio, + "Unsupported audio MIME type.", + ))); + } + if deps.blobs.stat_blob(&source).await?.byte_len > audio::MAX_AUDIO_BYTES { + return Err(ExecutionError::Terminal(failure( + Kind::InvalidAudio, + "Audio exceeds the 25 MiB limit.", + ))); + } + let bytes = deps.blobs.read_bytes(&source).await?; + check_duration(&mime, &bytes)?; + let prepared = audio::prepare_audio_for_transcription( + bytes, + &mime, + &request.audio.name, + deps.transcoder.as_deref(), + ) + .await + .map_err(|e| ExecutionError::Terminal(failure(Kind::InvalidAudio, e.to_string())))?; + check_duration(&prepared.mime, &prepared.bytes)?; + let result = deps + .transcriber + .transcribe(audio::AudioTranscriptionRequest { + bytes: prepared.bytes, + mime: prepared.mime, + name: prepared.name, + model: record.model.clone(), + language: request.language.clone(), + prompt: request.prompt.clone(), + }) + .await + .map_err(|e| { + if e.retryable { + ExecutionError::Retry(e.message) + } else { + ExecutionError::Terminal(failure( + if e.configuration { + Kind::Configuration + } else { + Kind::Provider + }, + e.message, + )) + } + })?; + if result.text.len() > 1024 * 1024 { + return Err(ExecutionError::Terminal(failure( + Kind::Provider, + "Transcript exceeds the 1 MiB limit.", + ))); + } + let result_ref = deps + .blobs + .put_bytes(result.text.trim().as_bytes().to_vec()) + .await?; + Ok(result_ref) +} + +fn check_duration(mime: &str, bytes: &[u8]) -> Result<(), ExecutionError> { + if audio::audio_duration_ms(mime, bytes) + .is_some_and(|duration| duration > audio::MAX_AUDIO_DURATION_MS) + { + return Err(ExecutionError::Terminal(failure( + Kind::InvalidAudio, + "Audio exceeds the ten-minute duration limit.", + ))); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn duration_limit_is_enforced_before_provider_call() { + // One complete Ogg page with a one-byte body and a 48 kHz granule clock. + let mut page = b"OggS".to_vec(); + page.resize(29, 0); + page[26] = 1; + page[27] = 1; + page[6..14].copy_from_slice(&(600_u64 * 48_000).to_le_bytes()); + assert!(check_duration("audio/ogg", &page).is_ok()); + page[6..14].copy_from_slice(&(601_u64 * 48_000).to_le_bytes()); + assert!(matches!( + check_duration("audio/ogg", &page), + Err(ExecutionError::Terminal(TranscriptionFailure { + kind: Kind::InvalidAudio, + .. + })) + )); + } +} diff --git a/crates/temporal-server/src/worker/channels.rs b/crates/temporal-server/src/worker/channels.rs index 64b4dcbf8..7a843a519 100644 --- a/crates/temporal-server/src/worker/channels.rs +++ b/crates/temporal-server/src/worker/channels.rs @@ -31,6 +31,51 @@ impl ChannelWorkerActivities { #[activities] impl ChannelWorkerActivities { + #[activity(name = ACTIVITY_CHAT_TRANSCRIBE_MEDIA)] + pub async fn transcribe_media( + self: Arc, + _ctx: ActivityContext, + request: ChatTranscribeMediaRequest, + ) -> Result { + let api = self.universes.api_for(request.active.universe_id).await?; + let active = + crate::channels::activities::assert_trigger_active(&api, request.active.clone()) + .await?; + if let ChatTriggerActiveResult::Inactive { reason } = active { + return Err(ActivityError::application( + temporalio_common::error::ApplicationFailure::non_retryable(anyhow::anyhow!( + reason + )), + )); + } + let owner = api::Attribution::Internal { + component: "channel".into(), + cause: request.active.trigger_id.to_string(), + }; + api.admit_transcription( + api::TranscriptionStartParams { + idempotency_key: request.idempotency_key, + audio: request.audio, + model: None, + language: None, + prompt: None, + }, + owner, + ) + .await + .map_err(|error| { + if matches!(error.kind, api::AgentApiErrorKind::Internal) { + ActivityError::from(anyhow::anyhow!(error.to_string())) + } else { + ActivityError::application( + temporalio_common::error::ApplicationFailure::non_retryable(anyhow::anyhow!( + error.to_string() + )), + ) + } + }) + } + #[activity(name = ACTIVITY_CHAT_TOOL_DECLARATIONS)] pub async fn chat_tool_declarations( self: Arc, diff --git a/crates/temporal-server/src/worker/mod.rs b/crates/temporal-server/src/worker/mod.rs index bfebe1f10..2775130b4 100644 --- a/crates/temporal-server/src/worker/mod.rs +++ b/crates/temporal-server/src/worker/mod.rs @@ -20,9 +20,9 @@ use temporal_workflow::{ }; pub use activities::{ - ActivityState, AudioTranscodeError, AudioTranscodeOutput, AudioTranscodeRequest, - AudioTranscoder, AudioTranscriber, AudioTranscription, AudioTranscriptionError, - AudioTranscriptionRequest, FfmpegAudioTranscoder, LlmActivityDeps, PreprocessActivityDeps, + ActivityState, AudioActivityDeps, AudioTranscodeError, AudioTranscodeOutput, + AudioTranscodeRequest, AudioTranscoder, AudioTranscriber, AudioTranscription, + AudioTranscriptionError, AudioTranscriptionRequest, FfmpegAudioTranscoder, LlmActivityDeps, RuntimeProjectionActivityDeps, StorageActivityDeps, ToolActivityDeps, WorkerActivities, default_audio_transcoder_from_env, subagent_catalog_snapshot, }; @@ -41,21 +41,19 @@ pub use temporal_workflow::{ ACTIVITY_CREATE_OR_LOAD_SESSION, ACTIVITY_ENVIRONMENT_JOB_CANCEL, ACTIVITY_ENVIRONMENT_JOB_POLL, ACTIVITY_ENVIRONMENT_JOB_PREPARE_WORKFLOW_TOOL, ACTIVITY_ENVIRONMENT_JOB_START, ACTIVITY_LLM_GENERATE, ACTIVITY_MATERIALIZE_AWAIT_RESULT, - ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PREPROCESS_RUN_INPUT, ACTIVITY_PUT_BLOB, - ACTIVITY_READ_BLOB, ACTIVITY_RUNTIME_PROJECTION_REFRESH, - ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, - ACTIVITY_SUBAGENT_RESOLVE, ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, - ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, - AgentSessionWorkflow, AppendEventsRequest, ContextCompactActivityRequest, - CreateOrLoadSessionRequest, CreateOrLoadSessionResult, DEFAULT_TASK_QUEUE, - DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, EnvironmentJobCancelActivityRequest, - EnvironmentJobPollActivityRequest, EnvironmentJobPollActivityResult, - EnvironmentJobStartActivityRequest, EnvironmentJobStartActivityResult, EnvironmentJobWorkflow, - EnvironmentJobWorkflowArgs, FAKE_TOOL_NAME, LlmGenerateActivityRequest, - PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult, PutBlobRequest, - ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, - RuntimeProjectionRefreshActivityResult, SubagentExecutionWorkflow, - ToolInvokeBatchActivityRequest, ToolInvokeCallActivityRequest, + ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PUT_BLOB, ACTIVITY_READ_BLOB, + ACTIVITY_RUNTIME_PROJECTION_REFRESH, ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, + ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, ACTIVITY_SUBAGENT_RESOLVE, + ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, + ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, AgentSessionWorkflow, AppendEventsRequest, + ContextCompactActivityRequest, CreateOrLoadSessionRequest, CreateOrLoadSessionResult, + DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, + EnvironmentJobCancelActivityRequest, EnvironmentJobPollActivityRequest, + EnvironmentJobPollActivityResult, EnvironmentJobStartActivityRequest, + EnvironmentJobStartActivityResult, EnvironmentJobWorkflow, EnvironmentJobWorkflowArgs, + FAKE_TOOL_NAME, LlmGenerateActivityRequest, PutBlobRequest, ReadBlobRequest, ReadBlobResult, + RuntimeProjectionRefreshActivityRequest, RuntimeProjectionRefreshActivityResult, + SubagentExecutionWorkflow, ToolInvokeBatchActivityRequest, ToolInvokeCallActivityRequest, ToolPreparePromiseControlsActivityRequest, connect_temporal, default_run_config, default_session_config, }; @@ -99,6 +97,7 @@ pub fn sessions_worker( ) -> anyhow::Result { let worker_options = WorkerOptions::new(task_queue) .register_workflow::() + .register_workflow::() .register_workflow::() .register_workflow::() .register_activities(activities) diff --git a/crates/temporal-server/tests/channels_live.rs b/crates/temporal-server/tests/channels_live.rs index 3f9921d13..cfbfd1c82 100644 --- a/crates/temporal-server/tests/channels_live.rs +++ b/crates/temporal-server/tests/channels_live.rs @@ -68,6 +68,8 @@ const WAIT: Duration = Duration::from_secs(90); pub struct FakeConnector { deliveries: Arc>>, typing_started: Arc>, + blobs: Option>, + transcription_calls: Arc, } #[activities] @@ -94,13 +96,23 @@ impl FakeConnector { pub async fn prepare_channel_media( self: Arc, _ctx: ActivityContext, - _input: PrepareChannelMediaInput, + input: PrepareChannelMediaInput, ) -> Result { - Err(ActivityError::application( - temporalio_common::error::ApplicationFailure::non_retryable(anyhow::anyhow!( - "the fake connector serves no media" - )), - )) + let blob_ref = self + .blobs + .as_ref() + .expect("fake CAS") + .put_bytes(b"OggS fake voice".to_vec()) + .await + .map_err(|e| ActivityError::from(anyhow::anyhow!(e)))?; + Ok(PrepareChannelMediaResult { + item: channels::media::PreparedMediaItem { + blob_ref: blob_ref.to_string(), + kind: input.media.kind, + mime: input.media.mime, + name: input.media.name, + }, + }) } #[activity(name = ACTIVITY_CONNECTOR_MAINTAIN_TYPING)] @@ -154,6 +166,10 @@ where .with_channel_task_queue(queues.channels.clone()) .build(), ); + let connector = FakeConnector { + blobs: Some(store.clone()), + ..Default::default() + }; let blobs: Arc = store.clone(); let llm = Arc::new(FakeLlm::new(blobs.clone()).with_tool_rounds(0)) as Arc; let tools = Arc::new(FakeTools::new(blobs)) as Arc; @@ -163,7 +179,9 @@ where queues.sessions.clone(), WorkerActivities::for_universe( universe, - ActivityState::from_pg_store(store.clone(), llm, tools), + ActivityState::from_pg_store(store.clone(), llm, tools).with_audio_transcriber( + Arc::new(FakeVoiceTranscriber(connector.transcription_calls.clone())), + ), ), )?; let mut bots = bots_worker( @@ -201,7 +219,6 @@ where }), ) .await?; - let connector = FakeConnector::default(); let connector_queue = connector_task_queue(universe, &ChannelProvider::new("telegram"), &account_id); let mut connector_worker = Worker::new( @@ -591,6 +608,7 @@ async fn temporal_live_chat_rebuilds_collected_declarations_and_retains_assets() let deleted = store.delete_dead_blobs(std::slice::from_ref(&tools_ref), 2, &[]).await?; assert_eq!(deleted.len(), 1, "a workflow's cached ref alone does not retain its declaration"); let result = emit_chat_event(&live.api, ChatEmitEventRequest { + transcript_refs: Default::default(), universe_id: store.config().universe_id, bot_id: bot_id.clone(), trigger_id, account_id: live.account_id.clone(), provider: ChannelProvider::new("telegram"), @@ -632,3 +650,105 @@ async fn temporal_live_chat_rebuilds_collected_declarations_and_retains_assets() Ok(()) }).await } + +struct FakeVoiceTranscriber(Arc); +#[async_trait::async_trait] +impl temporal_server::worker::AudioTranscriber for FakeVoiceTranscriber { + async fn transcribe( + &self, + request: temporal_server::worker::AudioTranscriptionRequest, + ) -> Result< + temporal_server::worker::AudioTranscription, + temporal_server::worker::AudioTranscriptionError, + > { + assert_eq!(request.model.model, "channel-speech"); + self.0.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + Ok(temporal_server::worker::AudioTranscription { + text: "/reset spoken words".into(), + }) + } +} + +#[tokio::test(flavor = "current_thread")] +#[ignore = "requires local Temporal and PostgreSQL"] +async fn channels_live_voice_preparation_preserves_provenance_and_redelivery_identity() +-> anyhow::Result<()> { + run_channels_live(|live| async move { + let (bot_id, _, _) = + create_bot_with_chat(&live.api, &live.account_id, ChatPairing::Open).await?; + let current = live + .api + .read_model_defaults(api::ModelDefaultsReadParams {}) + .await? + .result + .defaults; + live.api + .put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::SpeechToText, + model: Some(api::ModelConfig { + provider_id: "fake".into(), + api_kind: "openai:audio-transcriptions".into(), + model: "channel-speech".into(), + }), + expected_revision: current.revision, + }) + .await?; + let mut message = inbound("voice-chat", "voice-1", ""); + message.media.push(api::ChannelInboundMedia { + file_id: "voice-file".into(), + kind: api::ChannelMediaKind::Audio, + mime: "audio/ogg".into(), + name: Some("voice.ogg".into()), + byte_size: None, + }); + assert_eq!( + admit(&live.api, &live.account_id, message.clone()).await?, + ChannelInboundDecision::Bound + ); + wait_for_deliveries(&live.connector, 1).await?; + admit(&live.api, &live.account_id, message).await?; + // Wait behind the redelivery so its ordered processing has completed. + admit( + &live.api, + &live.account_id, + inbound("voice-chat", "text-2", "next message"), + ) + .await?; + wait_for_deliveries(&live.connector, 2).await?; + assert_eq!( + live.connector + .transcription_calls + .load(std::sync::atomic::Ordering::SeqCst), + 1 + ); + let events = live + .api + .list_bot_events(BotEventListParams { + bot_id, + limit: Some(10), + cursor: None, + }) + .await? + .result + .events; + let voice = events + .iter() + .find(|event| !event.media.is_empty()) + .expect("voice event"); + assert!(voice.media[0].text_ref.is_some()); + let store = pg_store_from_env().await?; + let doc: serde_json::Value = serde_json::from_slice( + &store + .read_bytes(&engine::BlobRef::parse(&voice.document_ref)?) + .await?, + )?; + assert_eq!(doc["data"]["message"]["text"], "/reset spoken words"); + assert_eq!( + events.iter().filter(|e| e.kind == "chat.message").count(), + 2, + "spoken command remains ordinary message content" + ); + Ok(()) + }) + .await +} diff --git a/crates/temporal-server/tests/mcp_live.rs b/crates/temporal-server/tests/mcp_live.rs index 3fc25f85d..503ff18c6 100644 --- a/crates/temporal-server/tests/mcp_live.rs +++ b/crates/temporal-server/tests/mcp_live.rs @@ -517,6 +517,7 @@ async fn run_matrix_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "Exercise the native MCP matrix".to_owned(), }], @@ -1406,6 +1407,7 @@ async fn run_approval_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("approval test {index}"), }], @@ -1578,6 +1580,7 @@ async fn run_native_mcp_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "List the configured models through the Configurator MCP".to_owned(), }], @@ -2247,6 +2250,7 @@ async fn run_mixed_batch_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "Schedule a timer, then await it while listing models".to_owned(), }], diff --git a/crates/temporal-server/tests/profiles_live.rs b/crates/temporal-server/tests/profiles_live.rs index 79fe0971a..7edbffe9b 100644 --- a/crates/temporal-server/tests/profiles_live.rs +++ b/crates/temporal-server/tests/profiles_live.rs @@ -464,6 +464,7 @@ async fn run_profiles_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "run after profile start".to_owned(), }], diff --git a/crates/temporal-server/tests/runs_live.rs b/crates/temporal-server/tests/runs_live.rs index 913dcf421..7ba0f0dab 100644 --- a/crates/temporal-server/tests/runs_live.rs +++ b/crates/temporal-server/tests/runs_live.rs @@ -456,6 +456,7 @@ async fn run_steering_live_client( session_id: session_id.as_str().to_owned(), run_id: run.id.clone(), items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "also mention the moon".to_owned(), }], @@ -489,6 +490,7 @@ async fn run_steering_live_client( session_id: session_id.as_str().to_owned(), run_id: run.id.clone(), items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "too late".to_owned(), }], @@ -521,6 +523,7 @@ async fn run_steering_final_turn_live_client( session_id: session_id.as_str().to_owned(), run_id: run.id.clone(), items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "one more thing".to_owned(), }], @@ -622,6 +625,7 @@ async fn run_queue_live_client( session_id: session_id.as_str().to_owned(), run_id: second.id.clone(), items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "nope".to_owned(), }], @@ -757,6 +761,7 @@ async fn run_parallel_tool_batch_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "run a parallel tool batch".to_owned(), }], @@ -877,6 +882,7 @@ async fn run_transient_llm_retry_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "retry through transient provider failures".to_owned(), }], @@ -958,6 +964,7 @@ async fn run_llm_retry_exhaustion_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "exhaust the provider retry budget".to_owned(), }], @@ -1007,6 +1014,7 @@ async fn run_llm_retry_exhaustion_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "recover after the provider outage".to_owned(), }], @@ -1084,6 +1092,7 @@ async fn run_unbounded_hosted_run_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "complete thirty verification tool rounds".to_owned(), }], diff --git a/crates/temporal-server/tests/runs_live_slow.rs b/crates/temporal-server/tests/runs_live_slow.rs index 6c31cac4d..c646fa3be 100644 --- a/crates/temporal-server/tests/runs_live_slow.rs +++ b/crates/temporal-server/tests/runs_live_slow.rs @@ -177,6 +177,7 @@ async fn start_text_run( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: text.to_owned(), }], diff --git a/crates/temporal-server/tests/sessions_live.rs b/crates/temporal-server/tests/sessions_live.rs index 2cefd34a5..34ac48917 100644 --- a/crates/temporal-server/tests/sessions_live.rs +++ b/crates/temporal-server/tests/sessions_live.rs @@ -526,6 +526,7 @@ async fn run_fake_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "hello temporal agent".to_owned(), }], @@ -544,6 +545,7 @@ async fn run_fake_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "second session-start input".to_owned(), }], @@ -564,6 +566,7 @@ async fn run_fake_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "second session-start input".to_owned(), }], @@ -581,6 +584,7 @@ async fn run_fake_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "different input".to_owned(), }], @@ -783,6 +787,7 @@ async fn run_continue_as_new_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "first run before continue as new".to_owned(), }], @@ -810,6 +815,7 @@ async fn run_continue_as_new_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "second run after continue as new".to_owned(), }], @@ -862,6 +868,7 @@ async fn run_missing_session_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "this should not create a session".to_owned(), }], @@ -922,6 +929,7 @@ async fn run_context_append_live_client( ContextAppendEntry { key: "channel.room.msg-1".to_owned(), item: InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: borrowed.to_string(), }, @@ -929,6 +937,7 @@ async fn run_context_append_live_client( ContextAppendEntry { key: "channel.room.msg-2".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: second_text.to_owned(), }, @@ -990,6 +999,7 @@ async fn run_context_append_live_client( ContextAppendEntry { key: "channel.room.msg-1".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: first_text.to_owned(), }, @@ -997,6 +1007,7 @@ async fn run_context_append_live_client( ContextAppendEntry { key: "channel.room.msg-2".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: second_text.to_owned(), }, @@ -1025,6 +1036,7 @@ async fn run_context_append_live_client( entries: vec![ContextAppendEntry { key: "channel.room.msg-2".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: "[telegram:group Engineering] Bob (12:02): edited message".to_owned(), }, @@ -1059,6 +1071,7 @@ async fn run_context_append_live_client( entries: vec![ContextAppendEntry { key: "channel.room.msg-3".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: " ".to_owned(), }, @@ -1079,6 +1092,7 @@ async fn run_context_append_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "summarize the room".to_owned(), }], @@ -1150,6 +1164,7 @@ async fn run_admission_failure_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "valid run after malformed command".to_owned(), }], @@ -1180,6 +1195,7 @@ async fn run_admission_failure_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "run after close should be rejected".to_owned(), }], @@ -1308,6 +1324,7 @@ async fn run_openai_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "Reply with exactly: real llm agent ok".to_owned(), }], @@ -1396,7 +1413,7 @@ async fn run_builtin_tool_live_client( submission_id: None, session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { - items: vec![InputItem::Text { origin: None, + items: vec![InputItem::Text { provenance_ref: None, origin: None, text: "Call sleep with delay_ms=1, await the returned promise, then reply exactly: temporal tool ok".to_owned(), }], }, diff --git a/crates/temporal-server/tests/subagents_live.rs b/crates/temporal-server/tests/subagents_live.rs index 4b4f09487..7ce8ec11d 100644 --- a/crates/temporal-server/tests/subagents_live.rs +++ b/crates/temporal-server/tests/subagents_live.rs @@ -665,6 +665,7 @@ async fn start_subagent_parent_with_features( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: script.to_owned(), }], @@ -1278,6 +1279,7 @@ async fn run_agent_run_inherit_environment_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("AGENT_RUN {child_profile_id}"), }], diff --git a/crates/temporal-server/tests/support/live.rs b/crates/temporal-server/tests/support/live.rs index 0b2d92ae8..1b46743bd 100644 --- a/crates/temporal-server/tests/support/live.rs +++ b/crates/temporal-server/tests/support/live.rs @@ -336,10 +336,10 @@ pub async fn fake_worker_activities_with_stall_switch( pub async fn fake_worker_activities_with_audio_transcriber( transcriber: Arc, ) -> anyhow::Result { - fake_worker_activities_with_audio_preprocessors(transcriber, None).await + fake_worker_activities_with_audio_processing(transcriber, None).await } -pub async fn fake_worker_activities_with_audio_preprocessors( +pub async fn fake_worker_activities_with_audio_processing( transcriber: Arc, transcoder: Option>, ) -> anyhow::Result { @@ -433,6 +433,7 @@ pub async fn start_text_run( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: text.to_owned(), }], diff --git a/crates/temporal-server/tests/preprocess_live.rs b/crates/temporal-server/tests/transcriptions_live.rs similarity index 55% rename from crates/temporal-server/tests/preprocess_live.rs rename to crates/temporal-server/tests/transcriptions_live.rs index 17dc34565..15491c3ad 100644 --- a/crates/temporal-server/tests/preprocess_live.rs +++ b/crates/temporal-server/tests/transcriptions_live.rs @@ -4,15 +4,14 @@ use std::sync::{Arc, Mutex}; use api::{ AgentApiService, BlobPutItem, BlobPutParams, ContextEntryKindView, ContextMessageRoleView, - InputItem, MediaKind, RunStartParams, RunStartSource, RunStatus, SessionConfig, - SessionStartParams, + InputItem, RunStartParams, RunStartSource, RunStatus, SessionConfig, SessionStartParams, }; use api_projection::model_to_api; use async_trait::async_trait; use base64::{Engine as _, engine::general_purpose::STANDARD as BASE64}; use engine::SessionId; use support::live::{ - LIVE_TEST_LOCK, fake_worker_activities_with_audio_preprocessors, + LIVE_TEST_LOCK, fake_worker_activities_with_audio_processing, fake_worker_activities_with_audio_transcriber, final_assistant_text, live_workflow_handle, require_storage_live_env, run_with_live_worker, wait_for_terminal_run, }; @@ -32,7 +31,7 @@ const TRANSCRIPT_TEXT: &str = "please file the deployment note from this audio"; #[tokio::test(flavor = "current_thread")] #[ignore = "requires ./dev.sh infra or compatible Temporal + Postgres env"] -async fn preprocess_live_audio_input_is_transcribed_before_admission() -> anyhow::Result<()> { +async fn transcription_live_prepared_input_retains_source_audio() -> anyhow::Result<()> { let _lock = LIVE_TEST_LOCK.lock().await; let _ = dotenvy::dotenv(); require_storage_live_env()?; @@ -40,14 +39,14 @@ async fn preprocess_live_audio_input_is_transcribed_before_admission() -> anyhow let transcriber = Arc::new(RecordingAudioTranscriber::new(TRANSCRIPT_TEXT)); let activities = fake_worker_activities_with_audio_transcriber(transcriber.clone()).await?; run_with_live_worker(activities, move |client, task_queue, session_id| { - run_audio_preprocess_live_client(client, task_queue, session_id, transcriber) + run_audio_transcription_live_client(client, task_queue, session_id, transcriber) }) .await } #[tokio::test(flavor = "current_thread")] #[ignore = "requires ./dev.sh infra or compatible Temporal + Postgres env"] -async fn preprocess_live_transcodable_audio_is_transcoded_before_admission() -> anyhow::Result<()> { +async fn transcription_live_transcodable_audio_is_prepared_outside_session() -> anyhow::Result<()> { let _lock = LIVE_TEST_LOCK.lock().await; let _ = dotenvy::dotenv(); require_storage_live_env()?; @@ -59,13 +58,11 @@ async fn preprocess_live_transcodable_audio_is_transcoded_before_admission() -> "audio/wav", "voice-note.wav", )); - let activities = fake_worker_activities_with_audio_preprocessors( - transcriber.clone(), - Some(transcoder.clone()), - ) - .await?; + let activities = + fake_worker_activities_with_audio_processing(transcriber.clone(), Some(transcoder.clone())) + .await?; run_with_live_worker(activities, move |client, task_queue, session_id| { - run_transcodable_audio_preprocess_live_client( + run_transcodable_audio_transcription_live_client( client, task_queue, session_id, @@ -77,7 +74,7 @@ async fn preprocess_live_transcodable_audio_is_transcoded_before_admission() -> .await } -async fn run_audio_preprocess_live_client( +async fn run_audio_transcription_live_client( client: Client, task_queue: String, session_id: SessionId, @@ -111,18 +108,23 @@ async fn run_audio_preprocess_live_client( }], }) .await?; + let transcript_ref = transcribe( + &api, + audio.result.blobs[0].blob_ref.clone(), + "audio/ogg", + "voice-note.ogg", + ) + .await?; let started = api .start_run(RunStartParams { notify_on_terminal: None, submission_id: None, session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { - items: vec![InputItem::Media { + items: vec![InputItem::TextRef { origin: None, - blob_ref: audio.result.blobs[0].blob_ref.clone(), - mime: "audio/ogg".to_owned(), - kind: MediaKind::Audio, - name: Some("voice-note.ogg".to_owned()), + blob_ref: transcript_ref, + provenance_ref: Some(audio.result.blobs[0].blob_ref.clone()), }], }, config: None, @@ -154,18 +156,18 @@ async fn run_audio_preprocess_live_client( .entries .iter() .find(|entry| { - entry.content.provider_kind.as_deref() - == Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) + entry.provenance_ref.as_deref() == Some(audio.result.blobs[0].blob_ref.as_str()) }) - .expect("structured transcript"); + .expect("text with source provenance"); assert_eq!( transcript_entry.content.media_type.as_deref(), - Some("application/json") + Some("text/plain") ); assert_eq!( transcript_entry.provenance_ref.as_deref(), Some(audio.result.blobs[0].blob_ref.as_str()) ); + assert!(transcript_entry.content.provider_kind.is_none()); assert_eq!(transcript_entry.text.as_deref(), Some(TRANSCRIPT_TEXT)); assert!(!transcript_entry.text_truncated); assert!( @@ -184,14 +186,14 @@ async fn run_audio_preprocess_live_client( let _ = handle .terminate( WorkflowTerminateOptions::builder() - .reason("agent audio preprocess live test cleanup") + .reason("agent audio transcription live test cleanup") .build(), ) .await; Ok(()) } -async fn run_transcodable_audio_preprocess_live_client( +async fn run_transcodable_audio_transcription_live_client( client: Client, task_queue: String, session_id: SessionId, @@ -227,18 +229,23 @@ async fn run_transcodable_audio_preprocess_live_client( }], }) .await?; + let transcript_ref = transcribe( + &api, + audio.result.blobs[0].blob_ref.clone(), + "audio/x-aac", + "voice-note.aac", + ) + .await?; let started = api .start_run(RunStartParams { notify_on_terminal: None, submission_id: None, session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { - items: vec![InputItem::Media { + items: vec![InputItem::TextRef { origin: None, - blob_ref: audio.result.blobs[0].blob_ref.clone(), - mime: "audio/x-aac".to_owned(), - kind: MediaKind::Audio, - name: Some("voice-note.aac".to_owned()), + blob_ref: transcript_ref, + provenance_ref: Some(audio.result.blobs[0].blob_ref.clone()), }], }, config: None, @@ -268,18 +275,18 @@ async fn run_transcodable_audio_preprocess_live_client( .entries .iter() .find(|entry| { - entry.content.provider_kind.as_deref() - == Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) + entry.provenance_ref.as_deref() == Some(audio.result.blobs[0].blob_ref.as_str()) }) - .expect("structured transcript"); + .expect("text with source provenance"); assert_eq!( transcript_entry.content.media_type.as_deref(), - Some("application/json") + Some("text/plain") ); assert_eq!( transcript_entry.provenance_ref.as_deref(), Some(audio.result.blobs[0].blob_ref.as_str()) ); + assert!(transcript_entry.content.provider_kind.is_none()); assert_eq!(transcript_entry.text.as_deref(), Some(TRANSCRIPT_TEXT)); assert!(!transcript_entry.text_truncated); assert!( @@ -307,7 +314,7 @@ async fn run_transcodable_audio_preprocess_live_client( let _ = handle .terminate( WorkflowTerminateOptions::builder() - .reason("agent audio transcode preprocess live test cleanup") + .reason("audio transcription live test cleanup") .build(), ) .await; @@ -420,3 +427,194 @@ fn tiny_wav_bytes() -> Vec { bytes.resize(44 + data_len, 0); bytes } + +async fn transcribe( + api: &GatewayAgentApi, + blob_ref: String, + mime: &str, + name: &str, +) -> anyhow::Result { + let model = api::ModelConfig { + provider_id: "test-speech".into(), + api_kind: "openai:audio-transcriptions".into(), + model: "pinned-speech-model".into(), + }; + let before = api + .read_model_defaults(api::ModelDefaultsReadParams {}) + .await? + .result + .defaults; + let updated = api + .put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::SpeechToText, + model: Some(model.clone()), + expected_revision: before.revision, + }) + .await? + .result + .defaults; + let request = api::TranscriptionStartParams { + idempotency_key: uuid::Uuid::new_v4().to_string(), + audio: api::TranscriptionAudio { + blob_ref, + mime: mime.into(), + name: name.into(), + }, + model: None, + language: Some("en".into()), + prompt: None, + }; + let initial = api + .start_transcription(request.clone()) + .await? + .result + .transcription; + api.put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::SpeechToText, + model: before.speech_to_text, + expected_revision: updated.revision, + }) + .await?; + let terminal = loop { + let view = api + .read_transcription(api::TranscriptionReadParams { + transcription_id: initial.transcription_id.clone(), + }) + .await? + .result + .transcription; + if view.status.is_terminal() { + break view; + } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + }; + assert_eq!( + terminal.status, + api::TranscriptionStatus::Succeeded, + "{:?}", + terminal.failure + ); + assert_eq!(terminal.model, model); + assert_eq!(terminal.text.as_deref(), Some(TRANSCRIPT_TEXT)); + let mut other = support::live::local_request_context().await?; + other.actor = Some("another-person".into()); + let denied = temporal_server::gateway::request_context::with_request_context( + other, + api.read_transcription(api::TranscriptionReadParams { + transcription_id: initial.transcription_id.clone(), + }), + ) + .await + .unwrap_err(); + assert_eq!(denied.kind, api::AgentApiErrorKind::Forbidden); + let retried = api + .start_transcription(request.clone()) + .await? + .result + .transcription; + assert_eq!(retried.transcript_ref, terminal.transcript_ref); + assert_eq!(terminal.audio, request.audio); + assert_eq!(retried.model, model); + let mut conflicting = request; + conflicting.prompt = Some("different options".into()); + assert_eq!( + api.start_transcription(conflicting).await.unwrap_err().kind, + api::AgentApiErrorKind::Conflict + ); + let cancelled = api + .cancel_transcription(api::TranscriptionCancelParams { + transcription_id: initial.transcription_id, + }) + .await? + .result + .transcription; + assert_eq!(cancelled.status, api::TranscriptionStatus::Succeeded); + Ok(terminal.transcript_ref.unwrap()) +} + +struct ControlledTranscriber { + text: String, + attempts: std::sync::atomic::AtomicUsize, + block: bool, +} + +#[async_trait] +impl AudioTranscriber for ControlledTranscriber { + async fn transcribe( + &self, + _request: AudioTranscriptionRequest, + ) -> Result { + let attempt = self + .attempts + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + if self.block { + return std::future::pending().await; + } + if attempt == 0 { + return Err(AudioTranscriptionError { + message: "temporary failure".into(), + retryable: true, + configuration: false, + }); + } + Ok(AudioTranscription { + text: self.text.clone(), + }) + } +} + +#[tokio::test(flavor = "current_thread")] +#[ignore = "requires ./dev.sh infra or compatible Temporal + Postgres env"] +async fn transcription_live_retries_transient_failure_and_cancels_running_activity() +-> anyhow::Result<()> { + let _lock = LIVE_TEST_LOCK.lock().await; + require_storage_live_env()?; + for block in [false, true] { + let transcriber = Arc::new(ControlledTranscriber { + // A plain text result can be shared with sessions from other jobs. + // The expiry assertion needs a unique, unsubmitted result blob. + text: format!("unsubmitted transcript {}", uuid::Uuid::new_v4()), + attempts: Default::default(), + block, + }); + let activities = fake_worker_activities_with_audio_transcriber(transcriber.clone()).await?; + run_with_live_worker(activities, move |client, queue, _| async move { + let store = pg_store_from_env().await?; + store.ensure_universe().await?; + let api = GatewayAgentApi::builder(client, store).with_task_queue(queue).build(); + let blob = api.put_blobs(BlobPutParams { blobs: vec![BlobPutItem { bytes_base64: BASE64.encode(AUDIO_BYTES) }] }).await?.result.blobs.remove(0).blob_ref; + let request = api::TranscriptionStartParams { + idempotency_key: uuid::Uuid::new_v4().to_string(), + audio: api::TranscriptionAudio { blob_ref: blob, mime: "audio/ogg".into(), name: "voice.ogg".into() }, + model: Some(api::ModelConfig { provider_id: "fake".into(), api_kind: "openai:audio-transcriptions".into(), model: "speech".into() }), + language: None, prompt: None, + }; + let started = api.start_transcription(request).await?.result.transcription; + while transcriber.attempts.load(std::sync::atomic::Ordering::SeqCst) == 0 { + tokio::time::sleep(std::time::Duration::from_millis(20)).await; + } + if block { + api.cancel_transcription(api::TranscriptionCancelParams { transcription_id: started.transcription_id.clone() }).await?; + } + let terminal = loop { + let view = api.read_transcription(api::TranscriptionReadParams { transcription_id: started.transcription_id.clone() }).await?.result.transcription; + if view.status.is_terminal() { break view; } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + }; + assert_eq!(terminal.status, if block { api::TranscriptionStatus::Cancelled } else { api::TranscriptionStatus::Succeeded }); + assert_eq!(transcriber.attempts.load(std::sync::atomic::Ordering::SeqCst), if block { 1 } else { 2 }); + if !block { + let store = pg_store_from_env().await?; + let reference = engine::BlobRef::parse(terminal.transcript_ref.as_ref().unwrap())?; + sqlx::query("UPDATE cas_blobs SET created_at_ms = 1, touched_at_ms = 1 WHERE universe_id = $1 AND digest = $2") + .bind(store.config().universe_id).bind(reference.as_str().trim_start_matches("sha256:")).execute(store.pool()).await?; + assert_eq!(store.delete_dead_blobs(&[reference], 2, &[]).await?.len(), 1, "unsubmitted workflows do not create retention roots"); + let expired = api.read_transcription(api::TranscriptionReadParams { transcription_id: started.transcription_id }).await?.result.transcription; + assert_eq!(expired.status, api::TranscriptionStatus::Expired); + } + + Ok(()) + }).await?; + } + Ok(()) +} diff --git a/crates/temporal-server/tests/vfs_transfer_live.rs b/crates/temporal-server/tests/vfs_transfer_live.rs index 2e1856974..3a6fea1da 100644 --- a/crates/temporal-server/tests/vfs_transfer_live.rs +++ b/crates/temporal-server/tests/vfs_transfer_live.rs @@ -357,6 +357,7 @@ async fn run_case( session_id: session.to_string(), source: api::RunStartSource::Input { items: vec![api::InputItem::Text { + provenance_ref: None, origin: None, text: "run transfer checks".into(), }], diff --git a/crates/temporal-server/tests/workflow_tool_plugins_live.rs b/crates/temporal-server/tests/workflow_tool_plugins_live.rs index 78b4f4ff9..2bd2b1ea2 100644 --- a/crates/temporal-server/tests/workflow_tool_plugins_live.rs +++ b/crates/temporal-server/tests/workflow_tool_plugins_live.rs @@ -935,6 +935,7 @@ async fn start_managed_session_and_run( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {call_tool}"), }], @@ -1157,6 +1158,7 @@ async fn workflow_tool_controller_self_receiver_resolves_before_run_terminal() - session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {MESSAGE_SEND_TOOL}"), }], @@ -1432,6 +1434,7 @@ async fn workflow_tool_controller_self_receiver_deadline_breaks_stalled_reply() session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {MESSAGE_SEND_TOOL}"), }], @@ -1552,6 +1555,7 @@ async fn workflow_tool_reply_requires_exact_stored_producer() -> anyhow::Result< session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {REQUEST_APPROVAL_TOOL}"), }], @@ -2005,6 +2009,7 @@ async fn workflow_tool_dead_receiver_fails_promise_terminally() -> anyhow::Resul session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {REQUEST_APPROVAL_TOOL}"), }], @@ -2183,6 +2188,7 @@ async fn workflow_tool_reply_schema_gates_resolutions() -> anyhow::Result<()> { session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {REQUEST_APPROVAL_TOOL}"), }], @@ -2511,6 +2517,7 @@ async fn workflow_tool_run_terminal_auto_cancel_notifies_bound_receiver() -> any session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL_NOWAIT {REQUEST_APPROVAL_TOOL}"), }], @@ -2606,7 +2613,7 @@ async fn workflow_tool_auto_cancel_cancels_started_execution() -> anyhow::Result submission_id: None, session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { - items: vec![InputItem::Text { origin: None, + items: vec![InputItem::Text { provenance_ref: None, origin: None, text: format!("CALL_SHORTWAIT {LAUNCH_JOB_TOOL}"), }], }, diff --git a/crates/temporal-workflow/contract/workflow-contract.md b/crates/temporal-workflow/contract/workflow-contract.md index dd4d29b02..902358c7f 100644 --- a/crates/temporal-workflow/contract/workflow-contract.md +++ b/crates/temporal-workflow/contract/workflow-contract.md @@ -72,5 +72,5 @@ fingerprinted over their exact raw bytes; canonical fingerprints begin with ## Schema inventory -The schema bundle contains 38 definitions. Its public roots -are: EmissionEnvelope, WorkflowToolStartArgs, WorkflowToolRecoveryResult, WorkflowToolRecipeV1, ConversationStart, ChannelDeliveryCommand, ChannelDeliveryResult, PrepareChannelMediaInput, PrepareChannelMediaResult. +The schema bundle contains 49 definitions. Its public roots +are: EmissionEnvelope, WorkflowToolStartArgs, WorkflowToolRecoveryResult, WorkflowToolRecipeV1, ConversationStart, ChannelDeliveryCommand, ChannelDeliveryResult, PrepareChannelMediaInput, PrepareChannelMediaResult, TranscriptionWorkflowArgs, TranscriptionSnapshot, TranscriptionActivityResult. diff --git a/crates/temporal-workflow/contract/workflow.json b/crates/temporal-workflow/contract/workflow.json index 47ea11762..c9baa301b 100644 --- a/crates/temporal-workflow/contract/workflow.json +++ b/crates/temporal-workflow/contract/workflow.json @@ -65,7 +65,10 @@ "ChannelDeliveryCommand", "ChannelDeliveryResult", "PrepareChannelMediaInput", - "PrepareChannelMediaResult" + "PrepareChannelMediaResult", + "TranscriptionWorkflowArgs", + "TranscriptionSnapshot", + "TranscriptionActivityResult" ], "signals": { "deliverEmission": "deliver_emission" diff --git a/crates/temporal-workflow/contract/workflow.schema.json b/crates/temporal-workflow/contract/workflow.schema.json index 624554450..214d658c8 100644 --- a/crates/temporal-workflow/contract/workflow.schema.json +++ b/crates/temporal-workflow/contract/workflow.schema.json @@ -1,6 +1,79 @@ { "$schema": "http://json-schema.org/draft-07/schema#", "definitions": { + "Attribution": { + "description": "Who created a resource or authored bytes. An actor is whatever a key\nallowed to assert one said; core compares it and never resolves it.", + "oneOf": [ + { + "description": "The actor a key asserted for its request.", + "properties": { + "id": { + "type": "string" + }, + "kind": { + "const": "actor", + "type": "string" + } + }, + "required": [ + "kind", + "id" + ], + "type": "object" + }, + { + "description": "A key acting for itself, named by its display prefix.", + "properties": { + "kind": { + "const": "key", + "type": "string" + }, + "prefix": { + "type": "string" + } + }, + "required": [ + "kind", + "prefix" + ], + "type": "object" + }, + { + "description": "An unauthenticated local development request, or an in-process call.", + "properties": { + "kind": { + "const": "local", + "type": "string" + } + }, + "required": [ + "kind" + ], + "type": "object" + }, + { + "description": "The runtime's own work: a bot, a delegated session, a registration,\nor host administration through the server CLI.", + "properties": { + "cause": { + "type": "string" + }, + "component": { + "type": "string" + }, + "kind": { + "const": "internal", + "type": "string" + } + }, + "required": [ + "kind", + "component", + "cause" + ], + "type": "object" + } + ] + }, "BotId": { "type": "string" }, @@ -614,6 +687,25 @@ ], "type": "object" }, + "ModelConfig": { + "properties": { + "apiKind": { + "type": "string" + }, + "model": { + "type": "string" + }, + "providerId": { + "type": "string" + } + }, + "required": [ + "providerId", + "apiKind", + "model" + ], + "type": "object" + }, "PrepareChannelMediaInput": { "description": "`prepare_channel_media`: the connector downloads the provider file and\nstores it in the universe's CAS.", "properties": { @@ -761,6 +853,243 @@ "ToolCallId": { "type": "string" }, + "TranscriptionActivityResult": { + "oneOf": [ + { + "properties": { + "kind": { + "const": "succeeded", + "type": "string" + }, + "transcript_ref": { + "type": "string" + } + }, + "required": [ + "kind", + "transcript_ref" + ], + "type": "object" + }, + { + "properties": { + "failure": { + "$ref": "#/definitions/TranscriptionFailure" + }, + "kind": { + "const": "failed", + "type": "string" + } + }, + "required": [ + "kind", + "failure" + ], + "type": "object" + } + ] + }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/definitions/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionSnapshot": { + "properties": { + "request": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "view": { + "$ref": "#/definitions/TranscriptionView" + } + }, + "required": [ + "request", + "view" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/definitions/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "status": { + "$ref": "#/definitions/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, + "TranscriptionWorkflowArgs": { + "properties": { + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "request": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "transcriptionId": { + "type": "string" + }, + "universeId": { + "format": "uuid", + "type": "string" + } + }, + "required": [ + "universeId", + "transcriptionId", + "createdBy", + "request", + "model", + "createdAtMs" + ], + "type": "object" + }, "WorkflowToolId": { "type": "string" }, diff --git a/crates/temporal-workflow/src/activities.rs b/crates/temporal-workflow/src/activities.rs index a2cba4134..f8b84639b 100644 --- a/crates/temporal-workflow/src/activities.rs +++ b/crates/temporal-workflow/src/activities.rs @@ -12,8 +12,7 @@ use crate::{ EnvironmentJobPollActivityRequest, EnvironmentJobPollActivityResult, EnvironmentJobPrepareWorkflowToolRequest, EnvironmentJobStartActivityRequest, EnvironmentJobStartActivityResult, JoinedContextPreparationRequest, LlmGenerateActivityRequest, - PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult, PutBlobRequest, - ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, + PutBlobRequest, ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, RuntimeProjectionRefreshActivityResult, SubagentCloseActivityRequest, SubagentPrepareActivityRequest, SubagentPrepareActivityResult, SubagentResolveActivityRequest, ToolInvokeBatchActivityRequest, ToolInvokeCallActivityRequest, ToolInvokeCallActivityResult, @@ -30,7 +29,6 @@ pub const ACTIVITY_MATERIALIZE_AWAIT_RESULT: &str = "WorkflowActivities::materia pub const ACTIVITY_PREPARE_JOINED_CONTEXT: &str = "WorkflowActivities::prepare_joined_context"; pub const ACTIVITY_APPEND_EVENTS: &str = "WorkflowActivities::append_events"; pub const ACTIVITY_LLM_GENERATE: &str = "WorkflowActivities::llm_generate"; -pub const ACTIVITY_PREPROCESS_RUN_INPUT: &str = "WorkflowActivities::preprocess_run_input"; pub const ACTIVITY_CONTEXT_COMPACT: &str = "WorkflowActivities::context_compact"; pub const ACTIVITY_TOOL_INVOKE_BATCH: &str = "WorkflowActivities::tool_invoke_batch"; pub const ACTIVITY_TOOL_INVOKE_CALL: &str = "WorkflowActivities::tool_invoke_call"; @@ -60,6 +58,14 @@ pub struct WorkflowActivities; #[activities] impl WorkflowActivities { + #[activity(name = "WorkflowActivities::execute_transcription")] + pub async fn execute_transcription( + _ctx: ActivityContext, + _args: crate::TranscriptionWorkflowArgs, + ) -> Result { + unimplemented!("workflow activity definition only") + } + #[activity(name = ACTIVITY_CREATE_OR_LOAD_SESSION)] pub async fn create_or_load_session( _ctx: ActivityContext, @@ -121,14 +127,6 @@ impl WorkflowActivities { unimplemented!("workflow activity definition only") } - #[activity(name = ACTIVITY_PREPROCESS_RUN_INPUT)] - pub async fn preprocess_run_input( - _ctx: ActivityContext, - _request: PreprocessRunInputActivityRequest, - ) -> Result { - unimplemented!("workflow activity definition only") - } - #[activity(name = ACTIVITY_CONTEXT_COMPACT)] pub async fn context_compact( _ctx: ActivityContext, diff --git a/crates/temporal-workflow/src/lib.rs b/crates/temporal-workflow/src/lib.rs index 823be6bb1..f45eb8034 100644 --- a/crates/temporal-workflow/src/lib.rs +++ b/crates/temporal-workflow/src/lib.rs @@ -16,12 +16,11 @@ pub use activities::{ ACTIVITY_CONTEXT_COMPACT, ACTIVITY_CREATE_OR_LOAD_SESSION, ACTIVITY_ENVIRONMENT_JOB_CANCEL, ACTIVITY_ENVIRONMENT_JOB_POLL, ACTIVITY_ENVIRONMENT_JOB_PREPARE_WORKFLOW_TOOL, ACTIVITY_ENVIRONMENT_JOB_START, ACTIVITY_LLM_GENERATE, ACTIVITY_MATERIALIZE_AWAIT_RESULT, - ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PREPROCESS_RUN_INPUT, ACTIVITY_PUT_BLOB, - ACTIVITY_READ_BLOB, ACTIVITY_RUNTIME_PROJECTION_REFRESH, - ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, - ACTIVITY_SUBAGENT_RESOLVE, ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, - ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, - WorkflowActivities, + ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PUT_BLOB, ACTIVITY_READ_BLOB, + ACTIVITY_RUNTIME_PROJECTION_REFRESH, ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, + ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, ACTIVITY_SUBAGENT_RESOLVE, + ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, + ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, WorkflowActivities, }; pub use config::{ ACTIVITY_CANCELLATION_HEARTBEAT_INTERVAL, ACTIVITY_CANCELLATION_HEARTBEAT_TIMEOUT, @@ -57,9 +56,7 @@ pub use types::{ LLM_PROVIDER_TRANSIENT_ERROR_TYPE, LLM_TRANSIENT_FAILURE_DETAILS_VERSION, LlmGenerateActivityRequest, LlmTransientFailureDetails, MaterializedAwaitPromiseResult, MaterializedAwaitResult, PendingEmission, PendingPromiseCancellation, PendingSourceResolution, - PendingToolBatchResume, PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult, - PreprocessRunInputFailure, PreprocessRunInputFailureKind, PreprocessRunInputOutcome, - PromiseSourcePoll, PutBlobRequest, ReadBlobRequest, ReadBlobResult, + PendingToolBatchResume, PromiseSourcePoll, PutBlobRequest, ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, RuntimeProjectionRefreshActivityResult, SessionBootstrapPayloadTooLarge, SubagentChildRef, SubagentCloseActivityRequest, SubagentExecutionPhase, SubagentExecutionSnapshot, SubagentPrepareActivityRequest, @@ -78,4 +75,6 @@ pub use workflows::channels; pub use workflows::{ AgentSessionWorkflow, BotControllerWorkflow, BotTriggerFireWorkflow, ChannelConversationWorkflow, EnvironmentJobWorkflow, SubagentExecutionWorkflow, + TranscriptionActivityResult, TranscriptionSnapshot, TranscriptionWorkflow, + TranscriptionWorkflowArgs, transcription_id, transcription_workflow_id, }; diff --git a/crates/temporal-workflow/src/types.rs b/crates/temporal-workflow/src/types.rs index aac3c03d1..6ede858c9 100644 --- a/crates/temporal-workflow/src/types.rs +++ b/crates/temporal-workflow/src/types.rs @@ -153,26 +153,10 @@ pub struct AgentAdmissionFailure { pub rejection: Option, } -impl AgentAdmissionFailure { - pub fn with_correlation_token(mut self, correlation_token: Option) -> Self { - if self.correlation_token.is_none() { - self.correlation_token = correlation_token; - } - self - } -} - #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "snake_case")] pub enum AgentAdmissionFailureKind { RejectedCommand, - UnsupportedAudioMime, - AudioBlobMissing, - AudioBlobTooLarge, - AudioDurationTooLong, - TranscoderUnavailable, - TranscodeFailure, - TranscriptionFailure, } #[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] @@ -755,42 +739,6 @@ pub struct LlmGenerateActivityRequest { pub request: engine::LlmGenerationRequest, } -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -pub struct PreprocessRunInputActivityRequest { - pub session_id: SessionId, - pub input: Vec, -} - -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -pub struct PreprocessRunInputActivityResult { - pub outcome: PreprocessRunInputOutcome, -} - -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case", tag = "status")] -pub enum PreprocessRunInputOutcome { - Succeeded { input: Vec }, - Failed { failure: PreprocessRunInputFailure }, -} - -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -pub struct PreprocessRunInputFailure { - pub kind: PreprocessRunInputFailureKind, - pub message: String, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum PreprocessRunInputFailureKind { - UnsupportedAudioMime, - AudioBlobMissing, - AudioBlobTooLarge, - AudioDurationTooLong, - TranscoderUnavailable, - TranscodeFailure, - TranscriptionFailure, -} - #[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] pub struct ContextCompactActivityRequest { pub request: engine::ContextCompactionRequest, diff --git a/crates/temporal-workflow/src/workflow_contract.rs b/crates/temporal-workflow/src/workflow_contract.rs index 5bbc9bb84..1861c69cf 100644 --- a/crates/temporal-workflow/src/workflow_contract.rs +++ b/crates/temporal-workflow/src/workflow_contract.rs @@ -42,7 +42,7 @@ pub const WORKFLOW_CONTRACT_VERSION: u32 = 2; pub const DELIVER_EMISSION_SIGNAL: &str = "deliver_emission"; /// Root types of the schema bundle; everything else is reachable from them. -pub const WORKFLOW_CONTRACT_ROOTS: [&str; 9] = [ +pub const WORKFLOW_CONTRACT_ROOTS: [&str; 12] = [ "EmissionEnvelope", "WorkflowToolStartArgs", "WorkflowToolRecoveryResult", @@ -52,6 +52,9 @@ pub const WORKFLOW_CONTRACT_ROOTS: [&str; 9] = [ "ChannelDeliveryResult", "PrepareChannelMediaInput", "PrepareChannelMediaResult", + "TranscriptionWorkflowArgs", + "TranscriptionSnapshot", + "TranscriptionActivityResult", ]; pub struct ExportedWorkflowContract { @@ -75,6 +78,9 @@ pub fn export() -> ExportedWorkflowContract { let _ = generator.subschema_for::(); let _ = generator.subschema_for::(); let _ = generator.subschema_for::(); + let _ = generator.subschema_for::(); + let _ = generator.subschema_for::(); + let _ = generator.subschema_for::(); let definitions: BTreeMap = generator.take_definitions(true).into_iter().collect(); for root in WORKFLOW_CONTRACT_ROOTS { diff --git a/crates/temporal-workflow/src/workflows/channels/activities.rs b/crates/temporal-workflow/src/workflows/channels/activities.rs index 7fe49c8b3..a97bce1b6 100644 --- a/crates/temporal-workflow/src/workflows/channels/activities.rs +++ b/crates/temporal-workflow/src/workflows/channels/activities.rs @@ -16,6 +16,7 @@ pub const ACTIVITY_CHAT_TOOL_DECLARATIONS: &str = "ChannelActivities::chat_tool_ pub const ACTIVITY_CHAT_READ_JSON_BLOB: &str = "ChannelActivities::read_json_blob"; pub const ACTIVITY_CHAT_PUT_JSON_BLOB: &str = "ChannelActivities::put_json_blob"; pub const ACTIVITY_CHAT_RECONCILE_DELIVERY: &str = "ChannelActivities::reconcile_delivery"; +pub const ACTIVITY_CHAT_TRANSCRIBE_MEDIA: &str = "ChannelActivities::transcribe_media"; pub const ACTIVITY_CHAT_EMIT_EVENT: &str = "ChannelActivities::emit_chat_event"; pub const ACTIVITY_CHAT_STORE_SENT: &str = "ChannelActivities::store_chat_sent"; pub const ACTIVITY_CHAT_RESOLVE_HANDLE: &str = "ChannelActivities::resolve_chat_handle"; @@ -30,6 +31,14 @@ pub struct ChannelActivities; #[activities] impl ChannelActivities { + #[activity(name = ACTIVITY_CHAT_TRANSCRIBE_MEDIA)] + pub async fn transcribe_media( + _ctx: ActivityContext, + _request: ChatTranscribeMediaRequest, + ) -> Result { + unimplemented!("workflow activity definition only") + } + /// Store the `message_*` declarations bound to this conversation as /// receiver; content-addressed, so stable per receiver. #[activity(name = ACTIVITY_CHAT_TOOL_DECLARATIONS)] diff --git a/crates/temporal-workflow/src/workflows/channels/conversation.rs b/crates/temporal-workflow/src/workflows/channels/conversation.rs index f5ae77038..a5f9f5576 100644 --- a/crates/temporal-workflow/src/workflows/channels/conversation.rs +++ b/crates/temporal-workflow/src/workflows/channels/conversation.rs @@ -545,31 +545,113 @@ async fn handle_inbound(ctx: &Ctx, inbound: AdmittedInbound) { tracing::debug!(message_id = %message(&inbound).message_id, ?reason, "inbound dropped"); } InboundPlan::Emit { text } => { - let media = match prepare_media(ctx, &inbound).await { - Ok(media) => media, - Err(error) => { - let message_id = message(&inbound).message_id.clone(); - ctx.state_mut(|wf| { - wf.state.messages.insert( - inbound_key, - ReceivedMessage { - message_id: message_id.clone(), - status: MessageStatus::Failed, - seq: None, - session_id: None, - error: Some(error.clone()), - }, - ); - wf.state - .protocol_errors - .push(format!("media {message_id}: {error}")); - }); - return; + let (media, transcript_refs) = + match prepare_message_media(ctx, &inbound, &inbound_key).await { + Ok(media) => media, + Err(error) => { + let message_id = message(&inbound).message_id.clone(); + ctx.state_mut(|wf| { + wf.state.messages.insert( + inbound_key, + ReceivedMessage { + message_id: message_id.clone(), + status: MessageStatus::Failed, + seq: None, + session_id: None, + error: Some(error.clone()), + }, + ); + wf.state + .protocol_errors + .push(format!("media {message_id}: {error}")); + }); + return; + } + }; + emit_message(ctx, inbound_key, &inbound, text, media, transcript_refs).await; + } + } +} + +async fn prepare_message_media( + ctx: &Ctx, + inbound: &AdmittedInbound, + key: &str, +) -> Result< + ( + Vec, + std::collections::BTreeMap, + ), + String, +> { + let media = prepare_media(ctx, inbound).await?; + let mut transcripts = std::collections::BTreeMap::new(); + for (index, item) in media.iter().enumerate() { + if item.kind != api::ChannelMediaKind::Audio { + continue; + } + let request = ctx.state(|wf| super::ChatTranscribeMediaRequest { + active: ChatAssertTriggerActiveRequest { + universe_id: wf.start.universe_id, + bot_id: wf.start.bot_id.clone(), + trigger_id: wf.start.trigger_id.clone(), + account_id: wf.start.account_id.clone(), + chat_id: wf.start.conversation.chat_id.clone(), + scope: wf.start.scope, + }, + idempotency_key: crate::transcription_id( + &api::Attribution::Local, + &format!("{}:{key}:{index}", wf.start.conversation.key()), + ), + audio: api::TranscriptionAudio { + blob_ref: item.blob_ref.clone(), + mime: item.mime.clone(), + name: item.name.clone().unwrap_or_else(|| "audio".into()), + }, + }); + loop { + let view = activity( + ctx, + ChannelActivities::transcribe_media, + request.clone(), + channel_activity_options(), + ) + .await?; + match view.status { + api::TranscriptionStatus::Pending | api::TranscriptionStatus::Running => { + let cancelled = { + let timer = ctx.timer(std::time::Duration::from_secs(2)); + let cancelled = ctx.cancelled(); + pin_mut!(timer, cancelled); + select! { _ = timer => false, _ = cancelled => true } + }; + if cancelled { + let _ = ctx + .external_workflow( + format!("{}/{}", request.active.universe_id, view.transcription_id), + None, + ) + .signal(crate::TranscriptionWorkflow::cancel, ()) + .await; + return Err("Audio preparation cancelled.".into()); + } } - }; - emit_message(ctx, inbound_key, &inbound, text, media).await; + api::TranscriptionStatus::Succeeded => { + transcripts.insert( + item.blob_ref.clone(), + view.transcript_ref.ok_or("missing transcript reference")?, + ); + break; + } + _ => { + return Err(view.failure.map(|f| f.message).unwrap_or_else(|| { + "Transcription unavailable; resend the recording.".into() + })); + } + } } } + Ok((media, transcripts)) } /// Download every attachment through the connector and put it in the CAS; @@ -621,6 +703,7 @@ async fn emit_message( inbound: &AdmittedInbound, text: String, media: Vec, + transcript_refs: std::collections::BTreeMap, ) { let chat = message(inbound); let request = ctx.state_mut(|wf| { @@ -648,6 +731,7 @@ async fn emit_message( is_reply_to_bot: chat.is_reply_to_bot, }, media, + transcript_refs, tools_ref: wf .state .tools_ref diff --git a/crates/temporal-workflow/src/workflows/channels/types.rs b/crates/temporal-workflow/src/workflows/channels/types.rs index 46fddafa0..16b4eea02 100644 --- a/crates/temporal-workflow/src/workflows/channels/types.rs +++ b/crates/temporal-workflow/src/workflows/channels/types.rs @@ -110,6 +110,8 @@ pub struct ChatEmitEventRequest { pub message: ChatMessage, #[serde(default)] pub media: Vec, + #[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")] + pub transcript_refs: std::collections::BTreeMap, /// CAS ref of the conversation's `message_*` declarations. pub tools_ref: String, /// This workflow, for `started` / `finished` receipts. @@ -211,3 +213,11 @@ pub enum ChatTriggerActiveResult { Active, Inactive { reason: String }, } + +/// Starts or reads independent transcription on the sessions worker queue. +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct ChatTranscribeMediaRequest { + pub active: ChatAssertTriggerActiveRequest, + pub idempotency_key: String, + pub audio: api::TranscriptionAudio, +} diff --git a/crates/temporal-workflow/src/workflows/mod.rs b/crates/temporal-workflow/src/workflows/mod.rs index ebb970f5d..7780f2851 100644 --- a/crates/temporal-workflow/src/workflows/mod.rs +++ b/crates/temporal-workflow/src/workflows/mod.rs @@ -3,9 +3,15 @@ pub mod channels; mod environment_job; mod session; mod subagent_execution; +mod transcriptions; pub use bots::{BotControllerWorkflow, BotTriggerFireWorkflow}; pub use channels::ChannelConversationWorkflow; pub use environment_job::EnvironmentJobWorkflow; pub use session::AgentSessionWorkflow; pub use subagent_execution::SubagentExecutionWorkflow; + +pub use transcriptions::{ + TranscriptionActivityResult, TranscriptionSnapshot, TranscriptionWorkflow, + TranscriptionWorkflowArgs, transcription_id, transcription_workflow_id, +}; diff --git a/crates/temporal-workflow/src/workflows/session/admissions.rs b/crates/temporal-workflow/src/workflows/session/admissions.rs index 2a7b3e8db..587170c67 100644 --- a/crates/temporal-workflow/src/workflows/session/admissions.rs +++ b/crates/temporal-workflow/src/workflows/session/admissions.rs @@ -38,7 +38,7 @@ pub(super) async fn admit_admissions( } }; let correlation_token = admission.correlation_token.clone(); - let mut command = admission.command; + let command = admission.command; if let CoreAgentCommand::ReplaceSessionConfig { config, expected_revision, @@ -61,19 +61,6 @@ pub(super) async fn admit_admissions( } continue; } - if observed_tools.is_none() && command_needs_input_preprocessing(&command) { - let session_id = drive.session_id().clone(); - match preprocess_input_entries(ctx, session_id, command).await? { - RunInputPreprocessResult::Succeeded { command: rewritten } => command = *rewritten, - RunInputPreprocessResult::Failed { failure } => { - record_admission_failure( - ctx, - failure.with_correlation_token(correlation_token), - ); - continue; - } - } - } let mut deferred_tools = None; if drive.state().lifecycle.status == CoreAgentStatus::Open && matches!(command, CoreAgentCommand::RequestRun(_)) @@ -256,172 +243,6 @@ pub(super) fn admissible_during_turn(command: &CoreAgentCommand) -> bool { ) } -enum RunInputPreprocessResult { - Succeeded { command: Box }, - Failed { failure: AgentAdmissionFailure }, -} - -pub(super) fn command_needs_input_preprocessing(command: &CoreAgentCommand) -> bool { - match command { - CoreAgentCommand::RequestRun(request) => request.source.input().iter().any(is_audio_input), - CoreAgentCommand::UpsertContext { entry, .. } => is_audio_input(entry), - _ => false, - } -} - -fn is_audio_input(input: &ContextEntryInput) -> bool { - input - .content - .media_type - .as_deref() - .map(|mime| mime.trim().to_ascii_lowercase().starts_with("audio/")) - .unwrap_or(false) -} - -async fn preprocess_input_entries( - ctx: &mut WorkflowContext, - session_id: SessionId, - command: CoreAgentCommand, -) -> anyhow::Result { - let (submission_id, input, rebuild) = match command { - CoreAgentCommand::RequestRun(request) => { - let engine::RunRequestSource::Input { input } = request.source; - ( - request.submission_id.clone(), - input, - InputPreprocessRebuild::RequestRun { - submission_id: request.submission_id, - run_config: request.run_config, - notify_on_terminal: request.notify_on_terminal, - requested_by: request.requested_by, - }, - ) - } - CoreAgentCommand::UpsertContext { - expected_revision, - key, - entry, - } => ( - None, - vec![entry], - InputPreprocessRebuild::UpsertContext { - expected_revision, - key, - }, - ), - command => { - return Ok(RunInputPreprocessResult::Succeeded { - command: Box::new(command), - }); - } - }; - - let result = ctx - .start_activity( - WorkflowActivities::preprocess_run_input, - PreprocessRunInputActivityRequest { session_id, input }, - activity_options(), - ) - .await - .map_err(|error| anyhow::anyhow!("{error}"))?; - - match result.outcome { - PreprocessRunInputOutcome::Succeeded { input } => Ok(RunInputPreprocessResult::Succeeded { - command: Box::new(rebuild.rebuild(input)?), - }), - PreprocessRunInputOutcome::Failed { failure } => Ok(RunInputPreprocessResult::Failed { - failure: preprocess_failure_to_admission_failure(submission_id, failure), - }), - } -} - -// Held only while one admission is preprocessed. -#[allow(clippy::large_enum_variant)] -enum InputPreprocessRebuild { - RequestRun { - submission_id: Option, - run_config: RunConfig, - notify_on_terminal: Vec, - requested_by: Option, - }, - UpsertContext { - expected_revision: Option, - key: ContextEntryKey, - }, -} - -impl InputPreprocessRebuild { - fn rebuild(self, input: Vec) -> anyhow::Result { - match self { - Self::RequestRun { - submission_id, - run_config, - notify_on_terminal, - requested_by, - } => Ok(CoreAgentCommand::RequestRun(engine::RunRequestCommand { - notify_on_terminal, - requested_by, - submission_id, - source: engine::RunRequestSource::Input { input }, - run_config, - })), - Self::UpsertContext { - expected_revision, - key, - } => { - let mut input = input; - let Some(entry) = input.pop() else { - anyhow::bail!("preprocessed context append returned no entry"); - }; - if !input.is_empty() { - anyhow::bail!("preprocessed context append returned multiple entries"); - } - Ok(CoreAgentCommand::UpsertContext { - expected_revision, - key, - entry, - }) - } - } - } -} - -pub(super) fn preprocess_failure_to_admission_failure( - submission_id: Option, - failure: PreprocessRunInputFailure, -) -> AgentAdmissionFailure { - AgentAdmissionFailure { - preparation_error: None, - submission_id, - correlation_token: None, - kind: match failure.kind { - PreprocessRunInputFailureKind::UnsupportedAudioMime => { - AgentAdmissionFailureKind::UnsupportedAudioMime - } - PreprocessRunInputFailureKind::AudioBlobMissing => { - AgentAdmissionFailureKind::AudioBlobMissing - } - PreprocessRunInputFailureKind::AudioBlobTooLarge => { - AgentAdmissionFailureKind::AudioBlobTooLarge - } - PreprocessRunInputFailureKind::AudioDurationTooLong => { - AgentAdmissionFailureKind::AudioDurationTooLong - } - PreprocessRunInputFailureKind::TranscoderUnavailable => { - AgentAdmissionFailureKind::TranscoderUnavailable - } - PreprocessRunInputFailureKind::TranscodeFailure => { - AgentAdmissionFailureKind::TranscodeFailure - } - PreprocessRunInputFailureKind::TranscriptionFailure => { - AgentAdmissionFailureKind::TranscriptionFailure - } - }, - message: failure.message, - rejection: None, - } -} - pub(super) fn should_refresh_runtime_projection_before_admitting( state: &CoreAgentState, command: &CoreAgentCommand, @@ -532,41 +353,3 @@ pub(super) fn active_instruction_inputs( }) .collect() } - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn upsert_preprocess_rebuild_preserves_expected_context_revision() { - let key = ContextEntryKey::new("client.audio"); - let entry = ContextEntryInput { - kind: engine::ContextEntryKind::ProviderOpaque, - content: engine::ContentRef { - content_ref: BlobRef::from_bytes(b"transcribed"), - media_type: Some("application/json".to_owned()), - provider_kind: None, - }, - preview: None, - origin: None, - provenance_ref: None, - token_estimate: None, - }; - - let command = InputPreprocessRebuild::UpsertContext { - expected_revision: Some(7), - key: key.clone(), - } - .rebuild(vec![entry.clone()]) - .expect("rebuild upsert"); - - assert_eq!( - command, - CoreAgentCommand::UpsertContext { - expected_revision: Some(7), - key, - entry, - } - ); - } -} diff --git a/crates/temporal-workflow/src/workflows/session/mod.rs b/crates/temporal-workflow/src/workflows/session/mod.rs index f3609d2bc..5522e79ab 100644 --- a/crates/temporal-workflow/src/workflows/session/mod.rs +++ b/crates/temporal-workflow/src/workflows/session/mod.rs @@ -29,7 +29,7 @@ use engine::{ BlobRef, CommandError, ContextEntryInput, ContextEntryKey, ContextEntryKind, ContextMessageRole, CoreAgentAction, CoreAgentCommand, CoreAgentDrive, CoreAgentDriveError, CoreAgentEntry, CoreAgentEvent, CoreAgentState, CoreAgentStatus, EmissionEnvelope, - LlmGenerationRequest, RunConfig, RunEvent, RunStatus, SessionId, SessionPosition, SubmissionId, + LlmGenerationRequest, RunEvent, RunStatus, SessionId, SessionPosition, SubmissionId, ToolInvocationBatchRequest, }; use futures::{FutureExt, pin_mut, select}; @@ -45,10 +45,8 @@ use crate::{ AwaitMaterializationRequest, AwaitOutcome, AwaitPromiseResult, CancellingWatchdog, CreateOrLoadSessionRequest, DEFAULT_CONTINUE_AS_NEW_HISTORY_THRESHOLD, JoinedContextPreparationRequest, LlmGenerateActivityRequest, PendingEmission, - PendingPromiseCancellation, PendingSourceResolution, PendingToolBatchResume, - PreprocessRunInputActivityRequest, PreprocessRunInputFailure, PreprocessRunInputFailureKind, - PreprocessRunInputOutcome, PromiseSourcePoll, PutBlobRequest, - RuntimeProjectionRefreshActivityRequest, ToolInvokeBatchActivityRequest, + PendingPromiseCancellation, PendingSourceResolution, PendingToolBatchResume, PromiseSourcePoll, + PutBlobRequest, RuntimeProjectionRefreshActivityRequest, ToolInvokeBatchActivityRequest, ToolPreparePromiseControlsActivityRequest, WorkflowActivities, activity_options, compose_workflow_id, default_instructions, split_workflow_id, }; diff --git a/crates/temporal-workflow/src/workflows/session/tests.rs b/crates/temporal-workflow/src/workflows/session/tests.rs index af973de3d..32fc99cc1 100644 --- a/crates/temporal-workflow/src/workflows/session/tests.rs +++ b/crates/temporal-workflow/src/workflows/session/tests.rs @@ -56,54 +56,6 @@ fn admission_failure_status_does_not_poison_later_admission() { assert_eq!(status.last_error, None); } -#[test] -fn request_run_with_audio_input_needs_preprocessing() { - let command = CoreAgentCommand::RequestRun(engine::RunRequestCommand { - requested_by: None, - notify_on_terminal: Vec::new(), - submission_id: Some(SubmissionId::new("submit_audio")), - source: engine::RunRequestSource::Input { - input: vec![ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: engine::BlobRef::from_bytes(b"audio"), - media_type: Some("audio/ogg".to_owned()), - provider_kind: None, - }, - preview: Some("[audio]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }], - }, - run_config: crate::default_run_config(), - }); - - assert!(admissions::command_needs_input_preprocessing(&command)); -} - -#[test] -fn preprocess_failures_preserve_submission_id_for_admission_failure() { - let failure = admissions::preprocess_failure_to_admission_failure( - Some(SubmissionId::new("submit_audio")), - PreprocessRunInputFailure { - kind: PreprocessRunInputFailureKind::TranscriptionFailure, - message: "missing OpenAI key".to_owned(), - }, - ); - - assert_eq!( - failure.submission_id.as_ref(), - Some(&SubmissionId::new("submit_audio")) - ); - assert_eq!( - failure.kind, - AgentAdmissionFailureKind::TranscriptionFailure - ); -} - #[test] fn source_resolution_emission_queues_pending_resolution_with_producer() { let mut workflow = AgentSessionWorkflow::default(); diff --git a/crates/temporal-workflow/src/workflows/transcriptions.rs b/crates/temporal-workflow/src/workflows/transcriptions.rs new file mode 100644 index 000000000..4f1c1c4b9 --- /dev/null +++ b/crates/temporal-workflow/src/workflows/transcriptions.rs @@ -0,0 +1,181 @@ +//! Session-independent transcription. Temporal owns request identity and state; +//! activities store audio and transcript content in CAS. +use api::{ + Attribution, ModelConfig, TranscriptionFailure, TranscriptionFailureKind, + TranscriptionStartParams, TranscriptionStatus, TranscriptionView, +}; +use futures::{FutureExt, select_biased}; +use schemars::JsonSchema; +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; +use std::time::Duration; +use temporalio_common::protos::{ + coresdk::workflow_commands::ActivityCancellationType, temporal::api::common::v1::RetryPolicy, +}; +use temporalio_macros::{workflow, workflow_methods}; +use temporalio_sdk::{ + ActivityCloseTimeouts, ActivityOptions, CancellableFuture, SyncWorkflowContext, + WorkflowContext, WorkflowContextView, WorkflowResult, +}; + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionWorkflowArgs { + pub universe_id: uuid::Uuid, + pub transcription_id: String, + pub created_by: Attribution, + pub request: TranscriptionStartParams, + pub model: ModelConfig, + pub created_at_ms: u64, +} + +/// Explicit caller keys, rather than audio hashes, define request identity. +pub fn transcription_id(owner: &Attribution, key: &str) -> String { + let bytes = serde_json::to_vec(&(owner, key)).expect("serializable transcription identity"); + format!("transcription_{}", hex::encode(Sha256::digest(bytes))) +} + +pub fn transcription_workflow_id(args: &TranscriptionWorkflowArgs) -> String { + format!("{}/{}", args.universe_id, args.transcription_id) +} + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionSnapshot { + pub request: TranscriptionStartParams, + pub view: TranscriptionView, +} + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +#[serde(tag = "kind", rename_all = "camelCase")] +pub enum TranscriptionActivityResult { + Succeeded { transcript_ref: String }, + Failed { failure: TranscriptionFailure }, +} + +impl TranscriptionWorkflowArgs { + pub fn pending(&self) -> TranscriptionView { + TranscriptionView { + transcription_id: self.transcription_id.clone(), + created_by: self.created_by.clone(), + audio: self.request.audio.clone(), + model: self.model.clone(), + status: TranscriptionStatus::Pending, + created_at_ms: self.created_at_ms, + transcript_ref: None, + text: None, + failure: None, + } + } +} + +#[workflow(name = "TranscriptionWorkflow")] +#[derive(Default)] +pub struct TranscriptionWorkflow { + snapshot: Option, + cancel_requested: bool, +} + +#[workflow_methods] +impl TranscriptionWorkflow { + #[run] + pub async fn run( + ctx: &mut WorkflowContext, + args: TranscriptionWorkflowArgs, + ) -> WorkflowResult<()> { + if ctx.workflow_id() != transcription_workflow_id(&args) { + return Err(anyhow::anyhow!("transcription workflow identity mismatch").into()); + } + let mut view = args.pending(); + view.status = TranscriptionStatus::Running; + ctx.state_mut(|state| { + state.snapshot = Some(TranscriptionSnapshot { + request: args.request.clone(), + view, + }) + }); + let options = ActivityOptions::with_close_timeouts(ActivityCloseTimeouts::Both { + start_to_close: Duration::from_secs(360), + schedule_to_close: Duration::from_secs(900), + }) + .heartbeat_timeout(Duration::from_secs(15)) + .cancellation_type(ActivityCancellationType::WaitCancellationCompleted) + .retry_policy(RetryPolicy { + initial_interval: Some(Duration::from_secs(2).try_into().unwrap()), + maximum_interval: Some(Duration::from_secs(15).try_into().unwrap()), + backoff_coefficient: 2.0, + maximum_attempts: 3, + non_retryable_error_types: vec![], + }) + .build(); + let mut activity = ctx.start_activity( + crate::WorkflowActivities::execute_transcription, + args, + options, + ); + let result = { + let cancellation = ctx.wait_condition(|state| state.cancel_requested).fuse(); + let external_cancel = ctx.cancelled().fuse(); + let mut work = (&mut activity).fuse(); + futures::pin_mut!(cancellation, external_cancel); + select_biased! { + _ = cancellation => None, + _ = external_cancel => None, + result = work => Some(result), + } + }; + if result.is_none() { + activity.cancel(); + let _ = activity.await; + } + ctx.state_mut(|state| { + let view = &mut state.snapshot.as_mut().expect("initialized").view; + match result { + None => view.status = TranscriptionStatus::Cancelled, + Some(Ok(TranscriptionActivityResult::Succeeded { transcript_ref })) => { + view.status = TranscriptionStatus::Succeeded; + view.transcript_ref = Some(transcript_ref); + } + Some(Ok(TranscriptionActivityResult::Failed { failure })) => { + view.status = TranscriptionStatus::Failed; + view.failure = Some(failure); + } + Some(Err(error)) => { + view.status = TranscriptionStatus::Failed; + view.failure = Some(TranscriptionFailure { + kind: if error.as_timeout().is_some() { + TranscriptionFailureKind::Timeout + } else { + TranscriptionFailureKind::Provider + }, + message: "Transcription exhausted its execution budget.".into(), + }); + } + } + }); + Ok(()) + } + + #[signal(name = "cancel")] + pub fn cancel(&mut self, _ctx: &mut SyncWorkflowContext) { + self.cancel_requested = true; + } + + #[query(name = "snapshot")] + pub fn snapshot(&self, _ctx: &WorkflowContextView) -> Option { + self.snapshot.clone() + } +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn identity_is_scoped_to_owner_and_explicit_key() { + let a = Attribution::Actor { id: "a".into() }; + let b = Attribution::Actor { id: "b".into() }; + assert_eq!(transcription_id(&a, "one"), transcription_id(&a, "one")); + assert_ne!(transcription_id(&a, "one"), transcription_id(&b, "one")); + assert_ne!(transcription_id(&a, "one"), transcription_id(&a, "two")); + } +} diff --git a/docs/roadmap/p184-universe-model-defaults-and-transcription.md b/docs/roadmap/p184-universe-model-defaults-and-transcription.md index 1688ece4e..73abf9209 100644 --- a/docs/roadmap/p184-universe-model-defaults-and-transcription.md +++ b/docs/roadmap/p184-universe-model-defaults-and-transcription.md @@ -1,8 +1,7 @@ # P184 — Universe model defaults and standalone transcription -**Status:** First slice implemented, 2026-09-28: universe defaults, session -resolution, CLI configuration, and development seeding. Platform settings, -route readiness, and standalone transcription remain to be implemented. +**Status:** Universe defaults, Platform model settings, standalone transcription, +and channel voice preparation implemented, 2026-09-28. Web dictation remains. Builds on [CLI session model routing](cli-session-model-routing.md) and [the runtime CLI](p183-first-class-runtime-cli.md). Supersedes the placement of transcription inside session admission in @@ -183,60 +182,56 @@ and cancellation races must converge on one recorded outcome. ### 5. Explicit request identity and result lifetime -Persist the admitted request identity, attribution, immutable resolved model, -and input/result references in a small transcription record. Temporal owns the -execution lifecycle. This record supports recovery, API lookup, and CAS -retention; it is not a generic job framework or a transcript cache. - -Within the caller's universe and ownership scope, an idempotency key identifies -one request. A matching retry rejoins that job; conflicting input or options -produce a conflict. Look up the original admission before resolving current -defaults. Fingerprint the submitted request, including whether its model was -omitted, rather than recomputing its identity from today's default. - -Persist the resolved route once. Credentials and endpoint configuration are -resolved through the provider record at execution time, following the same -policy as generation. Changes to defaults cannot select another model during -retry. A recoverable workflow-start boundary must handle crashes between -record creation and Temporal start. - -Do not use an audio-content hash as the public job identity. Explicitly -repeating a transcription can be intentional, and identical audio may belong -to different callers. Reuse an already persisted result on retry; do not claim -exactly-once upstream billing after an ambiguous provider response. - -Audio and transcript content live in CAS; workflow payloads carry bounded -metadata and references. A transcript artifact records text, source audio, -resolved model, and relevant options. Retain provider-native response material -separately if needed, without embedding credentials or transport headers. - -Active jobs root their required blobs. Completed jobs retain results and their -idempotency records for an explicit bounded interval, exposed through expiry -metadata. Define and test that interval before delivery. Expired results must -be distinguishable from provider failure. Once a transcript is admitted to a -session, session retention roots its content and source-audio provenance -independently of job expiry. Extend CAS reference traversal accordingly. - -Add a dedicated `transcriptions` method group. Platform Contributors can start -jobs; reads and cancellation respect requester ownership and administrator -access, including audio/result download paths. Dictation drafts are not exposed -as a universe-wide job listing. Preserve the existing distinction between -Platform person-level access and direct universe-key authority; a CAS hash is -not a substitute for an access check. +Temporal owns the admitted request, attribution, pinned model, status, and +result reference. There is no transcription table, schema migration, lease, or +separate retention service. + +Within a universe and requester scope, an explicit idempotency key identifies +one request. Matching retries query the original workflow before consulting +current defaults. Changed input or options conflict. The workflow start uses +reject-duplicate identity; concurrent admissions recover the winning request. +The original submitted request preserves whether its model was omitted. +Temporal's namespace history retention bounds this identity guarantee. + +Credentials and endpoint configuration are resolved through the provider record +at execution time. Default changes cannot select another model during retry. +An ambiguous upstream response can still result in another billed attempt; +this does not promise exactly-once provider execution. + +Audio and transcript content live in ordinary CAS. Workflow history carries +bounded metadata and references: source audio, resolved model, and transcription +options. The result blob contains only plain UTF-8 text. Input admission refreshes ordinary +CAS grace; it does not create a temporary retention root. Unsubmitted content +may be swept after that grace (seven days by default), and a missing completed +result is reported as expired. A delayed caller can upload again with a new key. + +Session admission sets the existing `provenance_ref` to the source audio. +Existing session roots retain both transcript and original recording. Bot-event +roots also retain prepared transcript references alongside their audio while +awaiting delivery. These extend existing reference enumeration, not storage +infrastructure. + +The `transcriptions` method group permits Contributors to start jobs through +person-level gateways. Asserted actors can only read or cancel their own drafts; +direct universe keys retain their method-group authority. There is no draft +listing. CAS download continues to use the existing universe-scoped blob access +policy. A separate person-level administrator draft browser is outside this slice. ### 6. Prepared input is the session boundary -The target session APIs accept ordinary text or an explicit transcript -reference for transcribed speech. Add `InputItem::Transcript { transcriptRef }` -to materialize the existing transcript context representation and source-audio -provenance. Validate and retain the artifact at admission. The engine receives -prepared content and references and performs no transcription orchestration. +Session APIs accept ordinary `Text` or `TextRef` input. Both support an optional +`provenanceRef` pointing to a source blob in the same universe. Admission checks +that the source exists and copies the reference into the existing context +`provenance_ref`. Projections preserve it. This generic metadata supports audio, +documents, or other sources without exposing transformation-specific formats to +sessions. The engine receives prepared text and performs no transcription +orchestration. There is no dedicated transcript input or JSON artifact. After callers migrate, raw audio `Media` is rejected consistently by run start, context append, and steering, with an actionable typed error directing callers to transcription. Context append retains its per-entry failure -semantics. Existing transcript history remains readable. Native audio model -input, if added later, will be a separate explicit capability. +semantics. Native audio model input, if added later, will be a separate explicit +capability. Do not build a general ingress coordinator merely to preserve the old single-call audio convenience. Audit existing clients before cutover. If a @@ -274,8 +269,8 @@ exit paths; failures preserve the user's existing text and support retry. The ordinary send controls determine whether reviewed text starts a run, queues it, or steers it. It is sent as text, without representing the edited -words as the original audio transcript. Keep temporary job retention separate -from the session's history. +words as the original audio transcript. Unsubmitted recordings and artifacts +use ordinary CAS grace rather than session retention. Show transcription availability and an actionable reason when unavailable. Readiness follows the selected transcription route, including providers that @@ -294,12 +289,10 @@ admission. Migrate first-party channel and web consumers to prepared input, then cut over the raw-audio API contract. CLI and direct-client migration guidance must describe the upload, transcription, and submission steps. -Treat removal of the old workflow activity calls as a Temporal compatibility -change. Inventory existing histories and pending admissions, then use a -supported workflow-versioning or drain/transition strategy with replay -coverage. Keep historical decoding and any required legacy activity handlers -until that transition is complete. Deleting PostgreSQL data is not a rollout -strategy and does not resolve Temporal histories. +This is a greenfield cutover. Remove session audio preprocessing, its activity +registration, request/result types, admission errors, and compatibility branches. +Channels always prepare audio through standalone transcription before session +admission. No legacy activity or replay adapter is retained. Existing sessions keep their persisted models. Upgrade tooling imports model defaults only for explicitly selected universes. Fresh universes remain @@ -321,17 +314,17 @@ this roadmap does not authorize unrelated documentation or root README edits. the CLI and seed untouched development universes through the API. - [x] Add Platform Models settings, per-selection configuration diagnostics, and effective-route readiness. -- [ ] Extend provider configuration/resolution and the audio client for +- [x] Extend provider configuration/resolution and the audio client for compatible transcription endpoints, including credentialless transport. -- [ ] Add the transcription record, workflow, start/read/cancel APIs, request - deduplication, access rules, and CAS lifetime handling. -- [ ] Add transcript artifacts and prepared transcript input with provenance. +- [x] Add the workflow-owned transcription state, start/read/cancel APIs, + request deduplication, and requester access rules. +- [x] Return plain transcript text and add generic source provenance to text inputs. - [ ] Add web dictation with draft preview, cancellation, and demo coverage. -- [ ] Move channel voice-message preparation before bot-event delivery and +- [x] Move channel voice-message preparation before bot-event delivery and validate ordering, failure handling, and retries. -- [ ] Audit raw-audio clients, implement the workflow-history transition, and - remove session preprocessing from new execution paths. -- [ ] Regenerate affected contracts, complete scoped checks, and record +- [x] Audit raw-audio clients, implement the workflow-history transition, and + remove session preprocessing and obsolete compatibility code. +- [x] Regenerate affected contracts, complete scoped checks, and record migration and validation results here. ### First slice @@ -434,8 +427,8 @@ bot setup editors label omitted selections as the universe default. Readiness checks the selected provider and API, distinguishes missing or disabled credentials from unknown availability, and accepts credentialless providers and models absent from discovery. This reports configuration status, -not proof that a future model call will succeed. The speech-to-text slot stays -out of this UI until standalone transcription consumes it. +not proof that a future model call will succeed. The speech-to-text slot is configurable through the runtime API and CLI; its +settings row will accompany web dictation. The demo uses the same defaults editor, revision checks, and creation policy, including bot sessions. Clearing a default leaves existing session models @@ -445,6 +438,68 @@ unset defaults, and demo isolation and preservation. The full web, Platform server, and TypeScript client suites, workspace typechecks, and production and demo builds pass. The Models page was also visually checked in the browser. +### Standalone transcription and channel preparation + +`TranscriptionWorkflow` lives under `temporal-workflow/src/workflows` and runs +on the sessions role's queue independently of session orchestration. The +start/read/cancel API carries requester identity, stable explicit request keys, +and a pinned model. Activity attempts are bounded to three, with a fifteen-minute +schedule budget and a sixteen-minute workflow deadline. Cancellation waits for +activity acknowledgment before recording the terminal state. Missing results +report expiry after an authoritative CAS metadata check, even when bytes remain +in a process cache. + +Provider resolution now accepts the audio-transcriptions protocol, including +custom authenticated and anonymous endpoints. Native requests use the selected +model, language, prompt, URL, and configured headers. Endpoint overrides exclude +deployment credentials and organization/project headers. Provider responses and +transcripts are bounded. Audio limits and the optional transcoder belong to +standalone transcription; the session workflow performs no audio processing. + +Channels authorize and prepare attachments before starting/joining standalone +transcription through a short activity bridge. The conversation awaits the +result before emitting a bot event; stable conversation/message/attachment keys +reuse the same job. Filters see the transcript in message text; the original +text is retained separately. Spoken commands are never reclassified as channel +commands. Bot attachment `textRef` becomes ordinary text-reference input with +its source attachment as provenance. Existing bot-event and session roots retain +both text and original audio. + +New run, context, and steering input rejects raw audio with guidance to use +`transcriptions/start`, poll `transcriptions/read`, then submit +`{type: "textRef", blobRef: ..., provenanceRef: ...}` or reviewed plain text. The in-tree +raw-audio producer was Channels; CLI chat sends text. Direct API callers must +migrate. The greenfield cleanup removes the legacy preprocessing activity, +session input rewriting, audio-specific admission failures, source-to-transcript +retry matching, and channel version branch. Audio helpers and their tests live +under the standalone transcription implementation. + +The subsequent simplification removes `Transcript`, `transcript_input`, the +JSON artifact, and transcript-specific provider rendering. The workflow returns +a plain text blob and source-audio metadata. Reviewed dictation can omit source +provenance; channel delivery supplies it through generic text input. + +The plain-text path passed API/projection/bot/model-adapter/workflow tests, +356 server unit tests, affected Rust target checks, TypeScript checks, and client +tests. Live transcription, channel redelivery, and PostgreSQL retention tests +passed together. The expiry fixture uses unique text because identical results +share a CAS blob that another session may legitimately retain. + +Cleanup validation passed: API and workflow tests, generated contract checks, +356 server unit tests, all server targets, and TypeScript checks. The three +standalone transcription live tests and the channel voice/redelivery live test +passed again against an isolated database, which was removed afterward. + +Validation passed: API/auth/model-runtime/bot/workflow suites and generated +contract checks; server unit tests; TypeScript checks, Platform/web/client tests, +and the web production build. Live tests used an isolated PostgreSQL database +and unique Temporal queues. They cover default pinning across changes, +idempotency conflicts, requester isolation, transcoding, session provenance, +transient retry, acknowledged cancellation, ordinary CAS expiry, bot-event +retention, and channel redelivery with spoken command text. A real OpenAI audio +transcription also passed. Web recording and dictation tests belong to the next +slice. + ## Acceptance and validation Use offline tests with fake provider transports for the default validation @@ -463,11 +518,11 @@ workflow boundaries require it; live/credentialed suites remain explicit. providers never fall back to OpenAI. Discovery absence does not prohibit valid manual selection. - Matching job retries survive default changes and gateway/workflow restarts; - conflicting payloads fail. Exercise the record/start crash boundary, + conflicting payloads fail. Exercise concurrent workflow admission, transient versus terminal failures, deadlines, and cancellation races. -- Active and retained jobs keep blobs alive. Job expiry releases temporary - roots while session-admitted transcripts retain their source audio. Verify - access isolation for job reads and content downloads. +- Unsubmitted content uses ordinary CAS grace. Missing results report expiry; + session-admitted transcripts and bot events retain their original audio and + transcript through existing roots. Verify requester isolation for job reads. - Channel redelivery and delivery failure after successful transcription do not repeat admitted work. Voice/text ordering and mixed-media failures are explicit. Authorization precedes model work, and transcripts are available @@ -476,8 +531,8 @@ workflow boundaries require it; live/credentialed suites remain explicit. preserved edits, cancellation, late completion after send/navigation, microphone cleanup, permission/format failures, and unavailable defaults. - Run start, context append, and steering consistently enforce prepared input. - Old transcript history remains readable and recorded workflow histories - remain replayable across the selected rollout strategy. + Text provenance survives projection and session retention without special + transcript rendering, artifacts, or legacy transcription activity calls. ## Scope boundary diff --git a/package-lock.json b/package-lock.json index 91350e537..4e325df0e 100644 --- a/package-lock.json +++ b/package-lock.json @@ -7350,6 +7350,19 @@ "node": ">=10.12.0" } }, + "node_modules/automation-events": { + "version": "7.1.19", + "resolved": "https://registry.npmjs.org/automation-events/-/automation-events-7.1.19.tgz", + "integrity": "sha512-cD+TLhJTI0q4AI3ktd353lrGZiVa9AchowSDzQzzGjSoYe22js4vlS32VUtWuaulghi1Yq0KYNWKk9wWuGymPA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.2", + "tslib": "^2.8.1" + }, + "engines": { + "node": ">=18.2.0" + } + }, "node_modules/axobject-query": { "version": "4.1.0", "resolved": "https://registry.npmjs.org/axobject-query/-/axobject-query-4.1.0.tgz", @@ -7640,6 +7653,18 @@ "node": ">=8" } }, + "node_modules/broker-factory": { + "version": "3.1.15", + "resolved": "https://registry.npmjs.org/broker-factory/-/broker-factory-3.1.15.tgz", + "integrity": "sha512-ko+aWvgNuP49meGrdjUu7rC+Y+Wai3cCPxP3xWwHsHfehFjOh5ZQM2yC4gEB2UddeZ/YXhm0K1eG/L6fxym2Og==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "fast-unique-numbers": "^9.0.27", + "tslib": "^2.8.1", + "worker-factory": "^7.0.50" + } + }, "node_modules/browserslist": { "version": "4.28.8", "funding": [ @@ -10331,6 +10356,44 @@ "version": "3.0.2", "license": "MIT" }, + "node_modules/extendable-media-recorder": { + "version": "9.2.40", + "resolved": "https://registry.npmjs.org/extendable-media-recorder/-/extendable-media-recorder-9.2.40.tgz", + "integrity": "sha512-7jU3W9vDzcb8V9wwy9TsZxq37NrhH8Ru0K5EfY6AGFmeohBYXsgmeDqK13F3tAGGf3uCYo48jfK4MdXTIvEwxA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "media-encoder-host": "^9.0.31", + "multi-buffer-data-view": "^6.0.27", + "recorder-audio-worklet": "^6.0.60", + "standardized-audio-context": "^25.3.77", + "subscribable-things": "^2.1.60", + "tslib": "^2.8.1" + } + }, + "node_modules/extendable-media-recorder-wav-encoder-broker": { + "version": "7.0.128", + "resolved": "https://registry.npmjs.org/extendable-media-recorder-wav-encoder-broker/-/extendable-media-recorder-wav-encoder-broker-7.0.128.tgz", + "integrity": "sha512-4pP+IBbU8n9aP6teDvewleI+QtYCQPaRiSDW12PuPkhjxc89QZkM56w1zXhm0M4niJiqzh8SoAs/ndlWVuQo9A==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "broker-factory": "^3.1.15", + "extendable-media-recorder-wav-encoder-worker": "^8.0.123", + "tslib": "^2.8.1" + } + }, + "node_modules/extendable-media-recorder-wav-encoder-worker": { + "version": "8.0.123", + "resolved": "https://registry.npmjs.org/extendable-media-recorder-wav-encoder-worker/-/extendable-media-recorder-wav-encoder-worker-8.0.123.tgz", + "integrity": "sha512-WRKybAbfSICvwCzkfvWIaHBClf5G9njUq1jyX8pFonzSv3koo5FqwKHv7e5jyGsd7mpDLAtOSZcPZOkM10SaRQ==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "tslib": "^2.8.1", + "worker-factory": "^7.0.50" + } + }, "node_modules/fast-deep-equal": { "version": "3.1.3", "license": "MIT" @@ -10364,6 +10427,19 @@ "fast-string-truncated-width": "^3.0.2" } }, + "node_modules/fast-unique-numbers": { + "version": "9.0.27", + "resolved": "https://registry.npmjs.org/fast-unique-numbers/-/fast-unique-numbers-9.0.27.tgz", + "integrity": "sha512-nDA9ADeINN8SA2u2wCtU+siWFTTDqQR37XvgPIDDmboWQeExz7X0mImxuaN+kJddliIqy2FpVRmnvRZ+j8i1/A==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.2", + "tslib": "^2.8.1" + }, + "engines": { + "node": ">=18.2.0" + } + }, "node_modules/fast-uri": { "version": "3.1.5", "funding": [ @@ -12597,6 +12673,43 @@ "integrity": "sha512-9Yubnt3e8A0OKwxYSXyhLymGW4sCufcLG6VdiDdUGVkPhpqLxlvP5vl1983gQjJl3tqbrM731mjaZaP68AgosQ==", "license": "CC0-1.0" }, + "node_modules/media-encoder-host": { + "version": "9.0.31", + "resolved": "https://registry.npmjs.org/media-encoder-host/-/media-encoder-host-9.0.31.tgz", + "integrity": "sha512-PTgBe18JxIAntMnxiiPU8bnWZi75ZVHp+ed9Zmm+ogUmtbEbCNkuuDu5T86FlEa2ekiQ0G7Lx0F7xA9EB+FKCA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "media-encoder-host-broker": "^8.0.29", + "media-encoder-host-worker": "^10.0.29", + "tslib": "^2.8.1" + } + }, + "node_modules/media-encoder-host-broker": { + "version": "8.0.29", + "resolved": "https://registry.npmjs.org/media-encoder-host-broker/-/media-encoder-host-broker-8.0.29.tgz", + "integrity": "sha512-ATJ1GRUPwoIiBeRXSBvfF96CIRBYjHKaRTg861CrcS+6pALgMMjTvn/XDIalkqjY2pXCnAWiXn5szRWDEFOflg==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "broker-factory": "^3.1.15", + "fast-unique-numbers": "^9.0.27", + "media-encoder-host-worker": "^10.0.29", + "tslib": "^2.8.1" + } + }, + "node_modules/media-encoder-host-worker": { + "version": "10.0.29", + "resolved": "https://registry.npmjs.org/media-encoder-host-worker/-/media-encoder-host-worker-10.0.29.tgz", + "integrity": "sha512-AT1X6mIp6W++OxMimGMm622GFkAARzny82T1k8jxZWK0rIZhVoro9FaqIx+xERdGSMPZioN4qj0MKAgY04XENA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "extendable-media-recorder-wav-encoder-broker": "^7.0.128", + "tslib": "^2.8.1", + "worker-factory": "^7.0.50" + } + }, "node_modules/media-typer": { "version": "2.0.0", "license": "MIT", @@ -13536,6 +13649,19 @@ "dev": true, "license": "MIT" }, + "node_modules/multi-buffer-data-view": { + "version": "6.0.27", + "resolved": "https://registry.npmjs.org/multi-buffer-data-view/-/multi-buffer-data-view-6.0.27.tgz", + "integrity": "sha512-hqbVIRFskEUX3PmHxGPGFOcLU93va4yfYkBBg4c1U0wK1Z/pDfIbYDexbIIWI+vtrXWkQAStCQ4drcQ1ooakiA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.2", + "tslib": "^2.8.1" + }, + "engines": { + "node": ">=18.2.0" + } + }, "node_modules/music-metadata": { "version": "11.14.0", "funding": [ @@ -14865,6 +14991,32 @@ "url": "https://opencollective.com/unified" } }, + "node_modules/recorder-audio-worklet": { + "version": "6.0.60", + "resolved": "https://registry.npmjs.org/recorder-audio-worklet/-/recorder-audio-worklet-6.0.60.tgz", + "integrity": "sha512-ZIDm5URZxsPH4bqr2BgGkSC5QLqtAKMSvovM9KeDsI2pCvZVpy+PSWZFrlpgJ/QeuAF/f9cbHX/s9GjFxm7Pjw==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "broker-factory": "^3.1.15", + "fast-unique-numbers": "^9.0.27", + "recorder-audio-worklet-processor": "^5.0.41", + "standardized-audio-context": "^25.3.77", + "subscribable-things": "^2.1.60", + "tslib": "^2.8.1", + "worker-factory": "^7.0.50" + } + }, + "node_modules/recorder-audio-worklet-processor": { + "version": "5.0.41", + "resolved": "https://registry.npmjs.org/recorder-audio-worklet-processor/-/recorder-audio-worklet-processor-5.0.41.tgz", + "integrity": "sha512-TXaP51ZloO7mDyi71pF6BVU1MenHTMbEOFCbY4KfQGBjOsPuvvvhJzj25b/9RjchaF857x6MGg2CT8Ai8XbtXg==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "tslib": "^2.8.1" + } + }, "node_modules/regex": { "version": "6.1.0", "resolved": "https://registry.npmjs.org/regex/-/regex-6.1.0.tgz", @@ -15350,6 +15502,12 @@ "tslib": "^2.1.0" } }, + "node_modules/rxjs-interop": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/rxjs-interop/-/rxjs-interop-2.0.0.tgz", + "integrity": "sha512-ASEq9atUw7lualXB+knvgtvwkCEvGWV2gDD/8qnASzBkzEARZck9JAyxmY8OS6Nc1pCPEgDTKNcx+YqqYfzArw==", + "license": "MIT" + }, "node_modules/safe-stable-stringify": { "version": "2.5.0", "license": "MIT", @@ -15867,6 +16025,17 @@ "devOptional": true, "license": "MIT" }, + "node_modules/standardized-audio-context": { + "version": "25.3.77", + "resolved": "https://registry.npmjs.org/standardized-audio-context/-/standardized-audio-context-25.3.77.tgz", + "integrity": "sha512-Ki9zNz6pKcC5Pi+QPjPyVsD9GwJIJWgryji0XL9cAJXMGyn+dPOf6Qik1AHei0+UNVcc4BOCa0hWLBzlwqsW/A==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.25.6", + "automation-events": "^7.0.9", + "tslib": "^2.7.0" + } + }, "node_modules/statuses": { "version": "2.0.2", "license": "MIT", @@ -16047,6 +16216,17 @@ "integrity": "sha512-5Z9ZpRzfuH6l/UAvCPAPUo3665Nk2wLaZU3x+TLHKVzIz33+sbJqbtrYoC3KD4/uVOr2Zp+L0LySezP9OHV9yA==", "license": "MIT" }, + "node_modules/subscribable-things": { + "version": "2.1.60", + "resolved": "https://registry.npmjs.org/subscribable-things/-/subscribable-things-2.1.60.tgz", + "integrity": "sha512-6gamStlTGrZAHmIMKUxnrZGW1y0b2N3ofQMX/ZrWdYCcNytk/wRyiRB8eGE4Jg+bYFLypoQAb7UpmWjqpiWCKA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "rxjs-interop": "^2.0.0", + "tslib": "^2.8.1" + } + }, "node_modules/supports-color": { "version": "8.1.1", "license": "MIT", @@ -17665,6 +17845,17 @@ "version": "0.2.1", "license": "MIT" }, + "node_modules/worker-factory": { + "version": "7.0.50", + "resolved": "https://registry.npmjs.org/worker-factory/-/worker-factory-7.0.50.tgz", + "integrity": "sha512-hhwc0G+sFwM4qBuhJIUBn2p1Jf8v/FwmLUANBf/Q+Lt2uI8mfIZQhXaZQACodQD4R7Zp6cn/6702bIvNn2puJQ==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "fast-unique-numbers": "^9.0.27", + "tslib": "^2.8.1" + } + }, "node_modules/wrap-ansi": { "version": "7.0.0", "license": "MIT", @@ -18103,6 +18294,7 @@ "better-auth": "1.7.6", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", + "extendable-media-recorder": "^9.2.40", "hono": "^4.7.0", "lucide-react": "^1.23.0", "react": "^19.1.0", diff --git a/platform/configurator-mcp/src/generated/tools.ts b/platform/configurator-mcp/src/generated/tools.ts index 6ea2d5906..9143ddad8 100644 --- a/platform/configurator-mcp/src/generated/tools.ts +++ b/platform/configurator-mcp/src/generated/tools.ts @@ -2354,7 +2354,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "method": "session/context/append", "group": "session", "summary": "Append keyed session context", - "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries.", + "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries.", "paramsType": "ContextAppendParams", "resultType": "AgentApiOutcome", "inputSchema": { @@ -2403,6 +2403,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -2429,6 +2436,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -2678,6 +2692,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -2704,6 +2725,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -3158,6 +3186,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -3184,6 +3219,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -5033,6 +5075,149 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "object" } }, + { + "name": "lightspeed_transcriptions_start", + "method": "transcriptions/start", + "group": "transcriptions", + "summary": "Start transcription", + "description": "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created.", + "paramsType": "TranscriptionStartParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object", + "definitions": { + "ModelConfig": { + "properties": { + "apiKind": { + "type": "string" + }, + "model": { + "type": "string" + }, + "providerId": { + "type": "string" + } + }, + "required": [ + "providerId", + "apiKind", + "model" + ], + "type": "object" + }, + "TranscriptionAudio": { + "additionalProperties": { + "not": {} + }, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + } + } + } + }, + { + "name": "lightspeed_transcriptions_read", + "method": "transcriptions/read", + "group": "transcriptions", + "summary": "Read transcription", + "description": "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire.", + "paramsType": "TranscriptionReadParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + } + }, + { + "name": "lightspeed_transcriptions_cancel", + "method": "transcriptions/cancel", + "group": "transcriptions", + "summary": "Cancel transcription", + "description": "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority.", + "paramsType": "TranscriptionCancelParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + } + }, { "name": "lightspeed_models_defaults_read", "method": "models/defaults/read", diff --git a/platform/server/src/routes/gateway.ts b/platform/server/src/routes/gateway.ts index 9341d466f..7a3f48c24 100644 --- a/platform/server/src/routes/gateway.ts +++ b/platform/server/src/routes/gateway.ts @@ -209,7 +209,7 @@ const modelEndpointSchema = z.object({ baseUrl: z.string().trim().url(), headers: z.record(z.string(), z.string()).optional(), apiKinds: z - .array(z.enum(["openai:responses", "openai:completions"])) + .array(z.enum(["openai:responses", "openai:completions", "openai:audio-transcriptions"])) .min(1) .refine((kinds) => new Set(kinds).size === kinds.length, "API kinds must be unique"), }); diff --git a/platform/server/src/routes/method-roles.ts b/platform/server/src/routes/method-roles.ts index 020646fc4..2089fd9a0 100644 --- a/platform/server/src/routes/method-roles.ts +++ b/platform/server/src/routes/method-roles.ts @@ -111,6 +111,9 @@ export const METHOD_ROLES: Readonly> = { "session/share": "contributor", "session/skills/list": "viewer", "session/start": "contributor", + "transcriptions/cancel": "contributor", + "transcriptions/read": "viewer", + "transcriptions/start": "contributor", "vfs/snapshots/commit": "contributor", "vfs/snapshots/read": "viewer", "vfs/workspaces/create": "contributor", diff --git a/platform/web/package.json b/platform/web/package.json index d8783b64a..3cf23ca30 100644 --- a/platform/web/package.json +++ b/platform/web/package.json @@ -23,6 +23,7 @@ "better-auth": "1.7.6", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", + "extendable-media-recorder": "^9.2.40", "hono": "^4.7.0", "lucide-react": "^1.23.0", "react": "^19.1.0", diff --git a/platform/web/src/api.ts b/platform/web/src/api.ts index 6fe988965..015444b4a 100644 --- a/platform/web/src/api.ts +++ b/platform/web/src/api.ts @@ -278,7 +278,7 @@ export interface SecretProvider { export interface ModelEndpointConfig { baseUrl: string; headers?: Record; - apiKinds: Array<"openai:responses" | "openai:completions">; + apiKinds: Array<"openai:responses" | "openai:completions" | "openai:audio-transcriptions">; } export interface SecretsInventory { diff --git a/platform/web/src/components/models/model-api-key.tsx b/platform/web/src/components/models/model-api-key.tsx index b6ed66fa9..2a27239f5 100644 --- a/platform/web/src/components/models/model-api-key.tsx +++ b/platform/web/src/components/models/model-api-key.tsx @@ -366,6 +366,7 @@ export function OpenAiCompatibleForm({ const [responses, setResponses] = useState( initialEndpoint?.apiKinds.includes("openai:responses") ?? false, ); + const [transcriptions, setTranscriptions] = useState(initialEndpoint?.apiKinds.includes("openai:audio-transcriptions") ?? false); const [completions, setCompletions] = useState( initialEndpoint?.apiKinds.includes("openai:completions") ?? true, ); @@ -386,12 +387,14 @@ export function OpenAiCompatibleForm({ setBaseUrl(preset.baseUrl); setResponses(false); setCompletions(true); + setTranscriptions(false); }; const save = useMutation({ mutationFn: () => { const apiKinds = [ ...(responses ? (["openai:responses"] as const) : []), ...(completions ? (["openai:completions"] as const) : []), + ...(transcriptions ? (["openai:audio-transcriptions"] as const) : []), ]; if (!apiKinds.length) throw new Error("select at least one API kind"); return api( @@ -525,6 +528,10 @@ export function OpenAiCompatibleForm({ />{" "} Responses + Extra headers diff --git a/platform/web/src/lib/method-groups.ts b/platform/web/src/lib/method-groups.ts index c9a89851d..9d61543ed 100644 --- a/platform/web/src/lib/method-groups.ts +++ b/platform/web/src/lib/method-groups.ts @@ -15,6 +15,7 @@ export const METHOD_GROUPS: Record Date: Tue, 29 Sep 2026 00:25:59 +0200 Subject: [PATCH 03/12] transcribe --- clients/typescript/schema/api.schema.json | 2 +- clients/typescript/src/generated/types.ts | 8 +- crates/api/contract/api.schema.json | 2 +- crates/api/contract/openrpc.json | 2 +- crates/api/src/models.rs | 8 +- .../src/gateway/service/models_api.rs | 109 ++++++++++++++--- ...iverse-model-defaults-and-transcription.md | 33 +++++- .../configurator-mcp/src/generated/tools.ts | 2 +- platform/server/src/routes/gateway.ts | 45 ++++++- .../server/src/routes/transcriptions.test.ts | 67 +++++++++++ platform/shared/src/index.ts | 1 + platform/shared/src/transcriptions.ts | 18 +++ platform/web/src/api.ts | 3 +- .../src/components/models/model-defaults.tsx | 70 ++++++----- .../session/composer.dictation.test.tsx | 109 +++++++++++++++++ .../web/src/components/session/composer.tsx | 44 ++++++- .../session/session-config-editor.test.ts | 6 + .../session/session-config-editor.tsx | 2 + .../src/demo/fixtures/personal-assistant.ts | 6 +- platform/web/src/demo/router.ts | 2 + .../web/src/demo/routes/transcriptions.ts | 60 ++++++++++ platform/web/src/demo/transcriptions.test.ts | 47 ++++++++ platform/web/src/lib/audio-capture.test.ts | 70 +++++++++++ platform/web/src/lib/audio-capture.ts | 86 ++++++++++++++ platform/web/src/lib/dictation.test.ts | 65 +++++++++++ platform/web/src/lib/dictation.ts | 76 ++++++++++++ platform/web/src/lib/provider-readiness.ts | 6 +- platform/web/src/lib/use-dictation.test.tsx | 44 +++++++ platform/web/src/lib/use-dictation.ts | 110 ++++++++++++++++++ platform/web/src/pages/ModelsPage.test.tsx | 25 +++- platform/web/src/pages/ModelsPage.tsx | 10 +- platform/web/src/pages/SessionsPage.tsx | 3 + 32 files changed, 1064 insertions(+), 77 deletions(-) create mode 100644 platform/server/src/routes/transcriptions.test.ts create mode 100644 platform/shared/src/transcriptions.ts create mode 100644 platform/web/src/components/session/composer.dictation.test.tsx create mode 100644 platform/web/src/demo/routes/transcriptions.ts create mode 100644 platform/web/src/demo/transcriptions.test.ts create mode 100644 platform/web/src/lib/audio-capture.test.ts create mode 100644 platform/web/src/lib/audio-capture.ts create mode 100644 platform/web/src/lib/dictation.test.ts create mode 100644 platform/web/src/lib/dictation.ts create mode 100644 platform/web/src/lib/use-dictation.test.tsx create mode 100644 platform/web/src/lib/use-dictation.ts diff --git a/clients/typescript/schema/api.schema.json b/clients/typescript/schema/api.schema.json index 0faeffd8a..ea2c97062 100644 --- a/clients/typescript/schema/api.schema.json +++ b/clients/typescript/schema/api.schema.json @@ -12001,7 +12001,7 @@ "description": "Direct provider model discovery. Results may be served from a brief\nprocess-local cache; clients refresh by calling this method.", "properties": { "selectableOnly": { - "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nremoves OpenAI model-id families that are clearly not text-generation\nroutes (embeddings, moderation, image/video, speech, and realtime).\nIt is an ID policy, not a provider capability claim.", + "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nkeeps supported file-transcription routes and filters clearly unrelated\nOpenAI families from agent suggestions (embeddings, moderation, image/video,\nspeech synthesis, and realtime). Agent suggestions also have an age limit.\nClients select routes by API kind for their intended use. This is an ID\npolicy, not a provider capability claim.", "type": "boolean" } }, diff --git a/clients/typescript/src/generated/types.ts b/clients/typescript/src/generated/types.ts index 4ff6f81db..81706c700 100644 --- a/clients/typescript/src/generated/types.ts +++ b/clients/typescript/src/generated/types.ts @@ -7688,9 +7688,11 @@ export interface ModelDefaultsReadParams {} export interface ModelListParams { /** * Apply Lightspeed's small, conservative selectable-model policy. It - * removes OpenAI model-id families that are clearly not text-generation - * routes (embeddings, moderation, image/video, speech, and realtime). - * It is an ID policy, not a provider capability claim. + * keeps supported file-transcription routes and filters clearly unrelated + * OpenAI families from agent suggestions (embeddings, moderation, image/video, + * speech synthesis, and realtime). Agent suggestions also have an age limit. + * Clients select routes by API kind for their intended use. This is an ID + * policy, not a provider capability claim. */ selectableOnly?: boolean; } diff --git a/crates/api/contract/api.schema.json b/crates/api/contract/api.schema.json index 0faeffd8a..ea2c97062 100644 --- a/crates/api/contract/api.schema.json +++ b/crates/api/contract/api.schema.json @@ -12001,7 +12001,7 @@ "description": "Direct provider model discovery. Results may be served from a brief\nprocess-local cache; clients refresh by calling this method.", "properties": { "selectableOnly": { - "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nremoves OpenAI model-id families that are clearly not text-generation\nroutes (embeddings, moderation, image/video, speech, and realtime).\nIt is an ID policy, not a provider capability claim.", + "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nkeeps supported file-transcription routes and filters clearly unrelated\nOpenAI families from agent suggestions (embeddings, moderation, image/video,\nspeech synthesis, and realtime). Agent suggestions also have an age limit.\nClients select routes by API kind for their intended use. This is an ID\npolicy, not a provider capability claim.", "type": "boolean" } }, diff --git a/crates/api/contract/openrpc.json b/crates/api/contract/openrpc.json index 7a4ac4cf9..d924c3f0d 100644 --- a/crates/api/contract/openrpc.json +++ b/crates/api/contract/openrpc.json @@ -12001,7 +12001,7 @@ "description": "Direct provider model discovery. Results may be served from a brief\nprocess-local cache; clients refresh by calling this method.", "properties": { "selectableOnly": { - "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nremoves OpenAI model-id families that are clearly not text-generation\nroutes (embeddings, moderation, image/video, speech, and realtime).\nIt is an ID policy, not a provider capability claim.", + "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nkeeps supported file-transcription routes and filters clearly unrelated\nOpenAI families from agent suggestions (embeddings, moderation, image/video,\nspeech synthesis, and realtime). Agent suggestions also have an age limit.\nClients select routes by API kind for their intended use. This is an ID\npolicy, not a provider capability claim.", "type": "boolean" } }, diff --git a/crates/api/src/models.rs b/crates/api/src/models.rs index 28ef814b2..dfb9b8e78 100644 --- a/crates/api/src/models.rs +++ b/crates/api/src/models.rs @@ -6,9 +6,11 @@ use super::*; #[serde(rename_all = "camelCase")] pub struct ModelListParams { /// Apply Lightspeed's small, conservative selectable-model policy. It - /// removes OpenAI model-id families that are clearly not text-generation - /// routes (embeddings, moderation, image/video, speech, and realtime). - /// It is an ID policy, not a provider capability claim. + /// keeps supported file-transcription routes and filters clearly unrelated + /// OpenAI families from agent suggestions (embeddings, moderation, image/video, + /// speech synthesis, and realtime). Agent suggestions also have an age limit. + /// Clients select routes by API kind for their intended use. This is an ID + /// policy, not a provider capability claim. #[serde(default, skip_serializing_if = "std::ops::Not::not")] pub selectable_only: bool, } diff --git a/crates/temporal-server/src/gateway/service/models_api.rs b/crates/temporal-server/src/gateway/service/models_api.rs index b79f1b91c..3280b782e 100644 --- a/crates/temporal-server/src/gateway/service/models_api.rs +++ b/crates/temporal-server/src/gateway/service/models_api.rs @@ -23,6 +23,7 @@ const OPENAI_PROVIDER_ID: &str = "openai"; const ANTHROPIC_PROVIDER_ID: &str = "anthropic"; const OPENAI_RESPONSES_API_KIND: &str = "openai:responses"; const OPENAI_COMPLETIONS_API_KIND: &str = "openai:completions"; +const OPENAI_AUDIO_API_KIND: &str = "openai:audio-transcriptions"; const ANTHROPIC_MESSAGES_API_KIND: &str = "anthropic:messages"; const MODEL_DISCOVERY_TIMEOUT: Duration = Duration::from_secs(15); const MODEL_DISCOVERY_CACHE_TTL: Duration = Duration::from_secs(10); @@ -184,11 +185,7 @@ impl ModelDiscoveryService { )) }); if selectable_only { - models.retain(|model| { - model.provider_id != OPENAI_PROVIDER_ID - || (is_openai_selectable_model(&model.model) - && is_openai_recent_model(model.created_at_ms, model.fetched_at_ms)) - }); + models.retain(is_selectable_model_route); } ModelListResponse { models, providers } } @@ -244,7 +241,11 @@ impl ModelDiscoveryService { models, provider_success( OPENAI_PROVIDER_ID, - &[OPENAI_RESPONSES_API_KIND, OPENAI_COMPLETIONS_API_KIND], + &[ + OPENAI_RESPONSES_API_KIND, + OPENAI_COMPLETIONS_API_KIND, + OPENAI_AUDIO_API_KIND, + ], fetched_at_ms, source, credential, @@ -255,7 +256,11 @@ impl ModelDiscoveryService { Vec::new(), provider_failure( OPENAI_PROVIDER_ID, - &[OPENAI_RESPONSES_API_KIND, OPENAI_COMPLETIONS_API_KIND], + &[ + OPENAI_RESPONSES_API_KIND, + OPENAI_COMPLETIONS_API_KIND, + OPENAI_AUDIO_API_KIND, + ], &error, source, ), @@ -492,19 +497,48 @@ fn model_endpoint(config: &auth::AuthProviderConfig) -> Option<&auth::ModelEndpo } } -fn openai_model_views(model: openai::Model, fetched_at_ms: i64) -> [ModelView; 2] { +fn is_selectable_model_route(model: &ModelView) -> bool { + model.provider_id != OPENAI_PROVIDER_ID + || model.api_kind == OPENAI_AUDIO_API_KIND + || (is_openai_selectable_model(&model.model) + && is_openai_recent_model(model.created_at_ms, model.fetched_at_ms)) +} + +// The Models API omits endpoint capabilities. These file-transcription families +// accept the plain JSON transcription request supported by our audio adapter. +// Diarization and realtime-only models require different request options. +fn is_openai_transcription_model(model: &str) -> bool { + [ + "whisper-1", + "gpt-transcribe", + "gpt-4o-transcribe", + "gpt-4o-mini-transcribe", + ] + .iter() + .any(|family| is_model_family(model, family)) +} + +fn openai_model_views(model: openai::Model, fetched_at_ms: i64) -> Vec { let created_at_ms = unix_seconds_to_millis(model.created); let capabilities = openai_model_capabilities(&model.id); - [OPENAI_RESPONSES_API_KIND, OPENAI_COMPLETIONS_API_KIND].map(|api_kind| ModelView { - provider_id: OPENAI_PROVIDER_ID.to_owned(), - api_kind: api_kind.to_owned(), - display_name: model.id.clone(), - model: model.id.clone(), - capabilities: capabilities.clone(), - created_at_ms, - source: ModelSource::Provider, - fetched_at_ms, - }) + let api_kinds: &[&str] = if is_openai_transcription_model(&model.id) { + &[OPENAI_AUDIO_API_KIND] + } else { + &[OPENAI_RESPONSES_API_KIND, OPENAI_COMPLETIONS_API_KIND] + }; + api_kinds + .iter() + .map(|api_kind| ModelView { + provider_id: OPENAI_PROVIDER_ID.to_owned(), + api_kind: (*api_kind).to_owned(), + display_name: model.id.clone(), + model: model.id.clone(), + capabilities: capabilities.clone(), + created_at_ms, + source: ModelSource::Provider, + fetched_at_ms, + }) + .collect() } fn unix_seconds_to_millis(seconds: Option) -> Option { @@ -1060,7 +1094,10 @@ mod tests { ); assert_eq!( - views.clone().map(|view| view.api_kind), + views + .iter() + .map(|view| view.api_kind.clone()) + .collect::>(), [ OPENAI_RESPONSES_API_KIND.to_owned(), OPENAI_COMPLETIONS_API_KIND.to_owned(), @@ -1079,6 +1116,40 @@ mod tests { } } + #[test] + fn speech_discovery_uses_the_audio_route_without_the_agent_age_filter() { + for id in [ + "whisper-1", + "gpt-transcribe", + "gpt-4o-transcribe", + "gpt-4o-mini-transcribe", + "gpt-4o-mini-transcribe-2025-12-15", + ] { + let views = openai_model_views( + openai::Model { + id: id.to_owned(), + created: Some(1), + object: None, + owned_by: None, + }, + OPENAI_SELECTABLE_MAX_AGE_MS * 2, + ); + assert_eq!(views.len(), 1, "{id}"); + assert_eq!(views[0].api_kind, OPENAI_AUDIO_API_KIND, "{id}"); + assert!(is_selectable_model_route(&views[0]), "{id}"); + assert_eq!(views[0].capabilities.reasoning_efforts, None); + } + for id in [ + "gpt-4o-transcribe-diarize", + "gpt-live-transcribe", + "gpt-realtime-whisper", + "gpt-6-sol", + "gpt-4o-mini-tts", + ] { + assert!(!is_openai_transcription_model(id), "{id}"); + } + } + #[test] fn openai_reasoning_catalog_covers_current_families_and_snapshots() { for model in ["gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"] { diff --git a/docs/roadmap/p184-universe-model-defaults-and-transcription.md b/docs/roadmap/p184-universe-model-defaults-and-transcription.md index 73abf9209..846509f1c 100644 --- a/docs/roadmap/p184-universe-model-defaults-and-transcription.md +++ b/docs/roadmap/p184-universe-model-defaults-and-transcription.md @@ -1,7 +1,7 @@ # P184 — Universe model defaults and standalone transcription **Status:** Universe defaults, Platform model settings, standalone transcription, -and channel voice preparation implemented, 2026-09-28. Web dictation remains. +channel voice preparation, and web dictation implemented, 2026-09-29. Builds on [CLI session model routing](cli-session-model-routing.md) and [the runtime CLI](p183-first-class-runtime-cli.md). Supersedes the placement of transcription inside session admission in @@ -497,8 +497,35 @@ and unique Temporal queues. They cover default pinning across changes, idempotency conflicts, requester isolation, transcoding, session provenance, transient retry, acknowledged cancellation, ordinary CAS expiry, bot-event retention, and channel redelivery with spoken command text. A real OpenAI audio -transcription also passed. Web recording and dictation tests belong to the next -slice. +transcription also passed. Web recording and dictation validation follows below. + +Web dictation now records through lazy-loaded `extendable-media-recorder`, +choosing a browser-supported WebM, MP4, or Ogg format. The composer requests +microphone access only on an explicit click, caps recordings at ten minutes +and 25 MiB, and releases tracks on completion, cancellation, errors, and +navigation. Upload/start/read/cancel routes use the member-scoped runtime client; +web admission always uses the universe speech default. + +The transcript appends to the latest editable draft and never sends itself. +Retry keeps the recording in memory and rejoins an admitted job after transport +failure. Cancel, send, and navigation prevent late completion from changing a +draft. Reviewed text uses ordinary session input without audio provenance. +The demo uses a clearly labeled sample recording and transcript. + +Models → Defaults contains separate agent-run and speech-to-text rows below +Providers. Operators can configure either slot; Contributors can transcribe. +Dictation is disabled until the speech default is set and is also subject to +session input permissions and known provider/browser readiness. OpenAI discovery +maps supported file-transcription families to the audio protocol, bypasses the +agent-only age filter for those routes, and keeps them out of agent pickers. +Custom providers continue to use their declared API kinds and manual choices. + +Validation passed: 572 web tests, 162 Platform server tests, 13 Rust model +discovery tests, eight API contract tests, TypeScript checks, and production +and demo builds. Browser checks covered speech suggestions, editable transcript +preview, clearing the default, and mobile layout. A real Chromium recorder +produced WebM audio from a generated audio stream and released its tracks; +physical microphones and Safari were not exercised in this slice. ## Acceptance and validation diff --git a/platform/configurator-mcp/src/generated/tools.ts b/platform/configurator-mcp/src/generated/tools.ts index 9143ddad8..323a5d682 100644 --- a/platform/configurator-mcp/src/generated/tools.ts +++ b/platform/configurator-mcp/src/generated/tools.ts @@ -5319,7 +5319,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "description": "Direct provider model discovery. Results may be served from a brief\nprocess-local cache; clients refresh by calling this method.", "properties": { "selectableOnly": { - "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nremoves OpenAI model-id families that are clearly not text-generation\nroutes (embeddings, moderation, image/video, speech, and realtime).\nIt is an ID policy, not a provider capability claim.", + "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nkeeps supported file-transcription routes and filters clearly unrelated\nOpenAI families from agent suggestions (embeddings, moderation, image/video,\nspeech synthesis, and realtime). Agent suggestions also have an age limit.\nClients select routes by API kind for their intended use. This is an ID\npolicy, not a provider capability claim.", "type": "boolean" } }, diff --git a/platform/server/src/routes/gateway.ts b/platform/server/src/routes/gateway.ts index 7a3f48c24..039ab7948 100644 --- a/platform/server/src/routes/gateway.ts +++ b/platform/server/src/routes/gateway.ts @@ -2,6 +2,7 @@ import { UniverseSlugCacheConflict } from "../universe-slugs.js"; import { deploymentClient, GatewayUnconfigured, GateRefusal, memberClient } from "../runtime-client.js"; import { Hono } from "hono"; import { z } from "zod"; +import { bodyLimit } from "hono/body-limit"; import { LightspeedClient, LightspeedRpcError, @@ -25,7 +26,7 @@ import { type SessionConfig, } from "@lightspeed-ai/agent-client"; import { schema } from "@lightspeed/platform-db"; -import { modelDefaultsPutSchema, roleAtLeast, slugify, workspaceCreateSchema } from "@lightspeed/platform-shared"; +import { MAX_DICTATION_AUDIO_BYTES, transcriptionStartSchema, transcriptionUploadSchema, modelDefaultsPutSchema, roleAtLeast, slugify, workspaceCreateSchema } from "@lightspeed/platform-shared"; import type { AppContext, ApiVariables } from "../context.js"; import { parseBody } from "../http.js"; import { @@ -939,6 +940,46 @@ export function gatewayRoutes(ctx: AppContext) { }); }); + app.post("/:id/transcriptions/audio", bodyLimit({ maxSize: 36 * 1024 * 1024 }), async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + const body = await parseBody(c, transcriptionUploadSchema); + if (!body.ok) return body.response; + if (Buffer.from(body.data.bytesBase64, "base64").length > MAX_DICTATION_AUDIO_BYTES) { + return c.json({ error: "Recording exceeds the 25 MiB limit." }, 413); + } + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("blobs/put", { blobs: [body.data] }); + return c.json(response.result); + }); + }); + app.post("/:id/transcriptions", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + const body = await parseBody(c, transcriptionStartSchema); + if (!body.ok) return body.response; + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("transcriptions/start", body.data); + return c.json(response.result.transcription); + }); + }); + app.get("/:id/transcriptions/:transcriptionId", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("transcriptions/read", { transcriptionId: c.req.param("transcriptionId") }); + return c.json(response.result.transcription); + }); + }); + app.post("/:id/transcriptions/:transcriptionId/cancel", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("transcriptions/cancel", { transcriptionId: c.req.param("transcriptionId") }); + return c.json(response.result.transcription); + }); + }); + app.get("/:id/models/defaults", async (c) => { const access = await universeForSession(ctx, c, c.req.param("id")); if (!access) return c.json({ error: "not found" }, 404); @@ -959,7 +1000,7 @@ export function gatewayRoutes(ctx: AppContext) { }); }); - /// Provider-discovered model routes for the session-config model picker. + /// Provider-discovered routes for agent and speech model pickers. /// Lightspeed owns credential injection and sanitizes per-provider errors. app.get("/:id/models", async (c) => { const access = await universeForSession(ctx, c, c.req.param("id")); diff --git a/platform/server/src/routes/transcriptions.test.ts b/platform/server/src/routes/transcriptions.test.ts new file mode 100644 index 000000000..515de4bdc --- /dev/null +++ b/platform/server/src/routes/transcriptions.test.ts @@ -0,0 +1,67 @@ +import { Hono } from "hono"; +import { afterEach, beforeEach, expect, it, vi } from "vitest"; +import type { ApiVariables, AppContext } from "../context.js"; +import { gatewayRoutes } from "./gateway.js"; +const auth = vi.hoisted(() => ({ role: "contributor" })); +vi.mock("./universes.js", () => ({ universeForSession: vi.fn(async (_ctx, _c, id: string) => ({ + universe: { lightspeedUniverseId: id, gatewayUrl: "https://engine.example/rpc" }, + slug: "test", role: auth.role, member: { userId: "member", role: auth.role }, +})) })); +beforeEach(() => { auth.role = "contributor"; }); +afterEach(() => vi.unstubAllGlobals()); +const audio = { blobRef: `sha256:${"a".repeat(64)}`, mime: "audio/webm", name: "dictation.webm" }; +function fixture() { + const requests: { method: string; params: Record }[] = []; + const fetch = vi.fn(async (_url: unknown, init: RequestInit) => { + expect(new Headers(init.headers).get("x-lightspeed-universe")).toBe("universe"); + expect(new Headers(init.headers).get("x-lightspeed-actor")).toBe("member"); + const rpc = JSON.parse(String(init.body)); + requests.push(rpc); + return Response.json({ id: rpc.id, result: { result: rpc.method === "blobs/put" ? { blobs: [{ blobRef: audio.blobRef, bytes: 5 }] } : { transcription: { transcriptionId: "job", status: "running" } }, notifications: [] } }); + }); + vi.stubGlobal("fetch", fetch); + const app = new Hono<{ Variables: ApiVariables }>(); + app.use("*", async (c, next) => { c.set("session", { user: { id: "member" } } as ApiVariables["session"]); await next(); }); + app.route("/", gatewayRoutes({ env: { lightspeedApiUrl: "https://engine.example/rpc", lightspeedApiKey: "lsk_fixture" } } as AppContext)); + const call = (method: string, path = "", body?: unknown) => app.request(`/universe/transcriptions${path}`, { + method, headers: { "content-type": "application/json" }, ...(body === undefined ? {} : { body: JSON.stringify(body) }), + }); + return { call, requests, fetch }; +} +it("lets contributors upload, transcribe, inspect, and cancel under their own identity", async () => { + const f = fixture(); + expect(await (await f.call("POST", "/audio", { bytesBase64: btoa("audio") })).json()).toMatchObject({ blobs: [{ blobRef: audio.blobRef }] }); + expect(await (await f.call("POST", "", { audio, idempotencyKey: "request" })).json()).toMatchObject({ transcriptionId: "job" }); + expect((await f.call("GET", "/job")).status).toBe(200); + expect((await f.call("POST", "/job/cancel")).status).toBe(200); + expect(f.requests.map(({ method, params }) => ({ method, params }))).toEqual([ + { method: "blobs/put", params: { blobs: [{ bytesBase64: btoa("audio") }] } }, + { method: "transcriptions/start", params: { audio, idempotencyKey: "request" } }, + { method: "transcriptions/read", params: { transcriptionId: "job" } }, + { method: "transcriptions/cancel", params: { transcriptionId: "job" } }, + ]); +}); +it("rejects viewer mutations before reaching the runtime", async () => { + auth.role = "viewer"; + const f = fixture(); + expect((await f.call("POST", "/audio", { bytesBase64: btoa("audio") })).status).toBe(403); + expect((await f.call("POST", "", { audio, idempotencyKey: "request" })).status).toBe(403); + expect((await f.call("POST", "/job/cancel")).status).toBe(403); + expect(f.fetch).not.toHaveBeenCalled(); +}); +it("rejects explicit model overrides and malformed audio before reaching the runtime", async () => { + const f = fixture(); + expect((await f.call("POST", "", { audio, idempotencyKey: "request", model: { providerId: "other", apiKind: "openai:audio-transcriptions", model: "override" } })).status).toBe(400); + expect((await f.call("POST", "/audio", { bytesBase64: "!!!" })).status).toBe(400); + expect(f.fetch).not.toHaveBeenCalled(); +}); +it("preserves the missing speech default error", async () => { + const f = fixture(); + f.fetch.mockImplementation(async (_url, init) => { + const rpc = JSON.parse(String(init.body)); + return Response.json({ id: rpc.id, error: { code: -32014, message: "Choose a speech model", data: { kind: "model_default_unset", message: "Choose a speech model", modelDefaultSlot: "speechToText" } } }); + }); + const response = await f.call("POST", "", { audio, idempotencyKey: "request" }); + expect(response.status).toBe(400); + expect(await response.json()).toMatchObject({ kind: "model_default_unset", modelDefaultSlot: "speechToText" }); +}); diff --git a/platform/shared/src/index.ts b/platform/shared/src/index.ts index 819af6589..cd73fb47a 100644 --- a/platform/shared/src/index.ts +++ b/platform/shared/src/index.ts @@ -137,3 +137,4 @@ export function mergeFeatureOverrides(stored: unknown, changes: FeatureOverrides return merged; } export { AGENT_MODEL_API_KINDS, modelDefaultsPutSchema } from "./model-defaults.js"; +export { MAX_DICTATION_AUDIO_BYTES, MAX_DICTATION_SECONDS, transcriptionStartSchema, transcriptionUploadSchema } from "./transcriptions.js"; diff --git a/platform/shared/src/transcriptions.ts b/platform/shared/src/transcriptions.ts new file mode 100644 index 000000000..eb0ca81ff --- /dev/null +++ b/platform/shared/src/transcriptions.ts @@ -0,0 +1,18 @@ +import { z } from "zod"; + +export const MAX_DICTATION_AUDIO_BYTES = 25 * 1024 * 1024; +export const MAX_DICTATION_SECONDS = 10 * 60; +export const transcriptionUploadSchema = z.object({ + bytesBase64: z.string().min(4).max(4 * Math.ceil(MAX_DICTATION_AUDIO_BYTES / 3)).regex(/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/), +}).strict(); + +// Web dictation always uses the universe default. Model overrides belong to +// the core API, not this browser route. +export const transcriptionStartSchema = z.object({ + idempotencyKey: z.string().min(1).max(200), + audio: z.object({ + blobRef: z.string().regex(/^sha256:[a-f0-9]{64}$/), + mime: z.string().min(1).max(128), + name: z.string().min(1).max(256), + }).strict(), +}).strict(); diff --git a/platform/web/src/api.ts b/platform/web/src/api.ts index 015444b4a..ed48239da 100644 --- a/platform/web/src/api.ts +++ b/platform/web/src/api.ts @@ -57,8 +57,9 @@ function extractMessage(body: unknown): string | null { return error; } -export async function api(method: string, path: string, body?: unknown): Promise { +export async function api(method: string, path: string, body?: unknown, signal?: AbortSignal): Promise { const res = await fetch(path, { + signal, method, headers: body !== undefined ? { "content-type": "application/json" } : undefined, body: body !== undefined ? JSON.stringify(body) : undefined, diff --git a/platform/web/src/components/models/model-defaults.tsx b/platform/web/src/components/models/model-defaults.tsx index 4ea976ca6..53a95bea8 100644 --- a/platform/web/src/components/models/model-defaults.tsx +++ b/platform/web/src/components/models/model-defaults.tsx @@ -15,42 +15,54 @@ const apiLabels: Record = { "openai:responses": "OpenAI Responses", "openai:completions": "OpenAI Chat Completions", "anthropic:messages": "Anthropic Messages", + "openai:audio-transcriptions": "OpenAI Audio Transcriptions", }; -export function ModelDefaultsSection({ universeId, writable, onEdit }: { universeId: string; writable: boolean; onEdit: (defaults: ModelDefaults) => void }) { +export type DefaultSlot = ModelDefaultsPutParams["slot"]; +const slots = [ + { slot: "agentRun", label: "Agent runs", description: "Used when a new session and its profile leave the model unset. Existing sessions keep their model." }, + { slot: "speechToText", label: "Speech-to-text", description: "Used for audio transcription and web dictation. Dictation is disabled until a default is selected." }, +] as const; + +export function ModelDefaultsSection({ universeId, writable, onEdit }: { universeId: string; writable: boolean; onEdit: (defaults: ModelDefaults, slot: DefaultSlot) => void }) { const defaults = useModelDefaults(universeId); const discovery = useModelDiscovery(universeId); const queryClient = useQueryClient(); const clear = useMutation({ - mutationFn: (revision: number) => api("PUT", `/api/v1/universes/${universeId}/models/defaults`, { - slot: "agentRun", model: null, expectedRevision: revision, + mutationFn: ({ revision, slot }: { revision: number; slot: DefaultSlot }) => api("PUT", `/api/v1/universes/${universeId}/models/defaults`, { + slot, model: null, expectedRevision: revision, } satisfies ModelDefaultsPutParams), onSuccess: (value) => queryClient.setQueryData(modelDefaultsKey(universeId), value), onError: () => { void queryClient.invalidateQueries({ queryKey: modelDefaultsKey(universeId) }); }, }); - const model = defaults.data?.agentRun; - const readiness = summarizeProviderReadiness(model, discovery.error ? undefined : discovery.data?.providers); + return (

Defaults

-

Used when a new session and its profile leave the model unset. Existing sessions keep their model.

-
-
-
-

Agent runs

-

{defaults.isLoading ? "Loading…" : defaults.error ? "Default unavailable" : model ? modelLabel(model) : "No default selected"}

- {model &&

{apiLabels[model.apiKind] ?? model.apiKind}

} - {model &&

- {discovery.isLoading ? "Checking provider…" : readiness.message} -

} -
- {writable && defaults.data && !defaults.error &&
- - {model && } -
} -
+
+ {slots.map(({ slot, label, description }) => { + const model = defaults.data?.[slot]; + const readiness = summarizeProviderReadiness(model, discovery.error ? undefined : discovery.data?.providers, slot); + return ( +
+
+

{label}

+

{description}

+

{defaults.isLoading ? "Loading…" : defaults.error ? "Default unavailable" : model ? modelLabel(model) : "No default selected"}

+ {model &&

{apiLabels[model.apiKind] ?? model.apiKind}

} + {model &&

+ {discovery.isLoading ? "Checking provider…" : readiness.message} +

} +
+ {writable && defaults.data && !defaults.error &&
+ + {model && } +
} +
+ ); + })} {defaults.error &&
Could not load defaults: {defaults.error.message} @@ -61,20 +73,22 @@ export function ModelDefaultsSection({ universeId, writable, onEdit }: { univers ); } -export function DefaultModelDialog({ universeId, initial, onClose }: { universeId: string; initial: ModelDefaults; onClose: () => void }) { +export function DefaultModelDialog({ universeId, initial, slot, onClose }: { universeId: string; initial: ModelDefaults; slot: DefaultSlot; onClose: () => void }) { + const apiKinds = slot === "agentRun" ? AGENT_MODEL_API_KINDS : ["openai:audio-transcriptions"] as const; + const emptyModel = { providerId: "", apiKind: apiKinds[0], model: "" }; const [revision, setRevision] = useState(initial.revision); - const [model, setModel] = useState(initial.agentRun ?? { providerId: "", apiKind: "openai:responses", model: "" }); + const [model, setModel] = useState(initial[slot] ?? emptyModel); const [reloadError, setReloadError] = useState(null); const [reloading, setReloading] = useState(false); const discovery = useModelDiscovery(universeId); const queryClient = useQueryClient(); const id = useId(); const [search, setSearch] = useState(""); - const routes = (discovery.data?.models ?? []).filter((route) => (AGENT_MODEL_API_KINDS as readonly string[]).includes(route.apiKind)); + const routes = (discovery.data?.models ?? []).filter((route) => (apiKinds as readonly string[]).includes(route.apiKind)); const choices = routes.map((route) => JSON.stringify([route.providerId, route.apiKind, route.model])); const routeFor = (key: string) => routes[choices.indexOf(key)]; const clean = { providerId: model.providerId.trim(), apiKind: model.apiKind, model: model.model.trim() }; - const params: ModelDefaultsPutParams = { slot: "agentRun", model: clean, expectedRevision: revision }; + const params: ModelDefaultsPutParams = { slot, model: clean, expectedRevision: revision }; const valid = modelDefaultsPutSchema.safeParse(params).success; const save = useMutation({ mutationFn: () => api("PUT", `/api/v1/universes/${universeId}/models/defaults`, params), @@ -88,7 +102,7 @@ export function DefaultModelDialog({ universeId, initial, onClose }: { universeI const latest = await api("GET", `/api/v1/universes/${universeId}/models/defaults`); queryClient.setQueryData(modelDefaultsKey(universeId), latest); setRevision(latest.revision); - setModel(latest.agentRun ?? { providerId: "", apiKind: "openai:responses", model: "" }); + setModel(latest[slot] ?? emptyModel); save.reset(); } catch (error) { setReloadError(error instanceof Error ? error.message : "Could not reload defaults."); } finally { setReloading(false); } @@ -96,7 +110,7 @@ export function DefaultModelDialog({ universeId, initial, onClose }: { universeI return { if (!open && !save.isPending) onClose(); }}> - Default model for agent runs + Default model for {slot === "agentRun" ? "agent runs" : "speech-to-text"} Choose a discovered model or enter your provider and model below.
{ event.preventDefault(); if (valid && !conflict && !save.isPending) save.mutate(); }}> @@ -120,7 +134,7 @@ export function DefaultModelDialog({ universeId, initial, onClose }: { universeI {discovery.data?.providers?.map((provider) => API
Model setModel({ ...model, model: event.target.value })} /> diff --git a/platform/web/src/components/session/composer.dictation.test.tsx b/platform/web/src/components/session/composer.dictation.test.tsx new file mode 100644 index 000000000..ae7054684 --- /dev/null +++ b/platform/web/src/components/session/composer.dictation.test.tsx @@ -0,0 +1,109 @@ +// @vitest-environment jsdom +import { act } from "react"; +import { createRoot, type Root } from "react-dom/client"; +import { afterEach, beforeEach, expect, it, vi } from "vitest"; +import { SessionComposer } from "./composer"; + +const mocks = vi.hoisted(() => ({ capture: vi.fn(), transcribe: vi.fn(), cancel: vi.fn() })); +vi.mock("@/lib/audio-capture", () => ({ startAudioCapture: mocks.capture, isDemoDictation: false })); +vi.mock("@/lib/dictation", () => ({ transcribeRecording: mocks.transcribe, cancelRecording: mocks.cancel })); +let root: Root; +let container: HTMLDivElement; +let finish: (text: string) => void; +let stop: ReturnType; +let onSend: ReturnType; +beforeEach(() => { + vi.stubGlobal("IS_REACT_ACT_ENVIRONMENT", true); + vi.clearAllMocks(); + stop = vi.fn(); + mocks.capture.mockResolvedValue({ stop, name: "dictation.webm", result: Promise.resolve(new Blob(["audio"], { type: "audio/webm" })) }); + mocks.transcribe.mockImplementation(() => new Promise((resolve) => { finish = resolve; })); + onSend = vi.fn(); + container = document.createElement("div"); + document.body.append(container); + root = createRoot(container); +}); +afterEach(async () => { + await act(async () => root.unmount()); + container.remove(); + localStorage.clear(); + vi.unstubAllGlobals(); +}); +async function show(disabledReason?: string, disabled = false) { + await act(async () => root.render()); +} +async function click(label: string) { + const button = [...container.querySelectorAll("button")].find((node) => node.getAttribute("aria-label") === label || node.textContent === label)!; + await act(async () => button.click()); +} +async function type(text: string) { + const input = container.querySelector("textarea")!; + await act(async () => { + Object.getOwnPropertyDescriptor(HTMLTextAreaElement.prototype, "value")!.set!.call(input, text); + input.dispatchEvent(new Event("input", { bubbles: true })); + }); +} +async function record() { await click("Dictate message"); await click("Stop recording"); } +it("appends to the latest edited draft and waits for an explicit send", async () => { + await show(); + await type("Before"); + await record(); + await type("Edited while transcribing."); + await act(async () => finish("Spoken text.")); + expect(container.querySelector("textarea")!.value).toBe("Edited while transcribing. Spoken text."); + expect(localStorage.getItem("voice-test")).toBe("Edited while transcribing. Spoken text."); + expect(container.textContent).toContain("Review and edit before sending"); + expect(onSend).not.toHaveBeenCalled(); + expect(stop).toHaveBeenCalled(); + await click("Send message"); + expect(onSend).toHaveBeenCalledWith("Edited while transcribing. Spoken text.", null); +}); +it.each(["Cancel dictation", "Send message"])("ignores late completion after %s", async (action) => { + await show(); + await type("Keep this"); + await record(); + await click(action); + expect(mocks.transcribe.mock.calls[0]![2].aborted).toBe(true); + await act(async () => finish("Late text")); + expect(container.querySelector("textarea")!.value).toBe(action === "Send message" ? "" : "Keep this"); + expect(onSend).toHaveBeenCalledTimes(action === "Send message" ? 1 : 0); +}); +it("cancels on navigation without modifying the saved draft", async () => { + await show(); + await type("Saved draft"); + await record(); + await act(async () => root.render(null)); + await act(async () => finish("Late text")); + expect(mocks.transcribe.mock.calls[0]![2].aborted).toBe(true); + expect(localStorage.getItem("voice-test")).toBe("Saved draft"); +}); +it.each([["Set a speech-to-text default", false], [undefined, true]] as const)("disables recording for missing defaults or session permissions", async (reason, disabled) => { + await show(reason, disabled); + const button = container.querySelector('[aria-label="Dictate message"]')!; + expect(button.disabled).toBe(true); + await click("Dictate message"); + expect(mocks.capture).not.toHaveBeenCalled(); +}); +it("retains the draft and recording for a transcription retry", async () => { + mocks.transcribe.mockRejectedValueOnce(new Error("Service unavailable")); + await show(); + await type("Draft"); + await record(); + expect(container.textContent).toContain("Service unavailable"); + expect(container.querySelector("textarea")!.value).toBe("Draft"); + await click("Retry transcription"); + expect(mocks.capture).toHaveBeenCalledTimes(1); + expect(mocks.transcribe.mock.calls[1]![1]).toBe(mocks.transcribe.mock.calls[0]![1]); + await act(async () => finish("Retry worked")); + expect(container.querySelector("textarea")!.value).toBe("Draft Retry worked"); +}); +it("explains permission denial without losing text", async () => { + mocks.capture.mockRejectedValueOnce(new DOMException("Denied", "NotAllowedError")); + await show(); + await type("Draft"); + await click("Dictate message"); + expect(container.textContent).toContain("Microphone access was denied"); + expect(container.querySelector("textarea")!.value).toBe("Draft"); + expect(mocks.transcribe).not.toHaveBeenCalled(); +}); diff --git a/platform/web/src/components/session/composer.tsx b/platform/web/src/components/session/composer.tsx index a0f9c7287..62594feed 100644 --- a/platform/web/src/components/session/composer.tsx +++ b/platform/web/src/components/session/composer.tsx @@ -1,5 +1,8 @@ -import { useState, type KeyboardEvent, type ReactNode } from "react"; -import { ArrowUp, LoaderCircle, Square } from "lucide-react"; +import { useRef, useState, type KeyboardEvent, type ReactNode } from "react"; +import { ArrowUp, LoaderCircle, Mic, Square, X } from "lucide-react"; +import { Link } from "react-router-dom"; +import { useDictation } from "@/lib/use-dictation"; +import { isDemoDictation } from "@/lib/audio-capture"; import { Button } from "@/components/ui/button"; import { readSessionDraft, writeSessionDraft } from "@/lib/sessions/draft"; @@ -20,6 +23,7 @@ const steerKeyLabel = isMac ? "⌘↵" : "Ctrl+↵"; /// the composer read-only for transcript inspection. export function SessionComposer({ draftKey, + dictation, runActive, canSteer, canStop = false, @@ -33,6 +37,7 @@ export function SessionComposer({ }: { /// Stable universe + session storage key; also used as the React key. draftKey: string; + dictation?: { universeId: string; disabledReason?: string; settingsHref?: string }; /// A run is running, cancelling, or queued: Enter queues, ⌘/Ctrl+Enter /// steers. runActive: boolean; @@ -53,12 +58,25 @@ export function SessionComposer({ onStop: () => void; }) { const [text, setText] = useState(() => readSessionDraft(draftKey)); + const textRef = useRef(text); + const textarea = useRef(null); + const [dictationNotice, setDictationNotice] = useState(); const updateText = (value: string) => { + textRef.current = value; + setDictationNotice(undefined); setText(value); // Write on input rather than on unmount, so navigation cannot lose edits. writeSessionDraft(draftKey, value); }; + const voice = useDictation(dictation?.universeId, Boolean(dictation && !dictation.disabledReason && !disabled), (transcript) => { + const current = textRef.current; + updateText(`${current}${current && !/\s$/.test(current) ? " " : ""}${transcript}`); + setDictationNotice("Dictation added. Review and edit before sending."); + textarea.current?.focus(); + }); + const voiceBusy = ["requesting", "recording", "transcribing"].includes(voice.phase); + const submit = (steer: boolean) => { const trimmed = text.trim(); if (!trimmed || disabled) { @@ -68,6 +86,7 @@ export function SessionComposer({ // (the text stays in the box) rather than silently queued — the // difference matters to the reader. const mode: ComposerMode | null = !runActive ? null : steer ? "steer" : "queue"; + if (mode !== "steer" || canSteer) voice.cancel(); onSend(trimmed, mode); if (mode !== "steer" || canSteer) { updateText(""); @@ -100,6 +119,7 @@ export function SessionComposer({ )}