diff --git a/AGENTS.md b/AGENTS.md index 8a839f7f3..dc38d96c8 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -61,6 +61,13 @@ boundaries. `npm run check` is the full TypeScript gate; its `check:generated` step fails while regenerated files are uncommitted, so run the remaining steps directly until they are committed. +Before opening or updating a PR with Rust changes, run the exact CI Clippy +command, including test targets and treating warnings as errors: + +```bash +cargo clippy --workspace --all-targets --locked -- -D warnings +``` + Testing rules: - Unit tests live beside the code in `mod tests`; use integration tests for diff --git a/clients/typescript/schema/api.schema.json b/clients/typescript/schema/api.schema.json index 59004bb8f..f1e9107e8 100644 --- a/clients/typescript/schema/api.schema.json +++ b/clients/typescript/schema/api.schema.json @@ -81,6 +81,17 @@ }, "message": { "type": "string" + }, + "modelDefaultSlot": { + "anyOf": [ + { + "$ref": "#/definitions/ModelDefaultSlot" + }, + { + "type": "null" + } + ], + "description": "Present for model_default_unset; clients need not parse the message." } }, "required": [ @@ -96,12 +107,7 @@ "invalid_request", "not_found", "conflict", - "unsupported_audio_mime", "audio_blob_too_large", - "audio_duration_too_long", - "transcoder_unavailable", - "transcode_failure", - "transcription_failure", "internal" ], "type": "string" @@ -121,6 +127,11 @@ "description": "The authenticated caller lacks permission for this operation or target.", "type": "string" }, + { + "const": "model_default_unset", + "description": "No model was supplied and this universe has no default for the requested use.", + "type": "string" + }, { "const": "session_bootstrap_failed", "description": "The session's agent workflow exists but failed during bootstrap\n(rehydration) and cannot serve runs. Distinct from `NotFound` (no\nworkflow) so clients/bridges treat it as a session recovery problem\nrather than an ordinary \"answer this message\" failure.", @@ -1719,6 +1730,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfModelDefaultsResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/definitions/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/definitions/ModelDefaultsResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfModelListResponse": { "properties": { "notifications": { @@ -2178,6 +2206,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfTranscriptionResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/definitions/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/definitions/TranscriptionResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfVfsSnapshotCommitResponse": { "properties": { "notifications": { @@ -4542,6 +4587,13 @@ "string", "null" ] + }, + "textRef": { + "description": "Optional prepared UTF-8 text; the source attachment remains in blobRef.", + "type": [ + "string", + "null" + ] } }, "required": [ @@ -7979,7 +8031,7 @@ "type": "string" }, "groups": { - "description": "The method groups the key may call; absent grants every group its\nscope allows. Keys never change: to change what a key may do, revoke\nit and mint another.", + "description": "The method groups the key may call; absent grants every group its\nscope allows. To change what a key may do, revoke\nit and mint another.", "items": { "$ref": "#/definitions/MethodGroup" }, @@ -8004,7 +8056,7 @@ "type": "object" }, "DeploymentApiKeyCreateResponse": { - "description": "A newly minted key. `secret` is returned only by create and cannot be\nrecovered later. Its custom `Debug` implementation redacts the DTO before\nJSON-RPC serialization; the serialized response payload remains sensitive\nand must not be logged.", + "description": "A newly minted or rotated key. `secret` is returned only by create or rotate and cannot be\nrecovered later. Its custom `Debug` implementation redacts the DTO before\nJSON-RPC serialization; the serialized response payload remains sensitive\nand must not be logged.", "properties": { "apiKey": { "$ref": "#/definitions/DeploymentApiKeyView" @@ -8071,6 +8123,18 @@ ], "type": "object" }, + "DeploymentApiKeyRotateParams": { + "additionalProperties": false, + "properties": { + "keyPrefix": { + "type": "string" + } + }, + "required": [ + "keyPrefix" + ], + "type": "object" + }, "DeploymentApiKeyView": { "description": "Non-secret key metadata: what the key reaches and may call, and who\nminted it.", "properties": { @@ -10592,13 +10656,7 @@ "InputAdmissionFailureKind": { "enum": [ "unsupportedMedia", - "unsupportedAudioMime", "blobMissing", - "blobTooLarge", - "audioDurationTooLong", - "transcoderUnavailable", - "transcodeFailure", - "transcriptionFailure", "admissionRejected" ], "type": "string" @@ -10629,6 +10687,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -10655,6 +10720,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -11701,6 +11773,7 @@ "enum": [ "vfs", "profiles", + "transcriptions", "models", "mcp", "bots", @@ -11823,6 +11896,95 @@ ], "type": "object" }, + "ModelDefaultSlot": { + "description": "A universe's model selection for a particular use. Protocol and purpose\nare separate: several purposes may use the same provider API.", + "enum": [ + "agentRun", + "speechToText" + ], + "type": "string" + }, + "ModelDefaults": { + "additionalProperties": false, + "description": "Persisted universe defaults. Revision zero means no update has been made.\nClearing a slot still advances the revision, so setup cannot undo a clear.", + "properties": { + "agentRun": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "revision": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "speechToText": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + } + }, + "required": [ + "revision" + ], + "type": "object" + }, + "ModelDefaultsPutParams": { + "additionalProperties": false, + "properties": { + "expectedRevision": { + "description": "Revision returned by read/put; zero for a universe with no updates.", + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ], + "description": "Complete selection, or explicit null to clear this slot. Required." + }, + "slot": { + "$ref": "#/definitions/ModelDefaultSlot" + } + }, + "required": [ + "slot", + "model", + "expectedRevision" + ], + "type": "object" + }, + "ModelDefaultsReadParams": { + "additionalProperties": false, + "type": "object" + }, + "ModelDefaultsResponse": { + "properties": { + "defaults": { + "$ref": "#/definitions/ModelDefaults" + } + }, + "required": [ + "defaults" + ], + "type": "object" + }, "ModelEndpointConfig": { "properties": { "apiKinds": { @@ -11851,7 +12013,7 @@ "description": "Direct provider model discovery. Results may be served from a brief\nprocess-local cache; clients refresh by calling this method.", "properties": { "selectableOnly": { - "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nremoves OpenAI model-id families that are clearly not text-generation\nroutes (embeddings, moderation, image/video, speech, and realtime).\nIt is an ID policy, not a provider capability claim.", + "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nkeeps supported file-transcription routes and filters clearly unrelated\nOpenAI families from agent suggestions (embeddings, moderation, image/video,\nspeech synthesis, and realtime). Agent suggestions also have an age limit.\nClients select routes by API kind for their intended use. This is an ID\npolicy, not a provider capability claim.", "type": "boolean" } }, @@ -13452,7 +13614,7 @@ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -16980,6 +17142,193 @@ ], "type": "object" }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionCancelParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/definitions/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionReadParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionResponse": { + "properties": { + "transcription": { + "$ref": "#/definitions/TranscriptionView" + } + }, + "required": [ + "transcription" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/definitions/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "status": { + "$ref": "#/definitions/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, "UniverseAction": { "description": "What a public method does, independent of its RPC spelling. Requests are\ngated by their key's groups; this classifies what the runtime's own work\nmay do and which role a gate built on the contract should require.", "oneOf": [ diff --git a/clients/typescript/schema/workflow.json b/clients/typescript/schema/workflow.json index 47ea11762..c9baa301b 100644 --- a/clients/typescript/schema/workflow.json +++ b/clients/typescript/schema/workflow.json @@ -65,7 +65,10 @@ "ChannelDeliveryCommand", "ChannelDeliveryResult", "PrepareChannelMediaInput", - "PrepareChannelMediaResult" + "PrepareChannelMediaResult", + "TranscriptionWorkflowArgs", + "TranscriptionSnapshot", + "TranscriptionActivityResult" ], "signals": { "deliverEmission": "deliver_emission" diff --git a/clients/typescript/schema/workflow.schema.json b/clients/typescript/schema/workflow.schema.json index 624554450..214d658c8 100644 --- a/clients/typescript/schema/workflow.schema.json +++ b/clients/typescript/schema/workflow.schema.json @@ -1,6 +1,79 @@ { "$schema": "http://json-schema.org/draft-07/schema#", "definitions": { + "Attribution": { + "description": "Who created a resource or authored bytes. An actor is whatever a key\nallowed to assert one said; core compares it and never resolves it.", + "oneOf": [ + { + "description": "The actor a key asserted for its request.", + "properties": { + "id": { + "type": "string" + }, + "kind": { + "const": "actor", + "type": "string" + } + }, + "required": [ + "kind", + "id" + ], + "type": "object" + }, + { + "description": "A key acting for itself, named by its display prefix.", + "properties": { + "kind": { + "const": "key", + "type": "string" + }, + "prefix": { + "type": "string" + } + }, + "required": [ + "kind", + "prefix" + ], + "type": "object" + }, + { + "description": "An unauthenticated local development request, or an in-process call.", + "properties": { + "kind": { + "const": "local", + "type": "string" + } + }, + "required": [ + "kind" + ], + "type": "object" + }, + { + "description": "The runtime's own work: a bot, a delegated session, a registration,\nor host administration through the server CLI.", + "properties": { + "cause": { + "type": "string" + }, + "component": { + "type": "string" + }, + "kind": { + "const": "internal", + "type": "string" + } + }, + "required": [ + "kind", + "component", + "cause" + ], + "type": "object" + } + ] + }, "BotId": { "type": "string" }, @@ -614,6 +687,25 @@ ], "type": "object" }, + "ModelConfig": { + "properties": { + "apiKind": { + "type": "string" + }, + "model": { + "type": "string" + }, + "providerId": { + "type": "string" + } + }, + "required": [ + "providerId", + "apiKind", + "model" + ], + "type": "object" + }, "PrepareChannelMediaInput": { "description": "`prepare_channel_media`: the connector downloads the provider file and\nstores it in the universe's CAS.", "properties": { @@ -761,6 +853,243 @@ "ToolCallId": { "type": "string" }, + "TranscriptionActivityResult": { + "oneOf": [ + { + "properties": { + "kind": { + "const": "succeeded", + "type": "string" + }, + "transcript_ref": { + "type": "string" + } + }, + "required": [ + "kind", + "transcript_ref" + ], + "type": "object" + }, + { + "properties": { + "failure": { + "$ref": "#/definitions/TranscriptionFailure" + }, + "kind": { + "const": "failed", + "type": "string" + } + }, + "required": [ + "kind", + "failure" + ], + "type": "object" + } + ] + }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/definitions/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionSnapshot": { + "properties": { + "request": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "view": { + "$ref": "#/definitions/TranscriptionView" + } + }, + "required": [ + "request", + "view" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/definitions/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "status": { + "$ref": "#/definitions/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, + "TranscriptionWorkflowArgs": { + "properties": { + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "request": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "transcriptionId": { + "type": "string" + }, + "universeId": { + "format": "uuid", + "type": "string" + } + }, + "required": [ + "universeId", + "transcriptionId", + "createdBy", + "request", + "model", + "createdAtMs" + ], + "type": "object" + }, "WorkflowToolId": { "type": "string" }, diff --git a/clients/typescript/src/errors.ts b/clients/typescript/src/errors.ts index eb3105f33..a00766d1e 100644 --- a/clients/typescript/src/errors.ts +++ b/clients/typescript/src/errors.ts @@ -11,6 +11,7 @@ export type LightspeedRpcErrorKind = | "conflict" | "rejected" | "environment_not_ready" + | "model_default_unset" | "internal" | "unknown"; @@ -34,6 +35,8 @@ export function lightspeedRpcErrorKind(code: number): LightspeedRpcErrorKind { return "environment_not_ready"; case -32603: return "internal"; + case -32014: + return "model_default_unset"; default: return "unknown"; } diff --git a/clients/typescript/src/generated/methods.ts b/clients/typescript/src/generated/methods.ts index f0837d6f7..93ca72acf 100644 --- a/clients/typescript/src/generated/methods.ts +++ b/clients/typescript/src/generated/methods.ts @@ -54,6 +54,11 @@ export const METHODS = [ "environments/registration-keys/read", "environments/registration-keys/list", "environments/registration-keys/revoke", + "transcriptions/start", + "transcriptions/read", + "transcriptions/cancel", + "models/defaults/read", + "models/defaults/put", "models/list", "profiles/create", "profiles/read", @@ -127,6 +132,7 @@ export const METHODS = [ "deployment/universes/delete", "deployment/api-keys/create", "deployment/api-keys/list", + "deployment/api-keys/rotate", "deployment/api-keys/revoke", "deployment/environment-providers/put", "deployment/environment-providers/list", @@ -179,7 +185,7 @@ export const METHOD_INFO = { scope: "universe", access: {"action":"control_session","kind":"universe"}, summary: "Replace session configuration", - description: "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + description: "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", }, "session/rename": { scope: "universe", @@ -227,7 +233,7 @@ export const METHOD_INFO = { scope: "universe", access: {"action":"control_session","kind":"universe"}, summary: "Append keyed session context", - description: "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries.", + description: "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries.", }, "session/context/remove": { scope: "universe", @@ -433,6 +439,36 @@ export const METHOD_INFO = { summary: "Revoke an environment registration key", description: "Stops the key from admitting new daemon identities; already registered daemons keep reconnecting. With closeEnvironments, also closes every non-closed environment the key admitted. Idempotent.", }, + "transcriptions/start": { + scope: "universe", + access: {"action":"use_resource","kind":"universe"}, + summary: "Start transcription", + description: "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created.", + }, + "transcriptions/read": { + scope: "universe", + access: {"action":"read","kind":"universe"}, + summary: "Read transcription", + description: "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire.", + }, + "transcriptions/cancel": { + scope: "universe", + access: {"action":"use_resource","kind":"universe"}, + summary: "Cancel transcription", + description: "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority.", + }, + "models/defaults/read": { + scope: "universe", + access: {"action":"read","kind":"universe"}, + summary: "Read universe model defaults", + description: "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers.", + }, + "models/defaults/put": { + scope: "universe", + access: {"action":"configure_resource","kind":"universe"}, + summary: "Set a universe model default", + description: "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery.", + }, "models/list": { scope: "universe", access: {"action":"read","kind":"universe"}, @@ -863,7 +899,7 @@ export const METHOD_INFO = { scope: "deployment", access: {"kind":"deployment"}, summary: "Create a scoped API key", - description: "Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Keys are immutable: revoke and mint to change what one may do.", + description: "Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Key authority is immutable: revoke and mint to change what one may do.", }, "deployment/api-keys/list": { scope: "deployment", @@ -871,6 +907,12 @@ export const METHOD_INFO = { summary: "List scoped API keys", description: "Returns non-secret key metadata, all keys or those of one scope, including groups, revocation and last-use timestamps. Plaintext secrets are never stored or returned.", }, + "deployment/api-keys/rotate": { + scope: "deployment", + access: {"kind":"deployment"}, + summary: "Rotate a scoped API key", + description: "Atomically replaces an active key secret and display prefix, immediately rejecting the old secret on subsequent requests. Preserves scope, groups, actor authority, name, creator and creation time; clears last use. Returns the new secret once. Unknown or revoked prefixes are not found. Already admitted work continues.", + }, "deployment/api-keys/revoke": { scope: "deployment", access: {"kind":"deployment"}, @@ -997,7 +1039,7 @@ export interface MethodMap { /** * Replace session configuration * - * Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op. + * Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op. */ "session/config/put": { params: Api.SessionConfigPutParams; @@ -1069,7 +1111,7 @@ export interface MethodMap { /** * Append keyed session context * - * Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries. + * Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries. */ "session/context/append": { params: Api.ContextAppendParams; @@ -1381,6 +1423,51 @@ export interface MethodMap { params: Api.EnvironmentRegistrationKeyRevokeParams; result: Api.AgentApiOutcomeOfEnvironmentRegistrationKeyRevokeResponse; }; + /** + * Start transcription + * + * Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created. + */ + "transcriptions/start": { + params: Api.TranscriptionStartParams; + result: Api.AgentApiOutcomeOfTranscriptionResponse; + }; + /** + * Read transcription + * + * Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire. + */ + "transcriptions/read": { + params: Api.TranscriptionReadParams; + result: Api.AgentApiOutcomeOfTranscriptionResponse; + }; + /** + * Cancel transcription + * + * Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority. + */ + "transcriptions/cancel": { + params: Api.TranscriptionCancelParams; + result: Api.AgentApiOutcomeOfTranscriptionResponse; + }; + /** + * Read universe model defaults + * + * Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers. + */ + "models/defaults/read": { + params: Api.ModelDefaultsReadParams; + result: Api.AgentApiOutcomeOfModelDefaultsResponse; + }; + /** + * Set a universe model default + * + * Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery. + */ + "models/defaults/put": { + params: Api.ModelDefaultsPutParams; + result: Api.AgentApiOutcomeOfModelDefaultsResponse; + }; /** * Discover available models * @@ -2023,7 +2110,7 @@ export interface MethodMap { /** * Create a scoped API key * - * Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Keys are immutable: revoke and mint to change what one may do. + * Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Key authority is immutable: revoke and mint to change what one may do. */ "deployment/api-keys/create": { params: Api.DeploymentApiKeyCreateParams; @@ -2038,6 +2125,15 @@ export interface MethodMap { params: Api.DeploymentApiKeyListParams; result: Api.AgentApiOutcomeOfDeploymentApiKeyListResponse; }; + /** + * Rotate a scoped API key + * + * Atomically replaces an active key secret and display prefix, immediately rejecting the old secret on subsequent requests. Preserves scope, groups, actor authority, name, creator and creation time; clears last use. Returns the new secret once. Unknown or revoked prefixes are not found. Already admitted work continues. + */ + "deployment/api-keys/rotate": { + params: Api.DeploymentApiKeyRotateParams; + result: Api.AgentApiOutcomeOfDeploymentApiKeyCreateResponse; + }; /** * Revoke a scoped API key * @@ -2180,7 +2276,7 @@ export const rpc = { /** * Replace session configuration * - * Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op. + * Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op. */ sessionConfigPut(client: RpcCaller, params: Api.SessionConfigPutParams): Promise { return client.call("session/config/put", params); @@ -2244,7 +2340,7 @@ export const rpc = { /** * Append keyed session context * - * Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries. + * Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries. */ sessionContextAppend(client: RpcCaller, params: Api.ContextAppendParams): Promise { return client.call("session/context/append", params); @@ -2521,6 +2617,46 @@ export const rpc = { environmentsRegistrationKeysRevoke(client: RpcCaller, params: Api.EnvironmentRegistrationKeyRevokeParams): Promise { return client.call("environments/registration-keys/revoke", params); }, + /** + * Start transcription + * + * Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created. + */ + transcriptionsStart(client: RpcCaller, params: Api.TranscriptionStartParams): Promise { + return client.call("transcriptions/start", params); + }, + /** + * Read transcription + * + * Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire. + */ + transcriptionsRead(client: RpcCaller, params: Api.TranscriptionReadParams): Promise { + return client.call("transcriptions/read", params); + }, + /** + * Cancel transcription + * + * Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority. + */ + transcriptionsCancel(client: RpcCaller, params: Api.TranscriptionCancelParams): Promise { + return client.call("transcriptions/cancel", params); + }, + /** + * Read universe model defaults + * + * Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers. + */ + modelsDefaultsRead(client: RpcCaller, params: Api.ModelDefaultsReadParams): Promise { + return client.call("models/defaults/read", params); + }, + /** + * Set a universe model default + * + * Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery. + */ + modelsDefaultsPut(client: RpcCaller, params: Api.ModelDefaultsPutParams): Promise { + return client.call("models/defaults/put", params); + }, /** * Discover available models * @@ -3092,7 +3228,7 @@ export const rpc = { /** * Create a scoped API key * - * Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Keys are immutable: revoke and mint to change what one may do. + * Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Key authority is immutable: revoke and mint to change what one may do. */ deploymentApiKeysCreate(client: RpcCaller, params: Api.DeploymentApiKeyCreateParams): Promise { return client.call("deployment/api-keys/create", params); @@ -3105,6 +3241,14 @@ export const rpc = { deploymentApiKeysList(client: RpcCaller, params: Api.DeploymentApiKeyListParams): Promise { return client.call("deployment/api-keys/list", params); }, + /** + * Rotate a scoped API key + * + * Atomically replaces an active key secret and display prefix, immediately rejecting the old secret on subsequent requests. Preserves scope, groups, actor authority, name, creator and creation time; clears last use. Returns the new secret once. Unknown or revoked prefixes are not found. Already admitted work continues. + */ + deploymentApiKeysRotate(client: RpcCaller, params: Api.DeploymentApiKeyRotateParams): Promise { + return client.call("deployment/api-keys/rotate", params); + }, /** * Revoke a scoped API key * diff --git a/clients/typescript/src/generated/types.ts b/clients/typescript/src/generated/types.ts index 0a092efa7..148a0bfe9 100644 --- a/clients/typescript/src/generated/types.ts +++ b/clients/typescript/src/generated/types.ts @@ -80,24 +80,22 @@ export type ToolParallelismView = "exclusive" | "parallelSafe"; * via the `definition` "AgentApiErrorKind". */ export type AgentApiErrorKind = - | ( - | "invalid_request" - | "not_found" - | "conflict" - | "unsupported_audio_mime" - | "audio_blob_too_large" - | "audio_duration_too_long" - | "transcoder_unavailable" - | "transcode_failure" - | "transcription_failure" - | "internal" - ) + | ("invalid_request" | "not_found" | "conflict" | "audio_blob_too_large" | "internal") | "rejected" | "unauthenticated" | "forbidden" + | "model_default_unset" | "session_bootstrap_failed" | "environment_not_ready" | "response_too_large"; +/** + * A universe's model selection for a particular use. Protocol and purpose + * are separate: several purposes may use the same provider API. + * + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaultSlot". + */ +export type ModelDefaultSlot = "agentRun" | "speechToText"; /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "AgentNotification". @@ -840,6 +838,11 @@ export type InputItem = * This metadata is not an authorization identity or model input text. */ origin?: string | null; + /** + * Optional source blob in this universe, retained with the session. + * Provenance is metadata, not model input or an authorization identity. + */ + provenanceRef?: string | null; text: string; type: "text"; } @@ -852,6 +855,11 @@ export type InputItem = * This metadata is not an authorization identity or model input text. */ origin?: string | null; + /** + * Optional source blob in this universe, retained with the session. + * Provenance is metadata, not model input or an authorization identity. + */ + provenanceRef?: string | null; type: "textRef"; } | { @@ -1278,16 +1286,7 @@ export type ChannelPairedVia = "open" | "code"; * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "InputAdmissionFailureKind". */ -export type InputAdmissionFailureKind = - | "unsupportedMedia" - | "unsupportedAudioMime" - | "blobMissing" - | "blobTooLarge" - | "audioDurationTooLong" - | "transcoderUnavailable" - | "transcodeFailure" - | "transcriptionFailure" - | "admissionRejected"; +export type InputAdmissionFailureKind = "unsupportedMedia" | "blobMissing" | "admissionRejected"; /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "ContextAppendStatus". @@ -1310,6 +1309,7 @@ export type MethodGroup = | ( | "vfs" | "profiles" + | "transcriptions" | "models" | "mcp" | "bots" @@ -1611,6 +1611,18 @@ export type SkillCatalogSource = environmentId: string; type: "environment"; }; +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionFailureKind". + */ +export type TranscriptionFailureKind = + "invalidAudio" | "configuration" | "provider" | "timeout" | "internal"; +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionStatus". + */ +export type TranscriptionStatus = + "pending" | "running" | "succeeded" | "failed" | "cancelled" | "expired"; /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "AuthProviderConfigInput". @@ -1865,6 +1877,10 @@ export interface ToolView { export interface AgentApiError { kind: AgentApiErrorKind; message: string; + /** + * Present for model_default_unset; clients need not parse the message. + */ + modelDefaultSlot?: ModelDefaultSlot | null; } /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema @@ -2154,7 +2170,9 @@ export interface SessionConfig { generation?: GenerationConfig | null; limits?: LimitsConfig | null; /** - * Absent on input means the deployment default model. Documents read + * At creation, omission uses the profile model or universe agentRun + * default. On configuration replacement or profile application to an + * existing session, omission preserves its current model. Documents read * back from a session always carry the model. Provider identity and API * kind are fixed for the session lifetime; the model name may change. */ @@ -3560,6 +3578,10 @@ export interface BotEventMedia { kind: BotEventMediaKind; mime: string; name?: string | null; + /** + * Optional prepared UTF-8 text; the source attachment remains in blobRef. + */ + textRef?: string | null; } /** * The routed session an event was admitted to. @@ -4357,7 +4379,7 @@ export interface AgentApiOutcomeOfDeploymentApiKeyCreateResponse { result: DeploymentApiKeyCreateResponse; } /** - * A newly minted key. `secret` is returned only by create and cannot be + * A newly minted or rotated key. `secret` is returned only by create or rotate and cannot be * recovered later. Its custom `Debug` implementation redacts the DTO before * JSON-RPC serialization; the serialized response payload remains sensitive * and must not be logged. @@ -5496,6 +5518,33 @@ export interface McpToolAnnotationsView { openWorldHint?: boolean | null; readOnlyHint?: boolean | null; } +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "AgentApiOutcomeOfModelDefaultsResponse". + */ +export interface AgentApiOutcomeOfModelDefaultsResponse { + notifications?: AgentNotification[]; + result: ModelDefaultsResponse; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaultsResponse". + */ +export interface ModelDefaultsResponse { + defaults: ModelDefaults; +} +/** + * Persisted universe defaults. Revision zero means no update has been made. + * Clearing a slot still advances the revision, so setup cannot undo a clear. + * + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaults". + */ +export interface ModelDefaults { + agentRun?: ModelConfig | null; + revision: number; + speechToText?: ModelConfig | null; +} /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "AgentApiOutcomeOfModelListResponse". @@ -6144,6 +6193,59 @@ export interface SkillLocationView { skillDirPath: string; skillDocPath: string; } +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "AgentApiOutcomeOfTranscriptionResponse". + */ +export interface AgentApiOutcomeOfTranscriptionResponse { + notifications?: AgentNotification[]; + result: TranscriptionResponse; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionResponse". + */ +export interface TranscriptionResponse { + transcription: TranscriptionView; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionView". + */ +export interface TranscriptionView { + audio: TranscriptionAudio; + createdAtMs: number; + createdBy: Attribution; + failure?: TranscriptionFailure | null; + model: ModelConfig; + status: TranscriptionStatus; + text?: string | null; + /** + * Plain UTF-8 transcript blob, usable as ordinary textRef input. + * Unsubmitted content can be swept after the ordinary CAS grace period. + */ + transcriptRef?: string | null; + transcriptionId: string; +} +/** + * Immutable audio input in this universe's content store. + * + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionAudio". + */ +export interface TranscriptionAudio { + blobRef: string; + mime: string; + name: string; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionFailure". + */ +export interface TranscriptionFailure { + kind: TranscriptionFailureKind; + message: string; +} /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "AgentApiOutcomeOfVfsSnapshotCommitResponse". @@ -6971,7 +7073,7 @@ export interface DeploymentApiKeyCreateParams { displayName: string; /** * The method groups the key may call; absent grants every group its - * scope allows. Keys never change: to change what a key may do, revoke + * scope allows. To change what a key may do, revoke * it and mint another. */ groups?: MethodGroup[] | null; @@ -6998,6 +7100,13 @@ export interface DeploymentApiKeyListParams { export interface DeploymentApiKeyRevokeParams { keyPrefix: string; } +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "DeploymentApiKeyRotateParams". + */ +export interface DeploymentApiKeyRotateParams { + keyPrefix: string; +} /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "DeploymentChannelAccountListParams". @@ -7556,6 +7665,26 @@ export interface McpServerReadParams { export interface McpServerToolsDiscoverParams { serverId: string; } +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaultsPutParams". + */ +export interface ModelDefaultsPutParams { + /** + * Revision returned by read/put; zero for a universe with no updates. + */ + expectedRevision: number; + /** + * Complete selection, or explicit null to clear this slot. Required. + */ + model: ModelConfig | null; + slot: ModelDefaultSlot; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "ModelDefaultsReadParams". + */ +export interface ModelDefaultsReadParams {} /** * Direct provider model discovery. Results may be served from a brief * process-local cache; clients refresh by calling this method. @@ -7566,9 +7695,11 @@ export interface McpServerToolsDiscoverParams { export interface ModelListParams { /** * Apply Lightspeed's small, conservative selectable-model policy. It - * removes OpenAI model-id families that are clearly not text-generation - * routes (embeddings, moderation, image/video, speech, and realtime). - * It is an ID policy, not a provider capability claim. + * keeps supported file-transcription routes and filters clearly unrelated + * OpenAI families from agent suggestions (embeddings, moderation, image/video, + * speech synthesis, and realtime). Agent suggestions also have an age limit. + * Clients select routes by API kind for their intended use. This is an ID + * policy, not a provider capability claim. */ selectableOnly?: boolean; } @@ -7955,6 +8086,36 @@ export interface SessionStartParams { export interface SkillListParams { sessionId: string; } +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionCancelParams". + */ +export interface TranscriptionCancelParams { + transcriptionId: string; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionReadParams". + */ +export interface TranscriptionReadParams { + transcriptionId: string; +} +/** + * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema + * via the `definition` "TranscriptionStartParams". + */ +export interface TranscriptionStartParams { + audio: TranscriptionAudio; + /** + * Scoped to the requester. Matching retries rejoin the original job, + * including after defaults change; changed requests conflict. Identity is + * retained for the Temporal namespace's workflow-history retention period. + */ + idempotencyKey: string; + language?: string | null; + model?: ModelConfig | null; + prompt?: string | null; +} /** * This interface was referenced by `LightspeedAgentAPI`'s JSON-Schema * via the `definition` "VfsSnapshotCommitParams". diff --git a/clients/typescript/src/generated/workflow-manifest.ts b/clients/typescript/src/generated/workflow-manifest.ts index 5d10a5448..b0c1da9ee 100644 --- a/clients/typescript/src/generated/workflow-manifest.ts +++ b/clients/typescript/src/generated/workflow-manifest.ts @@ -70,7 +70,10 @@ export const WORKFLOW_CONTRACT_MANIFEST = "ChannelDeliveryCommand", "ChannelDeliveryResult", "PrepareChannelMediaInput", - "PrepareChannelMediaResult" + "PrepareChannelMediaResult", + "TranscriptionWorkflowArgs", + "TranscriptionSnapshot", + "TranscriptionActivityResult" ], "signals": { "deliverEmission": "deliver_emission" diff --git a/clients/typescript/src/generated/workflow-types.ts b/clients/typescript/src/generated/workflow-types.ts index 9d847e8c4..2f2607331 100644 --- a/clients/typescript/src/generated/workflow-types.ts +++ b/clients/typescript/src/generated/workflow-types.ts @@ -3,6 +3,30 @@ * Do not edit by hand. */ +/** + * Who created a resource or authored bytes. An actor is whatever a key + * allowed to assert one said; core compares it and never resolves it. + * + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "Attribution". + */ +export type Attribution = + | { + id: string; + kind: "actor"; + } + | { + kind: "key"; + prefix: string; + } + | { + kind: "local"; + } + | { + cause: string; + component: string; + kind: "internal"; + }; /** * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema * via the `definition` "BotId". @@ -197,6 +221,31 @@ export type EmissionProducer = universe_id: string; workflow_id: string; }; +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionActivityResult". + */ +export type TranscriptionActivityResult = + | { + kind: "succeeded"; + transcript_ref: string; + } + | { + failure: TranscriptionFailure; + kind: "failed"; + }; +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionFailureKind". + */ +export type TranscriptionFailureKind = + "invalidAudio" | "configuration" | "provider" | "timeout" | "internal"; +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionStatus". + */ +export type TranscriptionStatus = + "pending" | "running" | "succeeded" | "failed" | "cancelled" | "expired"; /** * Envelope and start-on-call types of the fixed deliver_emission transport between sessions and receiver workflows. @@ -413,6 +462,15 @@ export interface EmissionEnvelope { export interface MaintainChannelTypingInput { route: ChannelRoute; } +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "ModelConfig". + */ +export interface ModelConfig { + apiKind: string; + model: string; + providerId: string; +} /** * `prepare_channel_media`: the connector downloads the provider file and * stores it in the universe's CAS. @@ -444,6 +502,80 @@ export interface PreparedMediaItem { mime: string; name?: string | null; } +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionFailure". + */ +export interface TranscriptionFailure { + kind: TranscriptionFailureKind; + message: string; +} +/** + * Immutable audio input in this universe's content store. + * + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionAudio". + */ +export interface TranscriptionAudio { + blobRef: string; + mime: string; + name: string; +} +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionSnapshot". + */ +export interface TranscriptionSnapshot { + request: TranscriptionStartParams; + view: TranscriptionView; +} +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionStartParams". + */ +export interface TranscriptionStartParams { + audio: TranscriptionAudio; + /** + * Scoped to the requester. Matching retries rejoin the original job, + * including after defaults change; changed requests conflict. Identity is + * retained for the Temporal namespace's workflow-history retention period. + */ + idempotencyKey: string; + language?: string | null; + model?: ModelConfig | null; + prompt?: string | null; +} +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionView". + */ +export interface TranscriptionView { + audio: TranscriptionAudio; + createdAtMs: number; + createdBy: Attribution; + failure?: TranscriptionFailure | null; + model: ModelConfig; + status: TranscriptionStatus; + text?: string | null; + /** + * Plain UTF-8 transcript blob, usable as ordinary textRef input. + * Unsubmitted content can be swept after the ordinary CAS grace period. + */ + transcriptRef?: string | null; + transcriptionId: string; +} +/** + * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema + * via the `definition` "TranscriptionWorkflowArgs". + */ +export interface TranscriptionWorkflowArgs { + createdAtMs: number; + createdBy: Attribution; + model: ModelConfig; + request: TranscriptionStartParams; + transcriptionId: string; + universeId: string; +} /** * This interface was referenced by `LightspeedWorkflowContract`'s JSON-Schema * via the `definition` "WorkflowToolRecipeV1". diff --git a/clients/typescript/test/client.test.ts b/clients/typescript/test/client.test.ts index 3337f67c9..4b40f8277 100644 --- a/clients/typescript/test/client.test.ts +++ b/clients/typescript/test/client.test.ts @@ -20,6 +20,12 @@ function decodeBody(init: RequestInit | undefined): Record { } describe("LightspeedClient", () => { + it("classifies missing model defaults and retains the affected slot", () => { + const error = new LightspeedRpcError({ code: -32014, message: "Choose a model", data: { kind: "model_default_unset", message: "Choose a model", modelDefaultSlot: "agentRun" } }); + expect(error.kind).toBe("model_default_unset"); + expect(error.data?.modelDefaultSlot).toBe("agentRun"); + }); + it("omits universe selection on connection and deployment calls", async () => { const captured: Headers[] = []; const fetchImpl = vi.fn(async (_input: RequestInfo | URL, init?: RequestInit) => { @@ -40,7 +46,7 @@ describe("LightspeedClient", () => { access: { kind: "universe", action: "control_session" }, summary: "Replace session configuration", description: - "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", }); expect(METHOD_INFO["deployment/api-keys/create"]).toMatchObject({ scope: "deployment", diff --git a/crates/api-projection/src/content.rs b/crates/api-projection/src/content.rs index 238969e55..a62abcd50 100644 --- a/crates/api-projection/src/content.rs +++ b/crates/api-projection/src/content.rs @@ -48,9 +48,6 @@ pub async fn project_content_text( Some(llm_clients::content::ANTHROPIC_THINKING_PROVIDER_KIND) => { llm_clients::content::anthropic_thinking } - Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) => { - llm_clients::content::audio_transcript - } Some(_) => { return Err(AgentApiError::invalid_request( "unsupported native content text projection", @@ -92,7 +89,7 @@ mod tests { } #[tokio::test(flavor = "current_thread")] - async fn audio_run_summary_previews_project_text_before_applying_the_limit() { + async fn run_summary_previews_preserve_unicode_at_the_limit() { let blobs = InMemoryBlobStore::new(); let projector = crate::CoreAgentProjector::new(&blobs); let boundary = format!("{}é", "🦀".repeat(127)); @@ -108,11 +105,7 @@ mod tests { true, ), ] { - let bytes = serde_json::to_vec(&AudioTranscript { - filename: "voice.ogg".to_owned(), - text, - }) - .unwrap(); + let bytes = text.into_bytes(); let reference = blobs.put_bytes(bytes.clone()).await.unwrap(); let source = engine::RunSource::Input { input: vec![engine::ContextEntryInput { @@ -121,10 +114,10 @@ mod tests { }, content: ContentRef { content_ref: reference.clone(), - media_type: Some("application/json".into()), - provider_kind: Some(AUDIO_TRANSCRIPT_PROVIDER_KIND.into()), + media_type: Some("text/plain".into()), + provider_kind: None, }, - preview: Some("short preprocessing preview".into()), + preview: None, origin: None, provenance_ref: None, token_estimate: None, @@ -158,8 +151,6 @@ mod tests { serde_json::to_vec(&json!({"type":"message","role":"assistant","content":[{"type":"output_text","text":full,"annotations":[]}]})).unwrap()), (Some(OPENAI_COMPLETIONS_MESSAGE_PROVIDER_KIND), serde_json::to_vec(&json!({"role":"assistant","content":full,"annotations":[]})).unwrap()), - (Some(AUDIO_TRANSCRIPT_PROVIDER_KIND), - serde_json::to_vec(&json!({"filename":"note.ogg","text":full})).unwrap()), ] { let content = ContentRef { content_ref: blobs.put_bytes(bytes.clone()).await.unwrap(), @@ -180,7 +171,7 @@ mod tests { assert_eq!(view.text.as_deref(), Some(full.as_str())); assert!(!view.text_truncated); assert_eq!(projector.project_input_entries(&[input]).await.unwrap(), - vec![api::InputItem::Text { origin: None, text: full.clone() }]); + vec![api::InputItem::Text { provenance_ref: None, origin: None, text: full.clone() }]); // The retained terminal output still resolves after its context is gone. let source = engine::RunSource::Input { input: Vec::new() }; diff --git a/crates/api-projection/src/lib.rs b/crates/api-projection/src/lib.rs index 4410178bb..cfb5077b4 100644 --- a/crates/api-projection/src/lib.rs +++ b/crates/api-projection/src/lib.rs @@ -611,6 +611,7 @@ impl<'a> CoreAgentProjector<'a> { // the blob as UTF-8 text would fail. if is_text_message_media_type(entry.content.media_type.as_deref()) { InputItem::Text { + provenance_ref: entry.provenance_ref.as_ref().map(ToString::to_string), origin: entry.origin.clone(), text: project_content_text(self.blobs, &entry.content) .await? @@ -628,6 +629,7 @@ impl<'a> CoreAgentProjector<'a> { } } _ => InputItem::TextRef { + provenance_ref: entry.provenance_ref.as_ref().map(ToString::to_string), origin: entry.origin.clone(), blob_ref: entry.content.content_ref.as_str().to_owned(), }, @@ -4562,14 +4564,17 @@ mod tests { fn input_text_joins_non_empty_text_items() { let text = input_text(&[ InputItem::Text { + provenance_ref: None, origin: None, text: " first ".to_owned(), }, InputItem::Text { + provenance_ref: None, origin: None, text: "".to_owned(), }, InputItem::Text { + provenance_ref: None, origin: None, text: "second".to_owned(), }, @@ -4582,6 +4587,7 @@ mod tests { #[test] fn input_text_rejects_unresolved_text_refs() { let error = input_text(&[InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: BlobRef::from_bytes(b"hello").as_str().to_owned(), }]) diff --git a/crates/api/contract/api-reference.md b/crates/api/contract/api-reference.md index ae515f8e6..2611dd036 100644 --- a/crates/api/contract/api-reference.md +++ b/crates/api/contract/api-reference.md @@ -94,7 +94,7 @@ Returns a cursor-paginated summary list ordered by most recent update, optionall **Replace session configuration** -Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op. +Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op. - Access: `{"kind":"universe","action":"control_session"}` - Group: `session` @@ -198,7 +198,7 @@ Returns chronological events. Forward (default) follows after and supports long- **Append keyed session context** -Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries. +Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries. - Access: `{"kind":"universe","action":"control_session"}` - Group: `session` @@ -649,6 +649,71 @@ Stops the key from admitting new daemon identities; already registered daemons k - Params: `EnvironmentRegistrationKeyRevokeParams` - Result: `AgentApiOutcome` +### `transcriptions/start` + +**Start transcription** + +Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created. + +- Access: `{"kind":"universe","action":"use_resource"}` +- Group: `transcriptions` +- Role: `contributor` +- Target: `none` +- Params: `TranscriptionStartParams` +- Result: `AgentApiOutcome` + +### `transcriptions/read` + +**Read transcription** + +Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire. + +- Access: `{"kind":"universe","action":"read"}` +- Group: `transcriptions` +- Role: `viewer` +- Target: `none` +- Params: `TranscriptionReadParams` +- Result: `AgentApiOutcome` + +### `transcriptions/cancel` + +**Cancel transcription** + +Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority. + +- Access: `{"kind":"universe","action":"use_resource"}` +- Group: `transcriptions` +- Role: `contributor` +- Target: `none` +- Params: `TranscriptionCancelParams` +- Result: `AgentApiOutcome` + +### `models/defaults/read` + +**Read universe model defaults** + +Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers. + +- Access: `{"kind":"universe","action":"read"}` +- Group: `models` +- Role: `viewer` +- Target: `none` +- Params: `ModelDefaultsReadParams` +- Result: `AgentApiOutcome` + +### `models/defaults/put` + +**Set a universe model default** + +Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery. + +- Access: `{"kind":"universe","action":"configure_resource"}` +- Group: `models` +- Role: `operator` +- Target: `none` +- Params: `ModelDefaultsPutParams` +- Result: `AgentApiOutcome` + ### `models/list` **Discover available models** @@ -1591,7 +1656,7 @@ Permanently terminates live session workflows, deletes external blob objects, an **Create a scoped API key** -Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Keys are immutable: revoke and mint to change what one may do. +Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Key authority is immutable: revoke and mint to change what one may do. - Access: `{"kind":"deployment"}` - Group: `deployment/api-keys` @@ -1613,6 +1678,19 @@ Returns non-secret key metadata, all keys or those of one scope, including group - Params: `DeploymentApiKeyListParams` - Result: `AgentApiOutcome` +### `deployment/api-keys/rotate` + +**Rotate a scoped API key** + +Atomically replaces an active key secret and display prefix, immediately rejecting the old secret on subsequent requests. Preserves scope, groups, actor authority, name, creator and creation time; clears last use. Returns the new secret once. Unknown or revoked prefixes are not found. Already admitted work continues. + +- Access: `{"kind":"deployment"}` +- Group: `deployment/api-keys` +- Role: `none` +- Target: `none` +- Params: `DeploymentApiKeyRotateParams` +- Result: `AgentApiOutcome` + ### `deployment/api-keys/revoke` **Revoke a scoped API key** diff --git a/crates/api/contract/api.schema.json b/crates/api/contract/api.schema.json index 59004bb8f..f1e9107e8 100644 --- a/crates/api/contract/api.schema.json +++ b/crates/api/contract/api.schema.json @@ -81,6 +81,17 @@ }, "message": { "type": "string" + }, + "modelDefaultSlot": { + "anyOf": [ + { + "$ref": "#/definitions/ModelDefaultSlot" + }, + { + "type": "null" + } + ], + "description": "Present for model_default_unset; clients need not parse the message." } }, "required": [ @@ -96,12 +107,7 @@ "invalid_request", "not_found", "conflict", - "unsupported_audio_mime", "audio_blob_too_large", - "audio_duration_too_long", - "transcoder_unavailable", - "transcode_failure", - "transcription_failure", "internal" ], "type": "string" @@ -121,6 +127,11 @@ "description": "The authenticated caller lacks permission for this operation or target.", "type": "string" }, + { + "const": "model_default_unset", + "description": "No model was supplied and this universe has no default for the requested use.", + "type": "string" + }, { "const": "session_bootstrap_failed", "description": "The session's agent workflow exists but failed during bootstrap\n(rehydration) and cannot serve runs. Distinct from `NotFound` (no\nworkflow) so clients/bridges treat it as a session recovery problem\nrather than an ordinary \"answer this message\" failure.", @@ -1719,6 +1730,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfModelDefaultsResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/definitions/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/definitions/ModelDefaultsResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfModelListResponse": { "properties": { "notifications": { @@ -2178,6 +2206,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfTranscriptionResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/definitions/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/definitions/TranscriptionResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfVfsSnapshotCommitResponse": { "properties": { "notifications": { @@ -4542,6 +4587,13 @@ "string", "null" ] + }, + "textRef": { + "description": "Optional prepared UTF-8 text; the source attachment remains in blobRef.", + "type": [ + "string", + "null" + ] } }, "required": [ @@ -7979,7 +8031,7 @@ "type": "string" }, "groups": { - "description": "The method groups the key may call; absent grants every group its\nscope allows. Keys never change: to change what a key may do, revoke\nit and mint another.", + "description": "The method groups the key may call; absent grants every group its\nscope allows. To change what a key may do, revoke\nit and mint another.", "items": { "$ref": "#/definitions/MethodGroup" }, @@ -8004,7 +8056,7 @@ "type": "object" }, "DeploymentApiKeyCreateResponse": { - "description": "A newly minted key. `secret` is returned only by create and cannot be\nrecovered later. Its custom `Debug` implementation redacts the DTO before\nJSON-RPC serialization; the serialized response payload remains sensitive\nand must not be logged.", + "description": "A newly minted or rotated key. `secret` is returned only by create or rotate and cannot be\nrecovered later. Its custom `Debug` implementation redacts the DTO before\nJSON-RPC serialization; the serialized response payload remains sensitive\nand must not be logged.", "properties": { "apiKey": { "$ref": "#/definitions/DeploymentApiKeyView" @@ -8071,6 +8123,18 @@ ], "type": "object" }, + "DeploymentApiKeyRotateParams": { + "additionalProperties": false, + "properties": { + "keyPrefix": { + "type": "string" + } + }, + "required": [ + "keyPrefix" + ], + "type": "object" + }, "DeploymentApiKeyView": { "description": "Non-secret key metadata: what the key reaches and may call, and who\nminted it.", "properties": { @@ -10592,13 +10656,7 @@ "InputAdmissionFailureKind": { "enum": [ "unsupportedMedia", - "unsupportedAudioMime", "blobMissing", - "blobTooLarge", - "audioDurationTooLong", - "transcoderUnavailable", - "transcodeFailure", - "transcriptionFailure", "admissionRejected" ], "type": "string" @@ -10629,6 +10687,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -10655,6 +10720,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -11701,6 +11773,7 @@ "enum": [ "vfs", "profiles", + "transcriptions", "models", "mcp", "bots", @@ -11823,6 +11896,95 @@ ], "type": "object" }, + "ModelDefaultSlot": { + "description": "A universe's model selection for a particular use. Protocol and purpose\nare separate: several purposes may use the same provider API.", + "enum": [ + "agentRun", + "speechToText" + ], + "type": "string" + }, + "ModelDefaults": { + "additionalProperties": false, + "description": "Persisted universe defaults. Revision zero means no update has been made.\nClearing a slot still advances the revision, so setup cannot undo a clear.", + "properties": { + "agentRun": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "revision": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "speechToText": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + } + }, + "required": [ + "revision" + ], + "type": "object" + }, + "ModelDefaultsPutParams": { + "additionalProperties": false, + "properties": { + "expectedRevision": { + "description": "Revision returned by read/put; zero for a universe with no updates.", + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ], + "description": "Complete selection, or explicit null to clear this slot. Required." + }, + "slot": { + "$ref": "#/definitions/ModelDefaultSlot" + } + }, + "required": [ + "slot", + "model", + "expectedRevision" + ], + "type": "object" + }, + "ModelDefaultsReadParams": { + "additionalProperties": false, + "type": "object" + }, + "ModelDefaultsResponse": { + "properties": { + "defaults": { + "$ref": "#/definitions/ModelDefaults" + } + }, + "required": [ + "defaults" + ], + "type": "object" + }, "ModelEndpointConfig": { "properties": { "apiKinds": { @@ -11851,7 +12013,7 @@ "description": "Direct provider model discovery. Results may be served from a brief\nprocess-local cache; clients refresh by calling this method.", "properties": { "selectableOnly": { - "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nremoves OpenAI model-id families that are clearly not text-generation\nroutes (embeddings, moderation, image/video, speech, and realtime).\nIt is an ID policy, not a provider capability claim.", + "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nkeeps supported file-transcription routes and filters clearly unrelated\nOpenAI families from agent suggestions (embeddings, moderation, image/video,\nspeech synthesis, and realtime). Agent suggestions also have an age limit.\nClients select routes by API kind for their intended use. This is an ID\npolicy, not a provider capability claim.", "type": "boolean" } }, @@ -13452,7 +13614,7 @@ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -16980,6 +17142,193 @@ ], "type": "object" }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionCancelParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/definitions/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionReadParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionResponse": { + "properties": { + "transcription": { + "$ref": "#/definitions/TranscriptionView" + } + }, + "required": [ + "transcription" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/definitions/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "status": { + "$ref": "#/definitions/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, "UniverseAction": { "description": "What a public method does, independent of its RPC spelling. Requests are\ngated by their key's groups; this classifies what the runtime's own work\nmay do and which role a gate built on the contract should require.", "oneOf": [ diff --git a/crates/api/contract/methods.json b/crates/api/contract/methods.json index 16bebb4f2..ad27618e8 100644 --- a/crates/api/contract/methods.json +++ b/crates/api/contract/methods.json @@ -154,7 +154,7 @@ "action": "control_session", "kind": "universe" }, - "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", "group": "session", "method": "session/config/put", "params": { @@ -354,7 +354,7 @@ "action": "control_session", "kind": "universe" }, - "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries.", + "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries.", "group": "session", "method": "session/context/append", "params": { @@ -1224,6 +1224,131 @@ "summary": "Revoke an environment registration key", "target": null }, + { + "access": { + "action": "use_resource", + "kind": "universe" + }, + "description": "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created.", + "group": "transcriptions", + "method": "transcriptions/start", + "params": { + "schema": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "type": "TranscriptionStartParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfTranscriptionResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "contributor", + "scope": "universe", + "summary": "Start transcription", + "target": null + }, + { + "access": { + "action": "read", + "kind": "universe" + }, + "description": "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire.", + "group": "transcriptions", + "method": "transcriptions/read", + "params": { + "schema": { + "$ref": "#/definitions/TranscriptionReadParams" + }, + "type": "TranscriptionReadParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfTranscriptionResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "viewer", + "scope": "universe", + "summary": "Read transcription", + "target": null + }, + { + "access": { + "action": "use_resource", + "kind": "universe" + }, + "description": "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority.", + "group": "transcriptions", + "method": "transcriptions/cancel", + "params": { + "schema": { + "$ref": "#/definitions/TranscriptionCancelParams" + }, + "type": "TranscriptionCancelParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfTranscriptionResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "contributor", + "scope": "universe", + "summary": "Cancel transcription", + "target": null + }, + { + "access": { + "action": "read", + "kind": "universe" + }, + "description": "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers.", + "group": "models", + "method": "models/defaults/read", + "params": { + "schema": { + "$ref": "#/definitions/ModelDefaultsReadParams" + }, + "type": "ModelDefaultsReadParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfModelDefaultsResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "viewer", + "scope": "universe", + "summary": "Read universe model defaults", + "target": null + }, + { + "access": { + "action": "configure_resource", + "kind": "universe" + }, + "description": "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery.", + "group": "models", + "method": "models/defaults/put", + "params": { + "schema": { + "$ref": "#/definitions/ModelDefaultsPutParams" + }, + "type": "ModelDefaultsPutParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfModelDefaultsResponse" + }, + "type": "AgentApiOutcome" + }, + "role": "operator", + "scope": "universe", + "summary": "Set a universe model default", + "target": null + }, { "access": { "action": "read", @@ -2995,7 +3120,7 @@ "access": { "kind": "deployment" }, - "description": "Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Keys are immutable: revoke and mint to change what one may do.", + "description": "Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Key authority is immutable: revoke and mint to change what one may do.", "group": "deployment/api-keys", "method": "deployment/api-keys/create", "params": { @@ -3039,6 +3164,30 @@ "summary": "List scoped API keys", "target": null }, + { + "access": { + "kind": "deployment" + }, + "description": "Atomically replaces an active key secret and display prefix, immediately rejecting the old secret on subsequent requests. Preserves scope, groups, actor authority, name, creator and creation time; clears last use. Returns the new secret once. Unknown or revoked prefixes are not found. Already admitted work continues.", + "group": "deployment/api-keys", + "method": "deployment/api-keys/rotate", + "params": { + "schema": { + "$ref": "#/definitions/DeploymentApiKeyRotateParams" + }, + "type": "DeploymentApiKeyRotateParams" + }, + "result": { + "schema": { + "$ref": "#/definitions/AgentApiOutcomeOfDeploymentApiKeyCreateResponse" + }, + "type": "AgentApiOutcome" + }, + "role": null, + "scope": "deployment", + "summary": "Rotate a scoped API key", + "target": null + }, { "access": { "kind": "deployment" diff --git a/crates/api/contract/openrpc.json b/crates/api/contract/openrpc.json index d900f24c6..73eae652f 100644 --- a/crates/api/contract/openrpc.json +++ b/crates/api/contract/openrpc.json @@ -81,6 +81,17 @@ }, "message": { "type": "string" + }, + "modelDefaultSlot": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelDefaultSlot" + }, + { + "type": "null" + } + ], + "description": "Present for model_default_unset; clients need not parse the message." } }, "required": [ @@ -96,12 +107,7 @@ "invalid_request", "not_found", "conflict", - "unsupported_audio_mime", "audio_blob_too_large", - "audio_duration_too_long", - "transcoder_unavailable", - "transcode_failure", - "transcription_failure", "internal" ], "type": "string" @@ -121,6 +127,11 @@ "description": "The authenticated caller lacks permission for this operation or target.", "type": "string" }, + { + "const": "model_default_unset", + "description": "No model was supplied and this universe has no default for the requested use.", + "type": "string" + }, { "const": "session_bootstrap_failed", "description": "The session's agent workflow exists but failed during bootstrap\n(rehydration) and cannot serve runs. Distinct from `NotFound` (no\nworkflow) so clients/bridges treat it as a session recovery problem\nrather than an ordinary \"answer this message\" failure.", @@ -1719,6 +1730,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfModelDefaultsResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/components/schemas/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/components/schemas/ModelDefaultsResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfModelListResponse": { "properties": { "notifications": { @@ -2178,6 +2206,23 @@ ], "type": "object" }, + "AgentApiOutcomeOfTranscriptionResponse": { + "properties": { + "notifications": { + "items": { + "$ref": "#/components/schemas/AgentNotification" + }, + "type": "array" + }, + "result": { + "$ref": "#/components/schemas/TranscriptionResponse" + } + }, + "required": [ + "result" + ], + "type": "object" + }, "AgentApiOutcomeOfVfsSnapshotCommitResponse": { "properties": { "notifications": { @@ -4542,6 +4587,13 @@ "string", "null" ] + }, + "textRef": { + "description": "Optional prepared UTF-8 text; the source attachment remains in blobRef.", + "type": [ + "string", + "null" + ] } }, "required": [ @@ -7979,7 +8031,7 @@ "type": "string" }, "groups": { - "description": "The method groups the key may call; absent grants every group its\nscope allows. Keys never change: to change what a key may do, revoke\nit and mint another.", + "description": "The method groups the key may call; absent grants every group its\nscope allows. To change what a key may do, revoke\nit and mint another.", "items": { "$ref": "#/components/schemas/MethodGroup" }, @@ -8004,7 +8056,7 @@ "type": "object" }, "DeploymentApiKeyCreateResponse": { - "description": "A newly minted key. `secret` is returned only by create and cannot be\nrecovered later. Its custom `Debug` implementation redacts the DTO before\nJSON-RPC serialization; the serialized response payload remains sensitive\nand must not be logged.", + "description": "A newly minted or rotated key. `secret` is returned only by create or rotate and cannot be\nrecovered later. Its custom `Debug` implementation redacts the DTO before\nJSON-RPC serialization; the serialized response payload remains sensitive\nand must not be logged.", "properties": { "apiKey": { "$ref": "#/components/schemas/DeploymentApiKeyView" @@ -8071,6 +8123,18 @@ ], "type": "object" }, + "DeploymentApiKeyRotateParams": { + "additionalProperties": false, + "properties": { + "keyPrefix": { + "type": "string" + } + }, + "required": [ + "keyPrefix" + ], + "type": "object" + }, "DeploymentApiKeyView": { "description": "Non-secret key metadata: what the key reaches and may call, and who\nminted it.", "properties": { @@ -10592,13 +10656,7 @@ "InputAdmissionFailureKind": { "enum": [ "unsupportedMedia", - "unsupportedAudioMime", "blobMissing", - "blobTooLarge", - "audioDurationTooLong", - "transcoderUnavailable", - "transcodeFailure", - "transcriptionFailure", "admissionRejected" ], "type": "string" @@ -10629,6 +10687,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -10655,6 +10720,13 @@ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -11701,6 +11773,7 @@ "enum": [ "vfs", "profiles", + "transcriptions", "models", "mcp", "bots", @@ -11823,6 +11896,95 @@ ], "type": "object" }, + "ModelDefaultSlot": { + "description": "A universe's model selection for a particular use. Protocol and purpose\nare separate: several purposes may use the same provider API.", + "enum": [ + "agentRun", + "speechToText" + ], + "type": "string" + }, + "ModelDefaults": { + "additionalProperties": false, + "description": "Persisted universe defaults. Revision zero means no update has been made.\nClearing a slot still advances the revision, so setup cannot undo a clear.", + "properties": { + "agentRun": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "revision": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "speechToText": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelConfig" + }, + { + "type": "null" + } + ] + } + }, + "required": [ + "revision" + ], + "type": "object" + }, + "ModelDefaultsPutParams": { + "additionalProperties": false, + "properties": { + "expectedRevision": { + "description": "Revision returned by read/put; zero for a universe with no updates.", + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "model": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelConfig" + }, + { + "type": "null" + } + ], + "description": "Complete selection, or explicit null to clear this slot. Required." + }, + "slot": { + "$ref": "#/components/schemas/ModelDefaultSlot" + } + }, + "required": [ + "slot", + "model", + "expectedRevision" + ], + "type": "object" + }, + "ModelDefaultsReadParams": { + "additionalProperties": false, + "type": "object" + }, + "ModelDefaultsResponse": { + "properties": { + "defaults": { + "$ref": "#/components/schemas/ModelDefaults" + } + }, + "required": [ + "defaults" + ], + "type": "object" + }, "ModelEndpointConfig": { "properties": { "apiKinds": { @@ -11851,7 +12013,7 @@ "description": "Direct provider model discovery. Results may be served from a brief\nprocess-local cache; clients refresh by calling this method.", "properties": { "selectableOnly": { - "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nremoves OpenAI model-id families that are clearly not text-generation\nroutes (embeddings, moderation, image/video, speech, and realtime).\nIt is an ID policy, not a provider capability claim.", + "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nkeeps supported file-transcription routes and filters clearly unrelated\nOpenAI families from agent suggestions (embeddings, moderation, image/video,\nspeech synthesis, and realtime). Agent suggestions also have an age limit.\nClients select routes by API kind for their intended use. This is an ID\npolicy, not a provider capability claim.", "type": "boolean" } }, @@ -13452,7 +13614,7 @@ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -16980,6 +17142,193 @@ ], "type": "object" }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionCancelParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/components/schemas/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionReadParams": { + "additionalProperties": false, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + }, + "TranscriptionResponse": { + "properties": { + "transcription": { + "$ref": "#/components/schemas/TranscriptionView" + } + }, + "required": [ + "transcription" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/components/schemas/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/components/schemas/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/components/schemas/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/components/schemas/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/components/schemas/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/components/schemas/ModelConfig" + }, + "status": { + "$ref": "#/components/schemas/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, "UniverseAction": { "description": "What a public method does, independent of its RPC spelling. Requests are\ngated by their key's groups; this classifies what the runtime's own work\nmay do and which role a gate built on the contract should require.", "oneOf": [ @@ -18074,7 +18423,7 @@ "x-lightspeed-target": null }, { - "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", "name": "session/config/put", "paramStructure": "by-name", "params": [ @@ -18298,7 +18647,7 @@ "x-lightspeed-target": "sessionId" }, { - "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries.", + "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries.", "name": "session/context/append", "paramStructure": "by-name", "params": [ @@ -19277,6 +19626,146 @@ "x-lightspeed-role": "operator", "x-lightspeed-target": null }, + { + "description": "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created.", + "name": "transcriptions/start", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/TranscriptionStartParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfTranscriptionResponse" + } + }, + "summary": "Start transcription", + "x-lightspeed-access": { + "action": "use_resource", + "kind": "universe" + }, + "x-lightspeed-group": "transcriptions", + "x-lightspeed-role": "contributor", + "x-lightspeed-target": null + }, + { + "description": "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire.", + "name": "transcriptions/read", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/TranscriptionReadParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfTranscriptionResponse" + } + }, + "summary": "Read transcription", + "x-lightspeed-access": { + "action": "read", + "kind": "universe" + }, + "x-lightspeed-group": "transcriptions", + "x-lightspeed-role": "viewer", + "x-lightspeed-target": null + }, + { + "description": "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority.", + "name": "transcriptions/cancel", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/TranscriptionCancelParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfTranscriptionResponse" + } + }, + "summary": "Cancel transcription", + "x-lightspeed-access": { + "action": "use_resource", + "kind": "universe" + }, + "x-lightspeed-group": "transcriptions", + "x-lightspeed-role": "contributor", + "x-lightspeed-target": null + }, + { + "description": "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers.", + "name": "models/defaults/read", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/ModelDefaultsReadParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfModelDefaultsResponse" + } + }, + "summary": "Read universe model defaults", + "x-lightspeed-access": { + "action": "read", + "kind": "universe" + }, + "x-lightspeed-group": "models", + "x-lightspeed-role": "viewer", + "x-lightspeed-target": null + }, + { + "description": "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery.", + "name": "models/defaults/put", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/ModelDefaultsPutParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfModelDefaultsResponse" + } + }, + "summary": "Set a universe model default", + "x-lightspeed-access": { + "action": "configure_resource", + "kind": "universe" + }, + "x-lightspeed-group": "models", + "x-lightspeed-role": "operator", + "x-lightspeed-target": null + }, { "description": "Queries supported providers directly, with a brief process-local burst cache, and returns best-effort selectable routes. One provider failure does not discard successful results from others.", "name": "models/list", @@ -21258,7 +21747,7 @@ "x-lightspeed-target": null }, { - "description": "Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Keys are immutable: revoke and mint to change what one may do.", + "description": "Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Key authority is immutable: revoke and mint to change what one may do.", "name": "deployment/api-keys/create", "paramStructure": "by-name", "params": [ @@ -21311,6 +21800,33 @@ "x-lightspeed-role": null, "x-lightspeed-target": null }, + { + "description": "Atomically replaces an active key secret and display prefix, immediately rejecting the old secret on subsequent requests. Preserves scope, groups, actor authority, name, creator and creation time; clears last use. Returns the new secret once. Unknown or revoked prefixes are not found. Already admitted work continues.", + "name": "deployment/api-keys/rotate", + "paramStructure": "by-name", + "params": [ + { + "name": "params", + "required": true, + "schema": { + "$ref": "#/components/schemas/DeploymentApiKeyRotateParams" + } + } + ], + "result": { + "name": "result", + "schema": { + "$ref": "#/components/schemas/AgentApiOutcomeOfDeploymentApiKeyCreateResponse" + } + }, + "summary": "Rotate a scoped API key", + "x-lightspeed-access": { + "kind": "deployment" + }, + "x-lightspeed-group": "deployment/api-keys", + "x-lightspeed-role": null, + "x-lightspeed-target": null + }, { "description": "Revokes the key with this display prefix; revoking a revoked key keeps its first revocation time. An unknown prefix is not found.", "name": "deployment/api-keys/revoke", diff --git a/crates/api/src/access.rs b/crates/api/src/access.rs index 58d6231f8..538310352 100644 --- a/crates/api/src/access.rs +++ b/crates/api/src/access.rs @@ -55,6 +55,8 @@ pub enum MethodGroup { Vfs, #[serde(rename = "profiles")] Profiles, + #[serde(rename = "transcriptions")] + Transcriptions, #[serde(rename = "models")] Models, #[serde(rename = "mcp")] @@ -89,12 +91,13 @@ pub enum MethodGroup { } impl MethodGroup { - pub const ALL: [MethodGroup; 16] = [ + pub const ALL: [MethodGroup; 17] = [ Self::Session, Self::BlobsPut, Self::Vfs, Self::Profiles, Self::Models, + Self::Transcriptions, Self::Mcp, Self::Environments, Self::Bots, @@ -116,6 +119,7 @@ impl MethodGroup { Self::Vfs => "vfs", Self::Profiles => "profiles", Self::Models => "models", + Self::Transcriptions => "transcriptions", Self::Mcp => "mcp", Self::Environments => "environments", Self::Bots => "bots", @@ -164,6 +168,8 @@ impl MethodGroup { Self::BlobsPut } else if group("session/") || group("blobs/") { Self::Session + } else if group("transcriptions/") { + Self::Transcriptions } else if group("vfs/") { Self::Vfs } else if group("profiles/") { diff --git a/crates/api/src/bots.rs b/crates/api/src/bots.rs index 6cef07f7e..1a471bdb7 100644 --- a/crates/api/src/bots.rs +++ b/crates/api/src/bots.rs @@ -699,6 +699,9 @@ pub enum BotEventMediaKind { #[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] #[serde(rename_all = "camelCase")] pub struct BotEventMedia { + /// Optional prepared UTF-8 text; the source attachment remains in blobRef. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub text_ref: Option, pub blob_ref: String, pub kind: BotEventMediaKind, pub mime: String, diff --git a/crates/api/src/constants.rs b/crates/api/src/constants.rs index 7832921f4..32d7ddaa4 100644 --- a/crates/api/src/constants.rs +++ b/crates/api/src/constants.rs @@ -60,6 +60,8 @@ pub const METHOD_ENVIRONMENTS_CREDENTIALS_UNBIND: &str = "environments/credentia // ── Universe: direct provider model discovery ─────────────────────────────── +pub const METHOD_MODELS_DEFAULTS_READ: &str = "models/defaults/read"; +pub const METHOD_MODELS_DEFAULTS_PUT: &str = "models/defaults/put"; pub const METHOD_MODELS_LIST: &str = "models/list"; pub const METHOD_PROFILES_CREATE: &str = "profiles/create"; @@ -175,3 +177,9 @@ pub const NOTIFY_SESSION_EVENT: &str = "session/event"; pub const NOTIFY_SESSION_RUNS_STARTED: &str = "session/runs/started"; pub const NOTIFY_SESSION_RUNS_COMPLETED: &str = "session/runs/completed"; pub const NOTIFY_ERROR: &str = "error"; + +// ── Transcriptions ─────────────────────────────────────────────────────────── + +pub const METHOD_TRANSCRIPTIONS_START: &str = "transcriptions/start"; +pub const METHOD_TRANSCRIPTIONS_READ: &str = "transcriptions/read"; +pub const METHOD_TRANSCRIPTIONS_CANCEL: &str = "transcriptions/cancel"; diff --git a/crates/api/src/deployment.rs b/crates/api/src/deployment.rs index 6409b8f00..ee5621c0f 100644 --- a/crates/api/src/deployment.rs +++ b/crates/api/src/deployment.rs @@ -23,6 +23,7 @@ pub const METHOD_DEPLOYMENT_PROVIDER_BINDINGS_LIST: &str = pub const METHOD_DEPLOYMENT_API_KEYS_CREATE: &str = "deployment/api-keys/create"; pub const METHOD_DEPLOYMENT_API_KEYS_LIST: &str = "deployment/api-keys/list"; +pub const METHOD_DEPLOYMENT_API_KEYS_ROTATE: &str = "deployment/api-keys/rotate"; pub const METHOD_DEPLOYMENT_API_KEYS_REVOKE: &str = "deployment/api-keys/revoke"; pub const METHOD_DEPLOYMENT_ENVIRONMENT_PROVIDERS_PUT: &str = "deployment/environment-providers/put"; @@ -162,7 +163,7 @@ pub struct DeploymentApiKeyCreateParams { /// with the `x-lightspeed-universe` header and may hold deployment groups. pub scope: AccessScope, /// The method groups the key may call; absent grants every group its - /// scope allows. Keys never change: to change what a key may do, revoke + /// scope allows. To change what a key may do, revoke /// it and mint another. #[serde(default, skip_serializing_if = "Option::is_none")] pub groups: Option>, @@ -175,7 +176,7 @@ pub struct DeploymentApiKeyCreateParams { pub display_name: String, } -/// A newly minted key. `secret` is returned only by create and cannot be +/// A newly minted or rotated key. `secret` is returned only by create or rotate and cannot be /// recovered later. Its custom `Debug` implementation redacts the DTO before /// JSON-RPC serialization; the serialized response payload remains sensitive /// and must not be logged. @@ -211,6 +212,12 @@ pub struct DeploymentApiKeyListResponse { pub api_keys: Vec, } +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct DeploymentApiKeyRotateParams { + pub key_prefix: String, +} + #[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] #[serde(rename_all = "camelCase", deny_unknown_fields)] pub struct DeploymentApiKeyRevokeParams { @@ -436,6 +443,11 @@ pub trait DeploymentApiService: Send + Sync { params: DeploymentApiKeyListParams, ) -> Result, AgentApiError>; + async fn rotate_api_key( + &self, + params: DeploymentApiKeyRotateParams, + ) -> Result, AgentApiError>; + async fn revoke_api_key( &self, params: DeploymentApiKeyRevokeParams, @@ -579,9 +591,11 @@ deployment_api_methods! { METHOD_DEPLOYMENT_UNIVERSES_DELETE => delete_universe(DeploymentUniverseDeleteParams) -> DeploymentUniverseDeleteResponse => ["Purge a universe", "Permanently terminates live session workflows, deletes external blob objects, and cascades universe data. The purge is resumable/idempotent after partial failure."], access: MethodAccess::Deployment, METHOD_DEPLOYMENT_API_KEYS_CREATE => create_api_key(DeploymentApiKeyCreateParams) -> DeploymentApiKeyCreateResponse => - ["Create a scoped API key", "Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Keys are immutable: revoke and mint to change what one may do."], access: MethodAccess::Deployment, + ["Create a scoped API key", "Mints a key for a universe or the deployment with the method groups it may call and whether it may assert actors. The plaintext secret is returned exactly once and cannot be recovered; persist only the displayed prefix for identification. Key authority is immutable: revoke and mint to change what one may do."], access: MethodAccess::Deployment, METHOD_DEPLOYMENT_API_KEYS_LIST => list_api_keys(DeploymentApiKeyListParams) -> DeploymentApiKeyListResponse => ["List scoped API keys", "Returns non-secret key metadata, all keys or those of one scope, including groups, revocation and last-use timestamps. Plaintext secrets are never stored or returned."], access: MethodAccess::Deployment, + METHOD_DEPLOYMENT_API_KEYS_ROTATE => rotate_api_key(DeploymentApiKeyRotateParams) -> DeploymentApiKeyCreateResponse => + ["Rotate a scoped API key", "Atomically replaces an active key secret and display prefix, immediately rejecting the old secret on subsequent requests. Preserves scope, groups, actor authority, name, creator and creation time; clears last use. Returns the new secret once. Unknown or revoked prefixes are not found. Already admitted work continues."], access: MethodAccess::Deployment, METHOD_DEPLOYMENT_API_KEYS_REVOKE => revoke_api_key(DeploymentApiKeyRevokeParams) -> DeploymentApiKeyRevokeResponse => ["Revoke a scoped API key", "Revokes the key with this display prefix; revoking a revoked key keeps its first revocation time. An unknown prefix is not found."], access: MethodAccess::Deployment, METHOD_DEPLOYMENT_ENVIRONMENT_PROVIDERS_PUT => put_environment_provider(DeploymentEnvironmentProviderPutParams) -> DeploymentEnvironmentProviderPutResponse => diff --git a/crates/api/src/lib.rs b/crates/api/src/lib.rs index 183679f3c..f7944e68e 100644 --- a/crates/api/src/lib.rs +++ b/crates/api/src/lib.rs @@ -27,6 +27,7 @@ mod handshake; mod ids; mod mcp; mod model; +mod model_defaults; mod models; mod notifications; mod profiles; @@ -37,6 +38,7 @@ mod service; mod sessions; mod skills; mod storage; +mod transcriptions; mod views; pub use access::*; @@ -50,6 +52,7 @@ pub use handshake::*; pub use ids::*; pub use mcp::*; pub use model::*; +pub use model_defaults::*; pub use models::*; pub use notifications::*; pub use profiles::*; @@ -60,6 +63,7 @@ pub use service::*; pub use sessions::*; pub use skills::*; pub use storage::*; +pub use transcriptions::*; pub use views::*; #[cfg(test)] diff --git a/crates/api/src/model_defaults.rs b/crates/api/src/model_defaults.rs new file mode 100644 index 000000000..1a25625e8 --- /dev/null +++ b/crates/api/src/model_defaults.rs @@ -0,0 +1,150 @@ +use super::*; + +/// A universe's model selection for a particular use. Protocol and purpose +/// are separate: several purposes may use the same provider API. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub enum ModelDefaultSlot { + AgentRun, + SpeechToText, +} + +impl ModelDefaultSlot { + pub fn as_str(self) -> &'static str { + match self { + Self::AgentRun => "agentRun", + Self::SpeechToText => "speechToText", + } + } + + pub fn validate_model(self, model: &ModelConfig) -> Result<(), AgentApiError> { + for (field, value) in [("providerId", &model.provider_id), ("model", &model.model)] { + if value.trim().is_empty() || value.trim() != value || value.len() > 512 { + return Err(AgentApiError::invalid_request(format!( + "{field} must contain 1..=512 bytes without surrounding whitespace" + ))); + } + } + let supported = match self { + Self::AgentRun => matches!( + model.api_kind.as_str(), + "openai:responses" | "openai:completions" | "anthropic:messages" + ), + Self::SpeechToText => model.api_kind == "openai:audio-transcriptions", + }; + if !supported { + return Err(AgentApiError::invalid_request(format!( + "API kind {} cannot be used for {}", + model.api_kind, + self.as_str() + ))); + } + Ok(()) + } +} + +/// Persisted universe defaults. Revision zero means no update has been made. +/// Clearing a slot still advances the revision, so setup cannot undo a clear. +#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct ModelDefaults { + pub revision: u64, + pub agent_run: Option, + pub speech_to_text: Option, +} + +impl ModelDefaults { + pub fn model(&self, slot: ModelDefaultSlot) -> Option<&ModelConfig> { + match slot { + ModelDefaultSlot::AgentRun => self.agent_run.as_ref(), + ModelDefaultSlot::SpeechToText => self.speech_to_text.as_ref(), + } + } +} + +#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct ModelDefaultsReadParams {} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct ModelDefaultsPutParams { + pub slot: ModelDefaultSlot, + /// Complete selection, or explicit null to clear this slot. Required. + #[serde(deserialize_with = "Option::deserialize")] + #[schemars(required, schema_with = "required_nullable_model_schema")] + pub model: Option, + /// Revision returned by read/put; zero for a universe with no updates. + pub expected_revision: u64, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct ModelDefaultsResponse { + pub defaults: ModelDefaults, +} + +fn required_nullable_model_schema(generator: &mut schemars::SchemaGenerator) -> schemars::Schema { + Option::::json_schema(generator) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn slots_validate_protocols_without_requiring_a_catalog_model() { + for (kind, slot) in [ + ("openai:responses", ModelDefaultSlot::AgentRun), + ("anthropic:messages", ModelDefaultSlot::AgentRun), + ("openai:completions", ModelDefaultSlot::AgentRun), + ( + "openai:audio-transcriptions", + ModelDefaultSlot::SpeechToText, + ), + ] { + let model = ModelConfig { + provider_id: "custom".into(), + api_kind: kind.into(), + model: "private-model".into(), + }; + slot.validate_model(&model).unwrap(); + let other = match slot { + ModelDefaultSlot::AgentRun => ModelDefaultSlot::SpeechToText, + ModelDefaultSlot::SpeechToText => ModelDefaultSlot::AgentRun, + }; + assert_eq!( + other.validate_model(&model).unwrap_err().kind, + AgentApiErrorKind::InvalidRequest + ); + } + } + + #[test] + fn clearing_requires_an_explicit_null_and_a_revision() { + let request = serde_json::json!({"slot":"agentRun", "model":null, "expectedRevision":4}); + assert_eq!( + serde_json::from_value::(request.clone()) + .unwrap() + .model, + None + ); + for field in ["model", "expectedRevision"] { + let mut missing = request.clone(); + missing.as_object_mut().unwrap().remove(field); + assert!(serde_json::from_value::(missing).is_err()); + } + } + + #[test] + fn missing_default_error_identifies_its_slot() { + let error = AgentApiError::model_default_unset(ModelDefaultSlot::AgentRun); + let json = serde_json::to_value(&error).unwrap(); + assert_eq!(json["kind"], "model_default_unset"); + assert_eq!(json["modelDefaultSlot"], "agentRun"); + assert_eq!( + serde_json::from_value::(json).unwrap(), + error + ); + } +} diff --git a/crates/api/src/models.rs b/crates/api/src/models.rs index 28ef814b2..dfb9b8e78 100644 --- a/crates/api/src/models.rs +++ b/crates/api/src/models.rs @@ -6,9 +6,11 @@ use super::*; #[serde(rename_all = "camelCase")] pub struct ModelListParams { /// Apply Lightspeed's small, conservative selectable-model policy. It - /// removes OpenAI model-id families that are clearly not text-generation - /// routes (embeddings, moderation, image/video, speech, and realtime). - /// It is an ID policy, not a provider capability claim. + /// keeps supported file-transcription routes and filters clearly unrelated + /// OpenAI families from agent suggestions (embeddings, moderation, image/video, + /// speech synthesis, and realtime). Agent suggestions also have an age limit. + /// Clients select routes by API kind for their intended use. This is an ID + /// policy, not a provider capability claim. #[serde(default, skip_serializing_if = "std::ops::Not::not")] pub selectable_only: bool, } diff --git a/crates/api/src/rpc.rs b/crates/api/src/rpc.rs index 396a83ef6..8c5839d70 100644 --- a/crates/api/src/rpc.rs +++ b/crates/api/src/rpc.rs @@ -13,12 +13,9 @@ pub enum AgentApiErrorKind { Unauthenticated, /// The authenticated caller lacks permission for this operation or target. Forbidden, - UnsupportedAudioMime, + /// No model was supplied and this universe has no default for the requested use. + ModelDefaultUnset, AudioBlobTooLarge, - AudioDurationTooLong, - TranscoderUnavailable, - TranscodeFailure, - TranscriptionFailure, /// The session's agent workflow exists but failed during bootstrap /// (rehydration) and cannot serve runs. Distinct from `NotFound` (no /// workflow) so clients/bridges treat it as a session recovery problem @@ -42,6 +39,9 @@ pub enum AgentApiErrorKind { pub struct AgentApiError { pub kind: AgentApiErrorKind, pub message: String, + /// Present for model_default_unset; clients need not parse the message. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_default_slot: Option, } impl AgentApiError { @@ -49,6 +49,18 @@ impl AgentApiError { Self { kind, message: message.into(), + model_default_slot: None, + } + } + + pub fn model_default_unset(slot: ModelDefaultSlot) -> Self { + Self { + kind: AgentApiErrorKind::ModelDefaultUnset, + message: format!( + "no universe model default is configured for {}; choose a model or set it with models/defaults/put", + slot.as_str() + ), + model_default_slot: Some(slot), } } @@ -79,30 +91,10 @@ impl AgentApiError { Self::new(AgentApiErrorKind::Forbidden, "request is not authorized") } - pub fn unsupported_audio_mime(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::UnsupportedAudioMime, message) - } - pub fn audio_blob_too_large(message: impl Into) -> Self { Self::new(AgentApiErrorKind::AudioBlobTooLarge, message) } - pub fn audio_duration_too_long(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::AudioDurationTooLong, message) - } - - pub fn transcoder_unavailable(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::TranscoderUnavailable, message) - } - - pub fn transcode_failure(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::TranscodeFailure, message) - } - - pub fn transcription_failure(message: impl Into) -> Self { - Self::new(AgentApiErrorKind::TranscriptionFailure, message) - } - pub fn session_bootstrap_failed(message: impl Into) -> Self { Self::new(AgentApiErrorKind::SessionBootstrapFailed, message) } @@ -121,21 +113,16 @@ impl AgentApiError { pub fn json_rpc_code(&self) -> i64 { match self.kind { - AgentApiErrorKind::InvalidRequest - | AgentApiErrorKind::UnsupportedAudioMime - | AgentApiErrorKind::AudioBlobTooLarge - | AgentApiErrorKind::AudioDurationTooLong - | AgentApiErrorKind::TranscoderUnavailable => -32602, + AgentApiErrorKind::InvalidRequest | AgentApiErrorKind::AudioBlobTooLarge => -32602, AgentApiErrorKind::Unauthenticated => -32001, AgentApiErrorKind::Forbidden => -32003, AgentApiErrorKind::NotFound => -32004, AgentApiErrorKind::Conflict => -32009, - AgentApiErrorKind::Rejected - | AgentApiErrorKind::TranscodeFailure - | AgentApiErrorKind::TranscriptionFailure => -32010, + AgentApiErrorKind::Rejected => -32010, AgentApiErrorKind::SessionBootstrapFailed => -32011, AgentApiErrorKind::EnvironmentNotReady => -32012, AgentApiErrorKind::ResponseTooLarge => -32013, + AgentApiErrorKind::ModelDefaultUnset => -32014, AgentApiErrorKind::Internal => -32603, } } @@ -358,7 +345,7 @@ api_methods! { METHOD_SESSION_LIST => list_sessions(SessionListParams) -> SessionListResponse => ["List sessions", "Returns a cursor-paginated summary list ordered by most recent update, optionally narrowed by the audience of each session's root: createdBy, visibility, or visibleTo (shared with the universe or created by that actor). Pages may shift while sessions are changing."], access: MethodAccess::Universe(UniverseAction::Read), METHOD_SESSION_CONFIG_PUT => put_session_config(SessionConfigPutParams) -> SessionConfigPutResponse => - ["Replace session configuration", "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op."], access: MethodAccess::Universe(UniverseAction::ControlSession), + ["Replace session configuration", "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op."], access: MethodAccess::Universe(UniverseAction::ControlSession), METHOD_SESSION_RENAME => rename_session(SessionRenameParams) -> SessionRenameResponse => ["Rename a session", "Sets the display name, or clears it when displayName is omitted."], access: MethodAccess::Universe(UniverseAction::ControlSession), METHOD_SESSION_METADATA_PUT => put_session_metadata(SessionMetadataPutParams) -> SessionMetadataPutResponse => @@ -374,7 +361,7 @@ api_methods! { METHOD_SESSION_EVENTS_READ => read_session_events(SessionEventsReadParams) -> SessionEventsReadResponse => ["Read the session event stream", "Returns chronological events. Forward (default) follows after and supports long-polling. Backward reads the latest window below before (or the head); pass nextCursor as before until complete. Follow live events after the initial backward headCursor. Windows may split runs/tool batches; keep historical reconstruction separate from live controls."], access: MethodAccess::Universe(UniverseAction::Read), METHOD_SESSION_CONTEXT_APPEND => append_context(ContextAppendParams) -> ContextAppendResponse => - ["Append keyed session context", "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries."], access: MethodAccess::Universe(UniverseAction::ControlSession), + ["Append keyed session context", "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries."], access: MethodAccess::Universe(UniverseAction::ControlSession), METHOD_SESSION_CONTEXT_REMOVE => remove_context(ContextRemoveParams) -> ContextRemoveResponse => ["Remove keyed session context", "Removes active entries by stable key with per-key results. Missing keys are idempotent no-ops; runtime-reserved run keys cannot be removed."], access: MethodAccess::Universe(UniverseAction::ControlSession), METHOD_SESSION_CONTEXT_COMPACT => compact_context(ContextCompactParams) -> ContextCompactResponse => @@ -443,6 +430,16 @@ api_methods! { ["List environment registration keys", "Lists this universe's registration keys with policy, status, and derived counts. Each key is the group of the environments it admitted."], access: MethodAccess::Universe(UniverseAction::ConfigureResource), METHOD_ENVIRONMENTS_REGISTRATION_KEYS_REVOKE => revoke_environment_registration_key(EnvironmentRegistrationKeyRevokeParams) -> EnvironmentRegistrationKeyRevokeResponse => ["Revoke an environment registration key", "Stops the key from admitting new daemon identities; already registered daemons keep reconnecting. With closeEnvironments, also closes every non-closed environment the key admitted. Idempotent."], access: MethodAccess::Universe(UniverseAction::ConfigureResource), + METHOD_TRANSCRIPTIONS_START => start_transcription(TranscriptionStartParams) -> TranscriptionResponse => + ["Start transcription", "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created."], access: MethodAccess::Universe(UniverseAction::UseResource), + METHOD_TRANSCRIPTIONS_READ => read_transcription(TranscriptionReadParams) -> TranscriptionResponse => + ["Read transcription", "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire."], access: MethodAccess::Universe(UniverseAction::Read), + METHOD_TRANSCRIPTIONS_CANCEL => cancel_transcription(TranscriptionCancelParams) -> TranscriptionResponse => + ["Cancel transcription", "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority."], access: MethodAccess::Universe(UniverseAction::UseResource), + METHOD_MODELS_DEFAULTS_READ => read_model_defaults(ModelDefaultsReadParams) -> ModelDefaultsResponse => + ["Read universe model defaults", "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers."], access: MethodAccess::Universe(UniverseAction::Read), + METHOD_MODELS_DEFAULTS_PUT => put_model_defaults(ModelDefaultsPutParams) -> ModelDefaultsResponse => + ["Set a universe model default", "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery."], access: MethodAccess::Universe(UniverseAction::ConfigureResource), METHOD_MODELS_LIST => list_models(ModelListParams) -> ModelListResponse => ["Discover available models", "Queries supported providers directly, with a brief process-local burst cache, and returns best-effort selectable routes. One provider failure does not discard successful results from others."], access: MethodAccess::Universe(UniverseAction::Read), METHOD_PROFILES_CREATE => create_profile(ProfileCreateParams) -> ProfileCreateResponse => diff --git a/crates/api/src/schema_export.rs b/crates/api/src/schema_export.rs index ffa10ec31..c135b713b 100644 --- a/crates/api/src/schema_export.rs +++ b/crates/api/src/schema_export.rs @@ -205,13 +205,13 @@ mod tests { methods.sort_unstable(); methods.dedup(); assert_eq!(methods.len(), total, "duplicate method in manifest"); - assert_eq!(total, 131); + assert_eq!(total, 137); assert_eq!( manifest .iter() .filter(|spec| spec.scope == crate::MethodScope::Deployment) .count(), - 17 + 18 ); } diff --git a/crates/api/src/service.rs b/crates/api/src/service.rs index 09cb6ce52..6d745b78f 100644 --- a/crates/api/src/service.rs +++ b/crates/api/src/service.rs @@ -2,6 +2,25 @@ use super::*; #[async_trait] pub trait AgentApiService: Send + Sync { + async fn start_transcription( + &self, + _params: TranscriptionStartParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("transcription is unavailable")) + } + async fn read_transcription( + &self, + _params: TranscriptionReadParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("transcription is unavailable")) + } + async fn cancel_transcription( + &self, + _params: TranscriptionCancelParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("transcription is unavailable")) + } + async fn read_vfs_workspace_file( &self, _params: VfsWorkspaceFileReadParams, @@ -15,6 +34,20 @@ pub trait AgentApiService: Send + Sync { params: InitializeParams, ) -> Result, AgentApiError>; + async fn read_model_defaults( + &self, + _params: ModelDefaultsReadParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("model defaults are unavailable")) + } + + async fn put_model_defaults( + &self, + _params: ModelDefaultsPutParams, + ) -> Result, AgentApiError> { + Err(AgentApiError::internal("model defaults are unavailable")) + } + async fn list_models( &self, params: ModelListParams, diff --git a/crates/api/src/sessions.rs b/crates/api/src/sessions.rs index 08300dc3c..0380ddbfe 100644 --- a/crates/api/src/sessions.rs +++ b/crates/api/src/sessions.rs @@ -264,7 +264,9 @@ fn default_feature_version() -> u32 { #[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] #[serde(rename_all = "camelCase", deny_unknown_fields)] pub struct SessionConfig { - /// Absent on input means the deployment default model. Documents read + /// At creation, omission uses the profile model or universe agentRun + /// default. On configuration replacement or profile application to an + /// existing session, omission preserves its current model. Documents read /// back from a session always carry the model. Provider identity and API /// kind are fixed for the session lifetime; the model name may change. #[serde(default, skip_serializing_if = "Option::is_none")] @@ -860,13 +862,7 @@ pub struct InputAdmissionFailureView { #[serde(rename_all = "camelCase")] pub enum InputAdmissionFailureKind { UnsupportedMedia, - UnsupportedAudioMime, BlobMissing, - BlobTooLarge, - AudioDurationTooLong, - TranscoderUnavailable, - TranscodeFailure, - TranscriptionFailure, AdmissionRejected, } diff --git a/crates/api/src/tests.rs b/crates/api/src/tests.rs index ca1de915a..c8179f108 100644 --- a/crates/api/src/tests.rs +++ b/crates/api/src/tests.rs @@ -59,6 +59,7 @@ fn notification_serializes_as_json_rpc_lite_shape() { completed_at_ms: Some(20), source: RunViewSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "hello".to_owned(), }], @@ -3464,6 +3465,17 @@ impl DeploymentApiService for TestDeploymentService { })) } + async fn rotate_api_key( + &self, + params: DeploymentApiKeyRotateParams, + ) -> Result, AgentApiError> { + assert_eq!(params.key_prefix, "lsk_ab12cd34"); + Ok(AgentApiOutcome::new(DeploymentApiKeyCreateResponse { + api_key: test_deployment_api_key("lsk_rotated1"), + secret: "lsk_rotated_secret".to_owned(), + })) + } + async fn revoke_api_key( &self, _params: DeploymentApiKeyRevokeParams, @@ -3685,6 +3697,19 @@ async fn dispatch_deployment_json_rpc_routes_scoped_api_key_management() { assert_eq!(result["result"]["apiKeys"].as_array().unwrap().len(), 1); assert!(result["result"]["apiKeys"][0].get("secret").is_none()); + let rotate = dispatch_deployment_json_rpc( + &TestDeploymentService, + JsonRpcRequest { + id: RequestId::Number(4), + method: METHOD_DEPLOYMENT_API_KEYS_ROTATE.to_owned(), + params: Some(json!({ "keyPrefix": "lsk_ab12cd34" })), + }, + ) + .await; + let result = rotate.result.expect("rotate result"); + assert_eq!(result["result"]["apiKey"]["keyPrefix"], "lsk_rotated1"); + assert_eq!(result["result"]["secret"], "lsk_rotated_secret"); + let revoke = dispatch_deployment_json_rpc( &TestDeploymentService, JsonRpcRequest { diff --git a/crates/api/src/transcriptions.rs b/crates/api/src/transcriptions.rs new file mode 100644 index 000000000..7eebb0d04 --- /dev/null +++ b/crates/api/src/transcriptions.rs @@ -0,0 +1,123 @@ +use super::*; + +/// Immutable audio input in this universe's content store. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct TranscriptionAudio { + pub blob_ref: String, + pub mime: String, + pub name: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct TranscriptionStartParams { + /// Scoped to the requester. Matching retries rejoin the original job, + /// including after defaults change; changed requests conflict. Identity is + /// retained for the Temporal namespace's workflow-history retention period. + pub idempotency_key: String, + pub audio: TranscriptionAudio, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub language: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub prompt: Option, +} + +impl TranscriptionStartParams { + pub fn validate(&self) -> Result<(), AgentApiError> { + for (name, value, limit) in [ + ("idempotencyKey", self.idempotency_key.as_str(), 200), + ("audio.mime", self.audio.mime.as_str(), 128), + ("audio.name", self.audio.name.as_str(), 256), + ] { + if value.trim().is_empty() || value.len() > limit { + return Err(AgentApiError::invalid_request(format!( + "{name} must contain 1–{limit} bytes" + ))); + } + } + if self + .language + .as_ref() + .is_some_and(|v| v.is_empty() || v.len() > 32) + || self.prompt.as_ref().is_some_and(|v| v.len() > 8192) + { + return Err(AgentApiError::invalid_request( + "language or prompt exceeds the transcription limit", + )); + } + if let Some(model) = &self.model { + ModelDefaultSlot::SpeechToText.validate_model(model)?; + } + Ok(()) + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct TranscriptionReadParams { + pub transcription_id: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub struct TranscriptionCancelParams { + pub transcription_id: String, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub enum TranscriptionStatus { + Pending, + Running, + Succeeded, + Failed, + Cancelled, + Expired, +} +impl TranscriptionStatus { + pub fn is_terminal(self) -> bool { + !matches!(self, Self::Pending | Self::Running) + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub enum TranscriptionFailureKind { + InvalidAudio, + Configuration, + Provider, + Timeout, + Internal, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionFailure { + pub kind: TranscriptionFailureKind, + pub message: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionView { + pub transcription_id: String, + pub created_by: Attribution, + pub audio: TranscriptionAudio, + pub model: ModelConfig, + pub status: TranscriptionStatus, + pub created_at_ms: u64, + /// Plain UTF-8 transcript blob, usable as ordinary textRef input. + /// Unsubmitted content can be swept after the ordinary CAS grace period. + pub transcript_ref: Option, + pub text: Option, + pub failure: Option, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionResponse { + pub transcription: TranscriptionView, +} diff --git a/crates/api/src/views.rs b/crates/api/src/views.rs index 766ffb7df..9d315566d 100644 --- a/crates/api/src/views.rs +++ b/crates/api/src/views.rs @@ -364,6 +364,10 @@ pub enum InputItem { /// bot deliveries; other values are allowed. Omitted means unknown. /// This metadata is not an authorization identity or model input text. origin: Option, + /// Optional source blob in this universe, retained with the session. + /// Provenance is metadata, not model input or an authorization identity. + #[serde(default, skip_serializing_if = "Option::is_none")] + provenance_ref: Option, text: String, }, TextRef { @@ -373,6 +377,10 @@ pub enum InputItem { /// bot deliveries; other values are allowed. Omitted means unknown. /// This metadata is not an authorization identity or model input text. origin: Option, + /// Optional source blob in this universe, retained with the session. + /// Provenance is metadata, not model input or an authorization identity. + #[serde(default, skip_serializing_if = "Option::is_none")] + provenance_ref: Option, blob_ref: String, }, Media { diff --git a/crates/api/tests/schema_artifacts.rs b/crates/api/tests/schema_artifacts.rs index d77fdcf52..f470bbcd0 100644 --- a/crates/api/tests/schema_artifacts.rs +++ b/crates/api/tests/schema_artifacts.rs @@ -36,6 +36,21 @@ fn committed_text(name: &str) -> String { }) } +#[test] +fn model_default_updates_accept_explicit_null_in_the_public_contract() { + let bundle = api::export_schemas().schema_bundle; + for model in [ + Value::Null, + json!({"providerId":"private", "apiKind":"openai:completions", "model":"custom"}), + ] { + let request = json!({"slot":"agentRun", "model":model, "expectedRevision":0}); + assert_validates(&bundle, "ModelDefaultsPutParams", &request); + serde_json::from_value::(request).unwrap(); + } + let required = &bundle["definitions"]["ModelDefaultsPutParams"]["required"]; + assert!(required.as_array().unwrap().contains(&json!("model"))); +} + fn assert_validates(bundle: &Value, definition: &str, instance: &Value) { let schema = json!({ "$schema": "http://json-schema.org/draft-07/schema#", @@ -62,6 +77,7 @@ fn serialized_fixtures_validate_against_exported_schemas() { session_id: "session_1".to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "hello".to_owned(), }], @@ -81,6 +97,7 @@ fn serialized_fixtures_validate_against_exported_schemas() { completed_at_ms: Some(20), source: RunViewSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "hello".to_owned(), }], diff --git a/crates/auth/src/api_keys.rs b/crates/auth/src/api_keys.rs index 8ce56202f..390d8b580 100644 --- a/crates/auth/src/api_keys.rs +++ b/crates/auth/src/api_keys.rs @@ -2,8 +2,8 @@ //! //! A key names what it reaches (one universe, or the deployment), the method //! groups it may call, and whether it may assert the actor a request acts -//! for. That is its whole authority. Keys are immutable apart from their -//! display name and revocation: changing what a key may do is revoking it and +//! for. That is its whole authority. Rotation replaces the secret without +//! changing authority: changing what a key may do is revoking it and //! minting another, so a running process never gains or loses rights //! silently. Persistence belongs to the deployment store, before universe //! resolution. diff --git a/crates/auth/src/providers.rs b/crates/auth/src/providers.rs index 72da1a26e..e53499f62 100644 --- a/crates/auth/src/providers.rs +++ b/crates/auth/src/providers.rs @@ -216,7 +216,10 @@ fn validate_model_endpoint(endpoint: &ModelEndpointConfig) -> Result<(), AuthReg } let mut kinds = BTreeSet::new(); for kind in &endpoint.api_kinds { - if !matches!(kind.as_str(), "openai:responses" | "openai:completions") { + if !matches!( + kind.as_str(), + "openai:responses" | "openai:completions" | "openai:audio-transcriptions" + ) { return Err(AuthRegistryError::InvalidInput { message: format!("model endpoint API kind {kind:?} is not supported"), }); @@ -613,6 +616,13 @@ mod tests { assert!(validate_model_endpoint(&reserved).is_err()); } + #[test] + fn speech_endpoints_declare_the_audio_protocol() { + let mut speech = endpoint("http://localhost:9000/v1"); + speech.api_kinds = vec!["openai:audio-transcriptions".into()]; + validate_model_endpoint(&speech).expect("speech endpoint"); + } + #[test] fn credentialless_model_endpoint_validates_and_round_trips() { let config = AuthProviderConfig::ModelEndpoint(ModelEndpointOnlyConfig { diff --git a/crates/bots/src/views.rs b/crates/bots/src/views.rs index dfa315d3d..fcf6fdd20 100644 --- a/crates/bots/src/views.rs +++ b/crates/bots/src/views.rs @@ -605,18 +605,29 @@ fn media_kind(kind: BotEventMediaKind) -> MediaKind { /// and its media. fn event_input_items(event: &BotEvent) -> impl Iterator + '_ { std::iter::once(InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: event .prompt_ref .clone() .unwrap_or_else(|| event.document_ref.clone()), }) - .chain(event.media.iter().map(|item| InputItem::Media { - origin: Some("event".to_owned()), - blob_ref: item.blob_ref.clone(), - mime: item.mime.clone(), - kind: media_kind(item.kind), - name: item.name.clone(), + .chain(event.media.iter().map(|item| { + if let Some(text_ref) = &item.text_ref { + InputItem::TextRef { + provenance_ref: Some(item.blob_ref.clone()), + origin: Some("event".into()), + blob_ref: text_ref.clone(), + } + } else { + InputItem::Media { + origin: Some("event".to_owned()), + blob_ref: item.blob_ref.clone(), + mime: item.mime.clone(), + kind: media_kind(item.kind), + name: item.name.clone(), + } + } })) } @@ -627,7 +638,7 @@ fn event_input_items(event: &BotEvent) -> impl Iterator + '_ { pub fn delivery_input_items(events: &[BotEvent]) -> Vec { let mut items = Vec::new(); if events.len() > 1 { - items.push(InputItem::Text { origin: Some("event".to_owned()), + items.push(InputItem::Text { provenance_ref: None, origin: Some("event".to_owned()), text: format!( "{} events delivered as one batch — handle them together and resolve the delivery once.", events.len() @@ -641,6 +652,7 @@ pub fn delivery_input_items(events: &[BotEvent]) -> Vec { /// Steering input for events folded into a running run. pub fn steer_input_items(events: &[BotEvent]) -> Vec { let mut items = vec![InputItem::Text { + provenance_ref: None, origin: Some("event".to_owned()), text: format!( "{} more event(s) arrived while you were working — fold them into your current work where relevant.", @@ -1471,6 +1483,7 @@ mod tests { assert_eq!( delivery_input_items(&[signal_event(&document_ref, Some(&prompt_ref))]), vec![InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: prompt_ref.clone(), }] @@ -1479,6 +1492,7 @@ mod tests { assert_eq!( delivery_input_items(&[signal_event(&document_ref, None)]), vec![InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: document_ref.clone(), }] @@ -1486,6 +1500,7 @@ mod tests { // Media follows its event's rendering. let mut with_media = signal_event(&document_ref, Some(&prompt_ref)); with_media.media = vec![BotEventMedia { + text_ref: None, blob_ref: format!("sha256:{}", "c".repeat(64)), kind: BotEventMediaKind::Image, mime: "image/png".to_owned(), @@ -1505,13 +1520,34 @@ mod tests { ); } + #[test] + fn prepared_attachment_delivery_uses_text_with_source_provenance() { + let reference = format!("sha256:{}", "d".repeat(64)); + let mut event = signal_event(&format!("sha256:{}", "a".repeat(64)), None); + event.media.push(BotEventMedia { + text_ref: Some(reference.clone()), + blob_ref: format!("sha256:{}", "c".repeat(64)), + kind: BotEventMediaKind::Audio, + mime: "audio/ogg".into(), + name: Some("voice.ogg".into()), + }); + assert_eq!( + delivery_input_items(&[event])[1], + InputItem::TextRef { + provenance_ref: Some(format!("sha256:{}", "c".repeat(64))), + origin: Some("event".into()), + blob_ref: reference + } + ); + } + #[test] fn frames_a_batch_with_one_header_line_binding_it_to_one_decision() { let a = format!("sha256:{}", "a".repeat(64)); let b = format!("sha256:{}", "b".repeat(64)); let items = delivery_input_items(&[signal_event(&a, Some(&b)), signal_event(&b, Some(&a))]); assert_eq!(items.len(), 3); - let InputItem::Text { text, origin } = &items[0] else { + let InputItem::Text { text, origin, .. } = &items[0] else { panic!("expected a text header, got {:?}", items[0]); }; assert_eq!(origin.as_deref(), Some("event")); @@ -1520,6 +1556,7 @@ mod tests { assert_eq!( items[1], InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: b.clone() } @@ -1527,6 +1564,7 @@ mod tests { assert_eq!( items[2], InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: a.clone() } @@ -1539,7 +1577,7 @@ mod tests { let b = format!("sha256:{}", "b".repeat(64)); let items = steer_input_items(&[signal_event(&a, Some(&b))]); assert_eq!(items.len(), 2); - let InputItem::Text { text, origin } = &items[0] else { + let InputItem::Text { text, origin, .. } = &items[0] else { panic!("expected a text header, got {:?}", items[0]); }; assert_eq!(origin.as_deref(), Some("event")); @@ -1548,6 +1586,7 @@ mod tests { assert_eq!( items[1], InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: b } diff --git a/crates/channels/src/media.rs b/crates/channels/src/media.rs index 1e2d12a80..57f772775 100644 --- a/crates/channels/src/media.rs +++ b/crates/channels/src/media.rs @@ -248,6 +248,7 @@ impl From for BotEventMedia { Self { blob_ref: item.blob_ref, kind: bot_event_media_kind(item.kind), + text_ref: None, mime: item.mime, name: item.name, } diff --git a/crates/cli/src/administration_cli.rs b/crates/cli/src/administration_cli.rs index ab7bcc35c..6318d3936 100644 --- a/crates/cli/src/administration_cli.rs +++ b/crates/cli/src/administration_cli.rs @@ -183,6 +183,10 @@ enum ApiKeyCommand { #[arg(long)] assert_actor: bool, }, + /// Replace an active key secret immediately; prints the new secret once. + Rotate { + key_prefix: String, + }, Revoke { key_prefix: String, }, @@ -261,6 +265,21 @@ pub async fn api_key(args: ApiKeyArgs) -> Result<()> { } crate::output::show(args.json, &response)?; } + ApiKeyCommand::Rotate { key_prefix } => { + let response: api::DeploymentApiKeyCreateResponse = client + .request( + api::METHOD_DEPLOYMENT_API_KEYS_ROTATE, + api::DeploymentApiKeyRotateParams { key_prefix }, + ) + .await? + .result; + if !args.json { + println!( + "API key rotated. The old secret no longer works. Save the new secret now." + ); + } + crate::output::show(args.json, &response)?; + } ApiKeyCommand::Revoke { key_prefix } => { let response: api::DeploymentApiKeyRevokeResponse = client .request( @@ -281,6 +300,8 @@ pub struct ModelsArgs { } #[derive(Debug, Subcommand)] enum ModelsCommand { + /// Inspect, set, or clear universe model defaults. + Defaults(crate::model_defaults_cli::ModelDefaultsArgs), /// Configure provider endpoints and their API-key or OAuth credentials. #[command(visible_alias = "providers")] Provider(crate::auth_cli::AuthModelArgs), @@ -293,6 +314,7 @@ enum ModelsCommand { } pub async fn models(args: ModelsArgs) -> Result<()> { let (json, all) = match args.command { + ModelsCommand::Defaults(args) => return crate::model_defaults_cli::run(args).await, ModelsCommand::Provider(args) => return crate::auth_cli::model(args).await, ModelsCommand::List { json, all } => (json, all), }; diff --git a/crates/cli/src/chat/driver.rs b/crates/cli/src/chat/driver.rs index 5d44408e7..e88688d15 100644 --- a/crates/cli/src/chat/driver.rs +++ b/crates/cli/src/chat/driver.rs @@ -581,7 +581,11 @@ impl ChatSessionDriver { notify_on_terminal: None, session_id, source: RunStartSource::Input { - items: vec![InputItem::Text { origin: None, text }], + items: vec![InputItem::Text { + provenance_ref: None, + origin: None, + text, + }], }, submission_id: Some(new_submission_id()), config: Some(config), @@ -648,7 +652,11 @@ impl ChatSessionDriver { .steer_run(api::RunSteerParams { session_id: self.session_id.clone(), run_id, - items: vec![InputItem::Text { origin: None, text }], + items: vec![InputItem::Text { + provenance_ref: None, + origin: None, + text, + }], }) .await .map_err(api_error)? diff --git a/crates/cli/src/main.rs b/crates/cli/src/main.rs index 46c282334..6bc8870f5 100644 --- a/crates/cli/src/main.rs +++ b/crates/cli/src/main.rs @@ -5,6 +5,7 @@ mod chat; mod connection; mod env_cli; mod mcp_cli; +mod model_defaults_cli; mod output; mod profile_cli; mod session_cli; diff --git a/crates/cli/src/model_defaults_cli.rs b/crates/cli/src/model_defaults_cli.rs new file mode 100644 index 000000000..843220e04 --- /dev/null +++ b/crates/cli/src/model_defaults_cli.rs @@ -0,0 +1,192 @@ +use anyhow::Result; +use clap::{Args, Subcommand, ValueEnum}; + +use crate::api_client::HttpAgentApi; + +#[derive(Debug, Args)] +pub struct ModelDefaultsArgs { + #[arg(long, global = true)] + json: bool, + #[command(subcommand)] + command: ModelDefaultsCommand, +} + +#[derive(Debug, Subcommand)] +enum ModelDefaultsCommand { + /// Read this universe's model selections and revision. + Read, + /// Set a complete provider/API/model route for one use. + Set { + #[arg(value_enum)] + slot: Slot, + #[arg(long)] + provider: String, + #[arg(long)] + api_kind: String, + #[arg(long)] + model: String, + /// Use this revision; otherwise read it immediately before updating. + #[arg(long)] + expected_revision: Option, + }, + /// Clear one default; existing sessions keep their model. + Clear { + #[arg(value_enum)] + slot: Slot, + #[arg(long)] + expected_revision: Option, + }, +} + +#[derive(Clone, Copy, Debug, ValueEnum)] +enum Slot { + #[value(alias = "agentRun")] + AgentRun, + #[value(alias = "speechToText")] + SpeechToText, +} + +impl From for api::ModelDefaultSlot { + fn from(slot: Slot) -> Self { + match slot { + Slot::AgentRun => Self::AgentRun, + Slot::SpeechToText => Self::SpeechToText, + } + } +} + +pub async fn run(args: ModelDefaultsArgs) -> Result<()> { + let client = HttpAgentApi::new(""); + let defaults = match args.command { + ModelDefaultsCommand::Read => read(&client).await?, + ModelDefaultsCommand::Set { + slot, + provider, + api_kind, + model, + expected_revision, + } => { + let slot = slot.into(); + let model = api::ModelConfig { + provider_id: provider, + api_kind, + model, + }; + api::ModelDefaultSlot::validate_model(slot, &model)?; + put(&client, slot, Some(model), expected_revision).await? + } + ModelDefaultsCommand::Clear { + slot, + expected_revision, + } => put(&client, slot.into(), None, expected_revision).await?, + }; + if args.json { + println!("{}", serde_json::to_string_pretty(&defaults)?); + } else { + println!("Model defaults (revision {})", defaults.revision); + for slot in [ + api::ModelDefaultSlot::AgentRun, + api::ModelDefaultSlot::SpeechToText, + ] { + match defaults.model(slot) { + Some(model) => println!( + "{}: {} / {} / {}", + slot.as_str(), + model.provider_id, + model.api_kind, + model.model + ), + None => println!("{}: unset", slot.as_str()), + } + } + } + Ok(()) +} + +async fn read(client: &HttpAgentApi) -> Result { + let response: api::AgentApiOutcome = client + .request( + api::METHOD_MODELS_DEFAULTS_READ, + api::ModelDefaultsReadParams {}, + ) + .await?; + Ok(response.result.defaults) +} + +async fn put( + client: &HttpAgentApi, + slot: api::ModelDefaultSlot, + model: Option, + revision: Option, +) -> Result { + let expected_revision = match revision { + Some(revision) => revision, + None => read(client).await?.revision, + }; + let response: api::AgentApiOutcome = client + .request( + api::METHOD_MODELS_DEFAULTS_PUT, + api::ModelDefaultsPutParams { + slot, + model, + expected_revision, + }, + ) + .await?; + // A concurrent change is returned to the caller; never retry a write + // against a newer revision without the caller reviewing that state. + Ok(response.result.defaults) +} + +#[cfg(test)] +mod tests { + use crate::Cli; + use clap::Parser; + + #[test] + fn defaults_commands_require_complete_routes_and_accept_revision_guards() { + for args in [ + vec!["lightspeed", "model", "defaults", "read", "--json"], + vec![ + "lightspeed", + "model", + "defaults", + "set", + "agent-run", + "--provider", + "custom", + "--api-kind", + "openai:completions", + "--model", + "model", + "--expected-revision", + "0", + ], + vec![ + "lightspeed", + "models", + "defaults", + "clear", + "speech-to-text", + "--json", + ], + ] { + Cli::try_parse_from(args).unwrap(); + } + assert!( + Cli::try_parse_from([ + "lightspeed", + "model", + "defaults", + "set", + "agent-run", + "--model", + "model" + ]) + .is_err() + ); + assert!( + Cli::try_parse_from(["lightspeed", "model", "defaults", "clear", "unknown"]).is_err() + ); + } +} diff --git a/crates/cli/src/skills_cli.rs b/crates/cli/src/skills_cli.rs index dcda7d99c..6698dbf94 100644 --- a/crates/cli/src/skills_cli.rs +++ b/crates/cli/src/skills_cli.rs @@ -128,7 +128,11 @@ async fn use_skill(args: SkillsUseArgs) -> Result<()> { .map_err(api_error)? .result .session; - let items = vec![api::InputItem::Text { origin: None, text }]; + let items = vec![api::InputItem::Text { + provenance_ref: None, + origin: None, + text, + }]; if let Some(run) = session.active_run { let response = api .steer_run(api::RunSteerParams { diff --git a/crates/cli/tests/resource_commands.rs b/crates/cli/tests/resource_commands.rs index 1d6d49a47..3fe1e2859 100644 --- a/crates/cli/tests/resource_commands.rs +++ b/crates/cli/tests/resource_commands.rs @@ -617,3 +617,101 @@ fn config_replacement_checks_the_requested_revision_and_outputs_one_document() { ]); assert_eq!(output["session"]["configRevision"], 43); } + +#[test] +fn model_defaults_commands_round_trip_routes_clear_explicitly_and_guard_revisions() { + let mut defaults = json!({"revision":0, "agentRun":null, "speechToText":null}); + let runtime = Runtime::start(move |method, params| { + match method { + "models/defaults/read" => {} + "models/defaults/put" => { + assert_eq!(params["expectedRevision"], defaults["revision"]); + let slot = params["slot"].as_str().unwrap(); + assert!( + params.get("model").is_some(), + "clearing must send explicit null" + ); + defaults[slot] = params["model"].clone(); + defaults["revision"] = json!(defaults["revision"].as_u64().unwrap() + 1); + } + other => panic!("unexpected method {other}"), + } + json!({"defaults":defaults}) + }); + let agent = runtime.json(&[ + "model", + "defaults", + "set", + "agent-run", + "--provider", + "anthropic", + "--api-kind", + "anthropic:messages", + "--model", + "chosen", + "--json", + ]); + assert_eq!(agent["agentRun"]["apiKind"], "anthropic:messages"); + assert_eq!(agent["revision"], 1); + let speech = runtime.json(&[ + "model", + "defaults", + "set", + "speech-to-text", + "--provider", + "custom-speech", + "--api-kind", + "openai:audio-transcriptions", + "--model", + "transcriber", + "--expected-revision", + "1", + "--json", + ]); + assert_eq!(speech["agentRun"], agent["agentRun"]); + let cleared = runtime.json(&["model", "defaults", "clear", "agent-run", "--json"]); + assert_eq!(cleared["agentRun"], Value::Null); + assert_eq!(cleared["speechToText"], speech["speechToText"]); + assert_eq!( + runtime.json(&["model", "defaults", "read", "--json"]), + cleared + ); + let requests = runtime.requests.lock().unwrap(); + assert_eq!( + requests + .iter() + .filter(|r| r["method"] == "models/defaults/read") + .count(), + 3 + ); +} + +#[test] +fn model_defaults_conflict_is_not_retried_with_a_new_revision() { + let runtime = Runtime::start_api(|method, _params| { + assert_eq!(method, "models/defaults/put"); + Err( + json!({"code":-32009, "message":"defaults changed", "data":{"kind":"conflict","message":"defaults changed"}}), + ) + }); + let output = runtime.run(&[ + "model", + "defaults", + "clear", + "agent-run", + "--expected-revision", + "4", + ]); + assert!(!output.status.success()); + assert!(String::from_utf8_lossy(&output.stderr).contains("defaults changed")); + assert_eq!( + runtime + .requests + .lock() + .unwrap() + .iter() + .filter(|r| r["method"] == "models/defaults/put") + .count(), + 1 + ); +} diff --git a/crates/llm-clients/src/content.rs b/crates/llm-clients/src/content.rs index f55224fe7..b21746daf 100644 --- a/crates/llm-clients/src/content.rs +++ b/crates/llm-clients/src/content.rs @@ -35,32 +35,6 @@ pub const OPENAI_COMPLETIONS_MESSAGE_PROVIDER_KIND: &str = "openai.completions.m pub const OPENAI_COMPLETIONS_REASONING_PROVIDER_KIND: &str = "openai.completions.reasoning_state"; pub const OPENAI_RESPONSES_REASONING_PROVIDER_KIND: &str = "openai.responses.reasoning"; pub const ANTHROPIC_THINKING_PROVIDER_KIND: &str = "anthropic.messages.thinking"; -pub const AUDIO_TRANSCRIPT_PROVIDER_KIND: &str = "lightspeed.audio.transcript"; - -/// Transcript content between audio preprocessing and model message lowering. -/// The source audio is recorded separately as the context entry's provenance. -#[derive(Clone, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub struct AudioTranscript { - pub filename: String, - pub text: String, -} - -impl AudioTranscript { - pub fn header(&self) -> String { - format!("[audio transcript: {}]", self.filename) - } - - pub fn model_text(&self) -> String { - format!("{}\n{}", self.header(), self.text) - } -} - -pub fn audio_transcript(raw: &Value) -> Option { - serde_json::from_value::(raw.clone()) - .ok() - .map(|transcript| transcript.text) -} - /// Only provider-exposed text participates in display. Signatures and encrypted /// continuation state remain in the payload for replay. pub fn anthropic_thinking(raw: &Value) -> Option { @@ -194,22 +168,4 @@ mod tests { Some("visible") ); } - - #[test] - fn transcript_labels_are_rendered_without_parsing_body_text() { - let transcript = AudioTranscript { - filename: "voice.ogg".into(), - text: "[audio transcript: quoted]\nkeep this header as speech".into(), - }; - let raw = serde_json::to_value(&transcript).unwrap(); - assert_eq!( - audio_transcript(&raw).as_deref(), - Some(transcript.text.as_str()) - ); - assert_eq!( - transcript.model_text(), - "[audio transcript: voice.ogg]\n[audio transcript: quoted]\nkeep this header as speech" - ); - assert!(audio_transcript(&json!({"text":"missing filename"})).is_none()); - } } diff --git a/crates/llm-clients/src/openai/audio.rs b/crates/llm-clients/src/openai/audio.rs index 23587e3b7..28a056a34 100644 --- a/crates/llm-clients/src/openai/audio.rs +++ b/crates/llm-clients/src/openai/audio.rs @@ -7,7 +7,9 @@ use crate::error::{ ConfigurationError, DecodeError, LlmApiError, ProviderHttpError, TransportError, }; use crate::transport::http::{join_url, normalize_base_url}; -use crate::transport::{ApiResponse, HeaderSnapshot, HttpClient, HttpClientConfig}; +use crate::transport::{ + ApiResponse, EndpointOverride, HeaderSnapshot, HttpClient, HttpClientConfig, +}; use reqwest::header::{AUTHORIZATION, HeaderValue}; use reqwest::{Method, StatusCode, Url}; use serde::de::DeserializeOwned; @@ -141,7 +143,20 @@ impl Client { request: CreateTranscriptionRequest, auth: Option>, ) -> Result, LlmApiError> { - let auth = self.auth_header(auth)?; + self.create_transcription_with_transport(request, auth, None) + .await + } + + pub async fn create_transcription_with_transport( + &self, + request: CreateTranscriptionRequest, + auth: Option>, + endpoint: Option<&EndpointOverride>, + ) -> Result, LlmApiError> { + let auth = match auth { + Some(crate::RequestAuth::None) if endpoint.is_some() => None, + other => Some(self.auth_header(other)?), + }; let file_part = reqwest::multipart::Part::bytes(request.file.bytes) .file_name(request.file.filename) .mime_str(&request.file.mime) @@ -159,10 +174,16 @@ impl Client { form = form.text("prompt", prompt); } - let response = self - .http - .request(Method::POST, self.transcriptions_url.clone()) - .header(AUTHORIZATION, auth) + let mut request_builder = self.http.request_with_endpoint( + Method::POST, + self.transcriptions_url.clone(), + "audio/transcriptions", + endpoint, + )?; + if let Some(auth) = auth { + request_builder = request_builder.header(AUTHORIZATION, auth); + } + let mut response = request_builder .multipart(form) .send() .await @@ -170,10 +191,21 @@ impl Client { let status = response.status(); let headers = HeaderSnapshot::from_headermap(response.headers()); - let body = response - .text() + let mut bytes = Vec::new(); + while let Some(chunk) = response + .chunk() .await - .map_err(|err| map_reqwest_error(err, self.http.config().request_timeout))?; + .map_err(|err| map_reqwest_error(err, self.http.config().request_timeout))? + { + if bytes.len().saturating_add(chunk.len()) > 2 * 1024 * 1024 { + return Err( + DecodeError::new("transcription response exceeds the 2 MiB limit").into(), + ); + } + bytes.extend_from_slice(&chunk); + } + let body = String::from_utf8(bytes) + .map_err(|_| DecodeError::new("transcription response is not UTF-8"))?; parse_json_response(status, headers, body, "OpenAI audio transcription") } } @@ -334,4 +366,73 @@ mod tests { assert!(matches!(error, LlmApiError::Configuration(_))); } + #[tokio::test(flavor = "current_thread")] + async fn custom_transport_isolated_from_deployment_headers_and_key() { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + for auth in [ + crate::RequestAuth::None, + crate::RequestAuth::Bearer("route-key"), + ] { + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = EndpointOverride::from_parts( + &format!("http://{}/custom/v1", listener.local_addr().unwrap()), + &BTreeMap::from([("x-route".into(), "speech".into())]), + ) + .unwrap(); + let server = tokio::spawn(async move { + let (mut stream, _) = listener.accept().await.unwrap(); + let mut bytes = Vec::new(); + loop { + let mut buffer = [0; 4096]; + let count = stream.read(&mut buffer).await.unwrap(); + assert!(count > 0); + bytes.extend_from_slice(&buffer[..count]); + if let Some(end) = bytes.windows(4).position(|v| v == b"\r\n\r\n") { + let headers = String::from_utf8_lossy(&bytes[..end]).to_ascii_lowercase(); + let size: usize = headers + .lines() + .find_map(|l| l.strip_prefix("content-length: ")) + .unwrap() + .parse() + .unwrap(); + if bytes.len() >= end + 4 + size { + break; + } + } + } + let body = r#"{"text":"hello"}"#; + stream.write_all(format!("HTTP/1.1 200 OK\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len()).as_bytes()).await.unwrap(); + String::from_utf8(bytes).unwrap() + }); + let mut config = Config::new("deployment-key"); + config.organization = Some("deployment-org".into()); + config.project = Some("deployment-project".into()); + let client = Client::new(config).unwrap(); + let mut request = CreateTranscriptionRequest::new(AudioFile { + bytes: b"voice".to_vec(), + filename: "voice.ogg".into(), + mime: "audio/ogg".into(), + }); + request.model = "custom-speech-model".into(); + request.language = Some("de".into()); + request.prompt = Some("Names".into()); + let result = client + .create_transcription_with_transport(request, Some(auth), Some(&endpoint)) + .await + .unwrap(); + assert_eq!(result.parsed.text, "hello"); + let request = server.await.unwrap(); + assert!(request.starts_with("POST /custom/v1/audio/transcriptions ")); + assert!(request.contains("x-route: speech")); + assert!(request.contains("custom-speech-model")); + assert!(request.contains("Names")); + assert!(!request.contains("deployment-")); + match auth { + crate::RequestAuth::None => { + assert!(!request.to_ascii_lowercase().contains("authorization:")) + } + _ => assert!(request.contains("Bearer route-key")), + } + } + } } diff --git a/crates/llm-runtime/src/anthropic_messages.rs b/crates/llm-runtime/src/anthropic_messages.rs index 0aa493b67..b7869368d 100644 --- a/crates/llm-runtime/src/anthropic_messages.rs +++ b/crates/llm-runtime/src/anthropic_messages.rs @@ -798,7 +798,7 @@ async fn materialize_block( }; return Ok((role, blocks)); } - let text = crate::blob_io::read_message_text(blobs, &entry.content).await?; + let text = crate::blob_io::read_text(blobs, &entry.content.content_ref).await?; Ok((role, vec![am::ContentBlockParam::text(text)])) } ContextEntryKind::ToolResult { call_id, is_error } => { diff --git a/crates/llm-runtime/src/blob_io.rs b/crates/llm-runtime/src/blob_io.rs index 1a6e01982..0cadb754f 100644 --- a/crates/llm-runtime/src/blob_io.rs +++ b/crates/llm-runtime/src/blob_io.rs @@ -71,26 +71,6 @@ pub fn document_entry(media_type: Option<&str>, preview: Option<&str>) -> Option }) } -/// Render a preprocessed transcript at the provider boundary. Ordinary authored -/// text stays byte-for-byte text, even when it resembles JSON or a transcript. -pub async fn read_message_text( - blobs: &dyn BlobStore, - content: &engine::ContentRef, -) -> LlmAdapterResult { - if content.provider_kind.as_deref() - == Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) - { - let raw = read_json(blobs, &content.content_ref).await?; - let transcript: llm_clients::content::AudioTranscript = serde_json::from_value(raw) - .map_err(|error| LlmAdapterError::InvalidJson { - blob_ref: content.content_ref.clone(), - message: error.to_string(), - })?; - return Ok(transcript.model_text()); - } - read_text(blobs, &content.content_ref).await -} - pub async fn read_base64(blobs: &dyn BlobStore, blob_ref: &BlobRef) -> LlmAdapterResult { use base64::Engine as _; let bytes = blobs.read_bytes(blob_ref).await?; diff --git a/crates/llm-runtime/src/lib.rs b/crates/llm-runtime/src/lib.rs index 93e22bf6f..ed8a2a5f2 100644 --- a/crates/llm-runtime/src/lib.rs +++ b/crates/llm-runtime/src/lib.rs @@ -36,7 +36,7 @@ pub use params::{ pub use provider_keys::{ ModelProviderResolver, NoStoredModelProviders, NoStoredProviderKeys, ProviderAuthScheme, ProviderKeyError, ResolvedEndpoint, ResolvedModelProvider, ResolvedProviderAuth, - StaticModelProviders, StaticProviderKeys, + StaticModelProviders, StaticProviderKeys, resolve_provider_route, }; pub use result::{LlmDebugDumps, LlmGenerationExecution, failed_generation_result}; pub use secrets::{ diff --git a/crates/llm-runtime/src/openai_completions.rs b/crates/llm-runtime/src/openai_completions.rs index 73fb69f3a..984e24ff0 100644 --- a/crates/llm-runtime/src/openai_completions.rs +++ b/crates/llm-runtime/src/openai_completions.rs @@ -688,7 +688,7 @@ async fn materialize_message( } } else { oai_c::CompletionMessageContent::Text( - crate::blob_io::read_message_text(blobs, &entry.content).await?, + crate::blob_io::read_text(blobs, &entry.content.content_ref).await?, ) }; Ok(oai_c::CompletionMessage { diff --git a/crates/llm-runtime/src/openai_responses.rs b/crates/llm-runtime/src/openai_responses.rs index 6b14eda2e..0122bca78 100644 --- a/crates/llm-runtime/src/openai_responses.rs +++ b/crates/llm-runtime/src/openai_responses.rs @@ -514,7 +514,7 @@ async fn materialize_input_item( extra: Default::default(), })); } - let text = crate::blob_io::read_message_text(blobs, &item.content).await?; + let text = crate::blob_io::read_text(blobs, &item.content.content_ref).await?; Ok(oai::ResponseInputItem::Message(oai::InputMessage { role, content: oai::InputMessageContent::Text(text), diff --git a/crates/llm-runtime/src/provider_keys.rs b/crates/llm-runtime/src/provider_keys.rs index c0a69f1b4..5084760fd 100644 --- a/crates/llm-runtime/src/provider_keys.rs +++ b/crates/llm-runtime/src/provider_keys.rs @@ -72,9 +72,8 @@ impl ResolvedEndpoint { }) } - fn supports(&self, api_kind: &ProviderApiKind) -> bool { - self.api_kinds - .contains_key(provider_api_kind_name(api_kind)) + pub fn supports(&self, api_kind: &str) -> bool { + self.api_kinds.contains_key(api_kind) } } @@ -133,42 +132,57 @@ pub(crate) async fn resolve_model_provider( resolver: &dyn ModelProviderResolver, model: &ModelSelection, ) -> Result, LlmAdapterError> { - let resolved = resolver - .resolve_model_provider(&model.provider_id) - .await - .map_err(|error| LlmAdapterError::ProviderKeyResolution { - message: error.to_string(), - })?; - if resolved.is_none() && !is_builtin_provider(&model.provider_id) { - return Err(LlmAdapterError::ProviderKeyResolution { + resolve_provider_route( + resolver, + &model.provider_id, + provider_api_kind_name(&model.api_kind), + ) + .await + .map_err(|error| LlmAdapterError::ProviderKeyResolution { + message: error.to_string(), + }) +} + +/// Resolve a protocol route without extending the engine's generation model enum. +/// Only built-in providers may use deployment transport defaults. +pub async fn resolve_provider_route( + resolver: &dyn ModelProviderResolver, + provider_id: &str, + api_kind: &str, +) -> Result, ProviderKeyError> { + let resolved = resolver.resolve_model_provider(provider_id).await?; + if resolved.is_none() && !is_builtin_provider(provider_id) { + return Err(ProviderKeyError::NotUsable { + provider_id: provider_id.into(), message: format!( "custom model provider {} has no universe model-provider record", - model.provider_id + provider_id ), }); } - if !is_builtin_provider(&model.provider_id) + if !is_builtin_provider(provider_id) && resolved .as_ref() .is_some_and(|provider| provider.endpoint.is_none()) { - return Err(LlmAdapterError::ProviderKeyResolution { + return Err(ProviderKeyError::NotUsable { + provider_id: provider_id.into(), message: format!( "custom model provider {} has no endpoint configuration", - model.provider_id + provider_id ), }); } if let Some(endpoint) = resolved .as_ref() .and_then(|provider| provider.endpoint.as_ref()) - && !endpoint.supports(&model.api_kind) + && !endpoint.supports(api_kind) { - return Err(LlmAdapterError::ProviderKeyResolution { + return Err(ProviderKeyError::NotUsable { + provider_id: provider_id.into(), message: format!( "model provider {} endpoint does not admit API kind {}", - model.provider_id, - provider_api_kind_name(&model.api_kind) + provider_id, api_kind ), }); } @@ -380,3 +394,61 @@ mod tests { )); } } + +#[cfg(test)] +mod route_tests { + use super::*; + #[tokio::test(flavor = "current_thread")] + async fn transcription_routes_require_matching_explicit_custom_endpoints() { + assert!( + resolve_provider_route( + &NoStoredModelProviders, + "openai", + "openai:audio-transcriptions" + ) + .await + .unwrap() + .is_none() + ); + assert!(matches!( + resolve_provider_route( + &NoStoredModelProviders, + "custom", + "openai:audio-transcriptions" + ) + .await, + Err(ProviderKeyError::NotUsable { .. }) + )); + let route = ResolvedModelProvider { + auth: None, + endpoint: Some( + ResolvedEndpoint::new( + "http://localhost:9999/v1", + &BTreeMap::new(), + ["openai:audio-transcriptions".into()], + ) + .unwrap(), + ), + }; + struct Resolver(ResolvedModelProvider); + #[async_trait] + impl ModelProviderResolver for Resolver { + async fn resolve_model_provider( + &self, + _: &str, + ) -> Result, ProviderKeyError> { + Ok(Some(self.0.clone())) + } + } + let resolver = Resolver(route); + let resolved = resolve_provider_route(&resolver, "custom", "openai:audio-transcriptions") + .await + .unwrap() + .unwrap(); + assert!(resolved.auth.is_none()); + assert!(matches!( + resolve_provider_route(&resolver, "custom", "openai:responses").await, + Err(ProviderKeyError::NotUsable { .. }) + )); + } +} diff --git a/crates/llm-runtime/tests/content_materialization.rs b/crates/llm-runtime/tests/content_materialization.rs index f63a622cb..43754cc9a 100644 --- a/crates/llm-runtime/tests/content_materialization.rs +++ b/crates/llm-runtime/tests/content_materialization.rs @@ -1,27 +1,20 @@ use engine::{ - BlobRef, ContentRef, ContextEntry, ContextEntryId, ContextEntryKind, ContextEntrySource, + ContentRef, ContextEntry, ContextEntryId, ContextEntryKind, ContextEntrySource, ContextMessageRole, ContextSnapshot, LlmRequest, ModelSelection, ProviderApiKind, storage::{BlobStore, InMemoryBlobStore}, }; -use llm_clients::content::{AUDIO_TRANSCRIPT_PROVIDER_KIND, AudioTranscript}; use serde_json::{Value, json}; #[tokio::test(flavor = "current_thread")] -async fn structured_transcripts_and_authored_json_lower_across_all_provider_apis() { +async fn text_with_provenance_is_unchanged_across_all_provider_apis() { let blobs = InMemoryBlobStore::new(); - let transcript = AudioTranscript { - filename: "voice.ogg".into(), - text: "[audio transcript: quoted]\nKeep these spoken words.".into(), - }; - let bytes = serde_json::to_vec(&transcript).unwrap(); - let reference = blobs.put_bytes(bytes.clone()).await.unwrap(); let source = blobs.put_bytes(b"source audio".to_vec()).await.unwrap(); - for structured in [true, false] { - let expected = if structured { - transcript.model_text() - } else { - String::from_utf8(bytes.clone()).unwrap() - }; + for expected in [ + "[audio transcript: quoted]\nKeep these spoken words.", + r#"{"filename":"voice.ogg","text":"ordinary authored JSON"}"#, + ] { + let bytes = expected.as_bytes().to_vec(); + let reference = blobs.put_bytes(bytes.clone()).await.unwrap(); for api_kind in [ ProviderApiKind::OpenAiResponses, ProviderApiKind::OpenAiCompletions, @@ -36,10 +29,10 @@ async fn structured_transcripts_and_authored_json_lower_across_all_provider_apis source: ContextEntrySource::ContextEdit, content: ContentRef { content_ref: reference.clone(), - media_type: Some("application/json".into()), - provider_kind: structured.then(|| AUDIO_TRANSCRIPT_PROVIDER_KIND.to_owned()), + media_type: Some("text/plain".into()), + provider_kind: None, }, - preview: structured.then(|| transcript.header()), + preview: None, origin: None, provenance_ref: Some(source.clone()), token_estimate: None, @@ -103,5 +96,4 @@ async fn structured_transcripts_and_authored_json_lower_across_all_provider_apis assert_eq!(blobs.read_bytes(&reference).await.unwrap(), bytes); } } - assert_eq!(reference, BlobRef::from_bytes(&bytes)); } diff --git a/crates/store-pg/migrations/010_model_defaults.sql b/crates/store-pg/migrations/010_model_defaults.sql new file mode 100644 index 000000000..38f3e3dec --- /dev/null +++ b/crates/store-pg/migrations/010_model_defaults.sql @@ -0,0 +1,8 @@ +-- Defaults are universe policy; provider transport and credentials stay in +-- auth_providers. Retain a row after clears to preserve its revision. +CREATE TABLE universe_model_defaults ( + universe_id uuid PRIMARY KEY REFERENCES universes (universe_id) ON DELETE CASCADE, + revision bigint NOT NULL CHECK (revision > 0), + agent_run jsonb, + speech_to_text jsonb +); diff --git a/crates/store-pg/src/api_keys.rs b/crates/store-pg/src/api_keys.rs index 6a5d3cd13..3943ae272 100644 --- a/crates/store-pg/src/api_keys.rs +++ b/crates/store-pg/src/api_keys.rs @@ -2,7 +2,7 @@ //! //! The store uses the deployment pool because authentication precedes //! universe resolution. A key's authority is its row: scope, groups and the -//! actor flag. Keys change only by revocation. +//! actor flag. Rotation replaces only the secret and its display prefix. use std::collections::BTreeSet; @@ -149,6 +149,46 @@ impl PgApiKeyStore { .collect() } + /// Replace an active key's secret atomically. The old prefix stops identifying + /// a key, so concurrent rotations cannot both succeed. Revoked keys stay revoked. + pub async fn rotate_api_key( + &self, + key_prefix: &str, + ) -> Result, ApiKeyError> { + for _ in 0..3 { + let secret = auth::generate_prefixed_secret(auth::API_KEY_SECRET_PREFIX); + let key_hash = auth::api_key_hash(&secret); + let new_prefix = auth::api_key_display_prefix(&secret); + if new_prefix == key_prefix { + continue; + } + let result = sqlx::query(&format!( + "UPDATE api_keys SET key_hash = $2, key_prefix = $3, last_used_at_ms = NULL + WHERE key_prefix = $1 AND revoked_at_ms IS NULL RETURNING {KEY_COLUMNS}" + )) + .bind(key_prefix) + .bind(&key_hash) + .bind(&new_prefix) + .fetch_optional(&self.pool) + .await; + match result { + Ok(Some(row)) => { + return Ok(Some(auth::MintedApiKey { + secret: auth::SecretValue::new(secret), + key_hash, + record: record_from_row(&row)?, + })); + } + Ok(None) => return Ok(None), + Err(sqlx::Error::Database(error)) if error.is_unique_violation() => continue, + Err(error) => return Err(map_sqlx_error(error)), + } + } + Err(ApiKeyError::Store { + message: "could not allocate a unique api key prefix".into(), + }) + } + /// Revoke a key by its display prefix. `None` for an unknown prefix; a /// revoked key stays revoked at its first revocation time. pub async fn revoke_api_key( diff --git a/crates/store-pg/src/bots.rs b/crates/store-pg/src/bots.rs index 7dcf5bd50..1ea875d51 100644 --- a/crates/store-pg/src/bots.rs +++ b/crates/store-pg/src/bots.rs @@ -981,6 +981,12 @@ impl BotEventStore for PgStore { let mut refs = std::collections::BTreeSet::from([record.document_ref.clone()]); refs.extend(record.prompt_ref.iter().cloned()); refs.extend(record.media.iter().map(|media| media.blob_ref.clone())); + refs.extend( + record + .media + .iter() + .filter_map(|media| media.text_ref.clone()), + ); refs.extend( record .receiver diff --git a/crates/store-pg/src/lib.rs b/crates/store-pg/src/lib.rs index c52f66e1f..73ab8cdc7 100644 --- a/crates/store-pg/src/lib.rs +++ b/crates/store-pg/src/lib.rs @@ -16,6 +16,7 @@ mod environment; mod environment_registration; mod mcp; mod migrations; +mod model_defaults; mod oauth; mod object; mod profile; @@ -36,6 +37,7 @@ use uuid::Uuid; pub use access::AccessFilter; pub use access::{AccessStoreError, PgAccessStore, ResourceAccess}; +pub use model_defaults::ModelDefaultsStoreError; /// A session page with the access summary each view carries. #[derive(Clone, Debug)] diff --git a/crates/store-pg/src/migrations.rs b/crates/store-pg/src/migrations.rs index d6632b55d..a8ac3f92b 100644 --- a/crates/store-pg/src/migrations.rs +++ b/crates/store-pg/src/migrations.rs @@ -44,6 +44,7 @@ const LIGHTSPEED_TABLES: &[&str] = &[ "session_checkpoints", "session_events", "sessions", + "universe_model_defaults", "universes", "vfs_snapshots", "vfs_workspaces", @@ -102,9 +103,14 @@ pub const MIGRATIONS: &[EmbeddedMigration] = &[ name: "channels", sql: include_str!("../migrations/009_channels.sql"), }, + EmbeddedMigration { + version: 10, + name: "model_defaults", + sql: include_str!("../migrations/010_model_defaults.sql"), + }, ]; -pub const REQUIRED_SCHEMA_REVISION: i64 = 9; +pub const REQUIRED_SCHEMA_REVISION: i64 = 10; #[derive(Clone, Debug, PartialEq, Eq)] pub struct SchemaStatus { diff --git a/crates/store-pg/src/model_defaults.rs b/crates/store-pg/src/model_defaults.rs new file mode 100644 index 000000000..f2eaf9dbd --- /dev/null +++ b/crates/store-pg/src/model_defaults.rs @@ -0,0 +1,90 @@ +use api::{ModelDefaultSlot, ModelDefaults, ModelDefaultsPutParams}; +use sqlx::Row; +use thiserror::Error; + +use crate::PgStore; + +#[derive(Debug, Error)] +pub enum ModelDefaultsStoreError { + #[error("model defaults revision conflict (expected {expected})")] + Conflict { expected: u64 }, + #[error(transparent)] + Invalid(#[from] api::AgentApiError), + #[error("model defaults storage failure: {0}")] + Postgres(#[from] sqlx::Error), + #[error("invalid stored model defaults: {0}")] + Decode(#[from] serde_json::Error), +} + +impl PgStore { + pub async fn read_model_defaults(&self) -> Result { + let row = sqlx::query("SELECT revision, agent_run, speech_to_text FROM universe_model_defaults WHERE universe_id = $1") + .bind(self.config.universe_id) + .fetch_optional(&self.pool) + .await?; + row.as_ref() + .map(decode) + .transpose() + .map(Option::unwrap_or_default) + } + + pub async fn put_model_defaults( + &self, + params: ModelDefaultsPutParams, + ) -> Result { + if let Some(model) = ¶ms.model { + params.slot.validate_model(model)?; + } + let expected = i64::try_from(params.expected_revision) + .ok() + .filter(|value| *value < i64::MAX) + .ok_or_else(|| { + api::AgentApiError::invalid_request("expectedRevision is out of range") + })?; + let model = params + .model + .as_ref() + .map(serde_json::to_value) + .transpose()?; + // An untouched universe uses a unique insert. Later writes use a + // revision guard; neither path can overwrite a concurrent winner. + let statement = if expected == 0 { + "INSERT INTO universe_model_defaults (universe_id, revision, agent_run, speech_to_text) + SELECT $1, 1, CASE WHEN $2 THEN $3::jsonb END, CASE WHEN NOT $2 THEN $3::jsonb END + WHERE $4::bigint = 0 + ON CONFLICT (universe_id) DO NOTHING + RETURNING revision, agent_run, speech_to_text" + } else { + "UPDATE universe_model_defaults SET revision = revision + 1, + agent_run = CASE WHEN $2 THEN $3::jsonb ELSE agent_run END, + speech_to_text = CASE WHEN NOT $2 THEN $3::jsonb ELSE speech_to_text END + WHERE universe_id = $1 AND revision = $4 + RETURNING revision, agent_run, speech_to_text" + }; + let row = sqlx::query(statement) + .bind(self.config.universe_id) + .bind(params.slot == ModelDefaultSlot::AgentRun) + .bind(model) + .bind(expected) + .fetch_optional(&self.pool) + .await? + .ok_or(ModelDefaultsStoreError::Conflict { + expected: params.expected_revision, + })?; + decode(&row) + } +} + +fn decode(row: &sqlx::postgres::PgRow) -> Result { + Ok(ModelDefaults { + revision: row.try_get::("revision")? as u64, + agent_run: row + .try_get::, _>("agent_run")? + .map(serde_json::from_value) + .transpose()?, + speech_to_text: row + .try_get::, _>("speech_to_text")? + .map(serde_json::from_value) + .transpose()?, + }) +} diff --git a/crates/store-pg/tests/api_keys_live.rs b/crates/store-pg/tests/api_keys_live.rs index b6a25e5e0..2785e95de 100644 --- a/crates/store-pg/tests/api_keys_live.rs +++ b/crates/store-pg/tests/api_keys_live.rs @@ -211,6 +211,92 @@ async fn exercise(pool: &sqlx::PgPool) { .is_none() ); + // Rotation updates the same row, preserving authority and creation metadata. + let rotated = api_keys + .rotate_api_key(&right_key.record.key_prefix) + .await + .expect("rotate") + .expect("active key"); + let mut expected = right_key.record.clone(); + expected.key_prefix = rotated.record.key_prefix.clone(); + assert_eq!(rotated.record, expected); + assert_ne!(rotated.record.key_prefix, right_key.record.key_prefix); + assert_ne!(rotated.secret.expose(), right_key.secret.expose()); + assert_eq!( + rotated.key_hash, + auth::api_key_hash(rotated.secret.expose()) + ); + assert!( + api_keys + .resolve_api_key(&right_key.key_hash, later) + .await + .unwrap() + .is_none() + ); + assert_eq!( + api_keys + .resolve_api_key(&rotated.key_hash, later) + .await + .unwrap(), + Some(expected) + ); + assert_eq!(api_keys.list_api_keys(Some(right)).await.unwrap().len(), 1); + assert!( + api_keys + .rotate_api_key(&right_key.record.key_prefix) + .await + .unwrap() + .is_none() + ); + assert!( + api_keys + .rotate_api_key(&left_key.record.key_prefix) + .await + .unwrap() + .is_none() + ); + assert!( + api_keys + .rotate_api_key("lsk_unknown00") + .await + .unwrap() + .is_none() + ); + + // Only one concurrent rotation can replace a given secret. + let (first, second) = tokio::join!( + api_keys.rotate_api_key(&rotated.record.key_prefix), + api_keys.rotate_api_key(&rotated.record.key_prefix), + ); + let winners: Vec<_> = [first.unwrap(), second.unwrap()] + .into_iter() + .flatten() + .collect(); + assert_eq!(winners.len(), 1); + assert!( + api_keys + .resolve_api_key(&rotated.key_hash, later) + .await + .unwrap() + .is_none() + ); + assert!( + api_keys + .resolve_api_key(&winners[0].key_hash, later) + .await + .unwrap() + .is_some() + ); + + let rotated_gate = api_keys + .rotate_api_key(&gate_key.record.key_prefix) + .await + .unwrap() + .unwrap(); + let mut expected_gate = gate_key.record.clone(); + expected_gate.key_prefix = rotated_gate.record.key_prefix.clone(); + assert_eq!(rotated_gate.record, expected_gate); + // Deleting a universe removes its keys. store_pg::delete_universe(pool, left_universe) .await @@ -220,13 +306,13 @@ async fn exercise(pool: &sqlx::PgPool) { .expect("delete right universe"); assert!( api_keys - .resolve_api_key(&right_key.key_hash, later) + .resolve_api_key(&winners[0].key_hash, later) .await .expect("resolve after delete") .is_none() ); api_keys - .revoke_api_key(&gate_key.record.key_prefix, later) + .revoke_api_key(&rotated_gate.record.key_prefix, later) .await .expect("revoke gate key"); } diff --git a/crates/store-pg/tests/bots_pg.rs b/crates/store-pg/tests/bots_pg.rs index 97361b5e9..0d45c2826 100644 --- a/crates/store-pg/tests/bots_pg.rs +++ b/crates/store-pg/tests/bots_pg.rs @@ -1253,6 +1253,10 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs .put_bytes(b"media attachment".to_vec()) .await .expect("media"); + let transcript = store + .put_bytes(b"prepared transcript".to_vec()) + .await + .expect("transcript"); let mut record = event(&bot_id, "with-tools", 1, 2, None, None); record.receiver = Some(bots::EventReceiver::Workflow { workflow_id: "conversation".to_owned(), @@ -1261,9 +1265,10 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs tools_ref: Some(tools.to_string()), }); record.media = vec![api::BotEventMedia { + text_ref: Some(transcript.to_string()), blob_ref: media.to_string(), - kind: api::BotEventMediaKind::Image, - mime: "image/png".to_owned(), + kind: api::BotEventMediaKind::Audio, + mime: "audio/ogg".to_owned(), name: None, }]; store @@ -1279,8 +1284,8 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs .await .expect("roots"); assert_eq!( - roots, 4, - "document, prompt, media, and receiver tools are retained" + roots, 5, + "document, prompt, source audio, transcript, and receiver tools are retained" ); sqlx::query("UPDATE cas_blobs SET created_at_ms = 1, touched_at_ms = 1 WHERE universe_id = $1") .bind(store.config().universe_id) @@ -1291,6 +1296,7 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs &record.document_ref, record.prompt_ref.as_ref().unwrap(), &media.to_string(), + &transcript.to_string(), &tools.to_string(), ] .into_iter() @@ -1342,7 +1348,7 @@ async fn pg_live_bot_event_roots_cover_receiver_tools_and_roll_back_missing_refs .await .expect("sweep released roots") .len(), - 4 + 5 ); drop_universe(&store).await; } diff --git a/crates/store-pg/tests/model_defaults_live.rs b/crates/store-pg/tests/model_defaults_live.rs new file mode 100644 index 000000000..eddc318b4 --- /dev/null +++ b/crates/store-pg/tests/model_defaults_live.rs @@ -0,0 +1,157 @@ +use futures_util::FutureExt as _; +use sqlx::{ + Executor as _, + postgres::{PgConnectOptions, PgPoolOptions}, +}; +use std::{panic::AssertUnwindSafe, str::FromStr}; +use store_pg::{PgStore, PgStoreConfig}; +use uuid::Uuid; + +#[tokio::test(flavor = "current_thread")] +#[ignore = "requires local LIGHTSPEED_TEST_POSTGRES_URL; isolated schema"] +async fn universe_defaults_are_revision_safe_and_isolated() { + let database_url = std::env::var("LIGHTSPEED_TEST_POSTGRES_URL") + .expect("LIGHTSPEED_TEST_POSTGRES_URL must be set; run ./dev.sh infra and source scripts/dev/env.sh"); + let admin = PgPoolOptions::new() + .max_connections(1) + .connect(&database_url) + .await + .expect("connect to live Postgres"); + let schema = format!("lightspeed_model_defaults_test_{}", Uuid::new_v4().simple()); + admin + .execute(format!("CREATE SCHEMA \"{schema}\"").as_str()) + .await + .expect("create isolated schema"); + let pool = PgPoolOptions::new() + .max_connections(2) + .connect_with( + PgConnectOptions::from_str(&database_url) + .expect("parse Postgres URL") + .options([("search_path", schema.as_str())]), + ) + .await + .expect("connect to isolated schema"); + let outcome = AssertUnwindSafe(exercise(&pool)).catch_unwind().await; + pool.close().await; + admin + .execute(format!("DROP SCHEMA \"{schema}\" CASCADE").as_str()) + .await + .expect("drop isolated schema"); + admin.close().await; + if let Err(panic) = outcome { + std::panic::resume_unwind(panic); + } +} + +async fn exercise(pool: &sqlx::PgPool) { + use api::{ModelConfig, ModelDefaultSlot as Slot, ModelDefaultsPutParams as Put}; + use store_pg::ModelDefaultsStoreError; + PgStore::migrate(pool).await.unwrap(); + let first = PgStore::new(pool.clone(), PgStoreConfig::new(Uuid::new_v4())); + let second = PgStore::new(pool.clone(), PgStoreConfig::new(Uuid::new_v4())); + first.ensure_universe().await.unwrap(); + second.ensure_universe().await.unwrap(); + assert_eq!( + first.read_model_defaults().await.unwrap(), + api::ModelDefaults::default() + ); + let route = |provider: &str, kind: &str| ModelConfig { + provider_id: provider.into(), + api_kind: kind.into(), + model: "private-model".into(), + }; + let request = Put { + slot: Slot::AgentRun, + model: Some(route("anthropic", "anthropic:messages")), + expected_revision: 0, + }; + let (left, right) = tokio::join!( + first.put_model_defaults(request.clone()), + first.put_model_defaults(request) + ); + let winner = match (left, right) { + (Ok(value), Err(ModelDefaultsStoreError::Conflict { expected: 0 })) + | (Err(ModelDefaultsStoreError::Conflict { expected: 0 }), Ok(value)) => value, + other => panic!("exactly one first writer must win: {other:?}"), + }; + assert_eq!(winner.revision, 1); + let other = second + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: Some(route("local", "openai:completions")), + expected_revision: 0, + }) + .await + .unwrap(); + let speech = first + .put_model_defaults(Put { + slot: Slot::SpeechToText, + model: Some(route("speech", "openai:audio-transcriptions")), + expected_revision: 1, + }) + .await + .unwrap(); + assert_eq!(speech.agent_run, winner.agent_run); + assert!(speech.speech_to_text.is_some()); + assert_eq!(second.read_model_defaults().await.unwrap(), other); + + let stale = first + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: None, + expected_revision: 1, + }) + .await + .unwrap_err(); + assert!(matches!( + stale, + ModelDefaultsStoreError::Conflict { expected: 1 } + )); + let invalid = first + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: speech.speech_to_text.clone(), + expected_revision: 2, + }) + .await + .unwrap_err(); + assert!( + matches!(invalid, ModelDefaultsStoreError::Invalid(error) if error.kind == api::AgentApiErrorKind::InvalidRequest) + ); + assert_eq!(first.read_model_defaults().await.unwrap(), speech); + + let cleared = first + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: None, + expected_revision: 2, + }) + .await + .unwrap(); + assert_eq!(cleared.revision, 3); + assert_eq!(cleared.agent_run, None); + assert_eq!(cleared.speech_to_text, speech.speech_to_text); + let retry_seed = first + .put_model_defaults(Put { + slot: Slot::AgentRun, + model: winner.agent_run, + expected_revision: 0, + }) + .await + .unwrap_err(); + assert!(matches!( + retry_seed, + ModelDefaultsStoreError::Conflict { expected: 0 } + )); + assert_eq!(first.read_model_defaults().await.unwrap(), cleared); + sqlx::query("DELETE FROM universes WHERE universe_id = $1") + .bind(first.config().universe_id) + .execute(pool) + .await + .unwrap(); + assert_eq!( + first.read_model_defaults().await.unwrap(), + api::ModelDefaults::default() + ); + assert_eq!(second.read_model_defaults().await.unwrap(), other); +} diff --git a/crates/temporal-server/src/bots/sessions.rs b/crates/temporal-server/src/bots/sessions.rs index 1509f0a1f..b8f62199a 100644 --- a/crates/temporal-server/src/bots/sessions.rs +++ b/crates/temporal-server/src/bots/sessions.rs @@ -672,6 +672,7 @@ pub async fn append_context( .map(|event| ContextAppendEntry { key: appended_event_context_key(&event.id), item: InputItem::TextRef { + provenance_ref: None, origin: Some("event".to_owned()), blob_ref: event .prompt_ref diff --git a/crates/temporal-server/src/channels/activities.rs b/crates/temporal-server/src/channels/activities.rs index 8b850d6b6..d7dd68544 100644 --- a/crates/temporal-server/src/channels/activities.rs +++ b/crates/temporal-server/src/channels/activities.rs @@ -664,9 +664,34 @@ pub async fn emit_chat_event( &request.conversation.key(), &request.message.message_id, ); - let mut input = StoreBotEventInput::new(event_id.clone(), chat_message_document(&request)); + let original_summary = chat_message_document(&request).summary; + let original_text = request.message.text.clone(); + let mut prepared_media = Vec::with_capacity(request.media.len()); + for item in &request.media { + let mut prepared = BotEventMedia::from(item.clone()); + if let Some(reference) = request.transcript_refs.get(&item.blob_ref) { + let text = store + .read_text(&parse_blob_ref(reference)?) + .await + .map_err(|e| blob_error("read transcript", e))?; + if !request.message.text.is_empty() { + request.message.text.push_str("\n\n"); + } + request.message.text.push_str(&text); + prepared.text_ref = Some(reference.clone()); + } + prepared_media.push(prepared); + } + let mut document = chat_message_document(&request); + document.summary = original_summary; + if !request.transcript_refs.is_empty() + && let Some(data) = document.data.as_mut() + { + data["message"]["originalText"] = serde_json::Value::String(original_text); + } + let mut input = StoreBotEventInput::new(event_id.clone(), document); input.prompt_data = Some(chat_prompt_data(&request.media)); - input.media = request.media.into_iter().map(BotEventMedia::from).collect(); + input.media = prepared_media; input.receiver = Some(EventReceiver::Workflow { workflow_id: request.notify.workflow_id, workflow_kind: request.notify.workflow_kind, @@ -913,6 +938,7 @@ mod tests { fn emit_request(text: &str, media: Vec) -> ChatEmitEventRequest { ChatEmitEventRequest { + transcript_refs: Default::default(), universe_id: Uuid::nil(), bot_id: BotId::new("triage"), trigger_id: BotTriggerId::new("tg"), diff --git a/crates/temporal-server/src/config.rs b/crates/temporal-server/src/config.rs index f0545b483..1cc633425 100644 --- a/crates/temporal-server/src/config.rs +++ b/crates/temporal-server/src/config.rs @@ -1,21 +1,29 @@ use std::{env, sync::Arc, time::Duration}; -use engine::{ModelSelection, ProviderApiKind}; use object_store::ObjectStore; use sqlx::{PgPool, postgres::PgPoolOptions}; use store_pg::{ BlobCache, PgStore, PgStoreConfig, PgStoreError, S3ObjectStoreConfig, SchemaStatus, SecretsMasterKey, build_s3_object_store, }; -use temporal_workflow::{DEFAULT_MODEL, DEFAULT_TASK_QUEUE, bots::DEFAULT_BOTS_TASK_QUEUE}; +use temporal_workflow::{DEFAULT_TASK_QUEUE, bots::DEFAULT_BOTS_TASK_QUEUE}; use uuid::Uuid; -pub fn default_model_from_env() -> ModelSelection { - ModelSelection { - api_kind: ProviderApiKind::OpenAiResponses, - provider_id: env::var("LIGHTSPEED_CHAT_PROVIDER").unwrap_or_else(|_| "openai".to_owned()), - model: env::var("LIGHTSPEED_CHAT_MODEL").unwrap_or_else(|_| DEFAULT_MODEL.to_owned()), +/// Retired deployment model policy must be migrated to universe defaults. +/// Credentials and provider transport still use their existing configuration. +pub fn validate_model_environment() -> anyhow::Result<()> { + validate_model_environment_with(|name| env::var_os(name).is_some()) +} + +fn validate_model_environment_with(is_set: impl Fn(&str) -> bool) -> anyhow::Result<()> { + for name in ["LIGHTSPEED_CHAT_PROVIDER", "LIGHTSPEED_CHAT_MODEL"] { + if is_set(name) { + anyhow::bail!( + "{name} is retired; remove it and configure each universe with `lightspeed model defaults set agent-run --provider --api-kind --model `" + ); + } } + Ok(()) } pub fn universe_id_from_env() -> anyhow::Result { @@ -413,6 +421,16 @@ fn optional_env(key: &str) -> Option { mod tests { use super::*; + #[test] + fn retired_model_environment_is_rejected_with_a_configuration_action() { + validate_model_environment_with(|_| false).unwrap(); + for name in ["LIGHTSPEED_CHAT_MODEL", "LIGHTSPEED_CHAT_PROVIDER"] { + let error = validate_model_environment_with(|candidate| candidate == name).unwrap_err(); + assert!(error.to_string().contains(name)); + assert!(error.to_string().contains("model defaults set")); + } + } + #[test] fn unledgered_schema_bypass_is_narrow() { let relations = vec!["sessions".to_owned(), "universes".to_owned()]; diff --git a/crates/temporal-server/src/gateway/deployment.rs b/crates/temporal-server/src/gateway/deployment.rs index 24e386bbe..698101186 100644 --- a/crates/temporal-server/src/gateway/deployment.rs +++ b/crates/temporal-server/src/gateway/deployment.rs @@ -17,8 +17,8 @@ use api::AccessScope; use api::{ AgentApiError, AgentApiOutcome, DeploymentApiKeyCreateParams, DeploymentApiKeyCreateResponse, DeploymentApiKeyListParams, DeploymentApiKeyListResponse, DeploymentApiKeyRevokeParams, - DeploymentApiKeyRevokeResponse, DeploymentApiKeyView, DeploymentApiService, - DeploymentEnvironmentAdoptParams, DeploymentEnvironmentAdoptResponse, + DeploymentApiKeyRevokeResponse, DeploymentApiKeyRotateParams, DeploymentApiKeyView, + DeploymentApiService, DeploymentEnvironmentAdoptParams, DeploymentEnvironmentAdoptResponse, DeploymentEnvironmentProviderConnection, DeploymentEnvironmentProviderDeleteParams, DeploymentEnvironmentProviderDeleteResponse, DeploymentEnvironmentProviderListParams, DeploymentEnvironmentProviderListResponse, DeploymentEnvironmentProviderPutParams, @@ -651,6 +651,30 @@ impl DeploymentApiService for GatewayDeploymentApi { .await } + async fn rotate_api_key( + &self, + params: DeploymentApiKeyRotateParams, + ) -> Result, AgentApiError> { + self.admitted(api::METHOD_DEPLOYMENT_API_KEYS_ROTATE, async { + let key_prefix = params.key_prefix.trim(); + if key_prefix.is_empty() { + return Err(AgentApiError::invalid_request( + "api key keyPrefix must not be empty", + )); + } + let rotated = store_pg::PgApiKeyStore::new(self.pool().clone()) + .rotate_api_key(key_prefix) + .await + .map_err(map_api_key_error)? + .ok_or_else(|| AgentApiError::not_found("unknown or revoked api key prefix"))?; + Ok(AgentApiOutcome::new(DeploymentApiKeyCreateResponse { + api_key: api_key_view(rotated.record), + secret: rotated.secret.expose().to_owned(), + })) + }) + .await + } + async fn revoke_api_key( &self, params: DeploymentApiKeyRevokeParams, diff --git a/crates/temporal-server/src/gateway/mod.rs b/crates/temporal-server/src/gateway/mod.rs index e8ecd246e..29d320d88 100644 --- a/crates/temporal-server/src/gateway/mod.rs +++ b/crates/temporal-server/src/gateway/mod.rs @@ -8,7 +8,7 @@ pub mod registration; pub mod request_context; pub(crate) mod service; -pub use crate::config::{default_model_from_env, pg_store_from_env}; +pub use crate::config::pg_store_from_env; pub use deployment::GatewayDeploymentApi; pub use http::{ ACTOR_HEADER, DEFAULT_GATEWAY_BIND, DEFAULT_MAX_REQUEST_BODY_BYTES, GatewayRoutes, @@ -21,6 +21,6 @@ pub use service::{ }; pub use temporal_workflow::{ AgentAdmission, AgentAdmissionFailure, AgentAdmissionFailureKind, AgentCompletedRunSummary, - AgentSessionArgs, AgentSessionStatus, AgentSessionWorkflow, DEFAULT_MODEL, DEFAULT_TASK_QUEUE, + AgentSessionArgs, AgentSessionStatus, AgentSessionWorkflow, DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, connect_temporal, default_session_config, }; diff --git a/crates/temporal-server/src/gateway/service/api_config.rs b/crates/temporal-server/src/gateway/service/api_config.rs index d4260c370..68fda99dd 100644 --- a/crates/temporal-server/src/gateway/service/api_config.rs +++ b/crates/temporal-server/src/gateway/service/api_config.rs @@ -5,10 +5,15 @@ impl GatewayAgentApi { &self, api_config: Option, ) -> Result { - let config = engine_session_config_from_api( - api_config.unwrap_or_default(), - self.default_model.clone(), - )?; + let api_config = api_config.unwrap_or_default(); + let model = model_defaults::creation_model(api_config.model.clone(), async { + self.store + .read_model_defaults() + .await + .map_err(model_defaults::map_store_error) + }) + .await?; + let config = engine_session_config_from_api(api_config, model)?; config .validate() .map_err(|error| AgentApiError::invalid_request(error.to_string()))?; @@ -67,14 +72,15 @@ pub(super) fn run_config_for_start( } /// Translate the wire config document into the engine document. An absent -/// `model` falls back to the deployment default; everything else maps 1:1. +/// `model` uses the caller's resolved creation model or current session model; +/// everything else maps 1:1. pub(super) fn engine_session_config_from_api( api_config: api::SessionConfig, - default_model: ModelSelection, + resolved_model: ModelSelection, ) -> Result { let model = match api_config.model { Some(model) => model_selection_from_api(model)?, - None => default_model, + None => resolved_model, }; let generation = generation_from_api(api_config.generation, &model)?; Ok(SessionConfig { diff --git a/crates/temporal-server/src/gateway/service/errors.rs b/crates/temporal-server/src/gateway/service/errors.rs index 265f812ce..5bd2800e6 100644 --- a/crates/temporal-server/src/gateway/service/errors.rs +++ b/crates/temporal-server/src/gateway/service/errors.rs @@ -19,27 +19,6 @@ pub(super) fn map_admission_failure_to_api_error(failure: &AgentAdmissionFailure AgentAdmissionFailureKind::RejectedCommand => { AgentApiError::rejected(failure.message.clone()) } - AgentAdmissionFailureKind::UnsupportedAudioMime => { - AgentApiError::unsupported_audio_mime(failure.message.clone()) - } - AgentAdmissionFailureKind::AudioBlobMissing => { - AgentApiError::invalid_request(failure.message.clone()) - } - AgentAdmissionFailureKind::AudioBlobTooLarge => { - AgentApiError::audio_blob_too_large(failure.message.clone()) - } - AgentAdmissionFailureKind::AudioDurationTooLong => { - AgentApiError::audio_duration_too_long(failure.message.clone()) - } - AgentAdmissionFailureKind::TranscoderUnavailable => { - AgentApiError::transcoder_unavailable(failure.message.clone()) - } - AgentAdmissionFailureKind::TranscodeFailure => { - AgentApiError::transcode_failure(failure.message.clone()) - } - AgentAdmissionFailureKind::TranscriptionFailure => { - AgentApiError::transcription_failure(failure.message.clone()) - } } } diff --git a/crates/temporal-server/src/gateway/service/input.rs b/crates/temporal-server/src/gateway/service/input.rs index 69b4e19c8..5045abad5 100644 --- a/crates/temporal-server/src/gateway/service/input.rs +++ b/crates/temporal-server/src/gateway/service/input.rs @@ -2,19 +2,6 @@ use super::*; /// Images and documents, bounded per run. const ALLOWED_IMAGE_MIMES: &[&str] = &["image/jpeg", "image/png", "image/webp", "image/gif"]; -/// Bounded audio blobs are accepted at admission, then rewritten by -/// workflow preprocessing before core planning. -const ALLOWED_AUDIO_MIMES: &[&str] = &[ - "audio/mpeg", - "audio/mp4", - "audio/wav", - "audio/webm", - "audio/ogg", - "audio/aac", - "audio/amr", - "audio/3gpp", - "audio/3gpp2", -]; /// PDF is the only document type both providers accept natively; the text /// MIMEs are inlined as text by the llm-runtime adapters. const PDF_MIME: &str = "application/pdf"; @@ -25,7 +12,6 @@ const TEXT_DOCUMENT_MIMES: &[&str] = &[ "application/json", ]; const MAX_IMAGE_BYTES: u64 = 10 * 1024 * 1024; -const MAX_AUDIO_BYTES: u64 = 25 * 1024 * 1024; const MAX_PDF_BYTES: u64 = 10 * 1024 * 1024; /// Text documents land in model context verbatim; keep them small. const MAX_TEXT_DOCUMENT_BYTES: u64 = 1024 * 1024; @@ -39,6 +25,7 @@ pub(super) async fn run_input_from_api( let mut media_items = 0usize; for item in input { let origin = input_origin_from_api(item)?; + let provenance_ref = input_provenance_from_api(store, item).await?; let first = entries.len(); match item { InputItem::Text { text, .. } => { @@ -87,6 +74,7 @@ pub(super) async fn run_input_from_api( } for entry in &mut entries[first..] { entry.origin = origin.clone(); + entry.provenance_ref = provenance_ref.clone(); } } @@ -103,17 +91,12 @@ async fn media_message_input( kind: MediaKind, name: Option<&str>, ) -> Result { - let raw_mime = mime.trim().to_ascii_lowercase(); - let mime = if matches!(kind, MediaKind::Audio) { - normalize_audio_mime(&raw_mime) - } else { - raw_mime - .split(';') - .next() - .unwrap_or_default() - .trim() - .to_owned() - }; + let mime = mime + .split(';') + .next() + .unwrap_or_default() + .trim() + .to_ascii_lowercase(); let (label, max_bytes) = match kind { MediaKind::Image => { if !ALLOWED_IMAGE_MIMES.contains(&mime.as_str()) { @@ -125,13 +108,9 @@ async fn media_message_input( ("image", MAX_IMAGE_BYTES) } MediaKind::Audio => { - if !ALLOWED_AUDIO_MIMES.contains(&mime.as_str()) { - return Err(AgentApiError::unsupported_audio_mime(format!( - "unsupported audio mime type {mime}; allowed: {}", - ALLOWED_AUDIO_MIMES.join(", ") - ))); - } - ("audio", MAX_AUDIO_BYTES) + return Err(AgentApiError::invalid_request( + "audio must be transcribed with transcriptions/start before session admission; submit text or a text reference", + )); } MediaKind::Document if mime == PDF_MIME => ("document", MAX_PDF_BYTES), MediaKind::Document if TEXT_DOCUMENT_MIMES.contains(&mime.as_str()) => { @@ -150,17 +129,10 @@ async fn media_message_input( .await .map_err(map_input_blob_store_error)?; if info.byte_len > max_bytes { - return if matches!(kind, MediaKind::Audio) { - Err(AgentApiError::audio_blob_too_large(format!( - "{label} blob is {} bytes; the limit is {max_bytes} bytes", - info.byte_len - ))) - } else { - Err(AgentApiError::invalid_request(format!( - "{label} blob is {} bytes; the limit is {max_bytes} bytes", - info.byte_len - ))) - }; + return Err(AgentApiError::invalid_request(format!( + "{label} blob is {} bytes; the limit is {max_bytes} bytes", + info.byte_len + ))); } if matches!(kind, MediaKind::Document) && mime != PDF_MIME { // Text documents reach the model as text; reject undecodable bytes @@ -244,6 +216,7 @@ pub(super) async fn context_entry_input_from_api( } }?; entry.origin = origin; + entry.provenance_ref = input_provenance_from_api(store, item).await?; Ok(entry) } @@ -288,26 +261,6 @@ pub(super) fn empty_run_input_error() -> AgentApiError { ) } -fn normalize_audio_mime(mime: &str) -> String { - let mime = mime - .split(';') - .next() - .unwrap_or_default() - .trim() - .to_ascii_lowercase(); - match mime.as_str() { - "audio/mp3" => "audio/mpeg", - "audio/x-m4a" | "audio/m4a" => "audio/mp4", - "audio/x-wav" | "audio/wave" | "audio/vnd.wave" => "audio/wav", - "audio/oga" | "audio/opus" => "audio/ogg", - "audio/x-aac" => "audio/aac", - "audio/3gp" => "audio/3gpp", - "audio/3gpp2" | "audio/3g2" => "audio/3gpp2", - other => other, - } - .to_owned() -} - fn input_origin_from_api(item: &InputItem) -> Result, AgentApiError> { let origin = match item { InputItem::Text { origin, .. } @@ -324,3 +277,24 @@ fn input_origin_from_api(item: &InputItem) -> Result, AgentApiErr } Ok(origin.clone()) } + +async fn input_provenance_from_api( + store: &dyn BlobStore, + item: &InputItem, +) -> Result, AgentApiError> { + let reference = match item { + InputItem::Text { provenance_ref, .. } | InputItem::TextRef { provenance_ref, .. } => { + provenance_ref + } + _ => return Ok(None), + }; + let Some(reference) = reference else { + return Ok(None); + }; + let reference = parse_blob_ref(reference)?; + store + .stat_blob(&reference) + .await + .map_err(map_input_blob_store_error)?; + Ok(Some(reference)) +} diff --git a/crates/temporal-server/src/gateway/service/mod.rs b/crates/temporal-server/src/gateway/service/mod.rs index 562fadec3..2fd14614d 100644 --- a/crates/temporal-server/src/gateway/service/mod.rs +++ b/crates/temporal-server/src/gateway/service/mod.rs @@ -19,10 +19,12 @@ mod github_api; mod input; mod mcp_api; pub(crate) mod mcp_discovery; +mod model_defaults; mod models_api; mod oauth_api; mod parse; mod profiles; +mod transcriptions; pub(crate) use crate::environments::provider_controllers; mod session_jobs; mod session_lifecycle; @@ -137,7 +139,7 @@ use vfs::{ use super::{ AgentAdmission, AgentAdmissionFailure, AgentAdmissionFailureKind, AgentSessionArgs, AgentSessionStatus, AgentSessionWorkflow, DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, - DEFAULT_TEMPORAL_TARGET, connect_temporal, default_model_from_env, pg_store_from_env, + DEFAULT_TEMPORAL_TARGET, connect_temporal, pg_store_from_env, }; const DEFAULT_POLL_INTERVAL: Duration = Duration::from_millis(500); @@ -393,9 +395,7 @@ async fn context_append_result( let entry = project_context_entry_inputs(std::slice::from_ref(input)) .into_iter() .next(); - let activation_text = if is_audio_transcript_entry(input) { - api_projection::project_content_text(store, &input.content).await? - } else if context_append_entry_has_activation_text(input) { + let activation_text = if context_append_entry_has_activation_text(input) { // The submitted text is reused when it produced this exact entry so // plain-text appends do not pay a blob read per response entry. match submitted_text { @@ -482,35 +482,11 @@ fn active_entry_input(entry: &ContextEntry) -> ContextEntryInput { } fn active_context_entry_matches_input(active: &ContextEntry, input: &ContextEntryInput) -> bool { - let active_input = active_entry_input(active); - active_input == *input || audio_input_matches_transcript(input, &active_input) -} - -fn audio_input_matches_transcript(input: &ContextEntryInput, active: &ContextEntryInput) -> bool { - input - .content - .media_type - .as_deref() - .is_some_and(|mime| mime.trim().to_ascii_lowercase().starts_with("audio/")) - && is_audio_transcript_entry(active) - && active.provenance_ref.as_ref() == Some(&input.content.content_ref) -} - -fn is_audio_transcript_entry(input: &ContextEntryInput) -> bool { - input.content.provider_kind.as_deref() - == Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) + active_entry_input(active) == *input } fn input_admission_failure_from_api_error(error: AgentApiError) -> InputAdmissionFailureView { let kind = match error.kind { - AgentApiErrorKind::UnsupportedAudioMime => InputAdmissionFailureKind::UnsupportedAudioMime, - AgentApiErrorKind::AudioBlobTooLarge => InputAdmissionFailureKind::BlobTooLarge, - AgentApiErrorKind::AudioDurationTooLong => InputAdmissionFailureKind::AudioDurationTooLong, - AgentApiErrorKind::TranscoderUnavailable => { - InputAdmissionFailureKind::TranscoderUnavailable - } - AgentApiErrorKind::TranscodeFailure => InputAdmissionFailureKind::TranscodeFailure, - AgentApiErrorKind::TranscriptionFailure => InputAdmissionFailureKind::TranscriptionFailure, AgentApiErrorKind::NotFound => InputAdmissionFailureKind::BlobMissing, _ => InputAdmissionFailureKind::UnsupportedMedia, }; @@ -524,21 +500,6 @@ fn input_admission_failure_from_workflow( failure: &AgentAdmissionFailure, ) -> InputAdmissionFailureView { let kind = match failure.kind { - AgentAdmissionFailureKind::UnsupportedAudioMime => { - InputAdmissionFailureKind::UnsupportedAudioMime - } - AgentAdmissionFailureKind::AudioBlobMissing => InputAdmissionFailureKind::BlobMissing, - AgentAdmissionFailureKind::AudioBlobTooLarge => InputAdmissionFailureKind::BlobTooLarge, - AgentAdmissionFailureKind::AudioDurationTooLong => { - InputAdmissionFailureKind::AudioDurationTooLong - } - AgentAdmissionFailureKind::TranscoderUnavailable => { - InputAdmissionFailureKind::TranscoderUnavailable - } - AgentAdmissionFailureKind::TranscodeFailure => InputAdmissionFailureKind::TranscodeFailure, - AgentAdmissionFailureKind::TranscriptionFailure => { - InputAdmissionFailureKind::TranscriptionFailure - } AgentAdmissionFailureKind::RejectedCommand => InputAdmissionFailureKind::AdmissionRejected, }; InputAdmissionFailureView { @@ -553,7 +514,6 @@ pub struct GatewayAgentApiBuilder { task_queue: String, bot_task_queue: String, channel_task_queue: String, - default_model: ModelSelection, continue_as_new_history_threshold: Option, poll_interval: Duration, operation_timeout: Duration, @@ -635,11 +595,6 @@ impl GatewayAgentApiBuilder { self } - pub fn with_default_model(mut self, model: ModelSelection) -> Self { - self.default_model = model; - self - } - pub fn with_environment_gateway( mut self, gateway: crate::environments::gateway::EnvironmentGatewayClientConfig, @@ -754,7 +709,6 @@ impl GatewayAgentApiBuilder { task_queue: self.task_queue, bot_task_queue: self.bot_task_queue, channel_task_queue: self.channel_task_queue, - default_model: self.default_model, continue_as_new_history_threshold: self.continue_as_new_history_threshold, poll_interval: self.poll_interval, operation_timeout: self.operation_timeout, @@ -781,7 +735,6 @@ pub struct GatewayAgentApi { task_queue: String, pub(crate) bot_task_queue: String, pub(crate) channel_task_queue: String, - default_model: ModelSelection, continue_as_new_history_threshold: Option, poll_interval: Duration, operation_timeout: Duration, @@ -818,7 +771,6 @@ impl GatewayAgentApi { task_queue: DEFAULT_TASK_QUEUE.to_owned(), bot_task_queue: temporal_workflow::bots::DEFAULT_BOTS_TASK_QUEUE.to_owned(), channel_task_queue: crate::config::DEFAULT_CHANNELS_TASK_QUEUE.to_owned(), - default_model: default_model_from_env(), continue_as_new_history_threshold: None, poll_interval: DEFAULT_POLL_INTERVAL, operation_timeout: DEFAULT_OPERATION_TIMEOUT, @@ -2022,6 +1974,57 @@ impl AgentApiService for GatewayAgentApi { .map(AgentApiOutcome::new) } + async fn start_transcription( + &self, + params: TranscriptionStartParams, + ) -> Result, AgentApiError> { + self.start_transcription_impl(params).await + } + async fn read_transcription( + &self, + params: TranscriptionReadParams, + ) -> Result, AgentApiError> { + self.read_transcription_impl(params).await + } + async fn cancel_transcription( + &self, + params: TranscriptionCancelParams, + ) -> Result, AgentApiError> { + self.cancel_transcription_impl(params).await + } + + async fn read_model_defaults( + &self, + _params: api::ModelDefaultsReadParams, + ) -> Result, AgentApiError> { + self.authorize_method(api::METHOD_MODELS_DEFAULTS_READ, None) + .await?; + let defaults = self + .store + .read_model_defaults() + .await + .map_err(model_defaults::map_store_error)?; + Ok(AgentApiOutcome::new(api::ModelDefaultsResponse { + defaults, + })) + } + + async fn put_model_defaults( + &self, + params: api::ModelDefaultsPutParams, + ) -> Result, AgentApiError> { + self.authorize_method(api::METHOD_MODELS_DEFAULTS_PUT, None) + .await?; + let defaults = self + .store + .put_model_defaults(params) + .await + .map_err(model_defaults::map_store_error)?; + Ok(AgentApiOutcome::new(api::ModelDefaultsResponse { + defaults, + })) + } + async fn list_models( &self, params: ModelListParams, @@ -2227,7 +2230,10 @@ impl AgentApiService for GatewayAgentApi { ))); } } - let config = engine_session_config_from_api(params.config, self.default_model.clone())?; + let config = engine_session_config_from_api( + params.config, + model_defaults::current_session_model(&loaded.state)?, + )?; config .validate() .map_err(|error| AgentApiError::invalid_request(error.to_string()))?; diff --git a/crates/temporal-server/src/gateway/service/model_defaults.rs b/crates/temporal-server/src/gateway/service/model_defaults.rs new file mode 100644 index 000000000..f7a7daf85 --- /dev/null +++ b/crates/temporal-server/src/gateway/service/model_defaults.rs @@ -0,0 +1,153 @@ +use super::*; + +pub(super) fn map_store_error(error: store_pg::ModelDefaultsStoreError) -> AgentApiError { + match error { + store_pg::ModelDefaultsStoreError::Conflict { .. } => { + AgentApiError::conflict(error.to_string()) + } + store_pg::ModelDefaultsStoreError::Invalid(error) => error, + other => AgentApiError::internal(other.to_string()), + } +} + +/// An explicit (already profile-merged) model never consults universe policy. +pub(super) async fn creation_model( + explicit: Option, + defaults: impl std::future::Future>, +) -> Result { + let model = match explicit { + Some(model) => model, + None => defaults + .await? + .agent_run + .ok_or_else(|| AgentApiError::model_default_unset(api::ModelDefaultSlot::AgentRun))?, + }; + api::ModelDefaultSlot::AgentRun.validate_model(&model)?; + api_config::model_selection_from_api(model) +} + +pub(super) fn current_session_model( + state: &engine::CoreAgentState, +) -> Result { + state + .lifecycle + .config + .as_ref() + .map(|config| config.model.clone()) + .ok_or_else(|| AgentApiError::rejected("session has no configuration")) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn route(provider: &str, kind: &str, model: &str) -> ModelConfig { + ModelConfig { + provider_id: provider.into(), + api_kind: kind.into(), + model: model.into(), + } + } + + #[tokio::test(flavor = "current_thread")] + async fn explicit_selection_does_not_read_universe_defaults() { + let explicit = route("private", "openai:completions", "custom"); + let model = creation_model(Some(explicit), async { + panic!("explicit model must not read defaults") + }) + .await + .unwrap(); + assert_eq!(model.provider_id, "private"); + assert_eq!(model.api_kind, ProviderApiKind::OpenAiCompletions); + } + + #[tokio::test(flavor = "current_thread")] + async fn profile_merge_precedes_universe_resolution() { + let profile = api::SessionConfig { + model: Some(route("anthropic", "anthropic:messages", "profile")), + ..Default::default() + }; + for explicit in [None, Some(route("local", "openai:completions", "explicit"))] { + let expected = explicit.clone().or(profile.model.clone()).unwrap(); + let merged = GatewayAgentApi::merge_profile_start_config( + Some(profile.clone()), + Some(api::SessionConfig { + model: explicit, + ..Default::default() + }), + ) + .unwrap(); + let model = creation_model(merged.model, async { panic!("merged selection must win") }) + .await + .unwrap(); + assert_eq!(model.provider_id, expected.provider_id); + assert_eq!(model.model, expected.model); + } + } + + #[test] + fn applying_sparse_profiles_preserves_model_but_replaces_features() { + let model = + api_config::model_selection_from_api(route("local", "openai:completions", "pinned")) + .unwrap(); + let mut previous = + engine_session_config_from_api(api::SessionConfig::default(), model.clone()).unwrap(); + previous.features.web = Some(engine::WebFeature { + version: 1, + fetch: Some(engine::WebFetchFeature {}), + search: None, + }); + let profile = api::ProfileDocument { + config: Some(api::SessionConfig::default()), + ..Default::default() + }; + let applied = + GatewayAgentApi::profile_intent(&profile, Some(previous.model.clone())).unwrap(); + let config = applied.config.unwrap(); + assert_eq!(config.model, model); + assert_eq!(config.features.web, None); + let replaced = + engine_session_config_from_api(api::SessionConfig::default(), previous.model).unwrap(); + assert_eq!(replaced, config); + assert!( + GatewayAgentApi::profile_intent(&profile, None) + .unwrap() + .config + .is_none(), + "creation already merged its profile configuration" + ); + } + + #[tokio::test(flavor = "current_thread")] + async fn universe_routes_are_independent_and_resolution_is_a_snapshot() { + let mut first = api::ModelDefaults { + revision: 1, + agent_run: Some(route("anthropic", "anthropic:messages", "first")), + speech_to_text: None, + }; + let second = api::ModelDefaults { + revision: 1, + agent_run: Some(route("private", "openai:completions", "second")), + speech_to_text: None, + }; + let first_model = creation_model(None, async { Ok(first.clone()) }) + .await + .unwrap(); + let second_model = creation_model(None, async { Ok(second) }).await.unwrap(); + let existing = + engine_session_config_from_api(api::SessionConfig::default(), first_model).unwrap(); + first.agent_run = None; + let error = creation_model(None, async { Ok(first) }).await.unwrap_err(); + assert_eq!(error.kind, AgentApiErrorKind::ModelDefaultUnset); + assert_eq!( + error.model_default_slot, + Some(api::ModelDefaultSlot::AgentRun) + ); + assert_eq!(existing.model.provider_id, "anthropic"); + assert_eq!(second_model.provider_id, "private"); + let replaced = + engine_session_config_from_api(api::SessionConfig::default(), existing.model.clone()) + .unwrap(); + assert_eq!(replaced.model, existing.model); + } +} diff --git a/crates/temporal-server/src/gateway/service/models_api.rs b/crates/temporal-server/src/gateway/service/models_api.rs index b79f1b91c..3280b782e 100644 --- a/crates/temporal-server/src/gateway/service/models_api.rs +++ b/crates/temporal-server/src/gateway/service/models_api.rs @@ -23,6 +23,7 @@ const OPENAI_PROVIDER_ID: &str = "openai"; const ANTHROPIC_PROVIDER_ID: &str = "anthropic"; const OPENAI_RESPONSES_API_KIND: &str = "openai:responses"; const OPENAI_COMPLETIONS_API_KIND: &str = "openai:completions"; +const OPENAI_AUDIO_API_KIND: &str = "openai:audio-transcriptions"; const ANTHROPIC_MESSAGES_API_KIND: &str = "anthropic:messages"; const MODEL_DISCOVERY_TIMEOUT: Duration = Duration::from_secs(15); const MODEL_DISCOVERY_CACHE_TTL: Duration = Duration::from_secs(10); @@ -184,11 +185,7 @@ impl ModelDiscoveryService { )) }); if selectable_only { - models.retain(|model| { - model.provider_id != OPENAI_PROVIDER_ID - || (is_openai_selectable_model(&model.model) - && is_openai_recent_model(model.created_at_ms, model.fetched_at_ms)) - }); + models.retain(is_selectable_model_route); } ModelListResponse { models, providers } } @@ -244,7 +241,11 @@ impl ModelDiscoveryService { models, provider_success( OPENAI_PROVIDER_ID, - &[OPENAI_RESPONSES_API_KIND, OPENAI_COMPLETIONS_API_KIND], + &[ + OPENAI_RESPONSES_API_KIND, + OPENAI_COMPLETIONS_API_KIND, + OPENAI_AUDIO_API_KIND, + ], fetched_at_ms, source, credential, @@ -255,7 +256,11 @@ impl ModelDiscoveryService { Vec::new(), provider_failure( OPENAI_PROVIDER_ID, - &[OPENAI_RESPONSES_API_KIND, OPENAI_COMPLETIONS_API_KIND], + &[ + OPENAI_RESPONSES_API_KIND, + OPENAI_COMPLETIONS_API_KIND, + OPENAI_AUDIO_API_KIND, + ], &error, source, ), @@ -492,19 +497,48 @@ fn model_endpoint(config: &auth::AuthProviderConfig) -> Option<&auth::ModelEndpo } } -fn openai_model_views(model: openai::Model, fetched_at_ms: i64) -> [ModelView; 2] { +fn is_selectable_model_route(model: &ModelView) -> bool { + model.provider_id != OPENAI_PROVIDER_ID + || model.api_kind == OPENAI_AUDIO_API_KIND + || (is_openai_selectable_model(&model.model) + && is_openai_recent_model(model.created_at_ms, model.fetched_at_ms)) +} + +// The Models API omits endpoint capabilities. These file-transcription families +// accept the plain JSON transcription request supported by our audio adapter. +// Diarization and realtime-only models require different request options. +fn is_openai_transcription_model(model: &str) -> bool { + [ + "whisper-1", + "gpt-transcribe", + "gpt-4o-transcribe", + "gpt-4o-mini-transcribe", + ] + .iter() + .any(|family| is_model_family(model, family)) +} + +fn openai_model_views(model: openai::Model, fetched_at_ms: i64) -> Vec { let created_at_ms = unix_seconds_to_millis(model.created); let capabilities = openai_model_capabilities(&model.id); - [OPENAI_RESPONSES_API_KIND, OPENAI_COMPLETIONS_API_KIND].map(|api_kind| ModelView { - provider_id: OPENAI_PROVIDER_ID.to_owned(), - api_kind: api_kind.to_owned(), - display_name: model.id.clone(), - model: model.id.clone(), - capabilities: capabilities.clone(), - created_at_ms, - source: ModelSource::Provider, - fetched_at_ms, - }) + let api_kinds: &[&str] = if is_openai_transcription_model(&model.id) { + &[OPENAI_AUDIO_API_KIND] + } else { + &[OPENAI_RESPONSES_API_KIND, OPENAI_COMPLETIONS_API_KIND] + }; + api_kinds + .iter() + .map(|api_kind| ModelView { + provider_id: OPENAI_PROVIDER_ID.to_owned(), + api_kind: (*api_kind).to_owned(), + display_name: model.id.clone(), + model: model.id.clone(), + capabilities: capabilities.clone(), + created_at_ms, + source: ModelSource::Provider, + fetched_at_ms, + }) + .collect() } fn unix_seconds_to_millis(seconds: Option) -> Option { @@ -1060,7 +1094,10 @@ mod tests { ); assert_eq!( - views.clone().map(|view| view.api_kind), + views + .iter() + .map(|view| view.api_kind.clone()) + .collect::>(), [ OPENAI_RESPONSES_API_KIND.to_owned(), OPENAI_COMPLETIONS_API_KIND.to_owned(), @@ -1079,6 +1116,40 @@ mod tests { } } + #[test] + fn speech_discovery_uses_the_audio_route_without_the_agent_age_filter() { + for id in [ + "whisper-1", + "gpt-transcribe", + "gpt-4o-transcribe", + "gpt-4o-mini-transcribe", + "gpt-4o-mini-transcribe-2025-12-15", + ] { + let views = openai_model_views( + openai::Model { + id: id.to_owned(), + created: Some(1), + object: None, + owned_by: None, + }, + OPENAI_SELECTABLE_MAX_AGE_MS * 2, + ); + assert_eq!(views.len(), 1, "{id}"); + assert_eq!(views[0].api_kind, OPENAI_AUDIO_API_KIND, "{id}"); + assert!(is_selectable_model_route(&views[0]), "{id}"); + assert_eq!(views[0].capabilities.reasoning_efforts, None); + } + for id in [ + "gpt-4o-transcribe-diarize", + "gpt-live-transcribe", + "gpt-realtime-whisper", + "gpt-6-sol", + "gpt-4o-mini-tts", + ] { + assert!(!is_openai_transcription_model(id), "{id}"); + } + } + #[test] fn openai_reasoning_catalog_covers_current_families_and_snapshots() { for model in ["gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"] { diff --git a/crates/temporal-server/src/gateway/service/profiles.rs b/crates/temporal-server/src/gateway/service/profiles.rs index 8e653e5a6..43d92698d 100644 --- a/crates/temporal-server/src/gateway/service/profiles.rs +++ b/crates/temporal-server/src/gateway/service/profiles.rs @@ -91,13 +91,25 @@ impl GatewayAgentApi { AgentApiError::invalid_request(format!("invalid session id: {error}")) })?; let resolved = self.resolve_profile_source(params.profile).await?; - let profile = self.profile_intent(&resolved, true)?; + let loaded = self.load_session_state(&session_id).await?; + let revision = loaded.state.lifecycle.config_revision; + if let Some(expected) = params.expected_config_revision + && expected != revision + { + return Err(AgentApiError::conflict(format!( + "expected config revision {expected}, got {revision}" + ))); + } + let profile = Self::profile_intent( + &resolved, + Some(model_defaults::current_session_model(&loaded.state)?), + )?; let applied = self .prepare_session_operation( &session_id, temporal_workflow::SessionOperation::ApplyProfile { profile, - expected_config_revision: params.expected_config_revision, + expected_config_revision: Some(revision), expected_tools_revision: params.expected_tools_revision, }, ) @@ -125,7 +137,6 @@ impl GatewayAgentApi { } pub(super) fn merge_profile_start_config( - &self, profile_config: Option, explicit_config: Option, ) -> Option { @@ -144,20 +155,18 @@ impl GatewayAgentApi { }) } - /// With `apply_config`, the profile's own configuration is applied and - /// its default attachment is the environment fill candidate. Without it - /// (session start), the caller has merged the effective configuration - /// and sets the candidate itself. + /// With a current model, apply the profile configuration and preserve an + /// omitted model. On creation the caller has already merged configuration + /// and sets the environment candidate itself. pub(super) fn profile_intent( - &self, profile: &ProfileDocument, - apply_config: bool, + current_model: Option, ) -> Result { - let config = if apply_config { + let config = if let Some(current_model) = current_model { profile .config .clone() - .map(|config| engine_session_config_from_api(config, self.default_model.clone())) + .map(|config| engine_session_config_from_api(config, current_model)) .transpose()? } else { None diff --git a/crates/temporal-server/src/gateway/service/session_lifecycle.rs b/crates/temporal-server/src/gateway/service/session_lifecycle.rs index 879b94762..e7405e673 100644 --- a/crates/temporal-server/src/gateway/service/session_lifecycle.rs +++ b/crates/temporal-server/src/gateway/service/session_lifecycle.rs @@ -333,7 +333,7 @@ impl GatewayAgentApi { delete_after_close_ms, ); validate_delete_after_close_ms(effective_delete_after_close_ms)?; - let start_config = self.merge_profile_start_config( + let start_config = Self::merge_profile_start_config( resolved_profile .as_ref() .and_then(|profile| profile.config.clone()), @@ -358,7 +358,7 @@ impl GatewayAgentApi { ); args.setup = match resolved_profile.as_ref() { Some(profile) => { - let mut intent = self.profile_intent(profile, false)?; + let mut intent = Self::profile_intent(profile, None)?; intent.environment = setup_environment; Some(intent) } diff --git a/crates/temporal-server/src/gateway/service/tests.rs b/crates/temporal-server/src/gateway/service/tests.rs index 3997e5954..aec1ca861 100644 --- a/crates/temporal-server/src/gateway/service/tests.rs +++ b/crates/temporal-server/src/gateway/service/tests.rs @@ -18,30 +18,6 @@ fn admission_failure_mapping_uses_gateway_error_kinds() { map_admission_failure_to_api_error(&revision_conflict).kind, AgentApiErrorKind::Conflict ); - assert_eq!( - map_admission_failure_to_api_error(&failure( - AgentAdmissionFailureKind::UnsupportedAudioMime - )) - .kind, - AgentApiErrorKind::UnsupportedAudioMime - ); - assert_eq!( - map_admission_failure_to_api_error(&failure(AgentAdmissionFailureKind::AudioBlobMissing)) - .kind, - AgentApiErrorKind::InvalidRequest - ); - assert_eq!( - map_admission_failure_to_api_error(&failure( - AgentAdmissionFailureKind::TranscriptionFailure - )) - .kind, - AgentApiErrorKind::TranscriptionFailure - ); - assert_eq!( - map_admission_failure_to_api_error(&failure(AgentAdmissionFailureKind::TranscodeFailure)) - .kind, - AgentApiErrorKind::TranscodeFailure - ); } #[test] @@ -1531,6 +1507,7 @@ async fn context_entry_input_from_api_stores_text_as_user_message() { let entry = context_entry_input_from_api( &store, &InputItem::Text { + provenance_ref: None, origin: None, text: " [telegram] Alice (12:01): hi ".to_owned(), }, @@ -1561,6 +1538,7 @@ async fn context_entry_input_from_api_rejects_empty_text() { let error = context_entry_input_from_api( &store, &InputItem::Text { + provenance_ref: None, origin: None, text: " ".to_owned(), }, @@ -1637,6 +1615,7 @@ async fn context_entry_input_from_api_preserves_text_ref() { let entry = context_entry_input_from_api( &store, &InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: blob_ref.as_str().to_owned(), }, @@ -1659,6 +1638,7 @@ async fn run_input_from_api_maps_image_media_to_user_message_entry() { &store, &[ InputItem::Text { + provenance_ref: None, origin: None, text: "what is this?".to_owned(), }, @@ -1775,30 +1755,28 @@ async fn run_input_from_api_rejects_unsupported_document_media() { } #[tokio::test(flavor = "current_thread")] -async fn run_input_from_api_maps_audio_media_to_user_message_entry() { +async fn run_input_from_api_rejects_unprepared_audio() { let store = engine::storage::InMemoryBlobStore::new(); let blob_ref = store .put_bytes(b"OggS fake voice note".to_vec()) .await .expect("store audio"); - let input = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: blob_ref.as_str().to_owned(), - mime: "audio/ogg".to_owned(), - kind: api::MediaKind::Audio, - name: Some("voice.ogg".to_owned()), - }], - ) - .await - .expect("input"); - - assert_eq!(input.len(), 1); - assert_eq!(input[0].content.content_ref, blob_ref); - assert_eq!(input[0].content.media_type.as_deref(), Some("audio/ogg")); - assert_eq!(input[0].preview.as_deref(), Some("[audio: voice.ogg]")); + let item = InputItem::Media { + origin: None, + blob_ref: blob_ref.to_string(), + mime: "audio/ogg".into(), + kind: api::MediaKind::Audio, + name: Some("voice.ogg".into()), + }; + let run_error = run_input_from_api(&store, std::slice::from_ref(&item)) + .await + .expect_err("audio requires transcription"); + let append_error = context_entry_input_from_api(&store, &item) + .await + .expect_err("context audio requires transcription"); + assert_eq!(run_error.kind, AgentApiErrorKind::InvalidRequest); + assert_eq!(append_error, run_error); } #[tokio::test(flavor = "current_thread")] @@ -1806,20 +1784,6 @@ async fn run_input_from_api_rejects_unsupported_media() { let store = engine::storage::InMemoryBlobStore::new(); let blob_ref = store.put_bytes(vec![1, 2, 3]).await.expect("store blob"); - let audio = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: blob_ref.as_str().to_owned(), - mime: "audio/flac".to_owned(), - kind: api::MediaKind::Audio, - name: None, - }], - ) - .await - .expect_err("unsupported audio mime must be rejected"); - assert_eq!(audio.kind, AgentApiErrorKind::UnsupportedAudioMime); - let bad_mime = run_input_from_api( &store, &[InputItem::Media { @@ -1835,73 +1799,6 @@ async fn run_input_from_api_rejects_unsupported_media() { assert_eq!(bad_mime.kind, AgentApiErrorKind::InvalidRequest); } -#[tokio::test(flavor = "current_thread")] -async fn run_input_from_api_accepts_transcodable_audio_media() { - let store = engine::storage::InMemoryBlobStore::new(); - let blob_ref = store.put_bytes(vec![1, 2, 3]).await.expect("store blob"); - - let input = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: blob_ref.as_str().to_owned(), - mime: "audio/x-aac".to_owned(), - kind: api::MediaKind::Audio, - name: Some("clip.aac".to_owned()), - }], - ) - .await - .expect("transcodable audio should be admitted"); - - assert_eq!(input[0].content.content_ref, blob_ref); - assert_eq!(input[0].content.media_type.as_deref(), Some("audio/aac")); - assert_eq!(input[0].preview.as_deref(), Some("[audio: clip.aac]")); -} - -#[tokio::test(flavor = "current_thread")] -async fn run_input_from_api_rejects_audio_over_byte_cap() { - let store = engine::storage::InMemoryBlobStore::new(); - let blob_ref = store - .put_bytes(vec![0; 25 * 1024 * 1024 + 1]) - .await - .expect("store large audio"); - - let error = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: blob_ref.as_str().to_owned(), - mime: "audio/ogg".to_owned(), - kind: api::MediaKind::Audio, - name: None, - }], - ) - .await - .expect_err("oversized audio must be rejected"); - - assert_eq!(error.kind, AgentApiErrorKind::AudioBlobTooLarge); -} - -#[tokio::test(flavor = "current_thread")] -async fn run_input_from_api_rejects_missing_audio_blob() { - let store = engine::storage::InMemoryBlobStore::new(); - - let error = run_input_from_api( - &store, - &[InputItem::Media { - origin: None, - blob_ref: BlobRef::from_bytes(b"missing-audio").as_str().to_owned(), - mime: "audio/ogg".to_owned(), - kind: api::MediaKind::Audio, - name: None, - }], - ) - .await - .expect_err("missing audio blob must be rejected"); - - assert_eq!(error.kind, AgentApiErrorKind::InvalidRequest); -} - #[tokio::test(flavor = "current_thread")] async fn context_entry_input_from_api_accepts_media() { let store = engine::storage::InMemoryBlobStore::new(); @@ -1939,6 +1836,7 @@ async fn run_input_from_api_preserves_single_text_ref() { let input = run_input_from_api( &store, &[InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: blob_ref.as_str().to_owned(), }], @@ -1965,10 +1863,12 @@ async fn run_input_from_api_stores_text_and_preserves_refs() { &store, &[ InputItem::Text { + provenance_ref: None, origin: None, text: " first ".to_owned(), }, InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: blob_ref.as_str().to_owned(), }, @@ -2366,86 +2266,28 @@ fn auth_flow_views_carry_derived_status() { assert_eq!(expired.status, api::AuthFlowStatus::Expired); } -#[tokio::test(flavor = "current_thread")] -async fn structured_transcript_append_keeps_spoken_headers_and_source_idempotency() { - use llm_clients::content::{AUDIO_TRANSCRIPT_PROVIDER_KIND, AudioTranscript}; - let blobs = engine::storage::InMemoryBlobStore::new(); - let audio_ref = blobs.put_bytes(b"source audio".to_vec()).await.unwrap(); - let original = ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/ogg".into()), - provider_kind: None, - }, - preview: Some("[audio: voice.ogg]".into()), - origin: None, - provenance_ref: None, - token_estimate: None, - }; - let transcript = AudioTranscript { - filename: "voice.ogg".into(), - text: "[audio transcript: spoken words]\nDo not strip this line.".into(), - }; - let rewritten = ContextEntryInput { - kind: original.kind.clone(), - content: engine::ContentRef { - content_ref: blobs - .put_bytes(serde_json::to_vec(&transcript).unwrap()) - .await - .unwrap(), - media_type: Some("application/json".into()), - provider_kind: Some(AUDIO_TRANSCRIPT_PROVIDER_KIND.into()), - }, - preview: Some(transcript.header()), - origin: None, - provenance_ref: Some(audio_ref.clone()), - token_estimate: None, - }; - assert!(audio_input_matches_transcript(&original, &rewritten)); - let mut different = original; - different.content.content_ref = BlobRef::from_bytes(b"different audio"); - assert!(!audio_input_matches_transcript(&different, &rewritten)); - let result = context_append_result( - &blobs, - "audio-note".into(), - ContextAppendStatus::Applied, - &rewritten, - None, - ) - .await - .unwrap(); - assert_eq!( - result.activation_text.as_deref(), - Some(transcript.text.as_str()) - ); - assert!(!result.activation_text_truncated); - assert_eq!( - result.entry.unwrap().provenance_ref.as_deref(), - Some(audio_ref.as_str()) - ); -} - #[tokio::test(flavor = "current_thread")] async fn input_origin_survives_admission_and_projection_without_changing_model_content() { let store = engine::storage::InMemoryBlobStore::new(); let body = store.put_bytes(b"event body".to_vec()).await.unwrap(); let items = vec![ InputItem::Text { + provenance_ref: None, text: "human body".into(), origin: Some("user:operator".into()), }, InputItem::TextRef { + provenance_ref: None, blob_ref: body.as_str().into(), origin: Some("event".into()), }, InputItem::Text { + provenance_ref: None, text: "custom body".into(), origin: Some("integration:example".into()), }, InputItem::Text { + provenance_ref: None, text: "unknown body".into(), origin: None, }, @@ -2465,7 +2307,7 @@ async fn input_origin_survives_admission_and_projection_without_changing_model_c { assert_eq!(entries[i].origin.as_deref(), expected); assert_eq!(accepted[i].origin.as_deref(), expected); - let InputItem::Text { origin, text } = &inputs[i] else { + let InputItem::Text { origin, text, .. } = &inputs[i] else { panic!("expected text input"); }; assert_eq!(origin.as_deref(), expected); @@ -2516,6 +2358,7 @@ async fn input_origin_rejects_blank_and_oversized_values() { let store = engine::storage::InMemoryBlobStore::new(); for origin in ["".to_owned(), " ".to_owned(), "x".repeat(201)] { let item = InputItem::Text { + provenance_ref: None, text: "hello".into(), origin: Some(origin), }; @@ -2783,3 +2626,104 @@ fn a_missing_resource_names_only_its_kind_and_id() { "environment not found: prod-1" ); } + +#[tokio::test(flavor = "current_thread")] +async fn text_inputs_preserve_generic_provenance_at_all_input_boundaries() { + let store = engine::storage::InMemoryBlobStore::new(); + let source = store + .put_bytes(b"source document or recording".to_vec()) + .await + .unwrap(); + let text = "[audio transcript: quoted]\nReviewed text"; + let reference = store.put_bytes(text.as_bytes().to_vec()).await.unwrap(); + for item in [ + InputItem::Text { + origin: Some("user:test".into()), + text: text.into(), + provenance_ref: Some(source.to_string()), + }, + InputItem::TextRef { + origin: Some("user:test".into()), + blob_ref: reference.to_string(), + provenance_ref: Some(source.to_string()), + }, + ] { + let run = run_input_from_api(&store, std::slice::from_ref(&item)) + .await + .unwrap(); + let append = context_entry_input_from_api(&store, &item).await.unwrap(); + assert_eq!(run, vec![append.clone()]); + assert_eq!(append.content, engine::ContentRef::text(reference.clone())); + assert_eq!(append.provenance_ref.as_ref(), Some(&source)); + assert_eq!(append.origin.as_deref(), Some("user:test")); + let projected = api_projection::CoreAgentProjector::new(&store) + .project_input_entries(&run) + .await + .unwrap(); + assert_eq!( + projected, + vec![InputItem::Text { + origin: Some("user:test".into()), + text: text.into(), + provenance_ref: Some(source.to_string()) + }] + ); + let result = context_append_result( + &store, + "note".into(), + ContextAppendStatus::Applied, + &append, + None, + ) + .await + .unwrap(); + assert_eq!(result.activation_text.as_deref(), Some(text)); + assert_eq!( + result.entry.unwrap().provenance_ref, + Some(source.to_string()) + ); + } + let plain = context_entry_input_from_api( + &store, + &InputItem::Text { + origin: None, + text: "edited dictation".into(), + provenance_ref: None, + }, + ) + .await + .unwrap(); + assert!(plain.provenance_ref.is_none()); +} + +#[tokio::test(flavor = "current_thread")] +async fn text_input_provenance_requires_an_existing_source_blob() { + let store = engine::storage::InMemoryBlobStore::new(); + let text_ref = store.put_bytes(b"text".to_vec()).await.unwrap(); + for reference in [ + "not-a-blob".into(), + BlobRef::from_bytes(b"missing source").to_string(), + ] { + for item in [ + InputItem::Text { + origin: None, + text: "text".into(), + provenance_ref: Some(reference.clone()), + }, + InputItem::TextRef { + origin: None, + blob_ref: text_ref.to_string(), + provenance_ref: Some(reference.clone()), + }, + ] { + let run_error = run_input_from_api(&store, std::slice::from_ref(&item)) + .await + .unwrap_err(); + let append_error = context_entry_input_from_api(&store, &item) + .await + .unwrap_err(); + assert_eq!(run_error.kind, AgentApiErrorKind::InvalidRequest); + assert_eq!(append_error.kind, run_error.kind); + } + } +} diff --git a/crates/temporal-server/src/gateway/service/transcriptions.rs b/crates/temporal-server/src/gateway/service/transcriptions.rs new file mode 100644 index 000000000..9ccfb3ed5 --- /dev/null +++ b/crates/temporal-server/src/gateway/service/transcriptions.rs @@ -0,0 +1,268 @@ +use super::*; +use temporal_workflow::{TranscriptionSnapshot, TranscriptionWorkflow, TranscriptionWorkflowArgs}; +use temporalio_common::protos::temporal::api::enums::v1::WorkflowIdReusePolicy; + +fn validate_id(id: &str) -> Result<(), AgentApiError> { + if id + .strip_prefix("transcription_") + .is_some_and(|s| s.len() == 64 && s.bytes().all(|b| b.is_ascii_hexdigit())) + { + Ok(()) + } else { + Err(AgentApiError::invalid_request("invalid transcription id")) + } +} + +impl GatewayAgentApi { + pub(crate) async fn transcription_snapshot( + &self, + id: &str, + ) -> Result, AgentApiError> { + validate_id(id)?; + let handle = self + .client + .get_workflow_handle::(format!("{}/{id}", self.universe_id())); + match handle + .query( + TranscriptionWorkflow::snapshot, + (), + WorkflowQueryOptions::default(), + ) + .await + { + Ok(Some(mut snapshot)) => { + if !snapshot.view.status.is_terminal() { + let description = handle + .describe(WorkflowDescribeOptions::default()) + .await + .map_err(map_workflow_interaction_error)?; + match description.status() { + WorkflowExecutionStatus::TimedOut => { + snapshot.view.status = TranscriptionStatus::Failed; + snapshot.view.failure = Some(TranscriptionFailure { + kind: TranscriptionFailureKind::Timeout, + message: "Transcription exceeded its workflow deadline.".into(), + }); + } + WorkflowExecutionStatus::Canceled | WorkflowExecutionStatus::Terminated => { + snapshot.view.status = TranscriptionStatus::Cancelled + } + WorkflowExecutionStatus::Failed => { + snapshot.view.status = TranscriptionStatus::Failed; + snapshot.view.failure = Some(TranscriptionFailure { + kind: TranscriptionFailureKind::Internal, + message: "Transcription workflow failed.".into(), + }); + } + _ => {} + } + } + Ok(Some(snapshot)) + } + Ok(None) => Err(AgentApiError::conflict( + "transcription is starting; retry shortly", + )), + Err(WorkflowQueryError::NotFound(_)) => Ok(None), + Err(error) => Err(map_workflow_query_error(error)), + } + } + + /// Trusted internal callers supply their controller attribution explicitly. + pub(crate) async fn admit_transcription( + &self, + request: TranscriptionStartParams, + owner: Attribution, + ) -> Result { + request.validate()?; + let source = parse_blob_ref(&request.audio.blob_ref)?; + let id = temporal_workflow::transcription_id(&owner, &request.idempotency_key); + if let Some(snapshot) = self.transcription_snapshot(&id).await? { + return matching_request(snapshot, &request); + } + let model = match &request.model { + Some(model) => model.clone(), + None => self + .store + .read_model_defaults() + .await + .map_err(model_defaults::map_store_error)? + .speech_to_text + .ok_or_else(|| { + AgentApiError::model_default_unset(ModelDefaultSlot::SpeechToText) + })?, + }; + ModelDefaultSlot::SpeechToText.validate_model(&model)?; + let info = self + .store + .stat_blob(&source) + .await + .map_err(map_input_blob_store_error)?; + if info.byte_len > 25 * 1024 * 1024 { + return Err(AgentApiError::audio_blob_too_large( + "audio exceeds the 25 MiB limit", + )); + } + self.store + .touch_blob_refs(&[source]) + .await + .map_err(map_input_blob_store_error)?; + let args = TranscriptionWorkflowArgs { + universe_id: self.universe_id(), + transcription_id: id.clone(), + created_by: owner, + request: request.clone(), + model, + created_at_ms: now_ms()? as u64, + }; + let pending = args.pending(); + match self + .client + .start_workflow( + TranscriptionWorkflow::run, + args, + WorkflowStartOptions::new( + self.task_queue.clone(), + format!("{}/{id}", self.universe_id()), + ) + .id_reuse_policy(WorkflowIdReusePolicy::RejectDuplicate) + .execution_timeout(Duration::from_secs(960)) + .build(), + ) + .await + { + Ok(_) => Ok(pending), + Err(WorkflowStartError::AlreadyStarted { .. }) => { + // Another admission won the race. Its pinned model is authoritative. + let snapshot = self.transcription_snapshot(&id).await?.ok_or_else(|| { + AgentApiError::conflict("transcription is starting; retry the same request") + })?; + matching_request(snapshot, &request) + } + Err(error) => Err(map_workflow_start_error(error)), + } + } + + pub(crate) async fn transcription_content( + &self, + mut view: TranscriptionView, + ) -> Result { + if let Some(reference) = &view.transcript_ref { + let reference = parse_blob_ref(reference)?; + let content = async { + // Check metadata before the byte cache: a swept result is no + // longer admissible even while this process has cached bytes. + self.store.stat_blob(&reference).await?; + self.store.read_bytes(&reference).await + } + .await; + match content { + Ok(bytes) => { + view.text = Some(String::from_utf8(bytes).map_err(|_| { + AgentApiError::internal("transcription result is not UTF-8") + })?); + } + Err(BlobStoreError::NotFound { .. }) => { + view.status = TranscriptionStatus::Expired; + } + Err(error) => return Err(map_blob_store_error(error)), + } + } + Ok(view) + } + + fn authorize_transcription_owner(&self, view: &TranscriptionView) -> Result<(), AgentApiError> { + // Person callers can only access their own drafts. Direct universe keys + // retain their method-group authority; core does not resolve person roles. + if let Some(actor) = self.caller()?.actor + && view.created_by != (Attribution::Actor { id: actor }) + { + return Err(AgentApiError::forbidden()); + } + Ok(()) + } + + pub(super) async fn start_transcription_impl( + &self, + params: TranscriptionStartParams, + ) -> Result, AgentApiError> { + self.authorize_method(METHOD_TRANSCRIPTIONS_START, None) + .await?; + let view = self + .admit_transcription(params, self.attribution()?) + .await?; + Ok(AgentApiOutcome::new(TranscriptionResponse { + transcription: self.transcription_content(view).await?, + })) + } + + pub(super) async fn read_transcription_impl( + &self, + params: TranscriptionReadParams, + ) -> Result, AgentApiError> { + self.authorize_method(METHOD_TRANSCRIPTIONS_READ, None) + .await?; + let view = self + .transcription_snapshot(¶ms.transcription_id) + .await? + .ok_or_else(|| AgentApiError::not_found("transcription not found"))? + .view; + self.authorize_transcription_owner(&view)?; + Ok(AgentApiOutcome::new(TranscriptionResponse { + transcription: self.transcription_content(view).await?, + })) + } + + pub(super) async fn cancel_transcription_impl( + &self, + params: TranscriptionCancelParams, + ) -> Result, AgentApiError> { + self.authorize_method(METHOD_TRANSCRIPTIONS_CANCEL, None) + .await?; + let view = self + .transcription_snapshot(¶ms.transcription_id) + .await? + .ok_or_else(|| AgentApiError::not_found("transcription not found"))? + .view; + self.authorize_transcription_owner(&view)?; + if !view.status.is_terminal() { + let handle = self + .client + .get_workflow_handle::(format!( + "{}/{}", + self.universe_id(), + params.transcription_id + )); + match handle + .signal( + TranscriptionWorkflow::cancel, + (), + WorkflowSignalOptions::default(), + ) + .await + { + Ok(()) | Err(WorkflowInteractionError::NotFound(_)) => {} + Err(error) => return Err(map_workflow_interaction_error(error)), + } + } + let view = self + .transcription_snapshot(¶ms.transcription_id) + .await? + .map(|s| s.view) + .unwrap_or(view); + Ok(AgentApiOutcome::new(TranscriptionResponse { + transcription: self.transcription_content(view).await?, + })) + } +} + +fn matching_request( + snapshot: TranscriptionSnapshot, + request: &TranscriptionStartParams, +) -> Result { + if snapshot.request != *request { + return Err(AgentApiError::conflict( + "transcription idempotency key was used with different input or options", + )); + } + Ok(snapshot.view) +} diff --git a/crates/temporal-server/src/lib.rs b/crates/temporal-server/src/lib.rs index 8eb12d690..1390551f2 100644 --- a/crates/temporal-server/src/lib.rs +++ b/crates/temporal-server/src/lib.rs @@ -17,7 +17,7 @@ pub mod universe; pub mod worker; pub use config::{ - DeploymentStores, GatewayAuthMode, default_model_from_env, gateway_auth_mode_from_env, - pg_store_from_env, task_queue_from_env, universe_id_from_env, + DeploymentStores, GatewayAuthMode, gateway_auth_mode_from_env, pg_store_from_env, + task_queue_from_env, universe_id_from_env, }; pub use universe::{UniverseError, UniverseRuntime, UniverseState}; diff --git a/crates/temporal-server/src/main.rs b/crates/temporal-server/src/main.rs index dc65ec14d..68a60e7ac 100644 --- a/crates/temporal-server/src/main.rs +++ b/crates/temporal-server/src/main.rs @@ -123,6 +123,8 @@ enum ApiKeyCommand { }, #[command(about = "List API keys (prefixes only; secrets are never stored)")] List, + /// Replace an active key secret immediately; prints the new secret once. + Rotate { key_prefix: String }, #[command(about = "Revoke an API key by its display prefix")] Revoke { key_prefix: String }, } @@ -466,6 +468,17 @@ async fn run_api_key_command(command: ApiKeyCommand) -> anyhow::Result<()> { } Ok(()) } + ApiKeyCommand::Rotate { key_prefix } => { + let key = api_keys + .rotate_api_key(&key_prefix) + .await? + .ok_or_else(|| anyhow::anyhow!("no active api key with prefix {key_prefix}"))?; + println!( + "{}", + serde_json::json!({ "keyPrefix": key.record.key_prefix, "secret": key.secret.expose() }) + ); + Ok(()) + } ApiKeyCommand::Revoke { key_prefix } => { if api_keys .revoke_api_key(&key_prefix, now_ms) @@ -517,6 +530,7 @@ async fn mint( /// Temporal worker on its own task queue; the gateway role adds the HTTP /// server and the deployment reconcilers. async fn run_roles(args: RunArgs) -> anyhow::Result<()> { + temporal_server::config::validate_model_environment()?; let roles = args.roles()?; let task_types = args.task_types()?; let task_queues = args.task_queues()?; diff --git a/crates/temporal-server/src/subagents.rs b/crates/temporal-server/src/subagents.rs index 8ca05d096..688f1835f 100644 --- a/crates/temporal-server/src/subagents.rs +++ b/crates/temporal-server/src/subagents.rs @@ -403,6 +403,7 @@ impl SubagentService { .start_run( &child_session_id, vec![InputItem::Text { + provenance_ref: None, origin: None, text: args.input, }], @@ -1122,6 +1123,7 @@ mod tests { assert_eq!( runs[0].1, vec![InputItem::Text { + provenance_ref: None, origin: None, text: "review the change".to_owned() }] diff --git a/crates/temporal-server/src/worker/activities/audio.rs b/crates/temporal-server/src/worker/activities/audio.rs new file mode 100644 index 000000000..3ac2d64be --- /dev/null +++ b/crates/temporal-server/src/worker/activities/audio.rs @@ -0,0 +1,803 @@ +use std::{ + env, + ffi::{OsStr, OsString}, + path::{Path, PathBuf}, + process::Stdio, + sync::Arc, + time::Duration, +}; + +use async_trait::async_trait; +use llm_clients::{LlmApiError, openai::audio as oai}; +use llm_runtime::ModelProviderResolver; +pub(super) const MAX_AUDIO_BYTES: u64 = 25 * 1024 * 1024; +pub(super) const MAX_AUDIO_DURATION_MS: u64 = 10 * 60 * 1000; +const OPENAI_PROVIDER_ID: &str = "openai"; +pub(super) const PROVIDER_ACCEPTED_AUDIO_MIMES: &[&str] = &[ + "audio/mpeg", + "audio/mp4", + "audio/wav", + "audio/webm", + "audio/ogg", +]; +pub(super) const TRANSCODABLE_AUDIO_MIMES: &[&str] = + &["audio/aac", "audio/amr", "audio/3gpp", "audio/3gpp2"]; +const TRANSCODED_AUDIO_MIME: &str = "audio/wav"; +const DEFAULT_FFMPEG_PATH: &str = "ffmpeg"; +const DEFAULT_TRANSCODE_TIMEOUT: Duration = Duration::from_secs(30); + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AudioTranscriptionRequest { + pub bytes: Vec, + pub mime: String, + pub name: String, + pub model: api::ModelConfig, + pub language: Option, + pub prompt: Option, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AudioTranscription { + pub text: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +#[error("audio transcription failed: {message}")] +pub struct AudioTranscriptionError { + pub message: String, + pub retryable: bool, + pub configuration: bool, +} + +#[async_trait] +pub trait AudioTranscriber: Send + Sync { + async fn transcribe( + &self, + request: AudioTranscriptionRequest, + ) -> Result; +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AudioTranscodeRequest { + pub bytes: Vec, + pub mime: String, + pub name: String, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AudioTranscodeOutput { + pub bytes: Vec, + pub mime: String, + pub name: String, +} + +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +#[error("audio transcode failed: {message}")] +pub struct AudioTranscodeError { + pub message: String, +} + +#[async_trait] +pub trait AudioTranscoder: Send + Sync { + async fn transcode( + &self, + request: AudioTranscodeRequest, + ) -> Result; +} + +pub struct UnavailableAudioTranscriber; + +#[async_trait] +impl AudioTranscriber for UnavailableAudioTranscriber { + async fn transcribe( + &self, + _request: AudioTranscriptionRequest, + ) -> Result { + Err(AudioTranscriptionError { + message: "audio transcriber is not configured".to_owned(), + retryable: false, + configuration: true, + }) + } +} + +#[derive(Clone, Debug)] +pub struct FfmpegAudioTranscoder { + ffmpeg_path: PathBuf, + timeout: Duration, + max_output_bytes: u64, +} + +impl FfmpegAudioTranscoder { + pub fn new(ffmpeg_path: impl Into) -> Self { + Self { + ffmpeg_path: ffmpeg_path.into(), + timeout: DEFAULT_TRANSCODE_TIMEOUT, + max_output_bytes: MAX_AUDIO_BYTES, + } + } + + pub fn with_timeout(mut self, timeout: Duration) -> Self { + self.timeout = timeout; + self + } + + pub fn from_env() -> Self { + let ffmpeg_path = env::var("LIGHTSPEED_FFMPEG_PATH") + .ok() + .filter(|value| !value.trim().is_empty()) + .unwrap_or_else(|| DEFAULT_FFMPEG_PATH.to_owned()); + let timeout = env::var("LIGHTSPEED_AUDIO_TRANSCODE_TIMEOUT_MS") + .ok() + .and_then(|value| value.parse::().ok()) + .filter(|millis| *millis > 0) + .map(Duration::from_millis) + .unwrap_or(DEFAULT_TRANSCODE_TIMEOUT); + Self::new(ffmpeg_path).with_timeout(timeout) + } +} + +#[async_trait] +impl AudioTranscoder for FfmpegAudioTranscoder { + async fn transcode( + &self, + request: AudioTranscodeRequest, + ) -> Result { + let temp_dir = tempfile::Builder::new() + .prefix("lightspeed-audio-transcode-") + .tempdir() + .map_err(|error| AudioTranscodeError { + message: format!("create transcode temp dir: {error}"), + })?; + let input_path = temp_dir + .path() + .join(format!("input.{}", extension_for_mime(&request.mime))); + let output_path = temp_dir.path().join("output.wav"); + tokio::fs::write(&input_path, &request.bytes) + .await + .map_err(|error| AudioTranscodeError { + message: format!("write transcode input: {error}"), + })?; + + let mut command = tokio::process::Command::new(&self.ffmpeg_path); + command + .args(ffmpeg_args(&input_path, &output_path)) + .kill_on_drop(true) + .stdin(Stdio::null()) + .stdout(Stdio::null()) + .stderr(Stdio::piped()); + let output = tokio::time::timeout(self.timeout, command.output()) + .await + .map_err(|_| AudioTranscodeError { + message: format!( + "ffmpeg did not finish within {}ms", + self.timeout.as_millis() + ), + })? + .map_err(|error| AudioTranscodeError { + message: format!("run ffmpeg: {error}"), + })?; + if !output.status.success() { + return Err(AudioTranscodeError { + message: ffmpeg_failure_message(&output.stderr), + }); + } + let metadata = + tokio::fs::metadata(&output_path) + .await + .map_err(|error| AudioTranscodeError { + message: format!("read ffmpeg output metadata: {error}"), + })?; + if metadata.len() > self.max_output_bytes { + return Err(AudioTranscodeError { + message: format!( + "transcoded audio {} is {} bytes; the limit is {} bytes", + output_path.display(), + metadata.len(), + self.max_output_bytes + ), + }); + } + let bytes = tokio::fs::read(&output_path) + .await + .map_err(|error| AudioTranscodeError { + message: format!("read ffmpeg output: {error}"), + })?; + Ok(AudioTranscodeOutput { + bytes, + mime: TRANSCODED_AUDIO_MIME.to_owned(), + name: transcoded_audio_filename(&request.name), + }) + } +} + +pub struct OpenAiAudioTranscriber { + client: Arc, + provider_keys: Arc, +} + +impl OpenAiAudioTranscriber { + pub fn new(client: Arc, provider_keys: Arc) -> Self { + Self { + client, + provider_keys, + } + } +} + +#[async_trait] +impl AudioTranscriber for OpenAiAudioTranscriber { + async fn transcribe( + &self, + request: AudioTranscriptionRequest, + ) -> Result { + api::ModelDefaultSlot::SpeechToText + .validate_model(&request.model) + .map_err(|error| AudioTranscriptionError { + message: error.to_string(), + retryable: false, + configuration: true, + })?; + let stored = llm_runtime::resolve_provider_route( + self.provider_keys.as_ref(), + &request.model.provider_id, + &request.model.api_kind, + ) + .await + .map_err(|error| { + let retryable = matches!(error, llm_runtime::ProviderKeyError::Backend { .. }); + AudioTranscriptionError { + message: "Transcription provider configuration is missing or unusable.".into(), + retryable, + configuration: !retryable, + } + })?; + if request.model.provider_id != OPENAI_PROVIDER_ID + && stored.as_ref().and_then(|p| p.endpoint.as_ref()).is_none() + { + return Err(AudioTranscriptionError { + message: "Transcription provider requires an explicit endpoint.".into(), + retryable: false, + configuration: true, + }); + } + let mut input = oai::CreateTranscriptionRequest::new(oai::AudioFile { + bytes: request.bytes, + filename: request.name, + mime: request.mime, + }); + input.model = request.model.model; + input.language = request.language; + input.prompt = request.prompt; + let response = self + .client + .create_transcription_with_transport( + input, + stored.as_ref().map(|provider| provider.as_request_auth()), + stored + .as_ref() + .and_then(|provider| provider.endpoint.as_ref()) + .map(|endpoint| &endpoint.transport), + ) + .await + .map_err(map_openai_error)?; + Ok(AudioTranscription { + text: response.parsed.text, + }) + } +} + +#[derive(Debug, thiserror::Error)] +pub(super) enum AudioPreparationError { + #[error("Audio requires a transcoder, but none is configured.")] + TranscoderUnavailable, + #[error(transparent)] + Transcode(#[from] AudioTranscodeError), + #[error("Transcoded audio exceeds the 25 MiB limit.")] + OutputTooLarge, + #[error("Audio transcoder returned unsupported MIME type {0}.")] + UnsupportedOutputMime(String), +} + +#[derive(Debug)] +pub(super) struct PreparedAudio { + pub bytes: Vec, + pub mime: String, + pub name: String, +} + +pub(super) async fn prepare_audio_for_transcription( + bytes: Vec, + mime: &str, + name: &str, + transcoder: Option<&dyn AudioTranscoder>, +) -> Result { + if PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&mime) { + return Ok(PreparedAudio { + bytes, + mime: mime.to_owned(), + name: audio_filename(name), + }); + } + let Some(transcoder) = transcoder else { + return Err(AudioPreparationError::TranscoderUnavailable); + }; + let output = transcoder + .transcode(AudioTranscodeRequest { + bytes, + mime: mime.to_owned(), + name: audio_filename(name), + }) + .await + .map_err(AudioPreparationError::Transcode)?; + if output.bytes.len() as u64 > MAX_AUDIO_BYTES { + return Err(AudioPreparationError::OutputTooLarge); + } + if !PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&output.mime.as_str()) { + return Err(AudioPreparationError::UnsupportedOutputMime(output.mime)); + } + Ok(PreparedAudio { + bytes: output.bytes, + mime: output.mime, + name: output.name, + }) +} + +pub(super) fn normalized_mime(mime: Option<&str>) -> String { + let mime = mime + .unwrap_or_default() + .split(';') + .next() + .unwrap_or_default() + .trim() + .to_ascii_lowercase(); + match mime.as_str() { + "audio/mp3" => "audio/mpeg", + "audio/x-m4a" | "audio/m4a" => "audio/mp4", + "audio/x-wav" | "audio/wave" | "audio/vnd.wave" => "audio/wav", + "audio/oga" | "audio/opus" => "audio/ogg", + "audio/x-aac" => "audio/aac", + "audio/3gp" => "audio/3gpp", + "audio/3g2" => "audio/3gpp2", + other => other, + } + .to_owned() +} + +fn audio_filename(label: &str) -> String { + let trimmed = label.trim(); + if trimmed.is_empty() || trimmed == "audio" { + "audio.ogg".to_owned() + } else { + trimmed.to_owned() + } +} + +fn transcoded_audio_filename(label: &str) -> String { + let filename = audio_filename(label); + let stem = Path::new(&filename) + .file_stem() + .and_then(OsStr::to_str) + .filter(|value| !value.trim().is_empty()) + .unwrap_or("audio"); + format!("{stem}.wav") +} + +fn extension_for_mime(mime: &str) -> &'static str { + match mime { + "audio/mpeg" => "mp3", + "audio/mp4" => "m4a", + "audio/wav" => "wav", + "audio/webm" => "webm", + "audio/ogg" => "ogg", + "audio/aac" => "aac", + "audio/amr" => "amr", + "audio/3gpp" => "3gp", + "audio/3gpp2" => "3g2", + _ => "bin", + } +} + +fn ffmpeg_args(input_path: &Path, output_path: &Path) -> Vec { + [ + OsString::from("-nostdin"), + OsString::from("-hide_banner"), + OsString::from("-loglevel"), + OsString::from("error"), + OsString::from("-y"), + OsString::from("-i"), + input_path.as_os_str().to_owned(), + OsString::from("-vn"), + OsString::from("-ac"), + OsString::from("1"), + OsString::from("-ar"), + OsString::from("16000"), + OsString::from("-acodec"), + OsString::from("pcm_s16le"), + OsString::from("-f"), + OsString::from("wav"), + output_path.as_os_str().to_owned(), + ] + .into_iter() + .collect() +} + +fn ffmpeg_failure_message(stderr: &[u8]) -> String { + let stderr = String::from_utf8_lossy(stderr); + let stderr = stderr.trim(); + if stderr.is_empty() { + "ffmpeg exited with a non-zero status".to_owned() + } else { + format!("ffmpeg exited with a non-zero status: {stderr}") + } +} + +pub(super) fn audio_duration_ms(mime: &str, bytes: &[u8]) -> Option { + // Cheap duration enforcement is intentionally narrow for the first cut: + // OGG/Opus and WAV are covered; MP3/M4A/WebM rely on the byte cap unless + // a non-provider container is transcoded to WAV. + match mime { + "audio/ogg" => ogg_opus_duration_ms(bytes), + "audio/wav" => wav_duration_ms(bytes), + _ => None, + } +} + +fn ogg_opus_duration_ms(bytes: &[u8]) -> Option { + let mut offset = 0usize; + let mut last_granule = None; + while offset + 27 <= bytes.len() { + if &bytes[offset..offset + 4] != b"OggS" { + return None; + } + let granule = u64::from_le_bytes(bytes[offset + 6..offset + 14].try_into().ok()?); + if granule != u64::MAX { + last_granule = Some(granule); + } + let segments = bytes[offset + 26] as usize; + let lacing_start = offset + 27; + let lacing_end = lacing_start.checked_add(segments)?; + if lacing_end > bytes.len() { + return None; + } + let mut body_len = 0usize; + for segment in &bytes[lacing_start..lacing_end] { + body_len = body_len.checked_add(*segment as usize)?; + } + offset = lacing_end.checked_add(body_len)?; + } + last_granule.map(|samples| samples.saturating_mul(1000) / 48_000) +} + +fn wav_duration_ms(bytes: &[u8]) -> Option { + if bytes.len() < 12 || &bytes[0..4] != b"RIFF" || &bytes[8..12] != b"WAVE" { + return None; + } + let mut offset = 12usize; + let mut byte_rate = None; + let mut data_len = None; + while offset + 8 <= bytes.len() { + let chunk_id = &bytes[offset..offset + 4]; + let chunk_len = u32::from_le_bytes(bytes[offset + 4..offset + 8].try_into().ok()?) as usize; + let data_start = offset + 8; + let data_end = data_start.checked_add(chunk_len)?; + if data_end > bytes.len() { + return None; + } + if chunk_id == b"fmt " && chunk_len >= 16 { + byte_rate = Some(u32::from_le_bytes( + bytes[data_start + 8..data_start + 12].try_into().ok()?, + ) as u64); + } else if chunk_id == b"data" { + data_len = Some(chunk_len as u64); + } + if byte_rate.is_some() && data_len.is_some() { + break; + } + offset = data_end.checked_add(chunk_len % 2)?; + } + let byte_rate = byte_rate?; + if byte_rate == 0 { + return None; + } + Some(data_len?.saturating_mul(1000) / byte_rate) +} + +fn map_openai_error(error: LlmApiError) -> AudioTranscriptionError { + let retryable = error.retryable(); + let configuration = matches!(error, LlmApiError::Configuration(_)); + AudioTranscriptionError { + message: if configuration { + "Transcription provider is not configured." + } else { + "Transcription provider request failed." + } + .into(), + retryable, + configuration, + } +} + +pub(super) fn default_openai_audio_transcriber( + provider_keys: Arc, +) -> Result, anyhow::Error> { + let client = oai::Client::new(oai::Config::from_env_allow_missing_key()) + .map_err(|error| anyhow::anyhow!("construct OpenAI audio client: {error}"))?; + Ok(Arc::new(OpenAiAudioTranscriber::new( + Arc::new(client), + provider_keys, + ))) +} + +pub fn default_audio_transcoder_from_env() -> anyhow::Result>> { + let Some(kind) = env::var("LIGHTSPEED_AUDIO_TRANSCODER") + .ok() + .map(|value| value.trim().to_ascii_lowercase()) + .filter(|value| !value.is_empty() && value != "none") + else { + return Ok(None); + }; + match kind.as_str() { + "ffmpeg" => Ok(Some(Arc::new(FfmpegAudioTranscoder::from_env()))), + other => anyhow::bail!( + "unsupported LIGHTSPEED_AUDIO_TRANSCODER value {other:?}; expected \"ffmpeg\" or \"none\"" + ), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::sync::Mutex; + + struct StaticTranscoder { + result: Result, + requests: Mutex>, + } + + impl StaticTranscoder { + fn new(result: Result) -> Self { + Self { + result, + requests: Mutex::new(Vec::new()), + } + } + + fn requests(&self) -> Vec { + self.requests + .lock() + .expect("static transcoder requests") + .clone() + } + } + + #[async_trait] + impl AudioTranscoder for StaticTranscoder { + async fn transcode( + &self, + request: AudioTranscodeRequest, + ) -> Result { + self.requests + .lock() + .expect("static transcoder requests") + .push(request); + self.result.clone() + } + } + + #[tokio::test(flavor = "current_thread")] + async fn accepted_audio_bypasses_transcoding() { + let transcoder = StaticTranscoder::new(Err(AudioTranscodeError { + message: "must not run".into(), + })); + let audio = prepare_audio_for_transcription( + b"audio".to_vec(), + "audio/ogg", + "voice.ogg", + Some(&transcoder), + ) + .await + .unwrap(); + assert_eq!(audio.bytes, b"audio"); + assert_eq!(audio.mime, "audio/ogg"); + assert_eq!(audio.name, "voice.ogg"); + assert!(transcoder.requests().is_empty()); + } + + #[tokio::test(flavor = "current_thread")] + async fn transcodable_audio_requires_transcoder() { + let error = prepare_audio_for_transcription(vec![], "audio/aac", "voice.aac", None) + .await + .unwrap_err(); + assert!(matches!( + error, + AudioPreparationError::TranscoderUnavailable + )); + } + + #[tokio::test(flavor = "current_thread")] + async fn transcodable_audio_is_prepared_for_provider() { + let transcoder = StaticTranscoder::new(Ok(AudioTranscodeOutput { + bytes: tiny_wav_bytes(), + mime: "audio/wav".into(), + name: "voice.wav".into(), + })); + let audio = prepare_audio_for_transcription( + b"aac fake".to_vec(), + "audio/aac", + "voice.aac", + Some(&transcoder), + ) + .await + .unwrap(); + assert_eq!(audio.bytes, tiny_wav_bytes()); + assert_eq!(audio.mime, "audio/wav"); + assert_eq!(audio.name, "voice.wav"); + assert_eq!( + transcoder.requests(), + vec![AudioTranscodeRequest { + bytes: b"aac fake".to_vec(), + mime: "audio/aac".into(), + name: "voice.aac".into(), + }] + ); + } + + #[tokio::test(flavor = "current_thread")] + async fn invalid_transcoder_output_is_rejected() { + for (output, too_large) in [ + ( + AudioTranscodeOutput { + bytes: vec![0; MAX_AUDIO_BYTES as usize + 1], + mime: "audio/wav".into(), + name: "voice.wav".into(), + }, + true, + ), + ( + AudioTranscodeOutput { + bytes: vec![], + mime: "audio/aac".into(), + name: "voice.aac".into(), + }, + false, + ), + ] { + let transcoder = StaticTranscoder::new(Ok(output)); + let error = prepare_audio_for_transcription( + vec![], + "audio/aac", + "voice.aac", + Some(&transcoder), + ) + .await + .unwrap_err(); + if too_large { + assert!(matches!(error, AudioPreparationError::OutputTooLarge)); + } else { + assert!(matches!( + error, + AudioPreparationError::UnsupportedOutputMime(_) + )); + } + } + } + + #[tokio::test(flavor = "current_thread")] + async fn transcode_failure_is_preserved() { + let transcoder = StaticTranscoder::new(Err(AudioTranscodeError { + message: "unsupported codec".into(), + })); + let error = + prepare_audio_for_transcription(vec![], "audio/aac", "voice.aac", Some(&transcoder)) + .await + .unwrap_err(); + assert!(matches!(error, AudioPreparationError::Transcode(_))); + } + + #[test] + fn duration_readers_measure_audio() { + assert_eq!( + audio_duration_ms("audio/wav", &tiny_wav_bytes()), + Some(1000) + ); + assert_eq!( + audio_duration_ms("audio/ogg", &ogg_page(601 * 48_000)), + Some(601_000) + ); + } + + #[test] + fn ffmpeg_command_args_normalize_audio_to_mono_wav() { + let args = ffmpeg_args(Path::new("/tmp/in.aac"), Path::new("/tmp/out.wav")); + let args: Vec = args + .iter() + .map(|arg| arg.to_string_lossy().into_owned()) + .collect(); + + assert_eq!( + args, + vec![ + "-nostdin", + "-hide_banner", + "-loglevel", + "error", + "-y", + "-i", + "/tmp/in.aac", + "-vn", + "-ac", + "1", + "-ar", + "16000", + "-acodec", + "pcm_s16le", + "-f", + "wav", + "/tmp/out.wav" + ] + ); + } + + #[tokio::test(flavor = "current_thread")] + #[ignore = "requires ffmpeg on PATH or LIGHTSPEED_FFMPEG_PATH"] + async fn ffmpeg_audio_transcoder_smoke_test() { + let transcoder = FfmpegAudioTranscoder::from_env(); + + let output = transcoder + .transcode(AudioTranscodeRequest { + bytes: tiny_wav_bytes(), + mime: "audio/wav".to_owned(), + name: "voice.wav".to_owned(), + }) + .await + .expect("ffmpeg should transcode tiny wav input"); + + assert_eq!(output.mime, "audio/wav"); + assert_eq!(output.name, "voice.wav"); + assert!(output.bytes.starts_with(b"RIFF")); + assert_eq!(audio_duration_ms(&output.mime, &output.bytes), Some(1000)); + } + + fn tiny_wav_bytes() -> Vec { + let sample_rate = 8_000u32; + let channels = 1u16; + let bits_per_sample = 16u16; + let sample_count = sample_rate as usize; + let byte_rate = sample_rate * channels as u32 * bits_per_sample as u32 / 8; + let block_align = channels * bits_per_sample / 8; + let data_len = sample_count * block_align as usize; + + let mut bytes = Vec::with_capacity(44 + data_len); + bytes.extend_from_slice(b"RIFF"); + bytes.extend_from_slice(&(36 + data_len as u32).to_le_bytes()); + bytes.extend_from_slice(b"WAVE"); + bytes.extend_from_slice(b"fmt "); + bytes.extend_from_slice(&16u32.to_le_bytes()); + bytes.extend_from_slice(&1u16.to_le_bytes()); + bytes.extend_from_slice(&channels.to_le_bytes()); + bytes.extend_from_slice(&sample_rate.to_le_bytes()); + bytes.extend_from_slice(&byte_rate.to_le_bytes()); + bytes.extend_from_slice(&block_align.to_le_bytes()); + bytes.extend_from_slice(&bits_per_sample.to_le_bytes()); + bytes.extend_from_slice(b"data"); + bytes.extend_from_slice(&(data_len as u32).to_le_bytes()); + bytes.resize(44 + data_len, 0); + bytes + } + + fn ogg_page(granule_position: u64) -> Vec { + let mut page = Vec::new(); + page.extend_from_slice(b"OggS"); + page.push(0); + page.push(0); + page.extend_from_slice(&granule_position.to_le_bytes()); + page.extend_from_slice(&1u32.to_le_bytes()); + page.extend_from_slice(&0u32.to_le_bytes()); + page.extend_from_slice(&0u32.to_le_bytes()); + page.push(1); + page.push(1); + page.push(0); + page + } +} diff --git a/crates/temporal-server/src/worker/activities/mod.rs b/crates/temporal-server/src/worker/activities/mod.rs index 296f753d8..9bc7a3827 100644 --- a/crates/temporal-server/src/worker/activities/mod.rs +++ b/crates/temporal-server/src/worker/activities/mod.rs @@ -16,29 +16,29 @@ use crate::worker::{ ACTIVITY_CONTEXT_COMPACT, ACTIVITY_CREATE_OR_LOAD_SESSION, ACTIVITY_ENVIRONMENT_JOB_CANCEL, ACTIVITY_ENVIRONMENT_JOB_POLL, ACTIVITY_ENVIRONMENT_JOB_PREPARE_WORKFLOW_TOOL, ACTIVITY_ENVIRONMENT_JOB_START, ACTIVITY_LLM_GENERATE, ACTIVITY_MATERIALIZE_AWAIT_RESULT, - ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PREPROCESS_RUN_INPUT, ACTIVITY_PUT_BLOB, - ACTIVITY_READ_BLOB, ACTIVITY_RUNTIME_PROJECTION_REFRESH, - ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, - ACTIVITY_SUBAGENT_RESOLVE, ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, - ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, - AppendEventsRequest, AwaitEnvironmentReadyActivityRequest, AwaitEnvironmentReadyActivityResult, + ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PUT_BLOB, ACTIVITY_READ_BLOB, + ACTIVITY_RUNTIME_PROJECTION_REFRESH, ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, + ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, ACTIVITY_SUBAGENT_RESOLVE, + ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, + ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, AppendEventsRequest, + AwaitEnvironmentReadyActivityRequest, AwaitEnvironmentReadyActivityResult, ContextCompactActivityRequest, CreateOrLoadSessionRequest, CreateOrLoadSessionResult, EnvironmentJobCancelActivityRequest, EnvironmentJobPollActivityRequest, EnvironmentJobPollActivityResult, EnvironmentJobStartActivityRequest, - EnvironmentJobStartActivityResult, LlmGenerateActivityRequest, - PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult, PutBlobRequest, - ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, + EnvironmentJobStartActivityResult, LlmGenerateActivityRequest, PutBlobRequest, ReadBlobRequest, + ReadBlobResult, RuntimeProjectionRefreshActivityRequest, RuntimeProjectionRefreshActivityResult, ToolInvokeBatchActivityRequest, ToolInvokeCallActivityRequest, ToolInvokeCallActivityResult, ToolPreparePromiseControlsActivityRequest, }; +mod audio; mod common; mod compaction; mod context_refresh; mod environment_jobs; mod llm; -mod preprocess; +mod transcriptions; pub use context_refresh::subagent_catalog_snapshot; mod state; mod storage; @@ -46,13 +46,13 @@ mod subagents; mod tools; mod workflow_tools; -pub use preprocess::{ +pub use audio::{ AudioTranscodeError, AudioTranscodeOutput, AudioTranscodeRequest, AudioTranscoder, AudioTranscriber, AudioTranscription, AudioTranscriptionError, AudioTranscriptionRequest, FfmpegAudioTranscoder, default_audio_transcoder_from_env, }; pub use state::{ - ActivityState, LlmActivityDeps, PreprocessActivityDeps, RuntimeProjectionActivityDeps, + ActivityState, AudioActivityDeps, LlmActivityDeps, RuntimeProjectionActivityDeps, StorageActivityDeps, ToolActivityDeps, }; @@ -226,8 +226,8 @@ mod tests { temporal_workflow::WorkflowActivities::llm_generate.name() ); assert_eq!( - WorkerActivities::preprocess_run_input.name(), - temporal_workflow::WorkflowActivities::preprocess_run_input.name() + WorkerActivities::execute_transcription.name(), + temporal_workflow::WorkflowActivities::execute_transcription.name() ); assert_eq!( WorkerActivities::context_compact.name(), @@ -463,6 +463,26 @@ mod tests { #[activities] impl WorkerActivities { + #[activity(name = "WorkflowActivities::execute_transcription")] + pub async fn execute_transcription( + self: Arc, + ctx: ActivityContext, + args: temporal_workflow::TranscriptionWorkflowArgs, + ) -> Result { + if ctx + .info() + .workflow_execution + .as_ref() + .map(|w| w.workflow_id.as_str()) + != Some(temporal_workflow::transcription_workflow_id(&args).as_str()) + { + return Err(common::activity_error(anyhow::anyhow!( + "transcription workflow identity mismatch" + ))); + } + let state = self.state_for(&ctx).await?; + common::cancellable(&ctx, transcriptions::execute(&state, args)).await + } #[activity(name = ACTIVITY_CREATE_OR_LOAD_SESSION)] pub async fn create_or_load_session( self: Arc, @@ -534,16 +554,6 @@ impl WorkerActivities { common::cancellable(&ctx, llm::generate(state.llm(), attempt, request)).await } - #[activity(name = ACTIVITY_PREPROCESS_RUN_INPUT)] - pub async fn preprocess_run_input( - self: Arc, - ctx: ActivityContext, - request: PreprocessRunInputActivityRequest, - ) -> Result { - let state = self.state_for(&ctx).await?; - preprocess::preprocess_run_input(state.preprocess(), request).await - } - #[activity(name = ACTIVITY_CONTEXT_COMPACT)] pub async fn context_compact( self: Arc, diff --git a/crates/temporal-server/src/worker/activities/preprocess.rs b/crates/temporal-server/src/worker/activities/preprocess.rs deleted file mode 100644 index 89a2640ee..000000000 --- a/crates/temporal-server/src/worker/activities/preprocess.rs +++ /dev/null @@ -1,1189 +0,0 @@ -use std::{ - env, - ffi::{OsStr, OsString}, - path::{Path, PathBuf}, - process::Stdio, - sync::Arc, - time::Duration, -}; - -use async_trait::async_trait; -use engine::{ - ContextEntryInput, ContextEntryKind, ContextMessageRole, - storage::{BlobStore, BlobStoreError}, -}; -use llm_clients::{LlmApiError, openai::audio as oai}; -use llm_runtime::ModelProviderResolver; -use temporalio_sdk::activities::ActivityError; - -use crate::worker::{PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult}; -use temporal_workflow::{ - PreprocessRunInputFailure, PreprocessRunInputFailureKind, PreprocessRunInputOutcome, -}; - -use super::state::PreprocessActivityDeps; -use llm_clients::content::{AUDIO_TRANSCRIPT_PROVIDER_KIND, AudioTranscript}; - -const MAX_AUDIO_BYTES: u64 = 25 * 1024 * 1024; -const MAX_AUDIO_DURATION_MS: u64 = 10 * 60 * 1000; -const OPENAI_PROVIDER_ID: &str = "openai"; -const PROVIDER_ACCEPTED_AUDIO_MIMES: &[&str] = &[ - "audio/mpeg", - "audio/mp4", - "audio/wav", - "audio/webm", - "audio/ogg", -]; -const TRANSCODABLE_AUDIO_MIMES: &[&str] = &["audio/aac", "audio/amr", "audio/3gpp", "audio/3gpp2"]; -const TRANSCODED_AUDIO_MIME: &str = "audio/wav"; -const DEFAULT_FFMPEG_PATH: &str = "ffmpeg"; -const DEFAULT_TRANSCODE_TIMEOUT: Duration = Duration::from_secs(30); - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct AudioTranscriptionRequest { - pub bytes: Vec, - pub mime: String, - pub name: String, -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct AudioTranscription { - pub text: String, -} - -#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] -#[error("audio transcription failed: {message}")] -pub struct AudioTranscriptionError { - pub message: String, -} - -#[async_trait] -pub trait AudioTranscriber: Send + Sync { - async fn transcribe( - &self, - request: AudioTranscriptionRequest, - ) -> Result; -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct AudioTranscodeRequest { - pub bytes: Vec, - pub mime: String, - pub name: String, -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct AudioTranscodeOutput { - pub bytes: Vec, - pub mime: String, - pub name: String, -} - -#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] -#[error("audio transcode failed: {message}")] -pub struct AudioTranscodeError { - pub message: String, -} - -#[async_trait] -pub trait AudioTranscoder: Send + Sync { - async fn transcode( - &self, - request: AudioTranscodeRequest, - ) -> Result; -} - -pub struct UnavailableAudioTranscriber; - -#[async_trait] -impl AudioTranscriber for UnavailableAudioTranscriber { - async fn transcribe( - &self, - _request: AudioTranscriptionRequest, - ) -> Result { - Err(AudioTranscriptionError { - message: "audio transcriber is not configured".to_owned(), - }) - } -} - -#[derive(Clone, Debug)] -pub struct FfmpegAudioTranscoder { - ffmpeg_path: PathBuf, - timeout: Duration, - max_output_bytes: u64, -} - -impl FfmpegAudioTranscoder { - pub fn new(ffmpeg_path: impl Into) -> Self { - Self { - ffmpeg_path: ffmpeg_path.into(), - timeout: DEFAULT_TRANSCODE_TIMEOUT, - max_output_bytes: MAX_AUDIO_BYTES, - } - } - - pub fn with_timeout(mut self, timeout: Duration) -> Self { - self.timeout = timeout; - self - } - - pub fn from_env() -> Self { - let ffmpeg_path = env::var("LIGHTSPEED_FFMPEG_PATH") - .ok() - .filter(|value| !value.trim().is_empty()) - .unwrap_or_else(|| DEFAULT_FFMPEG_PATH.to_owned()); - let timeout = env::var("LIGHTSPEED_AUDIO_TRANSCODE_TIMEOUT_MS") - .ok() - .and_then(|value| value.parse::().ok()) - .filter(|millis| *millis > 0) - .map(Duration::from_millis) - .unwrap_or(DEFAULT_TRANSCODE_TIMEOUT); - Self::new(ffmpeg_path).with_timeout(timeout) - } -} - -#[async_trait] -impl AudioTranscoder for FfmpegAudioTranscoder { - async fn transcode( - &self, - request: AudioTranscodeRequest, - ) -> Result { - let temp_dir = tempfile::Builder::new() - .prefix("lightspeed-audio-transcode-") - .tempdir() - .map_err(|error| AudioTranscodeError { - message: format!("create transcode temp dir: {error}"), - })?; - let input_path = temp_dir - .path() - .join(format!("input.{}", extension_for_mime(&request.mime))); - let output_path = temp_dir.path().join("output.wav"); - tokio::fs::write(&input_path, &request.bytes) - .await - .map_err(|error| AudioTranscodeError { - message: format!("write transcode input: {error}"), - })?; - - let mut command = tokio::process::Command::new(&self.ffmpeg_path); - command - .args(ffmpeg_args(&input_path, &output_path)) - .kill_on_drop(true) - .stdin(Stdio::null()) - .stdout(Stdio::null()) - .stderr(Stdio::piped()); - let output = tokio::time::timeout(self.timeout, command.output()) - .await - .map_err(|_| AudioTranscodeError { - message: format!( - "ffmpeg did not finish within {}ms", - self.timeout.as_millis() - ), - })? - .map_err(|error| AudioTranscodeError { - message: format!("run ffmpeg: {error}"), - })?; - if !output.status.success() { - return Err(AudioTranscodeError { - message: ffmpeg_failure_message(&output.stderr), - }); - } - let metadata = - tokio::fs::metadata(&output_path) - .await - .map_err(|error| AudioTranscodeError { - message: format!("read ffmpeg output metadata: {error}"), - })?; - if metadata.len() > self.max_output_bytes { - return Err(AudioTranscodeError { - message: format!( - "transcoded audio {} is {} bytes; the limit is {} bytes", - output_path.display(), - metadata.len(), - self.max_output_bytes - ), - }); - } - let bytes = tokio::fs::read(&output_path) - .await - .map_err(|error| AudioTranscodeError { - message: format!("read ffmpeg output: {error}"), - })?; - Ok(AudioTranscodeOutput { - bytes, - mime: TRANSCODED_AUDIO_MIME.to_owned(), - name: transcoded_audio_filename(&request.name), - }) - } -} - -pub struct OpenAiAudioTranscriber { - client: Arc, - provider_keys: Arc, -} - -impl OpenAiAudioTranscriber { - pub fn new(client: Arc, provider_keys: Arc) -> Self { - Self { - client, - provider_keys, - } - } -} - -#[async_trait] -impl AudioTranscriber for OpenAiAudioTranscriber { - async fn transcribe( - &self, - request: AudioTranscriptionRequest, - ) -> Result { - let stored_key = self - .provider_keys - .resolve_model_provider(OPENAI_PROVIDER_ID) - .await - .map_err(|error| AudioTranscriptionError { - message: error.to_string(), - })?; - let response = self - .client - .create_transcription_with_auth( - oai::CreateTranscriptionRequest::new(oai::AudioFile { - bytes: request.bytes, - filename: request.name, - mime: request.mime, - }), - stored_key - .as_ref() - .and_then(|provider| provider.auth.as_ref()) - .map(|auth| auth.as_request_auth()), - ) - .await - .map_err(map_openai_error)?; - Ok(AudioTranscription { - text: response.parsed.text, - }) - } -} - -pub(super) async fn preprocess_run_input( - deps: &PreprocessActivityDeps, - request: PreprocessRunInputActivityRequest, -) -> Result { - let outcome = match rewrite_run_input( - deps.blobs.as_ref(), - deps.transcriber.as_ref(), - deps.transcoder.as_deref(), - request.input, - ) - .await - { - Ok(input) => PreprocessRunInputOutcome::Succeeded { input }, - Err(failure) => PreprocessRunInputOutcome::Failed { failure }, - }; - Ok(PreprocessRunInputActivityResult { outcome }) -} - -async fn rewrite_run_input( - blobs: &dyn BlobStore, - transcriber: &dyn AudioTranscriber, - transcoder: Option<&dyn AudioTranscoder>, - input: Vec, -) -> Result, PreprocessRunInputFailure> { - let mut rewritten = Vec::with_capacity(input.len()); - for entry in input { - if !is_audio_entry(&entry) { - rewritten.push(entry); - continue; - } - rewritten.push(transcribe_entry(blobs, transcriber, transcoder, entry).await?); - } - Ok(rewritten) -} - -async fn transcribe_entry( - blobs: &dyn BlobStore, - transcriber: &dyn AudioTranscriber, - transcoder: Option<&dyn AudioTranscoder>, - entry: ContextEntryInput, -) -> Result { - let mime = normalized_mime(entry.content.media_type.as_deref()); - let name = audio_label(&entry); - if !PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&mime.as_str()) - && !TRANSCODABLE_AUDIO_MIMES.contains(&mime.as_str()) - { - return Err(failure( - PreprocessRunInputFailureKind::UnsupportedAudioMime, - format!( - "unsupported audio mime type {mime} for {name}; accepted: {}", - accepted_audio_mimes().join(", ") - ), - )); - } - - let info = blobs - .stat_blob(&entry.content.content_ref) - .await - .map_err(map_audio_blob_error)?; - if info.byte_len > MAX_AUDIO_BYTES { - return Err(failure( - PreprocessRunInputFailureKind::AudioBlobTooLarge, - format!( - "audio entry {name} blob {} is {} bytes; the limit is {MAX_AUDIO_BYTES} bytes", - entry.content.content_ref, info.byte_len - ), - )); - } - - let bytes = blobs - .read_bytes(&entry.content.content_ref) - .await - .map_err(map_audio_blob_error)?; - if let Some(duration_ms) = audio_duration_ms(&mime, &bytes) - && duration_ms > MAX_AUDIO_DURATION_MS - { - return Err(failure( - PreprocessRunInputFailureKind::AudioDurationTooLong, - format!( - "audio entry {name} duration is {}; the limit is {}", - format_duration_ms(duration_ms), - format_duration_ms(MAX_AUDIO_DURATION_MS) - ), - )); - } - let audio = prepare_audio_for_transcription(bytes, &mime, &name, transcoder).await?; - if let Some(duration_ms) = audio_duration_ms(&audio.mime, &audio.bytes) - && duration_ms > MAX_AUDIO_DURATION_MS - { - return Err(failure( - PreprocessRunInputFailureKind::AudioDurationTooLong, - format!( - "audio entry {name} duration is {}; the limit is {}", - format_duration_ms(duration_ms), - format_duration_ms(MAX_AUDIO_DURATION_MS) - ), - )); - } - let transcript = transcriber - .transcribe(AudioTranscriptionRequest { - bytes: audio.bytes, - mime: audio.mime, - name: audio.name, - }) - .await - .map_err(map_transcription_error)?; - let transcript = AudioTranscript { - filename: name, - text: transcript.text.trim().to_owned(), - }; - let transcript_bytes = serde_json::to_vec(&transcript).map_err(|error| { - failure( - PreprocessRunInputFailureKind::TranscriptionFailure, - format!("failed to encode transcript: {error}"), - ) - })?; - let transcript_ref = blobs.put_bytes(transcript_bytes).await.map_err(|error| { - failure( - PreprocessRunInputFailureKind::TranscriptionFailure, - format!("failed to store audio transcript: {error}"), - ) - })?; - - Ok(ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: transcript_ref, - media_type: Some("application/json".to_owned()), - provider_kind: Some(AUDIO_TRANSCRIPT_PROVIDER_KIND.to_owned()), - }, - preview: Some(transcript.header().chars().take(256).collect()), - origin: entry.origin.clone(), - provenance_ref: Some(entry.content.content_ref.clone()), - token_estimate: None, - }) -} - -struct PreparedAudio { - bytes: Vec, - mime: String, - name: String, -} - -async fn prepare_audio_for_transcription( - bytes: Vec, - mime: &str, - name: &str, - transcoder: Option<&dyn AudioTranscoder>, -) -> Result { - if PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&mime) { - return Ok(PreparedAudio { - bytes, - mime: mime.to_owned(), - name: audio_filename(name), - }); - } - let Some(transcoder) = transcoder else { - return Err(failure( - PreprocessRunInputFailureKind::TranscoderUnavailable, - format!( - "audio entry {name} has mime type {mime}, which requires a transcoder, but no audio transcoder is configured" - ), - )); - }; - let output = transcoder - .transcode(AudioTranscodeRequest { - bytes, - mime: mime.to_owned(), - name: audio_filename(name), - }) - .await - .map_err(map_transcode_error)?; - if output.bytes.len() as u64 > MAX_AUDIO_BYTES { - return Err(failure( - PreprocessRunInputFailureKind::TranscodeFailure, - format!( - "transcoded audio entry {name} is {} bytes; the limit is {MAX_AUDIO_BYTES} bytes", - output.bytes.len() - ), - )); - } - if !PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&output.mime.as_str()) { - return Err(failure( - PreprocessRunInputFailureKind::TranscodeFailure, - format!( - "audio transcoder returned unsupported mime type {} for {name}", - output.mime - ), - )); - } - Ok(PreparedAudio { - bytes: output.bytes, - mime: output.mime, - name: output.name, - }) -} - -fn is_audio_entry(entry: &ContextEntryInput) -> bool { - entry - .content - .media_type - .as_deref() - .map(|mime| mime.trim().to_ascii_lowercase().starts_with("audio/")) - .unwrap_or(false) -} - -fn normalized_mime(mime: Option<&str>) -> String { - let mime = mime - .unwrap_or_default() - .split(';') - .next() - .unwrap_or_default() - .trim() - .to_ascii_lowercase(); - match mime.as_str() { - "audio/mp3" => "audio/mpeg", - "audio/x-m4a" | "audio/m4a" => "audio/mp4", - "audio/x-wav" | "audio/wave" | "audio/vnd.wave" => "audio/wav", - "audio/oga" | "audio/opus" => "audio/ogg", - "audio/x-aac" => "audio/aac", - "audio/3gp" => "audio/3gpp", - "audio/3g2" => "audio/3gpp2", - other => other, - } - .to_owned() -} - -fn audio_label(entry: &ContextEntryInput) -> String { - let Some(preview) = entry.preview.as_deref().map(str::trim) else { - return "audio".to_owned(); - }; - preview - .strip_prefix("[audio: ") - .and_then(|value| value.strip_suffix(']')) - .map(str::trim) - .filter(|value| !value.is_empty()) - .unwrap_or("audio") - .to_owned() -} - -fn audio_filename(label: &str) -> String { - let trimmed = label.trim(); - if trimmed.is_empty() || trimmed == "audio" { - "audio.ogg".to_owned() - } else { - trimmed.to_owned() - } -} - -fn transcoded_audio_filename(label: &str) -> String { - let filename = audio_filename(label); - let stem = Path::new(&filename) - .file_stem() - .and_then(OsStr::to_str) - .filter(|value| !value.trim().is_empty()) - .unwrap_or("audio"); - format!("{stem}.wav") -} - -fn accepted_audio_mimes() -> Vec<&'static str> { - PROVIDER_ACCEPTED_AUDIO_MIMES - .iter() - .chain(TRANSCODABLE_AUDIO_MIMES.iter()) - .copied() - .collect() -} - -fn extension_for_mime(mime: &str) -> &'static str { - match mime { - "audio/mpeg" => "mp3", - "audio/mp4" => "m4a", - "audio/wav" => "wav", - "audio/webm" => "webm", - "audio/ogg" => "ogg", - "audio/aac" => "aac", - "audio/amr" => "amr", - "audio/3gpp" => "3gp", - "audio/3gpp2" => "3g2", - _ => "bin", - } -} - -fn ffmpeg_args(input_path: &Path, output_path: &Path) -> Vec { - [ - OsString::from("-nostdin"), - OsString::from("-hide_banner"), - OsString::from("-loglevel"), - OsString::from("error"), - OsString::from("-y"), - OsString::from("-i"), - input_path.as_os_str().to_owned(), - OsString::from("-vn"), - OsString::from("-ac"), - OsString::from("1"), - OsString::from("-ar"), - OsString::from("16000"), - OsString::from("-acodec"), - OsString::from("pcm_s16le"), - OsString::from("-f"), - OsString::from("wav"), - output_path.as_os_str().to_owned(), - ] - .into_iter() - .collect() -} - -fn ffmpeg_failure_message(stderr: &[u8]) -> String { - let stderr = String::from_utf8_lossy(stderr); - let stderr = stderr.trim(); - if stderr.is_empty() { - "ffmpeg exited with a non-zero status".to_owned() - } else { - format!("ffmpeg exited with a non-zero status: {stderr}") - } -} - -fn audio_duration_ms(mime: &str, bytes: &[u8]) -> Option { - // Cheap duration enforcement is intentionally narrow for the first cut: - // OGG/Opus and WAV are covered; MP3/M4A/WebM rely on the byte cap unless - // a non-provider container is transcoded to WAV. - match mime { - "audio/ogg" => ogg_opus_duration_ms(bytes), - "audio/wav" => wav_duration_ms(bytes), - _ => None, - } -} - -fn ogg_opus_duration_ms(bytes: &[u8]) -> Option { - let mut offset = 0usize; - let mut last_granule = None; - while offset + 27 <= bytes.len() { - if &bytes[offset..offset + 4] != b"OggS" { - return None; - } - let granule = u64::from_le_bytes(bytes[offset + 6..offset + 14].try_into().ok()?); - if granule != u64::MAX { - last_granule = Some(granule); - } - let segments = bytes[offset + 26] as usize; - let lacing_start = offset + 27; - let lacing_end = lacing_start.checked_add(segments)?; - if lacing_end > bytes.len() { - return None; - } - let mut body_len = 0usize; - for segment in &bytes[lacing_start..lacing_end] { - body_len = body_len.checked_add(*segment as usize)?; - } - offset = lacing_end.checked_add(body_len)?; - } - last_granule.map(|samples| samples.saturating_mul(1000) / 48_000) -} - -fn wav_duration_ms(bytes: &[u8]) -> Option { - if bytes.len() < 12 || &bytes[0..4] != b"RIFF" || &bytes[8..12] != b"WAVE" { - return None; - } - let mut offset = 12usize; - let mut byte_rate = None; - let mut data_len = None; - while offset + 8 <= bytes.len() { - let chunk_id = &bytes[offset..offset + 4]; - let chunk_len = u32::from_le_bytes(bytes[offset + 4..offset + 8].try_into().ok()?) as usize; - let data_start = offset + 8; - let data_end = data_start.checked_add(chunk_len)?; - if data_end > bytes.len() { - return None; - } - if chunk_id == b"fmt " && chunk_len >= 16 { - byte_rate = Some(u32::from_le_bytes( - bytes[data_start + 8..data_start + 12].try_into().ok()?, - ) as u64); - } else if chunk_id == b"data" { - data_len = Some(chunk_len as u64); - } - if byte_rate.is_some() && data_len.is_some() { - break; - } - offset = data_end.checked_add(chunk_len % 2)?; - } - let byte_rate = byte_rate?; - if byte_rate == 0 { - return None; - } - Some(data_len?.saturating_mul(1000) / byte_rate) -} - -fn format_duration_ms(duration_ms: u64) -> String { - format!("{}s", duration_ms.div_ceil(1000)) -} - -fn map_audio_blob_error(error: BlobStoreError) -> PreprocessRunInputFailure { - match error { - BlobStoreError::NotFound { blob_ref } => failure( - PreprocessRunInputFailureKind::AudioBlobMissing, - format!("audio blob not found: {blob_ref}"), - ), - BlobStoreError::Store { message } => { - failure(PreprocessRunInputFailureKind::AudioBlobMissing, message) - } - } -} - -fn map_transcription_error(error: AudioTranscriptionError) -> PreprocessRunInputFailure { - failure( - PreprocessRunInputFailureKind::TranscriptionFailure, - error.message, - ) -} - -fn map_transcode_error(error: AudioTranscodeError) -> PreprocessRunInputFailure { - failure( - PreprocessRunInputFailureKind::TranscodeFailure, - error.message, - ) -} - -fn map_openai_error(error: LlmApiError) -> AudioTranscriptionError { - AudioTranscriptionError { - message: error.to_string(), - } -} - -fn failure( - kind: PreprocessRunInputFailureKind, - message: impl Into, -) -> PreprocessRunInputFailure { - PreprocessRunInputFailure { - kind, - message: message.into(), - } -} - -pub(super) fn default_openai_audio_transcriber( - provider_keys: Arc, -) -> Result, anyhow::Error> { - let client = oai::Client::new(oai::Config::from_env_allow_missing_key()) - .map_err(|error| anyhow::anyhow!("construct OpenAI audio client: {error}"))?; - Ok(Arc::new(OpenAiAudioTranscriber::new( - Arc::new(client), - provider_keys, - ))) -} - -pub fn default_audio_transcoder_from_env() -> anyhow::Result>> { - let Some(kind) = env::var("LIGHTSPEED_AUDIO_TRANSCODER") - .ok() - .map(|value| value.trim().to_ascii_lowercase()) - .filter(|value| !value.is_empty() && value != "none") - else { - return Ok(None); - }; - match kind.as_str() { - "ffmpeg" => Ok(Some(Arc::new(FfmpegAudioTranscoder::from_env()))), - other => anyhow::bail!( - "unsupported LIGHTSPEED_AUDIO_TRANSCODER value {other:?}; expected \"ffmpeg\" or \"none\"" - ), - } -} - -#[cfg(test)] -mod tests { - use super::*; - use engine::storage::{BlobStore, InMemoryBlobStore}; - use std::sync::Mutex; - - #[derive(Clone)] - struct StaticTranscriber { - text: String, - } - - #[async_trait] - impl AudioTranscriber for StaticTranscriber { - async fn transcribe( - &self, - _request: AudioTranscriptionRequest, - ) -> Result { - Ok(AudioTranscription { - text: self.text.clone(), - }) - } - } - - struct RecordingTranscriber { - text: String, - requests: Mutex>, - } - - impl RecordingTranscriber { - fn new(text: impl Into) -> Self { - Self { - text: text.into(), - requests: Mutex::new(Vec::new()), - } - } - - fn requests(&self) -> Vec { - self.requests - .lock() - .expect("recording transcriber requests") - .clone() - } - } - - #[async_trait] - impl AudioTranscriber for RecordingTranscriber { - async fn transcribe( - &self, - request: AudioTranscriptionRequest, - ) -> Result { - self.requests - .lock() - .expect("recording transcriber requests") - .push(request); - Ok(AudioTranscription { - text: self.text.clone(), - }) - } - } - - struct StaticTranscoder { - result: Result, - requests: Mutex>, - } - - impl StaticTranscoder { - fn new(result: Result) -> Self { - Self { - result, - requests: Mutex::new(Vec::new()), - } - } - - fn requests(&self) -> Vec { - self.requests - .lock() - .expect("static transcoder requests") - .clone() - } - } - - #[async_trait] - impl AudioTranscoder for StaticTranscoder { - async fn transcode( - &self, - request: AudioTranscodeRequest, - ) -> Result { - self.requests - .lock() - .expect("static transcoder requests") - .push(request); - self.result.clone() - } - } - - #[tokio::test(flavor = "current_thread")] - async fn audio_entries_are_rewritten_to_transcript_text() { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(b"OggS fake".to_vec()) - .await - .expect("put audio"); - let input = vec![ - text_entry(blobs.put_bytes(b"before".to_vec()).await.expect("put text")), - ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/ogg".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: voice.ogg]".to_owned()), - origin: Some("user:operator".into()), - provenance_ref: None, - token_estimate: None, - }, - ]; - - let rewritten = rewrite_run_input( - &blobs, - &StaticTranscriber { - text: "please summarize this".to_owned(), - }, - None, - input, - ) - .await - .expect("rewrite"); - - assert_eq!(rewritten.len(), 2); - assert_eq!(rewritten[1].origin.as_deref(), Some("user:operator")); - assert_eq!( - rewritten[1].content.media_type.as_deref(), - Some("application/json") - ); - let raw = blobs - .read_text(&rewritten[1].content.content_ref) - .await - .expect("read transcript"); - assert_eq!( - serde_json::from_str::(&raw).unwrap(), - AudioTranscript { - filename: "voice.ogg".into(), - text: "please summarize this".into() - } - ); - assert_eq!(rewritten[1].provenance_ref.as_ref(), Some(&audio_ref)); - assert_eq!( - api_projection::project_content_text(&blobs, &rewritten[1].content) - .await - .unwrap() - .as_deref(), - Some("please summarize this") - ); - } - - #[tokio::test(flavor = "current_thread")] - async fn transcodable_audio_without_transcoder_fails_whole_group() { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(b"aac fake".to_vec()) - .await - .expect("put audio"); - let input = vec![ - text_entry(blobs.put_bytes(b"before".to_vec()).await.expect("put text")), - ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/aac".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: voice.aac]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }, - ]; - - let failure = rewrite_run_input( - &blobs, - &StaticTranscriber { - text: "unused".to_owned(), - }, - None, - input, - ) - .await - .expect_err("transcodable audio without transcoder must fail group"); - - assert_eq!( - failure.kind, - PreprocessRunInputFailureKind::TranscoderUnavailable - ); - } - - #[tokio::test(flavor = "current_thread")] - async fn transcodable_audio_is_transcoded_before_transcription() { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(b"aac fake".to_vec()) - .await - .expect("put audio"); - let input = vec![ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/aac".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: voice.aac]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }]; - let transcriber = RecordingTranscriber::new("transcoded request"); - let transcoder = StaticTranscoder::new(Ok(AudioTranscodeOutput { - bytes: b"RIFF\x24\x00\x00\x00WAVEfmt ".to_vec(), - mime: "audio/wav".to_owned(), - name: "voice.wav".to_owned(), - })); - - let rewritten = rewrite_run_input(&blobs, &transcriber, Some(&transcoder), input) - .await - .expect("rewrite"); - - let raw = blobs - .read_text(&rewritten[0].content.content_ref) - .await - .expect("read transcript"); - assert_eq!( - serde_json::from_str::(&raw).unwrap(), - AudioTranscript { - filename: "voice.aac".into(), - text: "transcoded request".into() - } - ); - assert_eq!(rewritten[0].provenance_ref.as_ref(), Some(&audio_ref)); - assert_eq!( - api_projection::project_content_text(&blobs, &rewritten[0].content) - .await - .unwrap() - .as_deref(), - Some("transcoded request") - ); - let transcode_requests = transcoder.requests(); - assert_eq!(transcode_requests.len(), 1); - assert_eq!(transcode_requests[0].mime, "audio/aac"); - assert_eq!(transcode_requests[0].name, "voice.aac"); - assert_eq!(transcode_requests[0].bytes, b"aac fake"); - let transcription_requests = transcriber.requests(); - assert_eq!(transcription_requests.len(), 1); - assert_eq!(transcription_requests[0].mime, "audio/wav"); - assert_eq!(transcription_requests[0].name, "voice.wav"); - assert_eq!( - transcription_requests[0].bytes, - b"RIFF\x24\x00\x00\x00WAVEfmt " - ); - } - - #[tokio::test(flavor = "current_thread")] - async fn transcode_failure_fails_whole_group() { - let failure = transcode_failure(AudioTranscodeError { - message: "unsupported codec".to_owned(), - }) - .await; - - assert_eq!( - failure.kind, - PreprocessRunInputFailureKind::TranscodeFailure - ); - } - - #[test] - fn ffmpeg_command_args_normalize_audio_to_mono_wav() { - let args = ffmpeg_args(Path::new("/tmp/in.aac"), Path::new("/tmp/out.wav")); - let args: Vec = args - .iter() - .map(|arg| arg.to_string_lossy().into_owned()) - .collect(); - - assert_eq!( - args, - vec![ - "-nostdin", - "-hide_banner", - "-loglevel", - "error", - "-y", - "-i", - "/tmp/in.aac", - "-vn", - "-ac", - "1", - "-ar", - "16000", - "-acodec", - "pcm_s16le", - "-f", - "wav", - "/tmp/out.wav" - ] - ); - } - - #[tokio::test(flavor = "current_thread")] - #[ignore = "requires ffmpeg on PATH or LIGHTSPEED_FFMPEG_PATH"] - async fn ffmpeg_audio_transcoder_smoke_test() { - let transcoder = FfmpegAudioTranscoder::from_env(); - - let output = transcoder - .transcode(AudioTranscodeRequest { - bytes: tiny_wav_bytes(), - mime: "audio/wav".to_owned(), - name: "voice.wav".to_owned(), - }) - .await - .expect("ffmpeg should transcode tiny wav input"); - - assert_eq!(output.mime, "audio/wav"); - assert_eq!(output.name, "voice.wav"); - assert!(output.bytes.starts_with(b"RIFF")); - assert_eq!(audio_duration_ms(&output.mime, &output.bytes), Some(1000)); - } - - #[tokio::test(flavor = "current_thread")] - async fn long_ogg_audio_fails_duration_cap() { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(ogg_page(((MAX_AUDIO_DURATION_MS / 1000) + 1) * 48_000)) - .await - .expect("put audio"); - let input = vec![ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/ogg".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: long.ogg]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }]; - - let failure = rewrite_run_input( - &blobs, - &StaticTranscriber { - text: "unused".to_owned(), - }, - None, - input, - ) - .await - .expect_err("long audio must fail group"); - - assert_eq!( - failure.kind, - PreprocessRunInputFailureKind::AudioDurationTooLong - ); - assert!(failure.message.contains("long.ogg")); - } - - fn text_entry(content_ref: engine::BlobRef) -> ContextEntryInput { - ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef::text(content_ref), - preview: None, - origin: None, - provenance_ref: None, - token_estimate: None, - } - } - - async fn transcode_failure(error: AudioTranscodeError) -> PreprocessRunInputFailure { - let blobs = InMemoryBlobStore::new(); - let audio_ref = blobs - .put_bytes(b"aac fake".to_vec()) - .await - .expect("put audio"); - let input = vec![ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: audio_ref.clone(), - media_type: Some("audio/aac".to_owned()), - provider_kind: None, - }, - preview: Some("[audio: voice.aac]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }]; - let transcoder = StaticTranscoder::new(Err(error)); - - rewrite_run_input( - &blobs, - &StaticTranscriber { - text: "unused".to_owned(), - }, - Some(&transcoder), - input, - ) - .await - .expect_err("transcode failure must reject group") - } - - fn tiny_wav_bytes() -> Vec { - let sample_rate = 8_000u32; - let channels = 1u16; - let bits_per_sample = 16u16; - let sample_count = sample_rate as usize; - let byte_rate = sample_rate * channels as u32 * bits_per_sample as u32 / 8; - let block_align = channels * bits_per_sample / 8; - let data_len = sample_count * block_align as usize; - - let mut bytes = Vec::with_capacity(44 + data_len); - bytes.extend_from_slice(b"RIFF"); - bytes.extend_from_slice(&(36 + data_len as u32).to_le_bytes()); - bytes.extend_from_slice(b"WAVE"); - bytes.extend_from_slice(b"fmt "); - bytes.extend_from_slice(&16u32.to_le_bytes()); - bytes.extend_from_slice(&1u16.to_le_bytes()); - bytes.extend_from_slice(&channels.to_le_bytes()); - bytes.extend_from_slice(&sample_rate.to_le_bytes()); - bytes.extend_from_slice(&byte_rate.to_le_bytes()); - bytes.extend_from_slice(&block_align.to_le_bytes()); - bytes.extend_from_slice(&bits_per_sample.to_le_bytes()); - bytes.extend_from_slice(b"data"); - bytes.extend_from_slice(&(data_len as u32).to_le_bytes()); - bytes.resize(44 + data_len, 0); - bytes - } - - fn ogg_page(granule_position: u64) -> Vec { - let mut page = Vec::new(); - page.extend_from_slice(b"OggS"); - page.push(0); - page.push(0); - page.extend_from_slice(&granule_position.to_le_bytes()); - page.extend_from_slice(&1u32.to_le_bytes()); - page.extend_from_slice(&0u32.to_le_bytes()); - page.extend_from_slice(&0u32.to_le_bytes()); - page.push(1); - page.push(1); - page.push(0); - page - } -} diff --git a/crates/temporal-server/src/worker/activities/state.rs b/crates/temporal-server/src/worker/activities/state.rs index f3c538bfb..014e496b3 100644 --- a/crates/temporal-server/src/worker/activities/state.rs +++ b/crates/temporal-server/src/worker/activities/state.rs @@ -31,7 +31,7 @@ use crate::{ worker::{BrokerSecretResolver, SessionTools, StoredProviderKeyResolver}, }; -use super::preprocess::{ +use super::audio::{ AudioTranscoder, AudioTranscriber, OpenAiAudioTranscriber, UnavailableAudioTranscriber, default_audio_transcoder_from_env, default_openai_audio_transcriber, }; @@ -77,8 +77,9 @@ pub struct RuntimeProjectionActivityDeps { pub(super) profiles: Option>, } +/// Audio I/O dependencies for standalone transcription. #[derive(Clone)] -pub struct PreprocessActivityDeps { +pub struct AudioActivityDeps { pub(super) blobs: Arc, pub(super) transcriber: Arc, pub(super) transcoder: Option>, @@ -119,7 +120,7 @@ pub struct ActivityState { tools: ToolActivityDeps, runtime_projection: Option, pub(super) preparation_store: Option>, - preprocess: PreprocessActivityDeps, + audio: AudioActivityDeps, environment_jobs: Option, workflow_tool_executions: Option, subagents: Option, @@ -151,7 +152,7 @@ impl ActivityState { }, runtime_projection: None, preparation_store: None, - preprocess: PreprocessActivityDeps { + audio: AudioActivityDeps { blobs: blobs.clone(), transcriber: Arc::new(UnavailableAudioTranscriber), transcoder: None, @@ -185,7 +186,7 @@ impl ActivityState { } pub fn with_audio_transcriber(mut self, transcriber: Arc) -> Self { - self.preprocess.transcriber = transcriber; + self.audio.transcriber = transcriber; self } @@ -207,7 +208,7 @@ impl ActivityState { } pub fn with_audio_transcoder(mut self, transcoder: Arc) -> Self { - self.preprocess.transcoder = Some(transcoder); + self.audio.transcoder = Some(transcoder); self } @@ -443,8 +444,8 @@ impl ActivityState { self.runtime_projection.as_ref() } - pub(super) fn preprocess(&self) -> &PreprocessActivityDeps { - &self.preprocess + pub(super) fn audio(&self) -> &AudioActivityDeps { + &self.audio } pub(super) fn environment_jobs(&self) -> Option<&EnvironmentJobActivityDeps> { diff --git a/crates/temporal-server/src/worker/activities/transcriptions.rs b/crates/temporal-server/src/worker/activities/transcriptions.rs new file mode 100644 index 000000000..81b4baa86 --- /dev/null +++ b/crates/temporal-server/src/worker/activities/transcriptions.rs @@ -0,0 +1,156 @@ +use super::{audio, state::ActivityState}; +use api::{TranscriptionFailure, TranscriptionFailureKind as Kind}; +use engine::{BlobRef, storage::BlobStore}; +use temporal_workflow::{TranscriptionActivityResult, TranscriptionWorkflowArgs}; +use temporalio_sdk::activities::ActivityError; + +fn activity_error(error: impl Into) -> ActivityError { + ActivityError::from(error.into()) +} +fn failure(kind: Kind, message: impl Into) -> TranscriptionFailure { + TranscriptionFailure { + kind, + message: message.into(), + } +} + +pub(super) async fn execute( + state: &ActivityState, + args: TranscriptionWorkflowArgs, +) -> Result { + match transcribe(state, &args).await { + Ok(transcript) => Ok(TranscriptionActivityResult::Succeeded { + transcript_ref: transcript.to_string(), + }), + Err(ExecutionError::Terminal(failure)) => { + Ok(TranscriptionActivityResult::Failed { failure }) + } + Err(ExecutionError::Retry(error)) => Err(activity_error(anyhow::anyhow!(error))), + } +} + +enum ExecutionError { + Terminal(TranscriptionFailure), + Retry(String), +} +impl From for ExecutionError { + fn from(error: engine::storage::BlobStoreError) -> Self { + if matches!(error, engine::storage::BlobStoreError::NotFound { .. }) { + Self::Terminal(failure( + Kind::InvalidAudio, + "Audio is unavailable; upload it again.", + )) + } else { + Self::Retry("Transcription content storage is unavailable.".into()) + } + } +} + +async fn transcribe( + state: &ActivityState, + record: &TranscriptionWorkflowArgs, +) -> Result { + let deps = state.audio(); + let request = &record.request; + let source = BlobRef::parse(&request.audio.blob_ref).map_err(|_| { + ExecutionError::Terminal(failure(Kind::InvalidAudio, "Invalid audio reference.")) + })?; + let mime = audio::normalized_mime(Some(&request.audio.mime)); + if !audio::PROVIDER_ACCEPTED_AUDIO_MIMES.contains(&mime.as_str()) + && !audio::TRANSCODABLE_AUDIO_MIMES.contains(&mime.as_str()) + { + return Err(ExecutionError::Terminal(failure( + Kind::InvalidAudio, + "Unsupported audio MIME type.", + ))); + } + if deps.blobs.stat_blob(&source).await?.byte_len > audio::MAX_AUDIO_BYTES { + return Err(ExecutionError::Terminal(failure( + Kind::InvalidAudio, + "Audio exceeds the 25 MiB limit.", + ))); + } + let bytes = deps.blobs.read_bytes(&source).await?; + check_duration(&mime, &bytes)?; + let prepared = audio::prepare_audio_for_transcription( + bytes, + &mime, + &request.audio.name, + deps.transcoder.as_deref(), + ) + .await + .map_err(|e| ExecutionError::Terminal(failure(Kind::InvalidAudio, e.to_string())))?; + check_duration(&prepared.mime, &prepared.bytes)?; + let result = deps + .transcriber + .transcribe(audio::AudioTranscriptionRequest { + bytes: prepared.bytes, + mime: prepared.mime, + name: prepared.name, + model: record.model.clone(), + language: request.language.clone(), + prompt: request.prompt.clone(), + }) + .await + .map_err(|e| { + if e.retryable { + ExecutionError::Retry(e.message) + } else { + ExecutionError::Terminal(failure( + if e.configuration { + Kind::Configuration + } else { + Kind::Provider + }, + e.message, + )) + } + })?; + if result.text.len() > 1024 * 1024 { + return Err(ExecutionError::Terminal(failure( + Kind::Provider, + "Transcript exceeds the 1 MiB limit.", + ))); + } + let result_ref = deps + .blobs + .put_bytes(result.text.trim().as_bytes().to_vec()) + .await?; + Ok(result_ref) +} + +fn check_duration(mime: &str, bytes: &[u8]) -> Result<(), ExecutionError> { + if audio::audio_duration_ms(mime, bytes) + .is_some_and(|duration| duration > audio::MAX_AUDIO_DURATION_MS) + { + return Err(ExecutionError::Terminal(failure( + Kind::InvalidAudio, + "Audio exceeds the ten-minute duration limit.", + ))); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn duration_limit_is_enforced_before_provider_call() { + // One complete Ogg page with a one-byte body and a 48 kHz granule clock. + let mut page = b"OggS".to_vec(); + page.resize(29, 0); + page[26] = 1; + page[27] = 1; + page[6..14].copy_from_slice(&(600_u64 * 48_000).to_le_bytes()); + assert!(check_duration("audio/ogg", &page).is_ok()); + page[6..14].copy_from_slice(&(601_u64 * 48_000).to_le_bytes()); + assert!(matches!( + check_duration("audio/ogg", &page), + Err(ExecutionError::Terminal(TranscriptionFailure { + kind: Kind::InvalidAudio, + .. + })) + )); + } +} diff --git a/crates/temporal-server/src/worker/channels.rs b/crates/temporal-server/src/worker/channels.rs index 64b4dcbf8..7a843a519 100644 --- a/crates/temporal-server/src/worker/channels.rs +++ b/crates/temporal-server/src/worker/channels.rs @@ -31,6 +31,51 @@ impl ChannelWorkerActivities { #[activities] impl ChannelWorkerActivities { + #[activity(name = ACTIVITY_CHAT_TRANSCRIBE_MEDIA)] + pub async fn transcribe_media( + self: Arc, + _ctx: ActivityContext, + request: ChatTranscribeMediaRequest, + ) -> Result { + let api = self.universes.api_for(request.active.universe_id).await?; + let active = + crate::channels::activities::assert_trigger_active(&api, request.active.clone()) + .await?; + if let ChatTriggerActiveResult::Inactive { reason } = active { + return Err(ActivityError::application( + temporalio_common::error::ApplicationFailure::non_retryable(anyhow::anyhow!( + reason + )), + )); + } + let owner = api::Attribution::Internal { + component: "channel".into(), + cause: request.active.trigger_id.to_string(), + }; + api.admit_transcription( + api::TranscriptionStartParams { + idempotency_key: request.idempotency_key, + audio: request.audio, + model: None, + language: None, + prompt: None, + }, + owner, + ) + .await + .map_err(|error| { + if matches!(error.kind, api::AgentApiErrorKind::Internal) { + ActivityError::from(anyhow::anyhow!(error.to_string())) + } else { + ActivityError::application( + temporalio_common::error::ApplicationFailure::non_retryable(anyhow::anyhow!( + error.to_string() + )), + ) + } + }) + } + #[activity(name = ACTIVITY_CHAT_TOOL_DECLARATIONS)] pub async fn chat_tool_declarations( self: Arc, diff --git a/crates/temporal-server/src/worker/mod.rs b/crates/temporal-server/src/worker/mod.rs index bfebe1f10..2775130b4 100644 --- a/crates/temporal-server/src/worker/mod.rs +++ b/crates/temporal-server/src/worker/mod.rs @@ -20,9 +20,9 @@ use temporal_workflow::{ }; pub use activities::{ - ActivityState, AudioTranscodeError, AudioTranscodeOutput, AudioTranscodeRequest, - AudioTranscoder, AudioTranscriber, AudioTranscription, AudioTranscriptionError, - AudioTranscriptionRequest, FfmpegAudioTranscoder, LlmActivityDeps, PreprocessActivityDeps, + ActivityState, AudioActivityDeps, AudioTranscodeError, AudioTranscodeOutput, + AudioTranscodeRequest, AudioTranscoder, AudioTranscriber, AudioTranscription, + AudioTranscriptionError, AudioTranscriptionRequest, FfmpegAudioTranscoder, LlmActivityDeps, RuntimeProjectionActivityDeps, StorageActivityDeps, ToolActivityDeps, WorkerActivities, default_audio_transcoder_from_env, subagent_catalog_snapshot, }; @@ -41,21 +41,19 @@ pub use temporal_workflow::{ ACTIVITY_CREATE_OR_LOAD_SESSION, ACTIVITY_ENVIRONMENT_JOB_CANCEL, ACTIVITY_ENVIRONMENT_JOB_POLL, ACTIVITY_ENVIRONMENT_JOB_PREPARE_WORKFLOW_TOOL, ACTIVITY_ENVIRONMENT_JOB_START, ACTIVITY_LLM_GENERATE, ACTIVITY_MATERIALIZE_AWAIT_RESULT, - ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PREPROCESS_RUN_INPUT, ACTIVITY_PUT_BLOB, - ACTIVITY_READ_BLOB, ACTIVITY_RUNTIME_PROJECTION_REFRESH, - ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, - ACTIVITY_SUBAGENT_RESOLVE, ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, - ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, - AgentSessionWorkflow, AppendEventsRequest, ContextCompactActivityRequest, - CreateOrLoadSessionRequest, CreateOrLoadSessionResult, DEFAULT_TASK_QUEUE, - DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, EnvironmentJobCancelActivityRequest, - EnvironmentJobPollActivityRequest, EnvironmentJobPollActivityResult, - EnvironmentJobStartActivityRequest, EnvironmentJobStartActivityResult, EnvironmentJobWorkflow, - EnvironmentJobWorkflowArgs, FAKE_TOOL_NAME, LlmGenerateActivityRequest, - PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult, PutBlobRequest, - ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, - RuntimeProjectionRefreshActivityResult, SubagentExecutionWorkflow, - ToolInvokeBatchActivityRequest, ToolInvokeCallActivityRequest, + ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PUT_BLOB, ACTIVITY_READ_BLOB, + ACTIVITY_RUNTIME_PROJECTION_REFRESH, ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, + ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, ACTIVITY_SUBAGENT_RESOLVE, + ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, + ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, AgentSessionWorkflow, AppendEventsRequest, + ContextCompactActivityRequest, CreateOrLoadSessionRequest, CreateOrLoadSessionResult, + DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, + EnvironmentJobCancelActivityRequest, EnvironmentJobPollActivityRequest, + EnvironmentJobPollActivityResult, EnvironmentJobStartActivityRequest, + EnvironmentJobStartActivityResult, EnvironmentJobWorkflow, EnvironmentJobWorkflowArgs, + FAKE_TOOL_NAME, LlmGenerateActivityRequest, PutBlobRequest, ReadBlobRequest, ReadBlobResult, + RuntimeProjectionRefreshActivityRequest, RuntimeProjectionRefreshActivityResult, + SubagentExecutionWorkflow, ToolInvokeBatchActivityRequest, ToolInvokeCallActivityRequest, ToolPreparePromiseControlsActivityRequest, connect_temporal, default_run_config, default_session_config, }; @@ -99,6 +97,7 @@ pub fn sessions_worker( ) -> anyhow::Result { let worker_options = WorkerOptions::new(task_queue) .register_workflow::() + .register_workflow::() .register_workflow::() .register_workflow::() .register_activities(activities) diff --git a/crates/temporal-server/src/worker/reaper.rs b/crates/temporal-server/src/worker/reaper.rs index 142f9f025..d4b582d00 100644 --- a/crates/temporal-server/src/worker/reaper.rs +++ b/crates/temporal-server/src/worker/reaper.rs @@ -1115,7 +1115,7 @@ mod tests { ActiveRun, ModelSelection, ProviderApiKind, RunId, RunSource, RunStatus, storage::{BlobGraphStore as _, BlobStore as _}, }; - use temporal_workflow::{DEFAULT_MODEL, default_run_config, default_session_config}; + use temporal_workflow::{default_run_config, default_session_config}; use super::*; @@ -1200,7 +1200,7 @@ mod tests { ModelSelection { api_kind: ProviderApiKind::OpenAiResponses, provider_id: "openai".to_owned(), - model: DEFAULT_MODEL.to_owned(), + model: "test-model".to_owned(), } } diff --git a/crates/temporal-server/tests/bots_live.rs b/crates/temporal-server/tests/bots_live.rs index fd8c6c031..5469acac0 100644 --- a/crates/temporal-server/tests/bots_live.rs +++ b/crates/temporal-server/tests/bots_live.rs @@ -30,8 +30,7 @@ use api::{ use bots::ids::{bot_controller_workflow_id, bot_main_session_id, bot_schedule_id}; use engine::{CoreAgentLlm, CoreAgentTools, storage::BlobStore}; use support::live::{ - LIVE_TEST_LOCK, live_universe_id, openai_live_model, require_openai_live_env, - require_storage_live_env, + LIVE_TEST_LOCK, live_universe_id, require_openai_live_env, require_storage_live_env, }; use temporal_server::{ config::TaskQueues, @@ -66,6 +65,7 @@ where let _guard = LIVE_TEST_LOCK.lock().await; let universe = live_universe_id()?; let store = pg_store_from_env().await?; + support::live::seed_agent_default(&store, &support::live::openai_live_model()).await?; let queues = TaskQueues::derived_from(format!( "lightspeed-bots-live-{}", uuid::Uuid::new_v4().simple() @@ -77,7 +77,7 @@ where let runtime = core_runtime()?; let client = connect_temporal(&temporal_target, &namespace).await?; - let mut builder = GatewayAgentApi::builder(client.clone(), store.clone()) + let builder = GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(queues.sessions.clone()) .with_bot_task_queue(queues.bots.clone()) .with_channel_task_queue(queues.channels.clone()); @@ -89,10 +89,7 @@ where let tools = Arc::new(FakeTools::new(blobs)) as Arc; ActivityState::from_pg_store(store.clone(), llm, tools) } - Llm::Real => { - builder = builder.with_default_model(openai_live_model()); - ActivityState::from_pg_store_with_default_runtime(store.clone())? - } + Llm::Real => ActivityState::from_pg_store_with_default_runtime(store.clone())?, }; let api = Arc::new(builder.build()); diff --git a/crates/temporal-server/tests/channels_live.rs b/crates/temporal-server/tests/channels_live.rs index e62c0427e..cfbfd1c82 100644 --- a/crates/temporal-server/tests/channels_live.rs +++ b/crates/temporal-server/tests/channels_live.rs @@ -68,6 +68,8 @@ const WAIT: Duration = Duration::from_secs(90); pub struct FakeConnector { deliveries: Arc>>, typing_started: Arc>, + blobs: Option>, + transcription_calls: Arc, } #[activities] @@ -94,13 +96,23 @@ impl FakeConnector { pub async fn prepare_channel_media( self: Arc, _ctx: ActivityContext, - _input: PrepareChannelMediaInput, + input: PrepareChannelMediaInput, ) -> Result { - Err(ActivityError::application( - temporalio_common::error::ApplicationFailure::non_retryable(anyhow::anyhow!( - "the fake connector serves no media" - )), - )) + let blob_ref = self + .blobs + .as_ref() + .expect("fake CAS") + .put_bytes(b"OggS fake voice".to_vec()) + .await + .map_err(|e| ActivityError::from(anyhow::anyhow!(e)))?; + Ok(PrepareChannelMediaResult { + item: channels::media::PreparedMediaItem { + blob_ref: blob_ref.to_string(), + kind: input.media.kind, + mime: input.media.mime, + name: input.media.name, + }, + }) } #[activity(name = ACTIVITY_CONNECTOR_MAINTAIN_TYPING)] @@ -136,6 +148,7 @@ where let _guard = LIVE_TEST_LOCK.lock().await; let universe = live_universe_id()?; let store = pg_store_from_env().await?; + support::live::seed_agent_default(&store, &support::live::openai_live_model()).await?; let queues = TaskQueues::derived_from(format!( "lightspeed-channels-live-{}", uuid::Uuid::new_v4().simple() @@ -153,6 +166,10 @@ where .with_channel_task_queue(queues.channels.clone()) .build(), ); + let connector = FakeConnector { + blobs: Some(store.clone()), + ..Default::default() + }; let blobs: Arc = store.clone(); let llm = Arc::new(FakeLlm::new(blobs.clone()).with_tool_rounds(0)) as Arc; let tools = Arc::new(FakeTools::new(blobs)) as Arc; @@ -162,7 +179,9 @@ where queues.sessions.clone(), WorkerActivities::for_universe( universe, - ActivityState::from_pg_store(store.clone(), llm, tools), + ActivityState::from_pg_store(store.clone(), llm, tools).with_audio_transcriber( + Arc::new(FakeVoiceTranscriber(connector.transcription_calls.clone())), + ), ), )?; let mut bots = bots_worker( @@ -200,7 +219,6 @@ where }), ) .await?; - let connector = FakeConnector::default(); let connector_queue = connector_task_queue(universe, &ChannelProvider::new("telegram"), &account_id); let mut connector_worker = Worker::new( @@ -590,6 +608,7 @@ async fn temporal_live_chat_rebuilds_collected_declarations_and_retains_assets() let deleted = store.delete_dead_blobs(std::slice::from_ref(&tools_ref), 2, &[]).await?; assert_eq!(deleted.len(), 1, "a workflow's cached ref alone does not retain its declaration"); let result = emit_chat_event(&live.api, ChatEmitEventRequest { + transcript_refs: Default::default(), universe_id: store.config().universe_id, bot_id: bot_id.clone(), trigger_id, account_id: live.account_id.clone(), provider: ChannelProvider::new("telegram"), @@ -631,3 +650,105 @@ async fn temporal_live_chat_rebuilds_collected_declarations_and_retains_assets() Ok(()) }).await } + +struct FakeVoiceTranscriber(Arc); +#[async_trait::async_trait] +impl temporal_server::worker::AudioTranscriber for FakeVoiceTranscriber { + async fn transcribe( + &self, + request: temporal_server::worker::AudioTranscriptionRequest, + ) -> Result< + temporal_server::worker::AudioTranscription, + temporal_server::worker::AudioTranscriptionError, + > { + assert_eq!(request.model.model, "channel-speech"); + self.0.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + Ok(temporal_server::worker::AudioTranscription { + text: "/reset spoken words".into(), + }) + } +} + +#[tokio::test(flavor = "current_thread")] +#[ignore = "requires local Temporal and PostgreSQL"] +async fn channels_live_voice_preparation_preserves_provenance_and_redelivery_identity() +-> anyhow::Result<()> { + run_channels_live(|live| async move { + let (bot_id, _, _) = + create_bot_with_chat(&live.api, &live.account_id, ChatPairing::Open).await?; + let current = live + .api + .read_model_defaults(api::ModelDefaultsReadParams {}) + .await? + .result + .defaults; + live.api + .put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::SpeechToText, + model: Some(api::ModelConfig { + provider_id: "fake".into(), + api_kind: "openai:audio-transcriptions".into(), + model: "channel-speech".into(), + }), + expected_revision: current.revision, + }) + .await?; + let mut message = inbound("voice-chat", "voice-1", ""); + message.media.push(api::ChannelInboundMedia { + file_id: "voice-file".into(), + kind: api::ChannelMediaKind::Audio, + mime: "audio/ogg".into(), + name: Some("voice.ogg".into()), + byte_size: None, + }); + assert_eq!( + admit(&live.api, &live.account_id, message.clone()).await?, + ChannelInboundDecision::Bound + ); + wait_for_deliveries(&live.connector, 1).await?; + admit(&live.api, &live.account_id, message).await?; + // Wait behind the redelivery so its ordered processing has completed. + admit( + &live.api, + &live.account_id, + inbound("voice-chat", "text-2", "next message"), + ) + .await?; + wait_for_deliveries(&live.connector, 2).await?; + assert_eq!( + live.connector + .transcription_calls + .load(std::sync::atomic::Ordering::SeqCst), + 1 + ); + let events = live + .api + .list_bot_events(BotEventListParams { + bot_id, + limit: Some(10), + cursor: None, + }) + .await? + .result + .events; + let voice = events + .iter() + .find(|event| !event.media.is_empty()) + .expect("voice event"); + assert!(voice.media[0].text_ref.is_some()); + let store = pg_store_from_env().await?; + let doc: serde_json::Value = serde_json::from_slice( + &store + .read_bytes(&engine::BlobRef::parse(&voice.document_ref)?) + .await?, + )?; + assert_eq!(doc["data"]["message"]["text"], "/reset spoken words"); + assert_eq!( + events.iter().filter(|e| e.kind == "chat.message").count(), + 2, + "spoken command remains ordinary message content" + ); + Ok(()) + }) + .await +} diff --git a/crates/temporal-server/tests/environment_provider_live.rs b/crates/temporal-server/tests/environment_provider_live.rs index f5ef6da0b..5b1bb6e3e 100644 --- a/crates/temporal-server/tests/environment_provider_live.rs +++ b/crates/temporal-server/tests/environment_provider_live.rs @@ -25,7 +25,7 @@ use support::live::{ run_with_live_worker, wait_for_environment_status, }; use temporal_server::{ - DeploymentStores, UniverseRuntime, default_model_from_env, + DeploymentStores, UniverseRuntime, gateway::{GatewayAgentApi, GatewayDeploymentApi}, pg_store_from_env, }; @@ -361,10 +361,10 @@ async fn run_environment_power_live_client( }; let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let suffix = uuid::Uuid::new_v4().simple().to_string(); let provider_id = format!("fake-power-{suffix}"); diff --git a/crates/temporal-server/tests/mcp_live.rs b/crates/temporal-server/tests/mcp_live.rs index 30633c41f..503ff18c6 100644 --- a/crates/temporal-server/tests/mcp_live.rs +++ b/crates/temporal-server/tests/mcp_live.rs @@ -38,7 +38,6 @@ use support::live::{ run_with_live_worker, wait_for_terminal_run, }; use temporal_server::{ - default_model_from_env, gateway::{DEFAULT_MAX_REQUEST_BODY_BYTES, GatewayAgentApi, GatewayState, gateway_router}, pg_store_from_env, worker::{ActivityState, FakeTools, SessionTools, WorkerActivities}, @@ -433,10 +432,10 @@ async fn run_matrix_client( ids: MatrixServerIds, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); put_fixture_server( @@ -518,6 +517,7 @@ async fn run_matrix_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "Exercise the native MCP matrix".to_owned(), }], @@ -1379,10 +1379,10 @@ async fn run_approval_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { access: None, @@ -1407,6 +1407,7 @@ async fn run_approval_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("approval test {index}"), }], @@ -1520,11 +1521,11 @@ async fn run_native_mcp_live_client( server_id: String, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = Arc::new( GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(), ); let configurator = LiveConfigurator::start(api.clone()).await?; @@ -1579,6 +1580,7 @@ async fn run_native_mcp_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "List the configured models through the Configurator MCP".to_owned(), }], @@ -1633,11 +1635,11 @@ async fn run_mcp_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = Arc::new( GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(), ); let configurator = LiveConfigurator::start(api.clone()).await?; @@ -2186,11 +2188,11 @@ async fn run_mixed_batch_live_client( server_id: String, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = Arc::new( GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(), ); let configurator = LiveConfigurator::start(api.clone()).await?; @@ -2248,6 +2250,7 @@ async fn run_mixed_batch_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "Schedule a timer, then await it while listing models".to_owned(), }], diff --git a/crates/temporal-server/tests/profiles_live.rs b/crates/temporal-server/tests/profiles_live.rs index 1dcb89b88..7edbffe9b 100644 --- a/crates/temporal-server/tests/profiles_live.rs +++ b/crates/temporal-server/tests/profiles_live.rs @@ -16,7 +16,7 @@ use support::live::{ read_session_view, require_storage_live_env, run_with_live_worker, wait_for_environment_status, wait_for_terminal_run, }; -use temporal_server::{default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env}; +use temporal_server::{gateway::GatewayAgentApi, pg_store_from_env}; use temporalio_client::{Client, WorkflowTerminateOptions}; #[tokio::test(flavor = "current_thread")] @@ -58,10 +58,10 @@ async fn run_profile_environment_selection_live_client( }; let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let suffix = uuid::Uuid::new_v4().simple().to_string(); let provider_id = format!("fake-profile-{suffix}"); @@ -242,10 +242,10 @@ async fn run_profiles_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let profile_id = ProfileId::new(format!("live_profile_{}", uuid::Uuid::new_v4().simple())); let server_id = format!("profile_crm_{}", uuid::Uuid::new_v4().simple()); @@ -464,6 +464,7 @@ async fn run_profiles_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "run after profile start".to_owned(), }], diff --git a/crates/temporal-server/tests/runs_live.rs b/crates/temporal-server/tests/runs_live.rs index f488c7e62..7ba0f0dab 100644 --- a/crates/temporal-server/tests/runs_live.rs +++ b/crates/temporal-server/tests/runs_live.rs @@ -19,7 +19,7 @@ use support::live::{ read_run, require_storage_live_env, run_with_live_worker, start_text_run, terminate_live_session, wait_for_terminal_run, wait_until, }; -use temporal_server::{default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env}; +use temporal_server::{gateway::GatewayAgentApi, pg_store_from_env}; use temporal_workflow::LLM_RETRY_MAX_ATTEMPTS; use temporalio_client::{Client, WorkflowDescribeOptions, WorkflowTerminateOptions}; @@ -220,10 +220,10 @@ async fn run_control_api( with_tools: bool, ) -> anyhow::Result { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let config = if with_tools { let workspace = api @@ -456,6 +456,7 @@ async fn run_steering_live_client( session_id: session_id.as_str().to_owned(), run_id: run.id.clone(), items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "also mention the moon".to_owned(), }], @@ -489,6 +490,7 @@ async fn run_steering_live_client( session_id: session_id.as_str().to_owned(), run_id: run.id.clone(), items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "too late".to_owned(), }], @@ -521,6 +523,7 @@ async fn run_steering_final_turn_live_client( session_id: session_id.as_str().to_owned(), run_id: run.id.clone(), items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "one more thing".to_owned(), }], @@ -622,6 +625,7 @@ async fn run_queue_live_client( session_id: session_id.as_str().to_owned(), run_id: second.id.clone(), items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "nope".to_owned(), }], @@ -725,10 +729,10 @@ async fn run_parallel_tool_batch_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); // A read-only workspace attachment derives parallel-safe function tools @@ -757,6 +761,7 @@ async fn run_parallel_tool_batch_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "run a parallel tool batch".to_owned(), }], @@ -853,10 +858,10 @@ async fn run_transient_llm_retry_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model) .build(); api.start_session(SessionStartParams { @@ -877,6 +882,7 @@ async fn run_transient_llm_retry_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "retry through transient provider failures".to_owned(), }], @@ -934,10 +940,10 @@ async fn run_llm_retry_exhaustion_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model) .build(); api.start_session(SessionStartParams { @@ -958,6 +964,7 @@ async fn run_llm_retry_exhaustion_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "exhaust the provider retry budget".to_owned(), }], @@ -1007,6 +1014,7 @@ async fn run_llm_retry_exhaustion_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "recover after the provider outage".to_owned(), }], @@ -1042,10 +1050,10 @@ async fn run_unbounded_hosted_run_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -1084,6 +1092,7 @@ async fn run_unbounded_hosted_run_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "complete thirty verification tool rounds".to_owned(), }], diff --git a/crates/temporal-server/tests/runs_live_slow.rs b/crates/temporal-server/tests/runs_live_slow.rs index a89a20a74..c646fa3be 100644 --- a/crates/temporal-server/tests/runs_live_slow.rs +++ b/crates/temporal-server/tests/runs_live_slow.rs @@ -27,10 +27,7 @@ use support::live::{ live_workflow_handle, require_storage_live_env, run_with_live_worker_timeout, wait_for_terminal_run, }; -use temporal_server::{ - default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env, - worker::FakeRuntimeCounters, -}; +use temporal_server::{gateway::GatewayAgentApi, pg_store_from_env, worker::FakeRuntimeCounters}; use temporal_workflow::{LLM_SCHEDULE_TO_CLOSE, LLM_START_TO_CLOSE}; use temporalio_client::{Client, WorkflowDescribeOptions, WorkflowTerminateOptions}; @@ -67,9 +64,9 @@ async fn run_llm_timeout_live_client( counters: FakeRuntimeCounters, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; + support::live::seed_agent_default(&store, &support::live::openai_live_model()).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(default_model_from_env()) .build(); api.start_session(SessionStartParams { @@ -180,6 +177,7 @@ async fn start_text_run( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: text.to_owned(), }], diff --git a/crates/temporal-server/tests/sessions_live.rs b/crates/temporal-server/tests/sessions_live.rs index 064df7805..34ac48917 100644 --- a/crates/temporal-server/tests/sessions_live.rs +++ b/crates/temporal-server/tests/sessions_live.rs @@ -23,9 +23,7 @@ use support::live::{ require_storage_live_env, run_with_live_worker, start_text_run, wait_for_admission_failure, wait_for_session_status, wait_for_terminal_run, }; -use temporal_server::{ - default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env, worker::WorkerActivities, -}; +use temporal_server::{gateway::GatewayAgentApi, pg_store_from_env, worker::WorkerActivities}; use temporal_workflow::{AgentAdmission, AgentAdmissionFailureKind, AgentSessionWorkflow}; use temporalio_client::{ Client, WorkflowDescribeOptions, WorkflowSignalOptions, WorkflowTerminateOptions, @@ -179,10 +177,10 @@ async fn run_checkpoint_and_bounded_reads_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store.clone()) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -372,10 +370,10 @@ async fn run_fake_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); let initialized = api.initialize(InitializeParams::default()).await?; @@ -528,6 +526,7 @@ async fn run_fake_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "hello temporal agent".to_owned(), }], @@ -546,6 +545,7 @@ async fn run_fake_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "second session-start input".to_owned(), }], @@ -566,6 +566,7 @@ async fn run_fake_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "second session-start input".to_owned(), }], @@ -583,6 +584,7 @@ async fn run_fake_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "different input".to_owned(), }], @@ -662,10 +664,10 @@ async fn run_lifecycle_delete_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store) .with_task_queue(task_queue) - .with_default_model(model) .build(); api.start_session(SessionStartParams { @@ -751,10 +753,10 @@ async fn run_continue_as_new_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .with_continue_as_new_history_threshold(1) .build(); @@ -785,6 +787,7 @@ async fn run_continue_as_new_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "first run before continue as new".to_owned(), }], @@ -812,6 +815,7 @@ async fn run_continue_as_new_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "second run after continue as new".to_owned(), }], @@ -851,10 +855,10 @@ async fn run_missing_session_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store) .with_task_queue(task_queue) - .with_default_model(model) .build(); let error = api @@ -864,6 +868,7 @@ async fn run_missing_session_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "this should not create a session".to_owned(), }], @@ -882,10 +887,10 @@ async fn run_context_append_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -924,6 +929,7 @@ async fn run_context_append_live_client( ContextAppendEntry { key: "channel.room.msg-1".to_owned(), item: InputItem::TextRef { + provenance_ref: None, origin: None, blob_ref: borrowed.to_string(), }, @@ -931,6 +937,7 @@ async fn run_context_append_live_client( ContextAppendEntry { key: "channel.room.msg-2".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: second_text.to_owned(), }, @@ -992,6 +999,7 @@ async fn run_context_append_live_client( ContextAppendEntry { key: "channel.room.msg-1".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: first_text.to_owned(), }, @@ -999,6 +1007,7 @@ async fn run_context_append_live_client( ContextAppendEntry { key: "channel.room.msg-2".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: second_text.to_owned(), }, @@ -1027,6 +1036,7 @@ async fn run_context_append_live_client( entries: vec![ContextAppendEntry { key: "channel.room.msg-2".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: "[telegram:group Engineering] Bob (12:02): edited message".to_owned(), }, @@ -1061,6 +1071,7 @@ async fn run_context_append_live_client( entries: vec![ContextAppendEntry { key: "channel.room.msg-3".to_owned(), item: InputItem::Text { + provenance_ref: None, origin: None, text: " ".to_owned(), }, @@ -1081,6 +1092,7 @@ async fn run_context_append_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "summarize the room".to_owned(), }], @@ -1101,10 +1113,10 @@ async fn run_admission_failure_live_client( session_id: SessionId, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -1152,6 +1164,7 @@ async fn run_admission_failure_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "valid run after malformed command".to_owned(), }], @@ -1182,6 +1195,7 @@ async fn run_admission_failure_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "run after close should be rejected".to_owned(), }], @@ -1217,25 +1231,23 @@ async fn run_openai_live_client( let store = pg_store_from_env().await?; let instructions = "You are Agent in a live integration test. Do not call tools for this test. Reply with the exact phrase requested by the user."; let model = openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); - api.start_session(SessionStartParams { + let start = SessionStartParams { access: None, metadata: Default::default(), session_id: Some(session_id.as_str().to_owned()), display_name: None, - config: Some(SessionConfig { - model: Some(model_to_api(&model)), - ..SessionConfig::default() - }), + config: None, profile: Some(ProfileSource::Inline { profile: Box::new(InlineAgentProfile { display_name: Some("OpenAI live test".to_owned()), description: None, document: ProfileDocument { + config: Some(SessionConfig::default()), instructions: Some(ProfileInstructions::Text { text: instructions.to_owned(), }), @@ -1244,8 +1256,66 @@ async fn run_openai_live_client( }), }), delete_after_close_ms: None, + }; + api.start_session(start.clone()).await?; + let session = read_session_view(&api, &session_id).await?; + assert_eq!( + session + .config + .as_ref() + .and_then(|config| config.model.as_ref()), + Some(&model_to_api(&model)), + "creation must persist the universe default" + ); + + let defaults = api + .read_model_defaults(api::ModelDefaultsReadParams {}) + .await? + .result + .defaults; + api.put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::AgentRun, + model: None, + expected_revision: defaults.revision, + }) + .await?; + let missing = api + .start_session(SessionStartParams { + session_id: Some(format!("{session_id}_without_default")), + ..start.clone() + }) + .await + .unwrap_err(); + assert_eq!(missing.kind, AgentApiErrorKind::ModelDefaultUnset); + assert_eq!( + missing.model_default_slot, + Some(api::ModelDefaultSlot::AgentRun) + ); + + // Reopening and sparse updates use the stored model after policy is cleared. + api.start_session(start.clone()).await?; + let replaced = api + .put_session_config(SessionConfigPutParams { + session_id: session_id.as_str().to_owned(), + config: SessionConfig::default(), + expected_config_revision: Some(session.config_revision), + }) + .await? + .result + .session; + assert_eq!( + read_session_view(&api, &session_id).await?.config, + session.config + ); + api.apply_profile(api::ProfileApplyParams { + session_id: session_id.as_str().to_owned(), + profile: start.profile.expect("inline test profile"), + expected_config_revision: Some(replaced.config_revision), + expected_tools_revision: None, }) .await?; + let reapplied = read_session_view(&api, &session_id).await?; + assert_eq!(reapplied.config, session.config); let run = api .start_run(RunStartParams { @@ -1254,6 +1324,7 @@ async fn run_openai_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: "Reply with exactly: real llm agent ok".to_owned(), }], @@ -1290,10 +1361,18 @@ async fn run_builtin_tool_live_client( session_id: SessionId, model: engine::ModelSelection, ) -> anyhow::Result<()> { + // gpt-6-sol requires reasoning to be disabled when combining function + // tools with Chat Completions. This fixture tests tool execution. + let generation = (model.api_kind == engine::ProviderApiKind::OpenAiCompletions).then(|| { + api::GenerationConfig { + reasoning_effort: Some("none".to_owned()), + ..Default::default() + } + }); let store = pg_store_from_env().await?; + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -1303,6 +1382,7 @@ async fn run_builtin_tool_live_client( display_name: None, config: Some(SessionConfig { model: Some(model_to_api(&model)), + generation, features: Some(FeaturesConfig { timers: Some(TimersFeature { version: api::CURRENT_FEATURE_VERSION, @@ -1333,7 +1413,7 @@ async fn run_builtin_tool_live_client( submission_id: None, session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { - items: vec![InputItem::Text { origin: None, + items: vec![InputItem::Text { provenance_ref: None, origin: None, text: "Call sleep with delay_ms=1, await the returned promise, then reply exactly: temporal tool ok".to_owned(), }], }, @@ -1341,11 +1421,6 @@ async fn run_builtin_tool_live_client( }) .await?; let run = wait_for_terminal_run(&api, &session_id, &run.result.run.id).await?; - let output = final_assistant_text(&run).expect("assistant output"); - assert!( - output.to_lowercase().contains("temporal tool ok"), - "expected completion marker: {output}" - ); let events = api .read_session_events(SessionEventsReadParams { direction: Default::default(), @@ -1358,6 +1433,21 @@ async fn run_builtin_tool_live_client( .await? .result .events; + anyhow::ensure!( + run.status == api::RunStatus::Completed, + "provider tool run ended with {:?}: {:?}", + run.status, + events + .iter() + .filter(|event| matches!(&event.kind, api::SessionEventKindView::RunFailed { run_id, .. } if run_id == &run.id)) + .map(|event| &event.kind) + .collect::>() + ); + let output = final_assistant_text(&run).expect("assistant output"); + assert!( + output.to_lowercase().contains("temporal tool ok"), + "expected completion marker: {output}" + ); let content = events .iter() .find_map(|event| match &event.kind { @@ -1438,10 +1528,10 @@ async fn run_session_metadata_live_client( use std::collections::BTreeMap; let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client, store) .with_task_queue(task_queue) - .with_default_model(model) .build(); let pair = |key: &str, value: &str| (key.to_owned(), value.to_owned()); // The job value is unique per run so the filter isolates this session diff --git a/crates/temporal-server/tests/subagents_live.rs b/crates/temporal-server/tests/subagents_live.rs index 3c25e903e..7ce8ec11d 100644 --- a/crates/temporal-server/tests/subagents_live.rs +++ b/crates/temporal-server/tests/subagents_live.rs @@ -23,7 +23,6 @@ use support::live::{ terminate_live_session, wait_for_terminal_run, wait_until, }; use temporal_server::{ - default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env, subagents::AgentApiSubagentRuntime, @@ -495,11 +494,11 @@ where let runtime = core_runtime()?; let client = connect_temporal(&temporal_target, &namespace).await?; let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = Arc::new( GatewayAgentApi::builder(client.clone(), store.clone()) .with_task_queue(task_queue.clone()) - .with_default_model(model.clone()) .build(), ); @@ -666,6 +665,7 @@ async fn start_subagent_parent_with_features( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: script.to_owned(), }], @@ -1279,6 +1279,7 @@ async fn run_agent_run_inherit_environment_live_client( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("AGENT_RUN {child_profile_id}"), }], diff --git a/crates/temporal-server/tests/support/live.rs b/crates/temporal-server/tests/support/live.rs index 014e0cccc..a10e25929 100644 --- a/crates/temporal-server/tests/support/live.rs +++ b/crates/temporal-server/tests/support/live.rs @@ -336,10 +336,10 @@ pub async fn fake_worker_activities_with_stall_switch( pub async fn fake_worker_activities_with_audio_transcriber( transcriber: Arc, ) -> anyhow::Result { - fake_worker_activities_with_audio_preprocessors(transcriber, None).await + fake_worker_activities_with_audio_processing(transcriber, None).await } -pub async fn fake_worker_activities_with_audio_preprocessors( +pub async fn fake_worker_activities_with_audio_processing( transcriber: Arc, transcoder: Option>, ) -> anyhow::Result { @@ -433,6 +433,7 @@ pub async fn start_text_run( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: text.to_owned(), }], @@ -610,7 +611,6 @@ pub fn openai_live_model() -> ModelSelection { model: env::var("LIGHTSPEED_OPENAI_MODEL") .or_else(|_| env::var("OPENAI_RESPONSES_MODEL")) .or_else(|_| env::var("OPENAI_LIVE_MODEL")) - .or_else(|_| env::var("LIGHTSPEED_CHAT_MODEL")) .unwrap_or_else(|_| "gpt-5.5".to_owned()), } } @@ -622,11 +622,36 @@ pub fn openai_completions_live_model() -> ModelSelection { model: env::var("LIGHTSPEED_OPENAI_MODEL") .or_else(|_| env::var("OPENAI_COMPLETIONS_MODEL")) .or_else(|_| env::var("OPENAI_LIVE_MODEL")) - .or_else(|_| env::var("LIGHTSPEED_CHAT_MODEL")) .unwrap_or_else(|_| "gpt-5.5".to_owned()), } } +/// Tests that exercise omitted models configure the same durable policy as clients. +pub async fn seed_agent_default( + store: &store_pg::PgStore, + model: &ModelSelection, +) -> anyhow::Result<()> { + store.ensure_universe().await?; + let current = store.read_model_defaults().await?; + store + .put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::AgentRun, + model: Some(api::ModelConfig { + provider_id: model.provider_id.clone(), + api_kind: match model.api_kind { + ProviderApiKind::OpenAiResponses => "openai:responses", + ProviderApiKind::OpenAiCompletions => "openai:completions", + ProviderApiKind::AnthropicMessages => "anthropic:messages", + } + .into(), + model: model.model.clone(), + }), + expected_revision: current.revision, + }) + .await?; + Ok(()) +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/temporal-server/tests/preprocess_live.rs b/crates/temporal-server/tests/transcriptions_live.rs similarity index 54% rename from crates/temporal-server/tests/preprocess_live.rs rename to crates/temporal-server/tests/transcriptions_live.rs index 0b3c47613..15491c3ad 100644 --- a/crates/temporal-server/tests/preprocess_live.rs +++ b/crates/temporal-server/tests/transcriptions_live.rs @@ -4,20 +4,18 @@ use std::sync::{Arc, Mutex}; use api::{ AgentApiService, BlobPutItem, BlobPutParams, ContextEntryKindView, ContextMessageRoleView, - InputItem, MediaKind, RunStartParams, RunStartSource, RunStatus, SessionConfig, - SessionStartParams, + InputItem, RunStartParams, RunStartSource, RunStatus, SessionConfig, SessionStartParams, }; use api_projection::model_to_api; use async_trait::async_trait; use base64::{Engine as _, engine::general_purpose::STANDARD as BASE64}; use engine::SessionId; use support::live::{ - LIVE_TEST_LOCK, fake_worker_activities_with_audio_preprocessors, + LIVE_TEST_LOCK, fake_worker_activities_with_audio_processing, fake_worker_activities_with_audio_transcriber, final_assistant_text, live_workflow_handle, require_storage_live_env, run_with_live_worker, wait_for_terminal_run, }; use temporal_server::{ - default_model_from_env, gateway::GatewayAgentApi, pg_store_from_env, worker::{ @@ -33,7 +31,7 @@ const TRANSCRIPT_TEXT: &str = "please file the deployment note from this audio"; #[tokio::test(flavor = "current_thread")] #[ignore = "requires ./dev.sh infra or compatible Temporal + Postgres env"] -async fn preprocess_live_audio_input_is_transcribed_before_admission() -> anyhow::Result<()> { +async fn transcription_live_prepared_input_retains_source_audio() -> anyhow::Result<()> { let _lock = LIVE_TEST_LOCK.lock().await; let _ = dotenvy::dotenv(); require_storage_live_env()?; @@ -41,14 +39,14 @@ async fn preprocess_live_audio_input_is_transcribed_before_admission() -> anyhow let transcriber = Arc::new(RecordingAudioTranscriber::new(TRANSCRIPT_TEXT)); let activities = fake_worker_activities_with_audio_transcriber(transcriber.clone()).await?; run_with_live_worker(activities, move |client, task_queue, session_id| { - run_audio_preprocess_live_client(client, task_queue, session_id, transcriber) + run_audio_transcription_live_client(client, task_queue, session_id, transcriber) }) .await } #[tokio::test(flavor = "current_thread")] #[ignore = "requires ./dev.sh infra or compatible Temporal + Postgres env"] -async fn preprocess_live_transcodable_audio_is_transcoded_before_admission() -> anyhow::Result<()> { +async fn transcription_live_transcodable_audio_is_prepared_outside_session() -> anyhow::Result<()> { let _lock = LIVE_TEST_LOCK.lock().await; let _ = dotenvy::dotenv(); require_storage_live_env()?; @@ -60,13 +58,11 @@ async fn preprocess_live_transcodable_audio_is_transcoded_before_admission() -> "audio/wav", "voice-note.wav", )); - let activities = fake_worker_activities_with_audio_preprocessors( - transcriber.clone(), - Some(transcoder.clone()), - ) - .await?; + let activities = + fake_worker_activities_with_audio_processing(transcriber.clone(), Some(transcoder.clone())) + .await?; run_with_live_worker(activities, move |client, task_queue, session_id| { - run_transcodable_audio_preprocess_live_client( + run_transcodable_audio_transcription_live_client( client, task_queue, session_id, @@ -78,17 +74,17 @@ async fn preprocess_live_transcodable_audio_is_transcoded_before_admission() -> .await } -async fn run_audio_preprocess_live_client( +async fn run_audio_transcription_live_client( client: Client, task_queue: String, session_id: SessionId, transcriber: Arc, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -112,18 +108,23 @@ async fn run_audio_preprocess_live_client( }], }) .await?; + let transcript_ref = transcribe( + &api, + audio.result.blobs[0].blob_ref.clone(), + "audio/ogg", + "voice-note.ogg", + ) + .await?; let started = api .start_run(RunStartParams { notify_on_terminal: None, submission_id: None, session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { - items: vec![InputItem::Media { + items: vec![InputItem::TextRef { origin: None, - blob_ref: audio.result.blobs[0].blob_ref.clone(), - mime: "audio/ogg".to_owned(), - kind: MediaKind::Audio, - name: Some("voice-note.ogg".to_owned()), + blob_ref: transcript_ref, + provenance_ref: Some(audio.result.blobs[0].blob_ref.clone()), }], }, config: None, @@ -155,18 +156,18 @@ async fn run_audio_preprocess_live_client( .entries .iter() .find(|entry| { - entry.content.provider_kind.as_deref() - == Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) + entry.provenance_ref.as_deref() == Some(audio.result.blobs[0].blob_ref.as_str()) }) - .expect("structured transcript"); + .expect("text with source provenance"); assert_eq!( transcript_entry.content.media_type.as_deref(), - Some("application/json") + Some("text/plain") ); assert_eq!( transcript_entry.provenance_ref.as_deref(), Some(audio.result.blobs[0].blob_ref.as_str()) ); + assert!(transcript_entry.content.provider_kind.is_none()); assert_eq!(transcript_entry.text.as_deref(), Some(TRANSCRIPT_TEXT)); assert!(!transcript_entry.text_truncated); assert!( @@ -185,14 +186,14 @@ async fn run_audio_preprocess_live_client( let _ = handle .terminate( WorkflowTerminateOptions::builder() - .reason("agent audio preprocess live test cleanup") + .reason("agent audio transcription live test cleanup") .build(), ) .await; Ok(()) } -async fn run_transcodable_audio_preprocess_live_client( +async fn run_transcodable_audio_transcription_live_client( client: Client, task_queue: String, session_id: SessionId, @@ -201,10 +202,10 @@ async fn run_transcodable_audio_preprocess_live_client( transcoded_bytes: Vec, ) -> anyhow::Result<()> { let store = pg_store_from_env().await?; - let model = default_model_from_env(); + let model = support::live::openai_live_model(); + support::live::seed_agent_default(&store, &model).await?; let api = GatewayAgentApi::builder(client.clone(), store) .with_task_queue(task_queue) - .with_default_model(model.clone()) .build(); api.start_session(SessionStartParams { @@ -228,18 +229,23 @@ async fn run_transcodable_audio_preprocess_live_client( }], }) .await?; + let transcript_ref = transcribe( + &api, + audio.result.blobs[0].blob_ref.clone(), + "audio/x-aac", + "voice-note.aac", + ) + .await?; let started = api .start_run(RunStartParams { notify_on_terminal: None, submission_id: None, session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { - items: vec![InputItem::Media { + items: vec![InputItem::TextRef { origin: None, - blob_ref: audio.result.blobs[0].blob_ref.clone(), - mime: "audio/x-aac".to_owned(), - kind: MediaKind::Audio, - name: Some("voice-note.aac".to_owned()), + blob_ref: transcript_ref, + provenance_ref: Some(audio.result.blobs[0].blob_ref.clone()), }], }, config: None, @@ -269,18 +275,18 @@ async fn run_transcodable_audio_preprocess_live_client( .entries .iter() .find(|entry| { - entry.content.provider_kind.as_deref() - == Some(llm_clients::content::AUDIO_TRANSCRIPT_PROVIDER_KIND) + entry.provenance_ref.as_deref() == Some(audio.result.blobs[0].blob_ref.as_str()) }) - .expect("structured transcript"); + .expect("text with source provenance"); assert_eq!( transcript_entry.content.media_type.as_deref(), - Some("application/json") + Some("text/plain") ); assert_eq!( transcript_entry.provenance_ref.as_deref(), Some(audio.result.blobs[0].blob_ref.as_str()) ); + assert!(transcript_entry.content.provider_kind.is_none()); assert_eq!(transcript_entry.text.as_deref(), Some(TRANSCRIPT_TEXT)); assert!(!transcript_entry.text_truncated); assert!( @@ -308,7 +314,7 @@ async fn run_transcodable_audio_preprocess_live_client( let _ = handle .terminate( WorkflowTerminateOptions::builder() - .reason("agent audio transcode preprocess live test cleanup") + .reason("audio transcription live test cleanup") .build(), ) .await; @@ -421,3 +427,194 @@ fn tiny_wav_bytes() -> Vec { bytes.resize(44 + data_len, 0); bytes } + +async fn transcribe( + api: &GatewayAgentApi, + blob_ref: String, + mime: &str, + name: &str, +) -> anyhow::Result { + let model = api::ModelConfig { + provider_id: "test-speech".into(), + api_kind: "openai:audio-transcriptions".into(), + model: "pinned-speech-model".into(), + }; + let before = api + .read_model_defaults(api::ModelDefaultsReadParams {}) + .await? + .result + .defaults; + let updated = api + .put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::SpeechToText, + model: Some(model.clone()), + expected_revision: before.revision, + }) + .await? + .result + .defaults; + let request = api::TranscriptionStartParams { + idempotency_key: uuid::Uuid::new_v4().to_string(), + audio: api::TranscriptionAudio { + blob_ref, + mime: mime.into(), + name: name.into(), + }, + model: None, + language: Some("en".into()), + prompt: None, + }; + let initial = api + .start_transcription(request.clone()) + .await? + .result + .transcription; + api.put_model_defaults(api::ModelDefaultsPutParams { + slot: api::ModelDefaultSlot::SpeechToText, + model: before.speech_to_text, + expected_revision: updated.revision, + }) + .await?; + let terminal = loop { + let view = api + .read_transcription(api::TranscriptionReadParams { + transcription_id: initial.transcription_id.clone(), + }) + .await? + .result + .transcription; + if view.status.is_terminal() { + break view; + } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + }; + assert_eq!( + terminal.status, + api::TranscriptionStatus::Succeeded, + "{:?}", + terminal.failure + ); + assert_eq!(terminal.model, model); + assert_eq!(terminal.text.as_deref(), Some(TRANSCRIPT_TEXT)); + let mut other = support::live::local_request_context().await?; + other.actor = Some("another-person".into()); + let denied = temporal_server::gateway::request_context::with_request_context( + other, + api.read_transcription(api::TranscriptionReadParams { + transcription_id: initial.transcription_id.clone(), + }), + ) + .await + .unwrap_err(); + assert_eq!(denied.kind, api::AgentApiErrorKind::Forbidden); + let retried = api + .start_transcription(request.clone()) + .await? + .result + .transcription; + assert_eq!(retried.transcript_ref, terminal.transcript_ref); + assert_eq!(terminal.audio, request.audio); + assert_eq!(retried.model, model); + let mut conflicting = request; + conflicting.prompt = Some("different options".into()); + assert_eq!( + api.start_transcription(conflicting).await.unwrap_err().kind, + api::AgentApiErrorKind::Conflict + ); + let cancelled = api + .cancel_transcription(api::TranscriptionCancelParams { + transcription_id: initial.transcription_id, + }) + .await? + .result + .transcription; + assert_eq!(cancelled.status, api::TranscriptionStatus::Succeeded); + Ok(terminal.transcript_ref.unwrap()) +} + +struct ControlledTranscriber { + text: String, + attempts: std::sync::atomic::AtomicUsize, + block: bool, +} + +#[async_trait] +impl AudioTranscriber for ControlledTranscriber { + async fn transcribe( + &self, + _request: AudioTranscriptionRequest, + ) -> Result { + let attempt = self + .attempts + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + if self.block { + return std::future::pending().await; + } + if attempt == 0 { + return Err(AudioTranscriptionError { + message: "temporary failure".into(), + retryable: true, + configuration: false, + }); + } + Ok(AudioTranscription { + text: self.text.clone(), + }) + } +} + +#[tokio::test(flavor = "current_thread")] +#[ignore = "requires ./dev.sh infra or compatible Temporal + Postgres env"] +async fn transcription_live_retries_transient_failure_and_cancels_running_activity() +-> anyhow::Result<()> { + let _lock = LIVE_TEST_LOCK.lock().await; + require_storage_live_env()?; + for block in [false, true] { + let transcriber = Arc::new(ControlledTranscriber { + // A plain text result can be shared with sessions from other jobs. + // The expiry assertion needs a unique, unsubmitted result blob. + text: format!("unsubmitted transcript {}", uuid::Uuid::new_v4()), + attempts: Default::default(), + block, + }); + let activities = fake_worker_activities_with_audio_transcriber(transcriber.clone()).await?; + run_with_live_worker(activities, move |client, queue, _| async move { + let store = pg_store_from_env().await?; + store.ensure_universe().await?; + let api = GatewayAgentApi::builder(client, store).with_task_queue(queue).build(); + let blob = api.put_blobs(BlobPutParams { blobs: vec![BlobPutItem { bytes_base64: BASE64.encode(AUDIO_BYTES) }] }).await?.result.blobs.remove(0).blob_ref; + let request = api::TranscriptionStartParams { + idempotency_key: uuid::Uuid::new_v4().to_string(), + audio: api::TranscriptionAudio { blob_ref: blob, mime: "audio/ogg".into(), name: "voice.ogg".into() }, + model: Some(api::ModelConfig { provider_id: "fake".into(), api_kind: "openai:audio-transcriptions".into(), model: "speech".into() }), + language: None, prompt: None, + }; + let started = api.start_transcription(request).await?.result.transcription; + while transcriber.attempts.load(std::sync::atomic::Ordering::SeqCst) == 0 { + tokio::time::sleep(std::time::Duration::from_millis(20)).await; + } + if block { + api.cancel_transcription(api::TranscriptionCancelParams { transcription_id: started.transcription_id.clone() }).await?; + } + let terminal = loop { + let view = api.read_transcription(api::TranscriptionReadParams { transcription_id: started.transcription_id.clone() }).await?.result.transcription; + if view.status.is_terminal() { break view; } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + }; + assert_eq!(terminal.status, if block { api::TranscriptionStatus::Cancelled } else { api::TranscriptionStatus::Succeeded }); + assert_eq!(transcriber.attempts.load(std::sync::atomic::Ordering::SeqCst), if block { 1 } else { 2 }); + if !block { + let store = pg_store_from_env().await?; + let reference = engine::BlobRef::parse(terminal.transcript_ref.as_ref().unwrap())?; + sqlx::query("UPDATE cas_blobs SET created_at_ms = 1, touched_at_ms = 1 WHERE universe_id = $1 AND digest = $2") + .bind(store.config().universe_id).bind(reference.as_str().trim_start_matches("sha256:")).execute(store.pool()).await?; + assert_eq!(store.delete_dead_blobs(&[reference], 2, &[]).await?.len(), 1, "unsubmitted workflows do not create retention roots"); + let expired = api.read_transcription(api::TranscriptionReadParams { transcription_id: started.transcription_id }).await?.result.transcription; + assert_eq!(expired.status, api::TranscriptionStatus::Expired); + } + + Ok(()) + }).await?; + } + Ok(()) +} diff --git a/crates/temporal-server/tests/vfs_transfer_live.rs b/crates/temporal-server/tests/vfs_transfer_live.rs index fd747c162..3a6fea1da 100644 --- a/crates/temporal-server/tests/vfs_transfer_live.rs +++ b/crates/temporal-server/tests/vfs_transfer_live.rs @@ -334,7 +334,7 @@ async fn run_case( features["environments"]["environments"][0]["workingDirectory"] = json!(root); features["environments"]["prompts"] = json!({"roots":[".agents/prompts"]}); } - let mut model = temporal_server::default_model_from_env(); + let mut model = support::live::openai_live_model(); model.api_kind = provider; let profile = api.create_profile(api::ProfileCreateParams { profile: api::AgentProfileInput { profile_id: api::ProfileId::new(format!("profile_{session}")), display_name: None, description: None, @@ -357,6 +357,7 @@ async fn run_case( session_id: session.to_string(), source: api::RunStartSource::Input { items: vec![api::InputItem::Text { + provenance_ref: None, origin: None, text: "run transfer checks".into(), }], diff --git a/crates/temporal-server/tests/workflow_tool_plugins_live.rs b/crates/temporal-server/tests/workflow_tool_plugins_live.rs index 4b41cec3a..2bd2b1ea2 100644 --- a/crates/temporal-server/tests/workflow_tool_plugins_live.rs +++ b/crates/temporal-server/tests/workflow_tool_plugins_live.rs @@ -835,9 +835,9 @@ where let store = pg_store_from_env().await?; let blobs: Arc = store.clone(); + support::live::seed_agent_default(&store, &support::live::openai_live_model()).await?; let mut api_builder = GatewayAgentApi::builder(client.clone(), store.clone()) - .with_task_queue(session_queue.clone()) - .with_default_model(temporal_server::default_model_from_env()); + .with_task_queue(session_queue.clone()); if let Some(threshold) = continue_as_new_history_threshold { api_builder = api_builder.with_continue_as_new_history_threshold(threshold); } @@ -935,6 +935,7 @@ async fn start_managed_session_and_run( session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {call_tool}"), }], @@ -1157,6 +1158,7 @@ async fn workflow_tool_controller_self_receiver_resolves_before_run_terminal() - session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {MESSAGE_SEND_TOOL}"), }], @@ -1432,6 +1434,7 @@ async fn workflow_tool_controller_self_receiver_deadline_breaks_stalled_reply() session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {MESSAGE_SEND_TOOL}"), }], @@ -1552,6 +1555,7 @@ async fn workflow_tool_reply_requires_exact_stored_producer() -> anyhow::Result< session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {REQUEST_APPROVAL_TOOL}"), }], @@ -2005,6 +2009,7 @@ async fn workflow_tool_dead_receiver_fails_promise_terminally() -> anyhow::Resul session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {REQUEST_APPROVAL_TOOL}"), }], @@ -2183,6 +2188,7 @@ async fn workflow_tool_reply_schema_gates_resolutions() -> anyhow::Result<()> { session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL {REQUEST_APPROVAL_TOOL}"), }], @@ -2511,6 +2517,7 @@ async fn workflow_tool_run_terminal_auto_cancel_notifies_bound_receiver() -> any session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { items: vec![InputItem::Text { + provenance_ref: None, origin: None, text: format!("CALL_NOWAIT {REQUEST_APPROVAL_TOOL}"), }], @@ -2606,7 +2613,7 @@ async fn workflow_tool_auto_cancel_cancels_started_execution() -> anyhow::Result submission_id: None, session_id: session_id.as_str().to_owned(), source: RunStartSource::Input { - items: vec![InputItem::Text { origin: None, + items: vec![InputItem::Text { provenance_ref: None, origin: None, text: format!("CALL_SHORTWAIT {LAUNCH_JOB_TOOL}"), }], }, diff --git a/crates/temporal-workflow/contract/workflow-contract.md b/crates/temporal-workflow/contract/workflow-contract.md index dd4d29b02..902358c7f 100644 --- a/crates/temporal-workflow/contract/workflow-contract.md +++ b/crates/temporal-workflow/contract/workflow-contract.md @@ -72,5 +72,5 @@ fingerprinted over their exact raw bytes; canonical fingerprints begin with ## Schema inventory -The schema bundle contains 38 definitions. Its public roots -are: EmissionEnvelope, WorkflowToolStartArgs, WorkflowToolRecoveryResult, WorkflowToolRecipeV1, ConversationStart, ChannelDeliveryCommand, ChannelDeliveryResult, PrepareChannelMediaInput, PrepareChannelMediaResult. +The schema bundle contains 49 definitions. Its public roots +are: EmissionEnvelope, WorkflowToolStartArgs, WorkflowToolRecoveryResult, WorkflowToolRecipeV1, ConversationStart, ChannelDeliveryCommand, ChannelDeliveryResult, PrepareChannelMediaInput, PrepareChannelMediaResult, TranscriptionWorkflowArgs, TranscriptionSnapshot, TranscriptionActivityResult. diff --git a/crates/temporal-workflow/contract/workflow.json b/crates/temporal-workflow/contract/workflow.json index 47ea11762..c9baa301b 100644 --- a/crates/temporal-workflow/contract/workflow.json +++ b/crates/temporal-workflow/contract/workflow.json @@ -65,7 +65,10 @@ "ChannelDeliveryCommand", "ChannelDeliveryResult", "PrepareChannelMediaInput", - "PrepareChannelMediaResult" + "PrepareChannelMediaResult", + "TranscriptionWorkflowArgs", + "TranscriptionSnapshot", + "TranscriptionActivityResult" ], "signals": { "deliverEmission": "deliver_emission" diff --git a/crates/temporal-workflow/contract/workflow.schema.json b/crates/temporal-workflow/contract/workflow.schema.json index 624554450..214d658c8 100644 --- a/crates/temporal-workflow/contract/workflow.schema.json +++ b/crates/temporal-workflow/contract/workflow.schema.json @@ -1,6 +1,79 @@ { "$schema": "http://json-schema.org/draft-07/schema#", "definitions": { + "Attribution": { + "description": "Who created a resource or authored bytes. An actor is whatever a key\nallowed to assert one said; core compares it and never resolves it.", + "oneOf": [ + { + "description": "The actor a key asserted for its request.", + "properties": { + "id": { + "type": "string" + }, + "kind": { + "const": "actor", + "type": "string" + } + }, + "required": [ + "kind", + "id" + ], + "type": "object" + }, + { + "description": "A key acting for itself, named by its display prefix.", + "properties": { + "kind": { + "const": "key", + "type": "string" + }, + "prefix": { + "type": "string" + } + }, + "required": [ + "kind", + "prefix" + ], + "type": "object" + }, + { + "description": "An unauthenticated local development request, or an in-process call.", + "properties": { + "kind": { + "const": "local", + "type": "string" + } + }, + "required": [ + "kind" + ], + "type": "object" + }, + { + "description": "The runtime's own work: a bot, a delegated session, a registration,\nor host administration through the server CLI.", + "properties": { + "cause": { + "type": "string" + }, + "component": { + "type": "string" + }, + "kind": { + "const": "internal", + "type": "string" + } + }, + "required": [ + "kind", + "component", + "cause" + ], + "type": "object" + } + ] + }, "BotId": { "type": "string" }, @@ -614,6 +687,25 @@ ], "type": "object" }, + "ModelConfig": { + "properties": { + "apiKind": { + "type": "string" + }, + "model": { + "type": "string" + }, + "providerId": { + "type": "string" + } + }, + "required": [ + "providerId", + "apiKind", + "model" + ], + "type": "object" + }, "PrepareChannelMediaInput": { "description": "`prepare_channel_media`: the connector downloads the provider file and\nstores it in the universe's CAS.", "properties": { @@ -761,6 +853,243 @@ "ToolCallId": { "type": "string" }, + "TranscriptionActivityResult": { + "oneOf": [ + { + "properties": { + "kind": { + "const": "succeeded", + "type": "string" + }, + "transcript_ref": { + "type": "string" + } + }, + "required": [ + "kind", + "transcript_ref" + ], + "type": "object" + }, + { + "properties": { + "failure": { + "$ref": "#/definitions/TranscriptionFailure" + }, + "kind": { + "const": "failed", + "type": "string" + } + }, + "required": [ + "kind", + "failure" + ], + "type": "object" + } + ] + }, + "TranscriptionAudio": { + "additionalProperties": false, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + }, + "TranscriptionFailure": { + "properties": { + "kind": { + "$ref": "#/definitions/TranscriptionFailureKind" + }, + "message": { + "type": "string" + } + }, + "required": [ + "kind", + "message" + ], + "type": "object" + }, + "TranscriptionFailureKind": { + "enum": [ + "invalidAudio", + "configuration", + "provider", + "timeout", + "internal" + ], + "type": "string" + }, + "TranscriptionSnapshot": { + "properties": { + "request": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "view": { + "$ref": "#/definitions/TranscriptionView" + } + }, + "required": [ + "request", + "view" + ], + "type": "object" + }, + "TranscriptionStartParams": { + "additionalProperties": false, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object" + }, + "TranscriptionStatus": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "expired" + ], + "type": "string" + }, + "TranscriptionView": { + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "failure": { + "anyOf": [ + { + "$ref": "#/definitions/TranscriptionFailure" + }, + { + "type": "null" + } + ] + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "status": { + "$ref": "#/definitions/TranscriptionStatus" + }, + "text": { + "type": [ + "string", + "null" + ] + }, + "transcriptRef": { + "description": "Plain UTF-8 transcript blob, usable as ordinary textRef input.\nUnsubmitted content can be swept after the ordinary CAS grace period.", + "type": [ + "string", + "null" + ] + }, + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId", + "createdBy", + "audio", + "model", + "status", + "createdAtMs" + ], + "type": "object" + }, + "TranscriptionWorkflowArgs": { + "properties": { + "createdAtMs": { + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "createdBy": { + "$ref": "#/definitions/Attribution" + }, + "model": { + "$ref": "#/definitions/ModelConfig" + }, + "request": { + "$ref": "#/definitions/TranscriptionStartParams" + }, + "transcriptionId": { + "type": "string" + }, + "universeId": { + "format": "uuid", + "type": "string" + } + }, + "required": [ + "universeId", + "transcriptionId", + "createdBy", + "request", + "model", + "createdAtMs" + ], + "type": "object" + }, "WorkflowToolId": { "type": "string" }, diff --git a/crates/temporal-workflow/src/activities.rs b/crates/temporal-workflow/src/activities.rs index a2cba4134..f8b84639b 100644 --- a/crates/temporal-workflow/src/activities.rs +++ b/crates/temporal-workflow/src/activities.rs @@ -12,8 +12,7 @@ use crate::{ EnvironmentJobPollActivityRequest, EnvironmentJobPollActivityResult, EnvironmentJobPrepareWorkflowToolRequest, EnvironmentJobStartActivityRequest, EnvironmentJobStartActivityResult, JoinedContextPreparationRequest, LlmGenerateActivityRequest, - PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult, PutBlobRequest, - ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, + PutBlobRequest, ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, RuntimeProjectionRefreshActivityResult, SubagentCloseActivityRequest, SubagentPrepareActivityRequest, SubagentPrepareActivityResult, SubagentResolveActivityRequest, ToolInvokeBatchActivityRequest, ToolInvokeCallActivityRequest, ToolInvokeCallActivityResult, @@ -30,7 +29,6 @@ pub const ACTIVITY_MATERIALIZE_AWAIT_RESULT: &str = "WorkflowActivities::materia pub const ACTIVITY_PREPARE_JOINED_CONTEXT: &str = "WorkflowActivities::prepare_joined_context"; pub const ACTIVITY_APPEND_EVENTS: &str = "WorkflowActivities::append_events"; pub const ACTIVITY_LLM_GENERATE: &str = "WorkflowActivities::llm_generate"; -pub const ACTIVITY_PREPROCESS_RUN_INPUT: &str = "WorkflowActivities::preprocess_run_input"; pub const ACTIVITY_CONTEXT_COMPACT: &str = "WorkflowActivities::context_compact"; pub const ACTIVITY_TOOL_INVOKE_BATCH: &str = "WorkflowActivities::tool_invoke_batch"; pub const ACTIVITY_TOOL_INVOKE_CALL: &str = "WorkflowActivities::tool_invoke_call"; @@ -60,6 +58,14 @@ pub struct WorkflowActivities; #[activities] impl WorkflowActivities { + #[activity(name = "WorkflowActivities::execute_transcription")] + pub async fn execute_transcription( + _ctx: ActivityContext, + _args: crate::TranscriptionWorkflowArgs, + ) -> Result { + unimplemented!("workflow activity definition only") + } + #[activity(name = ACTIVITY_CREATE_OR_LOAD_SESSION)] pub async fn create_or_load_session( _ctx: ActivityContext, @@ -121,14 +127,6 @@ impl WorkflowActivities { unimplemented!("workflow activity definition only") } - #[activity(name = ACTIVITY_PREPROCESS_RUN_INPUT)] - pub async fn preprocess_run_input( - _ctx: ActivityContext, - _request: PreprocessRunInputActivityRequest, - ) -> Result { - unimplemented!("workflow activity definition only") - } - #[activity(name = ACTIVITY_CONTEXT_COMPACT)] pub async fn context_compact( _ctx: ActivityContext, diff --git a/crates/temporal-workflow/src/config.rs b/crates/temporal-workflow/src/config.rs index e3c3d5e9e..4a0e667eb 100644 --- a/crates/temporal-workflow/src/config.rs +++ b/crates/temporal-workflow/src/config.rs @@ -8,7 +8,6 @@ use temporalio_sdk::{ActivityCloseTimeouts, ActivityOptions}; pub const DEFAULT_TASK_QUEUE: &str = "lightspeed-sessions"; pub const DEFAULT_TEMPORAL_TARGET: &str = "localhost:7233"; pub const DEFAULT_TEMPORAL_NAMESPACE: &str = "default"; -pub const DEFAULT_MODEL: &str = "gpt-5.5"; pub const DEFAULT_CONTINUE_AS_NEW_HISTORY_THRESHOLD: u32 = 10_000; pub const DEFAULT_ACTIVITY_START_TO_CLOSE_TIMEOUT: Duration = Duration::from_secs(360); diff --git a/crates/temporal-workflow/src/lib.rs b/crates/temporal-workflow/src/lib.rs index 441a7997f..f45eb8034 100644 --- a/crates/temporal-workflow/src/lib.rs +++ b/crates/temporal-workflow/src/lib.rs @@ -16,17 +16,16 @@ pub use activities::{ ACTIVITY_CONTEXT_COMPACT, ACTIVITY_CREATE_OR_LOAD_SESSION, ACTIVITY_ENVIRONMENT_JOB_CANCEL, ACTIVITY_ENVIRONMENT_JOB_POLL, ACTIVITY_ENVIRONMENT_JOB_PREPARE_WORKFLOW_TOOL, ACTIVITY_ENVIRONMENT_JOB_START, ACTIVITY_LLM_GENERATE, ACTIVITY_MATERIALIZE_AWAIT_RESULT, - ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PREPROCESS_RUN_INPUT, ACTIVITY_PUT_BLOB, - ACTIVITY_READ_BLOB, ACTIVITY_RUNTIME_PROJECTION_REFRESH, - ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, - ACTIVITY_SUBAGENT_RESOLVE, ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, - ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, - WorkflowActivities, + ACTIVITY_PREPARE_JOINED_CONTEXT, ACTIVITY_PUT_BLOB, ACTIVITY_READ_BLOB, + ACTIVITY_RUNTIME_PROJECTION_REFRESH, ACTIVITY_START_WORKFLOW_TOOL_EXECUTION, + ACTIVITY_SUBAGENT_CLOSE, ACTIVITY_SUBAGENT_PREPARE, ACTIVITY_SUBAGENT_RESOLVE, + ACTIVITY_TOOL_INVOKE_BATCH, ACTIVITY_TOOL_INVOKE_CALL, ACTIVITY_TOOL_PREPARE_PROMISE_CONTROLS, + ACTIVITY_VALIDATE_WORKFLOW_TOOL_REPLY, WorkflowActivities, }; pub use config::{ ACTIVITY_CANCELLATION_HEARTBEAT_INTERVAL, ACTIVITY_CANCELLATION_HEARTBEAT_TIMEOUT, DEFAULT_BOOTSTRAP_PAYLOAD_BUDGET_BYTES, DEFAULT_CONTINUE_AS_NEW_HISTORY_THRESHOLD, - DEFAULT_MODEL, DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, + DEFAULT_TASK_QUEUE, DEFAULT_TEMPORAL_NAMESPACE, DEFAULT_TEMPORAL_TARGET, ENVIRONMENT_READY_GRACE, ENVIRONMENT_READY_HEARTBEAT_TIMEOUT, ENVIRONMENT_READY_POLL_INTERVAL, ENVIRONMENT_READY_WAIT, FAKE_TOOL_NAME, LLM_RETRY_MAX_ATTEMPTS, LLM_RETRY_MAX_INTERVAL, LLM_SCHEDULE_TO_CLOSE, LLM_START_TO_CLOSE, MAX_CONCURRENT_TOOL_CALLS_PER_BATCH, @@ -57,9 +56,7 @@ pub use types::{ LLM_PROVIDER_TRANSIENT_ERROR_TYPE, LLM_TRANSIENT_FAILURE_DETAILS_VERSION, LlmGenerateActivityRequest, LlmTransientFailureDetails, MaterializedAwaitPromiseResult, MaterializedAwaitResult, PendingEmission, PendingPromiseCancellation, PendingSourceResolution, - PendingToolBatchResume, PreprocessRunInputActivityRequest, PreprocessRunInputActivityResult, - PreprocessRunInputFailure, PreprocessRunInputFailureKind, PreprocessRunInputOutcome, - PromiseSourcePoll, PutBlobRequest, ReadBlobRequest, ReadBlobResult, + PendingToolBatchResume, PromiseSourcePoll, PutBlobRequest, ReadBlobRequest, ReadBlobResult, RuntimeProjectionRefreshActivityRequest, RuntimeProjectionRefreshActivityResult, SessionBootstrapPayloadTooLarge, SubagentChildRef, SubagentCloseActivityRequest, SubagentExecutionPhase, SubagentExecutionSnapshot, SubagentPrepareActivityRequest, @@ -78,4 +75,6 @@ pub use workflows::channels; pub use workflows::{ AgentSessionWorkflow, BotControllerWorkflow, BotTriggerFireWorkflow, ChannelConversationWorkflow, EnvironmentJobWorkflow, SubagentExecutionWorkflow, + TranscriptionActivityResult, TranscriptionSnapshot, TranscriptionWorkflow, + TranscriptionWorkflowArgs, transcription_id, transcription_workflow_id, }; diff --git a/crates/temporal-workflow/src/types.rs b/crates/temporal-workflow/src/types.rs index aac3c03d1..6ede858c9 100644 --- a/crates/temporal-workflow/src/types.rs +++ b/crates/temporal-workflow/src/types.rs @@ -153,26 +153,10 @@ pub struct AgentAdmissionFailure { pub rejection: Option, } -impl AgentAdmissionFailure { - pub fn with_correlation_token(mut self, correlation_token: Option) -> Self { - if self.correlation_token.is_none() { - self.correlation_token = correlation_token; - } - self - } -} - #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "snake_case")] pub enum AgentAdmissionFailureKind { RejectedCommand, - UnsupportedAudioMime, - AudioBlobMissing, - AudioBlobTooLarge, - AudioDurationTooLong, - TranscoderUnavailable, - TranscodeFailure, - TranscriptionFailure, } #[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] @@ -755,42 +739,6 @@ pub struct LlmGenerateActivityRequest { pub request: engine::LlmGenerationRequest, } -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -pub struct PreprocessRunInputActivityRequest { - pub session_id: SessionId, - pub input: Vec, -} - -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -pub struct PreprocessRunInputActivityResult { - pub outcome: PreprocessRunInputOutcome, -} - -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case", tag = "status")] -pub enum PreprocessRunInputOutcome { - Succeeded { input: Vec }, - Failed { failure: PreprocessRunInputFailure }, -} - -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -pub struct PreprocessRunInputFailure { - pub kind: PreprocessRunInputFailureKind, - pub message: String, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum PreprocessRunInputFailureKind { - UnsupportedAudioMime, - AudioBlobMissing, - AudioBlobTooLarge, - AudioDurationTooLong, - TranscoderUnavailable, - TranscodeFailure, - TranscriptionFailure, -} - #[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] pub struct ContextCompactActivityRequest { pub request: engine::ContextCompactionRequest, diff --git a/crates/temporal-workflow/src/workflow_contract.rs b/crates/temporal-workflow/src/workflow_contract.rs index 5bbc9bb84..1861c69cf 100644 --- a/crates/temporal-workflow/src/workflow_contract.rs +++ b/crates/temporal-workflow/src/workflow_contract.rs @@ -42,7 +42,7 @@ pub const WORKFLOW_CONTRACT_VERSION: u32 = 2; pub const DELIVER_EMISSION_SIGNAL: &str = "deliver_emission"; /// Root types of the schema bundle; everything else is reachable from them. -pub const WORKFLOW_CONTRACT_ROOTS: [&str; 9] = [ +pub const WORKFLOW_CONTRACT_ROOTS: [&str; 12] = [ "EmissionEnvelope", "WorkflowToolStartArgs", "WorkflowToolRecoveryResult", @@ -52,6 +52,9 @@ pub const WORKFLOW_CONTRACT_ROOTS: [&str; 9] = [ "ChannelDeliveryResult", "PrepareChannelMediaInput", "PrepareChannelMediaResult", + "TranscriptionWorkflowArgs", + "TranscriptionSnapshot", + "TranscriptionActivityResult", ]; pub struct ExportedWorkflowContract { @@ -75,6 +78,9 @@ pub fn export() -> ExportedWorkflowContract { let _ = generator.subschema_for::(); let _ = generator.subschema_for::(); let _ = generator.subschema_for::(); + let _ = generator.subschema_for::(); + let _ = generator.subschema_for::(); + let _ = generator.subschema_for::(); let definitions: BTreeMap = generator.take_definitions(true).into_iter().collect(); for root in WORKFLOW_CONTRACT_ROOTS { diff --git a/crates/temporal-workflow/src/workflows/channels/activities.rs b/crates/temporal-workflow/src/workflows/channels/activities.rs index 7fe49c8b3..a97bce1b6 100644 --- a/crates/temporal-workflow/src/workflows/channels/activities.rs +++ b/crates/temporal-workflow/src/workflows/channels/activities.rs @@ -16,6 +16,7 @@ pub const ACTIVITY_CHAT_TOOL_DECLARATIONS: &str = "ChannelActivities::chat_tool_ pub const ACTIVITY_CHAT_READ_JSON_BLOB: &str = "ChannelActivities::read_json_blob"; pub const ACTIVITY_CHAT_PUT_JSON_BLOB: &str = "ChannelActivities::put_json_blob"; pub const ACTIVITY_CHAT_RECONCILE_DELIVERY: &str = "ChannelActivities::reconcile_delivery"; +pub const ACTIVITY_CHAT_TRANSCRIBE_MEDIA: &str = "ChannelActivities::transcribe_media"; pub const ACTIVITY_CHAT_EMIT_EVENT: &str = "ChannelActivities::emit_chat_event"; pub const ACTIVITY_CHAT_STORE_SENT: &str = "ChannelActivities::store_chat_sent"; pub const ACTIVITY_CHAT_RESOLVE_HANDLE: &str = "ChannelActivities::resolve_chat_handle"; @@ -30,6 +31,14 @@ pub struct ChannelActivities; #[activities] impl ChannelActivities { + #[activity(name = ACTIVITY_CHAT_TRANSCRIBE_MEDIA)] + pub async fn transcribe_media( + _ctx: ActivityContext, + _request: ChatTranscribeMediaRequest, + ) -> Result { + unimplemented!("workflow activity definition only") + } + /// Store the `message_*` declarations bound to this conversation as /// receiver; content-addressed, so stable per receiver. #[activity(name = ACTIVITY_CHAT_TOOL_DECLARATIONS)] diff --git a/crates/temporal-workflow/src/workflows/channels/conversation.rs b/crates/temporal-workflow/src/workflows/channels/conversation.rs index f5ae77038..a5f9f5576 100644 --- a/crates/temporal-workflow/src/workflows/channels/conversation.rs +++ b/crates/temporal-workflow/src/workflows/channels/conversation.rs @@ -545,31 +545,113 @@ async fn handle_inbound(ctx: &Ctx, inbound: AdmittedInbound) { tracing::debug!(message_id = %message(&inbound).message_id, ?reason, "inbound dropped"); } InboundPlan::Emit { text } => { - let media = match prepare_media(ctx, &inbound).await { - Ok(media) => media, - Err(error) => { - let message_id = message(&inbound).message_id.clone(); - ctx.state_mut(|wf| { - wf.state.messages.insert( - inbound_key, - ReceivedMessage { - message_id: message_id.clone(), - status: MessageStatus::Failed, - seq: None, - session_id: None, - error: Some(error.clone()), - }, - ); - wf.state - .protocol_errors - .push(format!("media {message_id}: {error}")); - }); - return; + let (media, transcript_refs) = + match prepare_message_media(ctx, &inbound, &inbound_key).await { + Ok(media) => media, + Err(error) => { + let message_id = message(&inbound).message_id.clone(); + ctx.state_mut(|wf| { + wf.state.messages.insert( + inbound_key, + ReceivedMessage { + message_id: message_id.clone(), + status: MessageStatus::Failed, + seq: None, + session_id: None, + error: Some(error.clone()), + }, + ); + wf.state + .protocol_errors + .push(format!("media {message_id}: {error}")); + }); + return; + } + }; + emit_message(ctx, inbound_key, &inbound, text, media, transcript_refs).await; + } + } +} + +async fn prepare_message_media( + ctx: &Ctx, + inbound: &AdmittedInbound, + key: &str, +) -> Result< + ( + Vec, + std::collections::BTreeMap, + ), + String, +> { + let media = prepare_media(ctx, inbound).await?; + let mut transcripts = std::collections::BTreeMap::new(); + for (index, item) in media.iter().enumerate() { + if item.kind != api::ChannelMediaKind::Audio { + continue; + } + let request = ctx.state(|wf| super::ChatTranscribeMediaRequest { + active: ChatAssertTriggerActiveRequest { + universe_id: wf.start.universe_id, + bot_id: wf.start.bot_id.clone(), + trigger_id: wf.start.trigger_id.clone(), + account_id: wf.start.account_id.clone(), + chat_id: wf.start.conversation.chat_id.clone(), + scope: wf.start.scope, + }, + idempotency_key: crate::transcription_id( + &api::Attribution::Local, + &format!("{}:{key}:{index}", wf.start.conversation.key()), + ), + audio: api::TranscriptionAudio { + blob_ref: item.blob_ref.clone(), + mime: item.mime.clone(), + name: item.name.clone().unwrap_or_else(|| "audio".into()), + }, + }); + loop { + let view = activity( + ctx, + ChannelActivities::transcribe_media, + request.clone(), + channel_activity_options(), + ) + .await?; + match view.status { + api::TranscriptionStatus::Pending | api::TranscriptionStatus::Running => { + let cancelled = { + let timer = ctx.timer(std::time::Duration::from_secs(2)); + let cancelled = ctx.cancelled(); + pin_mut!(timer, cancelled); + select! { _ = timer => false, _ = cancelled => true } + }; + if cancelled { + let _ = ctx + .external_workflow( + format!("{}/{}", request.active.universe_id, view.transcription_id), + None, + ) + .signal(crate::TranscriptionWorkflow::cancel, ()) + .await; + return Err("Audio preparation cancelled.".into()); + } } - }; - emit_message(ctx, inbound_key, &inbound, text, media).await; + api::TranscriptionStatus::Succeeded => { + transcripts.insert( + item.blob_ref.clone(), + view.transcript_ref.ok_or("missing transcript reference")?, + ); + break; + } + _ => { + return Err(view.failure.map(|f| f.message).unwrap_or_else(|| { + "Transcription unavailable; resend the recording.".into() + })); + } + } } } + Ok((media, transcripts)) } /// Download every attachment through the connector and put it in the CAS; @@ -621,6 +703,7 @@ async fn emit_message( inbound: &AdmittedInbound, text: String, media: Vec, + transcript_refs: std::collections::BTreeMap, ) { let chat = message(inbound); let request = ctx.state_mut(|wf| { @@ -648,6 +731,7 @@ async fn emit_message( is_reply_to_bot: chat.is_reply_to_bot, }, media, + transcript_refs, tools_ref: wf .state .tools_ref diff --git a/crates/temporal-workflow/src/workflows/channels/types.rs b/crates/temporal-workflow/src/workflows/channels/types.rs index 46fddafa0..16b4eea02 100644 --- a/crates/temporal-workflow/src/workflows/channels/types.rs +++ b/crates/temporal-workflow/src/workflows/channels/types.rs @@ -110,6 +110,8 @@ pub struct ChatEmitEventRequest { pub message: ChatMessage, #[serde(default)] pub media: Vec, + #[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")] + pub transcript_refs: std::collections::BTreeMap, /// CAS ref of the conversation's `message_*` declarations. pub tools_ref: String, /// This workflow, for `started` / `finished` receipts. @@ -211,3 +213,11 @@ pub enum ChatTriggerActiveResult { Active, Inactive { reason: String }, } + +/// Starts or reads independent transcription on the sessions worker queue. +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct ChatTranscribeMediaRequest { + pub active: ChatAssertTriggerActiveRequest, + pub idempotency_key: String, + pub audio: api::TranscriptionAudio, +} diff --git a/crates/temporal-workflow/src/workflows/mod.rs b/crates/temporal-workflow/src/workflows/mod.rs index ebb970f5d..7780f2851 100644 --- a/crates/temporal-workflow/src/workflows/mod.rs +++ b/crates/temporal-workflow/src/workflows/mod.rs @@ -3,9 +3,15 @@ pub mod channels; mod environment_job; mod session; mod subagent_execution; +mod transcriptions; pub use bots::{BotControllerWorkflow, BotTriggerFireWorkflow}; pub use channels::ChannelConversationWorkflow; pub use environment_job::EnvironmentJobWorkflow; pub use session::AgentSessionWorkflow; pub use subagent_execution::SubagentExecutionWorkflow; + +pub use transcriptions::{ + TranscriptionActivityResult, TranscriptionSnapshot, TranscriptionWorkflow, + TranscriptionWorkflowArgs, transcription_id, transcription_workflow_id, +}; diff --git a/crates/temporal-workflow/src/workflows/session/admissions.rs b/crates/temporal-workflow/src/workflows/session/admissions.rs index 2a7b3e8db..587170c67 100644 --- a/crates/temporal-workflow/src/workflows/session/admissions.rs +++ b/crates/temporal-workflow/src/workflows/session/admissions.rs @@ -38,7 +38,7 @@ pub(super) async fn admit_admissions( } }; let correlation_token = admission.correlation_token.clone(); - let mut command = admission.command; + let command = admission.command; if let CoreAgentCommand::ReplaceSessionConfig { config, expected_revision, @@ -61,19 +61,6 @@ pub(super) async fn admit_admissions( } continue; } - if observed_tools.is_none() && command_needs_input_preprocessing(&command) { - let session_id = drive.session_id().clone(); - match preprocess_input_entries(ctx, session_id, command).await? { - RunInputPreprocessResult::Succeeded { command: rewritten } => command = *rewritten, - RunInputPreprocessResult::Failed { failure } => { - record_admission_failure( - ctx, - failure.with_correlation_token(correlation_token), - ); - continue; - } - } - } let mut deferred_tools = None; if drive.state().lifecycle.status == CoreAgentStatus::Open && matches!(command, CoreAgentCommand::RequestRun(_)) @@ -256,172 +243,6 @@ pub(super) fn admissible_during_turn(command: &CoreAgentCommand) -> bool { ) } -enum RunInputPreprocessResult { - Succeeded { command: Box }, - Failed { failure: AgentAdmissionFailure }, -} - -pub(super) fn command_needs_input_preprocessing(command: &CoreAgentCommand) -> bool { - match command { - CoreAgentCommand::RequestRun(request) => request.source.input().iter().any(is_audio_input), - CoreAgentCommand::UpsertContext { entry, .. } => is_audio_input(entry), - _ => false, - } -} - -fn is_audio_input(input: &ContextEntryInput) -> bool { - input - .content - .media_type - .as_deref() - .map(|mime| mime.trim().to_ascii_lowercase().starts_with("audio/")) - .unwrap_or(false) -} - -async fn preprocess_input_entries( - ctx: &mut WorkflowContext, - session_id: SessionId, - command: CoreAgentCommand, -) -> anyhow::Result { - let (submission_id, input, rebuild) = match command { - CoreAgentCommand::RequestRun(request) => { - let engine::RunRequestSource::Input { input } = request.source; - ( - request.submission_id.clone(), - input, - InputPreprocessRebuild::RequestRun { - submission_id: request.submission_id, - run_config: request.run_config, - notify_on_terminal: request.notify_on_terminal, - requested_by: request.requested_by, - }, - ) - } - CoreAgentCommand::UpsertContext { - expected_revision, - key, - entry, - } => ( - None, - vec![entry], - InputPreprocessRebuild::UpsertContext { - expected_revision, - key, - }, - ), - command => { - return Ok(RunInputPreprocessResult::Succeeded { - command: Box::new(command), - }); - } - }; - - let result = ctx - .start_activity( - WorkflowActivities::preprocess_run_input, - PreprocessRunInputActivityRequest { session_id, input }, - activity_options(), - ) - .await - .map_err(|error| anyhow::anyhow!("{error}"))?; - - match result.outcome { - PreprocessRunInputOutcome::Succeeded { input } => Ok(RunInputPreprocessResult::Succeeded { - command: Box::new(rebuild.rebuild(input)?), - }), - PreprocessRunInputOutcome::Failed { failure } => Ok(RunInputPreprocessResult::Failed { - failure: preprocess_failure_to_admission_failure(submission_id, failure), - }), - } -} - -// Held only while one admission is preprocessed. -#[allow(clippy::large_enum_variant)] -enum InputPreprocessRebuild { - RequestRun { - submission_id: Option, - run_config: RunConfig, - notify_on_terminal: Vec, - requested_by: Option, - }, - UpsertContext { - expected_revision: Option, - key: ContextEntryKey, - }, -} - -impl InputPreprocessRebuild { - fn rebuild(self, input: Vec) -> anyhow::Result { - match self { - Self::RequestRun { - submission_id, - run_config, - notify_on_terminal, - requested_by, - } => Ok(CoreAgentCommand::RequestRun(engine::RunRequestCommand { - notify_on_terminal, - requested_by, - submission_id, - source: engine::RunRequestSource::Input { input }, - run_config, - })), - Self::UpsertContext { - expected_revision, - key, - } => { - let mut input = input; - let Some(entry) = input.pop() else { - anyhow::bail!("preprocessed context append returned no entry"); - }; - if !input.is_empty() { - anyhow::bail!("preprocessed context append returned multiple entries"); - } - Ok(CoreAgentCommand::UpsertContext { - expected_revision, - key, - entry, - }) - } - } - } -} - -pub(super) fn preprocess_failure_to_admission_failure( - submission_id: Option, - failure: PreprocessRunInputFailure, -) -> AgentAdmissionFailure { - AgentAdmissionFailure { - preparation_error: None, - submission_id, - correlation_token: None, - kind: match failure.kind { - PreprocessRunInputFailureKind::UnsupportedAudioMime => { - AgentAdmissionFailureKind::UnsupportedAudioMime - } - PreprocessRunInputFailureKind::AudioBlobMissing => { - AgentAdmissionFailureKind::AudioBlobMissing - } - PreprocessRunInputFailureKind::AudioBlobTooLarge => { - AgentAdmissionFailureKind::AudioBlobTooLarge - } - PreprocessRunInputFailureKind::AudioDurationTooLong => { - AgentAdmissionFailureKind::AudioDurationTooLong - } - PreprocessRunInputFailureKind::TranscoderUnavailable => { - AgentAdmissionFailureKind::TranscoderUnavailable - } - PreprocessRunInputFailureKind::TranscodeFailure => { - AgentAdmissionFailureKind::TranscodeFailure - } - PreprocessRunInputFailureKind::TranscriptionFailure => { - AgentAdmissionFailureKind::TranscriptionFailure - } - }, - message: failure.message, - rejection: None, - } -} - pub(super) fn should_refresh_runtime_projection_before_admitting( state: &CoreAgentState, command: &CoreAgentCommand, @@ -532,41 +353,3 @@ pub(super) fn active_instruction_inputs( }) .collect() } - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn upsert_preprocess_rebuild_preserves_expected_context_revision() { - let key = ContextEntryKey::new("client.audio"); - let entry = ContextEntryInput { - kind: engine::ContextEntryKind::ProviderOpaque, - content: engine::ContentRef { - content_ref: BlobRef::from_bytes(b"transcribed"), - media_type: Some("application/json".to_owned()), - provider_kind: None, - }, - preview: None, - origin: None, - provenance_ref: None, - token_estimate: None, - }; - - let command = InputPreprocessRebuild::UpsertContext { - expected_revision: Some(7), - key: key.clone(), - } - .rebuild(vec![entry.clone()]) - .expect("rebuild upsert"); - - assert_eq!( - command, - CoreAgentCommand::UpsertContext { - expected_revision: Some(7), - key, - entry, - } - ); - } -} diff --git a/crates/temporal-workflow/src/workflows/session/mod.rs b/crates/temporal-workflow/src/workflows/session/mod.rs index f3609d2bc..5522e79ab 100644 --- a/crates/temporal-workflow/src/workflows/session/mod.rs +++ b/crates/temporal-workflow/src/workflows/session/mod.rs @@ -29,7 +29,7 @@ use engine::{ BlobRef, CommandError, ContextEntryInput, ContextEntryKey, ContextEntryKind, ContextMessageRole, CoreAgentAction, CoreAgentCommand, CoreAgentDrive, CoreAgentDriveError, CoreAgentEntry, CoreAgentEvent, CoreAgentState, CoreAgentStatus, EmissionEnvelope, - LlmGenerationRequest, RunConfig, RunEvent, RunStatus, SessionId, SessionPosition, SubmissionId, + LlmGenerationRequest, RunEvent, RunStatus, SessionId, SessionPosition, SubmissionId, ToolInvocationBatchRequest, }; use futures::{FutureExt, pin_mut, select}; @@ -45,10 +45,8 @@ use crate::{ AwaitMaterializationRequest, AwaitOutcome, AwaitPromiseResult, CancellingWatchdog, CreateOrLoadSessionRequest, DEFAULT_CONTINUE_AS_NEW_HISTORY_THRESHOLD, JoinedContextPreparationRequest, LlmGenerateActivityRequest, PendingEmission, - PendingPromiseCancellation, PendingSourceResolution, PendingToolBatchResume, - PreprocessRunInputActivityRequest, PreprocessRunInputFailure, PreprocessRunInputFailureKind, - PreprocessRunInputOutcome, PromiseSourcePoll, PutBlobRequest, - RuntimeProjectionRefreshActivityRequest, ToolInvokeBatchActivityRequest, + PendingPromiseCancellation, PendingSourceResolution, PendingToolBatchResume, PromiseSourcePoll, + PutBlobRequest, RuntimeProjectionRefreshActivityRequest, ToolInvokeBatchActivityRequest, ToolPreparePromiseControlsActivityRequest, WorkflowActivities, activity_options, compose_workflow_id, default_instructions, split_workflow_id, }; diff --git a/crates/temporal-workflow/src/workflows/session/tests.rs b/crates/temporal-workflow/src/workflows/session/tests.rs index af973de3d..32fc99cc1 100644 --- a/crates/temporal-workflow/src/workflows/session/tests.rs +++ b/crates/temporal-workflow/src/workflows/session/tests.rs @@ -56,54 +56,6 @@ fn admission_failure_status_does_not_poison_later_admission() { assert_eq!(status.last_error, None); } -#[test] -fn request_run_with_audio_input_needs_preprocessing() { - let command = CoreAgentCommand::RequestRun(engine::RunRequestCommand { - requested_by: None, - notify_on_terminal: Vec::new(), - submission_id: Some(SubmissionId::new("submit_audio")), - source: engine::RunRequestSource::Input { - input: vec![ContextEntryInput { - kind: ContextEntryKind::Message { - role: ContextMessageRole::User, - }, - content: engine::ContentRef { - content_ref: engine::BlobRef::from_bytes(b"audio"), - media_type: Some("audio/ogg".to_owned()), - provider_kind: None, - }, - preview: Some("[audio]".to_owned()), - origin: None, - provenance_ref: None, - token_estimate: None, - }], - }, - run_config: crate::default_run_config(), - }); - - assert!(admissions::command_needs_input_preprocessing(&command)); -} - -#[test] -fn preprocess_failures_preserve_submission_id_for_admission_failure() { - let failure = admissions::preprocess_failure_to_admission_failure( - Some(SubmissionId::new("submit_audio")), - PreprocessRunInputFailure { - kind: PreprocessRunInputFailureKind::TranscriptionFailure, - message: "missing OpenAI key".to_owned(), - }, - ); - - assert_eq!( - failure.submission_id.as_ref(), - Some(&SubmissionId::new("submit_audio")) - ); - assert_eq!( - failure.kind, - AgentAdmissionFailureKind::TranscriptionFailure - ); -} - #[test] fn source_resolution_emission_queues_pending_resolution_with_producer() { let mut workflow = AgentSessionWorkflow::default(); diff --git a/crates/temporal-workflow/src/workflows/transcriptions.rs b/crates/temporal-workflow/src/workflows/transcriptions.rs new file mode 100644 index 000000000..4f1c1c4b9 --- /dev/null +++ b/crates/temporal-workflow/src/workflows/transcriptions.rs @@ -0,0 +1,181 @@ +//! Session-independent transcription. Temporal owns request identity and state; +//! activities store audio and transcript content in CAS. +use api::{ + Attribution, ModelConfig, TranscriptionFailure, TranscriptionFailureKind, + TranscriptionStartParams, TranscriptionStatus, TranscriptionView, +}; +use futures::{FutureExt, select_biased}; +use schemars::JsonSchema; +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; +use std::time::Duration; +use temporalio_common::protos::{ + coresdk::workflow_commands::ActivityCancellationType, temporal::api::common::v1::RetryPolicy, +}; +use temporalio_macros::{workflow, workflow_methods}; +use temporalio_sdk::{ + ActivityCloseTimeouts, ActivityOptions, CancellableFuture, SyncWorkflowContext, + WorkflowContext, WorkflowContextView, WorkflowResult, +}; + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionWorkflowArgs { + pub universe_id: uuid::Uuid, + pub transcription_id: String, + pub created_by: Attribution, + pub request: TranscriptionStartParams, + pub model: ModelConfig, + pub created_at_ms: u64, +} + +/// Explicit caller keys, rather than audio hashes, define request identity. +pub fn transcription_id(owner: &Attribution, key: &str) -> String { + let bytes = serde_json::to_vec(&(owner, key)).expect("serializable transcription identity"); + format!("transcription_{}", hex::encode(Sha256::digest(bytes))) +} + +pub fn transcription_workflow_id(args: &TranscriptionWorkflowArgs) -> String { + format!("{}/{}", args.universe_id, args.transcription_id) +} + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptionSnapshot { + pub request: TranscriptionStartParams, + pub view: TranscriptionView, +} + +#[derive(Clone, Debug, Serialize, Deserialize, JsonSchema)] +#[serde(tag = "kind", rename_all = "camelCase")] +pub enum TranscriptionActivityResult { + Succeeded { transcript_ref: String }, + Failed { failure: TranscriptionFailure }, +} + +impl TranscriptionWorkflowArgs { + pub fn pending(&self) -> TranscriptionView { + TranscriptionView { + transcription_id: self.transcription_id.clone(), + created_by: self.created_by.clone(), + audio: self.request.audio.clone(), + model: self.model.clone(), + status: TranscriptionStatus::Pending, + created_at_ms: self.created_at_ms, + transcript_ref: None, + text: None, + failure: None, + } + } +} + +#[workflow(name = "TranscriptionWorkflow")] +#[derive(Default)] +pub struct TranscriptionWorkflow { + snapshot: Option, + cancel_requested: bool, +} + +#[workflow_methods] +impl TranscriptionWorkflow { + #[run] + pub async fn run( + ctx: &mut WorkflowContext, + args: TranscriptionWorkflowArgs, + ) -> WorkflowResult<()> { + if ctx.workflow_id() != transcription_workflow_id(&args) { + return Err(anyhow::anyhow!("transcription workflow identity mismatch").into()); + } + let mut view = args.pending(); + view.status = TranscriptionStatus::Running; + ctx.state_mut(|state| { + state.snapshot = Some(TranscriptionSnapshot { + request: args.request.clone(), + view, + }) + }); + let options = ActivityOptions::with_close_timeouts(ActivityCloseTimeouts::Both { + start_to_close: Duration::from_secs(360), + schedule_to_close: Duration::from_secs(900), + }) + .heartbeat_timeout(Duration::from_secs(15)) + .cancellation_type(ActivityCancellationType::WaitCancellationCompleted) + .retry_policy(RetryPolicy { + initial_interval: Some(Duration::from_secs(2).try_into().unwrap()), + maximum_interval: Some(Duration::from_secs(15).try_into().unwrap()), + backoff_coefficient: 2.0, + maximum_attempts: 3, + non_retryable_error_types: vec![], + }) + .build(); + let mut activity = ctx.start_activity( + crate::WorkflowActivities::execute_transcription, + args, + options, + ); + let result = { + let cancellation = ctx.wait_condition(|state| state.cancel_requested).fuse(); + let external_cancel = ctx.cancelled().fuse(); + let mut work = (&mut activity).fuse(); + futures::pin_mut!(cancellation, external_cancel); + select_biased! { + _ = cancellation => None, + _ = external_cancel => None, + result = work => Some(result), + } + }; + if result.is_none() { + activity.cancel(); + let _ = activity.await; + } + ctx.state_mut(|state| { + let view = &mut state.snapshot.as_mut().expect("initialized").view; + match result { + None => view.status = TranscriptionStatus::Cancelled, + Some(Ok(TranscriptionActivityResult::Succeeded { transcript_ref })) => { + view.status = TranscriptionStatus::Succeeded; + view.transcript_ref = Some(transcript_ref); + } + Some(Ok(TranscriptionActivityResult::Failed { failure })) => { + view.status = TranscriptionStatus::Failed; + view.failure = Some(failure); + } + Some(Err(error)) => { + view.status = TranscriptionStatus::Failed; + view.failure = Some(TranscriptionFailure { + kind: if error.as_timeout().is_some() { + TranscriptionFailureKind::Timeout + } else { + TranscriptionFailureKind::Provider + }, + message: "Transcription exhausted its execution budget.".into(), + }); + } + } + }); + Ok(()) + } + + #[signal(name = "cancel")] + pub fn cancel(&mut self, _ctx: &mut SyncWorkflowContext) { + self.cancel_requested = true; + } + + #[query(name = "snapshot")] + pub fn snapshot(&self, _ctx: &WorkflowContextView) -> Option { + self.snapshot.clone() + } +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn identity_is_scoped_to_owner_and_explicit_key() { + let a = Attribution::Actor { id: "a".into() }; + let b = Attribution::Actor { id: "b".into() }; + assert_eq!(transcription_id(&a, "one"), transcription_id(&a, "one")); + assert_ne!(transcription_id(&a, "one"), transcription_id(&b, "one")); + assert_ne!(transcription_id(&a, "one"), transcription_id(&a, "two")); + } +} diff --git a/docs/documentation/access-and-security/api-keys-and-service-access.md b/docs/documentation/access-and-security/api-keys-and-service-access.md index 4c85ea8ae..4e94c7942 100644 --- a/docs/documentation/access-and-security/api-keys-and-service-access.md +++ b/docs/documentation/access-and-security/api-keys-and-service-access.md @@ -102,7 +102,9 @@ If provisioning a universe directly through the server host, use printed UUID. Platform-created universes are provisioned through its deployment key. See [People and roles](people-and-roles.md) for account administration. -## Create and revoke keys + + +## Create, rotate and revoke keys Universe Admins can open their universe's **API keys** page, choose **Create key**, supply a name, and choose the method groups the key may call. The form @@ -136,16 +138,37 @@ does not mean read-only access. Verify it with a profile read through the an operation outside that group is refused. Core stores a SHA-256 hash of the random secret, a display prefix, and key -metadata. The scope, groups, and actor flag cannot be edited. To rotate or -change authority, create a replacement, update and verify its consumer, then -revoke the old key in the UI or by prefix: +metadata. Choose **Rotate** on an active key to replace its secret immediately. +The new secret is shown once and the old secret stops working for subsequent +requests, with no grace period. Update every client using that key. Rotation +changes the display prefix and clears last use, while preserving the name, +scope, groups, actor flag, creator and creation time. Revoked keys cannot be +rotated. The Platform records the old and new prefixes in its access audit. +Keys that assert actors must be rotated from **Platform admin → API keys**. + +Both CLIs support rotation by the current prefix: + +```bash +lightspeed api-key rotate "" +lightspeed-server api-key rotate "" +``` + +The first command uses the configured connection; the second accesses the +store from the server host. Each returns the new secret once. If you rotate +the credential used by your CLI connection or Platform service, update that +configuration before making further requests. + +The scope, groups, and actor flag cannot be edited. To change authority, +create a replacement, update and verify its consumer, then revoke the old +key in the UI or by prefix: ```bash lightspeed-server api-key list lightspeed-server api-key revoke "" ``` -Revocation rejects subsequent requests. Already admitted requests, including +Rotation and revocation reject subsequent requests using the old secret. +Already admitted requests, including parked long-poll reads, are not reauthenticated while waiting. Revocation does not cancel admitted work or recall credentials delivered to another process. diff --git a/docs/documentation/using-lightspeed/cli.md b/docs/documentation/using-lightspeed/cli.md index 4c1f3fcd8..0a6b14ade 100644 --- a/docs/documentation/using-lightspeed/cli.md +++ b/docs/documentation/using-lightspeed/cli.md @@ -246,9 +246,14 @@ A deployment key with `deployment/api-keys` can issue narrower keys: lightspeed api-key create --name "Agent client" --universe-id UNIVERSE_UUID \ --group session --group models --group vfs --group blobs/put lightspeed api-key list +lightspeed api-key rotate KEY_PREFIX lightspeed api-key revoke KEY_PREFIX ``` +Rotation replaces the secret immediately and prints its replacement once. +Use the returned prefix for later key-management commands and update every +client using the old secret. Scope and permissions stay the same. + Create prints the secret once; use `--json` for machine-readable output. Add `--group auth` if the client needs to configure provider credentials, and other groups for the resources it manages. Omitting groups grants every group allowed by the chosen scope. diff --git a/docs/roadmap/p179-core-universes-keys-and-actors.md b/docs/roadmap/p179-core-universes-keys-and-actors.md index 814f30927..c08a45aa0 100644 --- a/docs/roadmap/p179-core-universes-keys-and-actors.md +++ b/docs/roadmap/p179-core-universes-keys-and-actors.md @@ -343,6 +343,21 @@ but the tables changed, so a development database needs `./dev.sh reset`. Step 4 is the one the Platform half waits for. +## Immediate secret rotation follow-up + +Implemented on 2026-09-29: `deployment/api-keys/rotate`, both CLI commands, +and both Platform key-management pages replace the hash and display prefix +atomically in the existing row. There is no grace period or schema migration. +Authority and creation metadata stay unchanged; last use resets. Revoked and +stale prefixes cannot rotate, and concurrent rotations of one prefix have a +single winner. The Platform audit records both prefixes without the secret. +Universe admins cannot obtain actor-asserting credentials through rotation; +those keys use Platform administration. + +Validation: API dispatch and schema freshness tests, an isolated PostgreSQL +test covering immediate invalidation and concurrent rotation, Platform route +and UI tests, generated-client checks, and production/demo builds pass. + ## Validation - Unit: group derivation covers every manifest method; key checks for diff --git a/docs/roadmap/p184-universe-model-defaults-and-transcription.md b/docs/roadmap/p184-universe-model-defaults-and-transcription.md new file mode 100644 index 000000000..846509f1c --- /dev/null +++ b/docs/roadmap/p184-universe-model-defaults-and-transcription.md @@ -0,0 +1,570 @@ +# P184 — Universe model defaults and standalone transcription + +**Status:** Universe defaults, Platform model settings, standalone transcription, +channel voice preparation, and web dictation implemented, 2026-09-29. +Builds on [CLI session model routing](cli-session-model-routing.md) and +[the runtime CLI](p183-first-class-runtime-cli.md). Supersedes the placement +of transcription inside session admission in +[audio transcription preprocessing](archive/p72-audio-transcription-preprocessing.md). + +## Outcome + +A universe chooses its default agent model and speech-to-text model through +the public runtime API. These choices work for the web, CLI, bots, and direct +API clients, including installations using custom provider endpoints. + +Transcription is an independent operation. Channels transcribe voice messages +before delivering prepared input to a bot. Web dictation produces an editable +draft; only the user's ordinary send action submits it to a session. + +The design stays small: two default slots, shared provider resolution, and one +typed transcription workflow. Future capabilities, such as image generation, +can add a slot, adapter, and operation without extending session orchestration. + +## Baseline before implementation + +- `temporal-server/src/config.rs` resolves an omitted session model from + `LIGHTSPEED_CHAT_PROVIDER` and `LIGHTSPEED_CHAT_MODEL`, with built-in defaults + of `openai` and `gpt-5.5`. The API kind is fixed to OpenAI Responses. +- Session creation, configuration replacement, and profile application use + that deployment fallback. Existing sessions pin their provider identity and + API kind; model changes within the route remain supported. +- Session admission detects audio in run requests and context upserts, awaits + preprocessing, and admits the rewritten transcript. Steering follows a + separate path. Audio preparation therefore has inconsistent entry points and + runs inside the session's admission loop. +- The transcription activity uses the built-in `openai` provider and the audio + client's `gpt-4o-transcribe` default. It resolves a stored credential but does + not apply a stored custom endpoint. Provider failures become ordinary failed + preprocessing outcomes rather than retryable activity failures. +- Provider records and generation clients already support custom endpoints, + API-kind declarations, and credential resolution. Custom endpoint validation + currently admits only Responses and Chat Completions protocols. +- The web composer sends text. Its provider-readiness helper considers any + provider with a configured credential sufficient, independently of the route + a session will actually use. + +## Decisions + +### 1. Universe-owned defaults with a focused API + +Store a small, typed model-defaults record in runtime PostgreSQL, keyed by +universe and protected by an optimistic revision. Platform reads and changes +this record through the runtime; it does not keep another copy of the policy. + +The initial slots are: + +| Slot | Selection | When the default is resolved | +| --- | --- | --- | +| `agentRun` | Agent generation route | Creation of a new session without an explicit or profile model | +| `speechToText` | Transcription route | Admission of a new transcription job without an explicit model | + +Every selection contains `{ providerId, apiKind, model }`. Endpoint URLs, +headers, and credentials remain in the existing provider configuration. +Slots can be unset independently; a universe may use agent runs without +configuring transcription, or transcription with no default agent model. + +Expose: + +- `models/defaults/read`: returns the revision, selections, and per-selection + configuration status. Keep persisted choices distinct from current provider + diagnostics. +- `models/defaults/put`: sets or clears one named slot, using an expected + revision. Clearing uses `null`; unknown slots and incompatible API kinds are + rejected. An update must not overwrite another slot inadvertently. + +Defaults use the existing model-resource policy: members can inspect them; +Operators and Admins can configure them through Platform. Direct runtime keys +continue to use method-group authority. Define these rules in the method +manifest and regenerate its consumers. + +A missing default fails with a typed `model_default_unset` error identifying +the slot, but only when the request needs that default. A valid explicit model +does not require a configured default. Invalid or unavailable selections never +silently switch to another provider or model. + +### 2. Resolve agent defaults at session creation + +New-session precedence is: + +1. Explicit session model. +2. Model supplied by the selected profile. +3. Universe `agentRun` default. + +Persist the resulting complete model selection in the session. Run precedence +remains explicit run override, then stored session model. Changing or clearing +a universe default does not change an existing session or an admitted run. +Provider identity and API kind remain pinned for the session's lifetime. + +Use the same resolution boundary for web, CLI, bot, and sub-agent creation. +Preserve established profile and inheritance rules around that boundary. +Reopening an existing session recovers its stored configuration before reading +mutable defaults. Clones and forks retain the configuration they already copy +or inherit; they do not acquire a new default as a side effect. + +For existing sessions, an omitted model in `session/config/put` or an applied +profile means **preserve the current model**. Make this an explicit contract +rule and resolve it against the current configuration revision. The remainder +of configuration replacement keeps its existing semantics, including revoking +omitted features. An explicit incompatible provider route remains an error; +bot profile reconciliation retains its existing session-rotation behavior. + +Remove `LIGHTSPEED_CHAT_PROVIDER` and `LIGHTSPEED_CHAT_MODEL` as runtime model +fallbacks. Setup and a deliberate upgrade/import step may write the old +effective selection into specific universe records. Do not populate every +universe with an assumed OpenAI route or overwrite configured values during +startup. Retired variables should produce an actionable configuration message. +Existing deployment credential fallbacks are a separate concern and retain +their current provider-resolution policy. + +The development launcher explicitly seeds defaults for its development +universes through the API, without resetting subsequent user choices. The CLI +must support reading, setting, and clearing defaults so setup stays usable +without Platform. Adding a provider in the web UI offers selection of a +default; adding a credential alone does not silently choose a model. + +### 3. Separate purpose, protocol, and provider + +Purpose names the use within Lightspeed. API kind names the protocol spoken by +an adapter. Provider ID names an endpoint and its authentication configuration. +Several purposes may use the same protocol and still have different defaults. + +Keep broader route selection and purpose validation outside the deterministic +engine. The engine's model types continue to describe agent generation only. +Share provider resolution and route validation where useful; provider requests +and responses stay native to their adapters. + +The first transcription protocol is the existing +`openai:audio-transcriptions`. Support both the built-in provider and configured +compatible endpoints, including explicit credentialless endpoints. Extend +endpoint validation, the resolver's API-kind check, and the audio client's +per-request transport support together. A custom provider must never fall back +to the built-in endpoint or credential when its configuration is missing or +unusable. FFmpeg remains an optional adapter. + +Add a purpose filter to model discovery and use it in the relevant pickers. +An endpoint's protocol declaration does not establish that every model it lists +supports that protocol. Use provider metadata or conservative suggestions and +retain manual model entry. Discovery failure is distinct from missing +configuration and must not prevent saving an otherwise valid explicit route. +Transition existing `selectableOnly` consumers deliberately. + +### 4. One standalone transcription workflow + +Expose a universe-scoped API independent of sessions: + +| Method | Behavior | +| --- | --- | +| `transcriptions/start` | Accepts an audio CAS reference with MIME/name, optional model, language/prompt options, and an idempotency key; returns a transcription ID, resolved model, and status | +| `transcriptions/read` | Returns pending/running/succeeded/failed/cancelled status and, on success, transcript reference and text | +| `transcriptions/cancel` | Requests cancellation of an unfinished job; repeated requests are safe | + +The gateway admits the request and starts a `TranscriptionWorkflow`. The +workflow owns validation, optional transcoding, provider execution, result +storage, and cancellation. Activities perform all provider, filesystem, and +store I/O. Keep its dependencies and registration separate from session +admission. + +Initially host it in the existing server deployment and sessions worker role +and queue. This is a hosting choice, not session-workflow ownership. Channel +controllers start/join the transcription workflow through the workflow +boundary; they do not dispatch its activities onto another role's queue. +A new worker role or independent capacity can be introduced when operationally +needed, without changing the public operation. + +Use bounded provider attempts and an overall deadline, with activity and HTTP +timeouts aligned. Classify transient provider failures for durable retry; +configuration, authentication, unsupported input, and other terminal failures +return typed outcomes. Preserve the existing byte/duration admission limits +and optional transcoder behavior, with provider-specific format support in +the adapter. Cancellation must reach in-flight activities; terminal completion +and cancellation races must converge on one recorded outcome. + +### 5. Explicit request identity and result lifetime + +Temporal owns the admitted request, attribution, pinned model, status, and +result reference. There is no transcription table, schema migration, lease, or +separate retention service. + +Within a universe and requester scope, an explicit idempotency key identifies +one request. Matching retries query the original workflow before consulting +current defaults. Changed input or options conflict. The workflow start uses +reject-duplicate identity; concurrent admissions recover the winning request. +The original submitted request preserves whether its model was omitted. +Temporal's namespace history retention bounds this identity guarantee. + +Credentials and endpoint configuration are resolved through the provider record +at execution time. Default changes cannot select another model during retry. +An ambiguous upstream response can still result in another billed attempt; +this does not promise exactly-once provider execution. + +Audio and transcript content live in ordinary CAS. Workflow history carries +bounded metadata and references: source audio, resolved model, and transcription +options. The result blob contains only plain UTF-8 text. Input admission refreshes ordinary +CAS grace; it does not create a temporary retention root. Unsubmitted content +may be swept after that grace (seven days by default), and a missing completed +result is reported as expired. A delayed caller can upload again with a new key. + +Session admission sets the existing `provenance_ref` to the source audio. +Existing session roots retain both transcript and original recording. Bot-event +roots also retain prepared transcript references alongside their audio while +awaiting delivery. These extend existing reference enumeration, not storage +infrastructure. + +The `transcriptions` method group permits Contributors to start jobs through +person-level gateways. Asserted actors can only read or cancel their own drafts; +direct universe keys retain their method-group authority. There is no draft +listing. CAS download continues to use the existing universe-scoped blob access +policy. A separate person-level administrator draft browser is outside this slice. + +### 6. Prepared input is the session boundary + +Session APIs accept ordinary `Text` or `TextRef` input. Both support an optional +`provenanceRef` pointing to a source blob in the same universe. Admission checks +that the source exists and copies the reference into the existing context +`provenance_ref`. Projections preserve it. This generic metadata supports audio, +documents, or other sources without exposing transformation-specific formats to +sessions. The engine receives prepared text and performs no transcription +orchestration. There is no dedicated transcript input or JSON artifact. + +After callers migrate, raw audio `Media` is rejected consistently by run +start, context append, and steering, with an actionable typed error directing +callers to transcription. Context append retains its per-entry failure +semantics. Native audio model input, if added later, will be a separate explicit +capability. + +Do not build a general ingress coordinator merely to preserve the old +single-call audio convenience. Audit existing clients before cutover. If a +specific compatibility requirement emerges, document its scope and retirement +separately rather than making it part of the target architecture. + +For Telegram and WhatsApp, the conversation workflow starts transcription after +channel authorization and media download/upload, before emitting the bot +event. Retain the original message and attach the transcript explicitly so +downstream routing and filters can inspect its words. Spoken text must not +automatically become a channel-management command. + +The conversation controller owns ordering, failure reporting, cancellation, +and delivery retries. Stable inbound identity recovers the same transcription +job. A retry after transcription succeeds reuses the result and does not emit +a duplicate event or run. Preserve the conversation's ordering contract when +audio and text arrive together. Connectors stay transport-only and acquire no +model credentials or routing authority. + +### 7. Web dictation edits a draft + +The interaction is: + +```text +Record → stop → upload/transcribe → review and edit → send +``` + +The microphone uses browser-supported recording formats selected by feature +detection. Recording and transcription never start, queue, or steer a run. +Successful transcription inserts text into the composer while preserving +existing draft content and edits made during the request. Late completion +after cancellation, navigation, draft submission, or universe/session change +must not restore or modify a stale draft. Release microphone resources on all +exit paths; failures preserve the user's existing text and support retry. + +The ordinary send controls determine whether reviewed text starts a run, +queues it, or steers it. It is sent as text, without representing the edited +words as the original audio transcript. Unsubmitted recordings and artifacts +use ordinary CAS grace rather than session retention. + +Show transcription availability and an actionable reason when unavailable. +Readiness follows the selected transcription route, including providers that +need no credential. For new sessions, check the effective explicit/profile/ +universe selection; for existing sessions, check their stored route. A key on +some other provider does not make the selected route ready. Unknown provider +health remains distinct from invalid configuration. + +Add matching demo routes, fixtures, and composer behavior so the browser demo +exercises the same draft flow without requiring a real microphone or provider. + +## Migration and rollout + +Introduce defaults and the standalone operation before removing old audio +admission. Migrate first-party channel and web consumers to prepared input, +then cut over the raw-audio API contract. CLI and direct-client migration +guidance must describe the upload, transcription, and submission steps. + +This is a greenfield cutover. Remove session audio preprocessing, its activity +registration, request/result types, admission errors, and compatibility branches. +Channels always prepare audio through standalone transcription before session +admission. No legacy activity or replay adapter is retained. + +Existing sessions keep their persisted models. Upgrade tooling imports model +defaults only for explicitly selected universes. Fresh universes remain +unconfigured until setup supplies their choices. Document failures caused by +unset defaults and retired environment variables before release. + +Regenerate public API artifacts, TypeScript clients, method roles, and the +workflow integration contract when changed. Keep schema revision metadata +aligned. Update affected product/development documentation with user review; +this roadmap does not authorize unrelated documentation or root README edits. + +## Implementation progress + +- [x] Add revisioned universe model defaults, APIs, typed failures, purpose + validation, and CLI read/set/clear commands. +- [x] Apply creation-time default resolution across session entry points; + preserve omitted models on existing-session changes and retain route pins. +- [x] Remove environment model fallbacks; support explicit universe setup with + the CLI and seed untouched development universes through the API. +- [x] Add Platform Models settings, per-selection configuration diagnostics, + and effective-route readiness. +- [x] Extend provider configuration/resolution and the audio client for + compatible transcription endpoints, including credentialless transport. +- [x] Add the workflow-owned transcription state, start/read/cancel APIs, + request deduplication, and requester access rules. +- [x] Return plain transcript text and add generic source provenance to text inputs. +- [ ] Add web dictation with draft preview, cancellation, and demo coverage. +- [x] Move channel voice-message preparation before bot-event delivery and + validate ordering, failure handling, and retries. +- [x] Audit raw-audio clients, implement the workflow-history transition, and + remove session preprocessing and obsolete compatibility code. +- [x] Regenerate affected contracts, complete scoped checks, and record + migration and validation results here. + +### First slice + +Migration `010_model_defaults.sql` adds one row per configured universe, with +an optimistic revision and independent nullable selections. An untouched +universe reads as revision zero. Setting or clearing either slot advances the +shared revision, and stale writes fail without changing either selection. +Clearing retains the row so development setup cannot undo an intentional +clear. No model policy is inferred or backfilled by migration or server startup. + +The read/put APIs use the model method group's existing authority, with read +and configure-resource actions respectively. Reads return persisted selections +and the revision; the web combines these with separate provider diagnostics. +The `speechToText` slot can be configured now, but the existing transcription +path will begin consuming it only when standalone transcription is implemented. + +Creation merges explicit and profile configuration before consulting the +universe default. An explicit or profile model bypasses the defaults lookup. +A missing required default returns `model_default_unset` with +`modelDefaultSlot`. Existing-session replacement and profile application resolve +omitted models from current session state and guard the resulting write with +that configuration revision. The engine's deterministic behavior is unchanged. + +The CLI exposes: + +```bash +lightspeed model defaults read --json +lightspeed model defaults set agent-run \ + --provider --api-kind --model +lightspeed model defaults clear agent-run +``` + +Use `speech-to-text` for the other slot. Set and clear accept +`--expected-revision`; otherwise the CLI reads the current revision once before +writing. A conflict is reported without automatically retrying over a newer +choice. `--json` returns the defaults record for each command. + +Before starting an upgraded runtime, apply schema revision 10 with +`cargo run -p temporal-server -- migrate`. Remove `LIGHTSPEED_CHAT_PROVIDER` +and `LIGHTSPEED_CHAT_MODEL`; their presence now produces an actionable startup +error. Select each universe explicitly in the CLI and set its intended route. +Existing sessions remain usable with their stored models, and explicit-model +creation remains available before a default is configured. + +The development launcher seeds its development universe and, when enabled, the +Platform Test universe after readiness. This fixture chooses the existing +OpenAI development route only at revision zero. Subsequent settings or clears +are preserved, including concurrent changes. Runtime startup itself never +chooses a default. Live-test fixtures now seed their model policy explicitly. + +Validation covers API wire schemas, protocol/purpose checks, model precedence, +omission semantics, CLI round trips and conflicts, launcher behavior, generated +TypeScript consumers, and the runtime library suite. PostgreSQL tests use +temporary isolated schemas to check migrations, concurrent first writes, +universe isolation, slot preservation, clears, stale revisions, and deletion. +Public API contracts and TypeScript consumers were regenerated; the workflow +contract exporter produced no contract change. + +Follow-up live validation ran seven selected Temporal tests serially against +temporary local databases and a separate test universe. All passed: the fake +session lifecycle, OpenAI default-backed session execution, OpenAI Responses +and Chat Completions tool round trips, an Anthropic Messages tool round trip, +and both profile integration tests. OpenAI calls used `gpt-6-sol`, the model +now explicitly seeded by the development launcher. Each temporary database +was migrated to revision 10 and removed after testing. + +The OpenAI session test now verifies creation from the universe default, +clearing through the API, the typed missing-default failure for new sessions, +and preservation of the original model across reopening, configuration +replacement, and profile application before a real model call. + +OpenAI rejected `gpt-6-sol` function tools with its default reasoning setting +on Chat Completions. That protocol's tool fixture now explicitly sets +`reasoning_effort: "none"`, and failure diagnostics report the recorded run +error. Responses passed with its default reasoning settings; production route +selection and generation policy were not changed to hide the provider error. +The slow live suite remains unrun. + +### Web defaults slice + +Setup → Models (`/u/:slug/models`) now places Defaults below Providers. The +Agent runs selection supports choosing a discovered model, entering a manual +provider/API/model route, and explicitly clearing the slot. Operators and +universe admins can edit; contributors and viewers can inspect. Platform +admins retain their admin access. The Platform routes use the member-scoped +runtime client and its method permission gate. + +Edits retain the revision loaded when the dialog opened. A conflict preserves +the draft and requires an explicit reload and review before another write; +there is no automatic overwrite. Discovery failures leave manual selection +available. Adding a provider offers default selection as a separate action. + +New-session creation previews the effective session, profile, or universe +model without copying the preview into the request. An unset universe default +blocks creation only when no model was supplied. Existing-session readiness +uses the session's stored model, including embedded bot sessions. Profile and +bot setup editors label omitted selections as the universe default. + +Readiness checks the selected provider and API, distinguishes missing or +disabled credentials from unknown availability, and accepts credentialless +providers and models absent from discovery. This reports configuration status, +not proof that a future model call will succeed. The speech-to-text slot is configurable through the runtime API and CLI; its +settings row will accompany web dictation. + +The demo uses the same defaults editor, revision checks, and creation policy, +including bot sessions. Clearing a default leaves existing session models +intact. Tests cover API permissions and typed failures, manual selection +during discovery failure, conflicts, provider setup, effective-model previews, +unset defaults, and demo isolation and preservation. The full web, Platform +server, and TypeScript client suites, workspace typechecks, and production and +demo builds pass. The Models page was also visually checked in the browser. + +### Standalone transcription and channel preparation + +`TranscriptionWorkflow` lives under `temporal-workflow/src/workflows` and runs +on the sessions role's queue independently of session orchestration. The +start/read/cancel API carries requester identity, stable explicit request keys, +and a pinned model. Activity attempts are bounded to three, with a fifteen-minute +schedule budget and a sixteen-minute workflow deadline. Cancellation waits for +activity acknowledgment before recording the terminal state. Missing results +report expiry after an authoritative CAS metadata check, even when bytes remain +in a process cache. + +Provider resolution now accepts the audio-transcriptions protocol, including +custom authenticated and anonymous endpoints. Native requests use the selected +model, language, prompt, URL, and configured headers. Endpoint overrides exclude +deployment credentials and organization/project headers. Provider responses and +transcripts are bounded. Audio limits and the optional transcoder belong to +standalone transcription; the session workflow performs no audio processing. + +Channels authorize and prepare attachments before starting/joining standalone +transcription through a short activity bridge. The conversation awaits the +result before emitting a bot event; stable conversation/message/attachment keys +reuse the same job. Filters see the transcript in message text; the original +text is retained separately. Spoken commands are never reclassified as channel +commands. Bot attachment `textRef` becomes ordinary text-reference input with +its source attachment as provenance. Existing bot-event and session roots retain +both text and original audio. + +New run, context, and steering input rejects raw audio with guidance to use +`transcriptions/start`, poll `transcriptions/read`, then submit +`{type: "textRef", blobRef: ..., provenanceRef: ...}` or reviewed plain text. The in-tree +raw-audio producer was Channels; CLI chat sends text. Direct API callers must +migrate. The greenfield cleanup removes the legacy preprocessing activity, +session input rewriting, audio-specific admission failures, source-to-transcript +retry matching, and channel version branch. Audio helpers and their tests live +under the standalone transcription implementation. + +The subsequent simplification removes `Transcript`, `transcript_input`, the +JSON artifact, and transcript-specific provider rendering. The workflow returns +a plain text blob and source-audio metadata. Reviewed dictation can omit source +provenance; channel delivery supplies it through generic text input. + +The plain-text path passed API/projection/bot/model-adapter/workflow tests, +356 server unit tests, affected Rust target checks, TypeScript checks, and client +tests. Live transcription, channel redelivery, and PostgreSQL retention tests +passed together. The expiry fixture uses unique text because identical results +share a CAS blob that another session may legitimately retain. + +Cleanup validation passed: API and workflow tests, generated contract checks, +356 server unit tests, all server targets, and TypeScript checks. The three +standalone transcription live tests and the channel voice/redelivery live test +passed again against an isolated database, which was removed afterward. + +Validation passed: API/auth/model-runtime/bot/workflow suites and generated +contract checks; server unit tests; TypeScript checks, Platform/web/client tests, +and the web production build. Live tests used an isolated PostgreSQL database +and unique Temporal queues. They cover default pinning across changes, +idempotency conflicts, requester isolation, transcoding, session provenance, +transient retry, acknowledged cancellation, ordinary CAS expiry, bot-event +retention, and channel redelivery with spoken command text. A real OpenAI audio +transcription also passed. Web recording and dictation validation follows below. + +Web dictation now records through lazy-loaded `extendable-media-recorder`, +choosing a browser-supported WebM, MP4, or Ogg format. The composer requests +microphone access only on an explicit click, caps recordings at ten minutes +and 25 MiB, and releases tracks on completion, cancellation, errors, and +navigation. Upload/start/read/cancel routes use the member-scoped runtime client; +web admission always uses the universe speech default. + +The transcript appends to the latest editable draft and never sends itself. +Retry keeps the recording in memory and rejoins an admitted job after transport +failure. Cancel, send, and navigation prevent late completion from changing a +draft. Reviewed text uses ordinary session input without audio provenance. +The demo uses a clearly labeled sample recording and transcript. + +Models → Defaults contains separate agent-run and speech-to-text rows below +Providers. Operators can configure either slot; Contributors can transcribe. +Dictation is disabled until the speech default is set and is also subject to +session input permissions and known provider/browser readiness. OpenAI discovery +maps supported file-transcription families to the audio protocol, bypasses the +agent-only age filter for those routes, and keeps them out of agent pickers. +Custom providers continue to use their declared API kinds and manual choices. + +Validation passed: 572 web tests, 162 Platform server tests, 13 Rust model +discovery tests, eight API contract tests, TypeScript checks, and production +and demo builds. Browser checks covered speech suggestions, editable transcript +preview, clearing the default, and mobile layout. A real Chromium recorder +produced WebM audio from a generated audio stream and released its tracks; +physical microphones and Safari were not exercised in this slice. + +## Acceptance and validation + +Use offline tests with fake provider transports for the default validation +loop. Add focused Temporal/PostgreSQL integration coverage where storage or +workflow boundaries require it; live/credentialed suites remain explicit. + +- Defaults are universe-isolated and revision-safe. Explicit and profile + models take precedence; unset defaults only reject requests that need them. + Updating defaults leaves existing sessions and admitted jobs unchanged. +- Session creation, reopening, profile application, configuration replacement, + clones/forks, bot rotation, and sub-agent creation preserve the documented + model semantics. Include engine replay coverage if deterministic behavior + or event handling changes. +- Custom transcription routes use their configured endpoint, model, headers, + and authentication. Anonymous endpoints work; missing/disabled custom + providers never fall back to OpenAI. Discovery absence does not prohibit + valid manual selection. +- Matching job retries survive default changes and gateway/workflow restarts; + conflicting payloads fail. Exercise concurrent workflow admission, + transient versus terminal failures, deadlines, and cancellation races. +- Unsubmitted content uses ordinary CAS grace. Missing results report expiry; + session-admitted transcripts and bot events retain their original audio and + transcript through existing roots. Verify requester isolation for job reads. +- Channel redelivery and delivery failure after successful transcription do + not repeat admitted work. Voice/text ordering and mixed-media failures are + explicit. Authorization precedes model work, and transcripts are available + to downstream routing without invoking channel-control commands. +- Web recording/transcription never sends a message automatically. Test + preserved edits, cancellation, late completion after send/navigation, + microphone cleanup, permission/format failures, and unavailable defaults. +- Run start, context append, and steering consistently enforce prepared input. + Text provenance survives projection and session retention without special + transcript rendering, artifacts, or legacy transcription activity calls. + +## Scope boundary + +This delivery implements agent defaults and batch speech-to-text. Image +generation, realtime transcription, diarization, automatic model failover, +cross-request transcript caching, and a general processing-pipeline framework +remain outside it. Shared abstractions cover model selection and provider +resolution; new capabilities get typed operations as their requirements arise. diff --git a/docs/roadmap/p185-external-harness-sessions.md b/docs/roadmap/p185-external-harness-sessions.md new file mode 100644 index 000000000..0670708b3 --- /dev/null +++ b/docs/roadmap/p185-external-harness-sessions.md @@ -0,0 +1,495 @@ +# P185 — External harness sessions in environments + +**Status:** Proposed, 2026-09-29. Repository and upstream documentation reviewed; +no bridge, compatibility trial, or runtime changes implemented. + +## Outcome + +Run third-party agent harnesses inside Lightspeed environments and expose their +conversations through the existing session API, event log, and clients. A user +can choose a harness, submit work, follow messages and tool activity, answer +permission requests, cancel work, and return to the conversation later where +the harness supports persistence. + +Lightspeed owns admission, orchestration, access policy, retained observations, +and supervision. The external harness owns its model calls, context, tools, +and internal execution loop. Recovering Lightspeed's workflow or transcript +does not imply recovering a harness's interrupted execution. + +Use Agent Client Protocol (ACP) as the first integration driver to validate. +Keep the internal session model and environment supervision independent of +ACP so a harness can use a native interface when required. + +Implement this in modules within existing workspace crates. Begin with the +environment protocol, its client, and envd, then drive one real harness through +that boundary before changing Lightspeed session storage, workflows, or UI. + +This is the opposite direction from the +[editor ACP adapter](later/pNNN-editor-acp-adapter.md): here Lightspeed is the +ACP client and an external harness is the agent. That proposal exposes +Lightspeed as an agent to editors. The +[A2A adapter](later/pNNN-a2a-protocol-adapter.md) addresses delegation between +agent services and remains a separate integration. + +## Protocol choice and adoption + +The relevant alternatives are ACP, harness-native interfaces, and integration +with an existing agent host. A2A addresses independently operated agent +services rather than the detailed control of a harness launched on a machine. + +Upstream evidence reviewed on 2026-09-29: + +| Integration | Documented approach | Implication | +| --- | --- | --- | +| [Zed external agents](https://zed.dev/docs/ai/external-agents) | ACP and an agent registry | ACP is an established external-agent path in a shipping editor. | +| [JetBrains AI Assistant](https://www.jetbrains.com/help/ai-assistant/acp.html) | Registry and manually configured ACP agents | A second editor family consumes the same integration surface. | +| [VS Code Agent Host](https://code.visualstudio.com/blogs/2026/08/26/agent-host-architecture) | Harness-specific adapters, including Copilot SDK and Claude Agent SDK; AHP between host and clients | A shared session experience does not require every harness to speak the same protocol. | +| [Copilot CLI](https://docs.github.com/en/copilot/reference/copilot-cli-reference/acp-server) | An ACP server, currently public preview | ACP is also exposed by a major harness vendor. | + +This establishes concrete implementations, not measured usage share or a +guarantee of future convergence. Compatibility must be demonstrated against +the versions Lightspeed intends to ship. + +[Agent Host Protocol (AHP)](https://code.visualstudio.com/docs/agents/concepts/agent-host) +provides synchronized session state for multiple clients connected to a host. +Its host-owned state, snapshots, and ordered updates are relevant design +references. Adopting it would be a separate decision to reuse an existing host +or expose Lightspeed to AHP clients; it is not required for the initial bridge. + +Native interfaces remain valid alternatives. For example, +[Codex App Server](https://learn.chatgpt.com/docs/app-server) exposes rich +Codex integration, while Claude Agent SDK and Pi RPC provide their respective +harness boundaries. A direct driver can avoid adapter feature gaps at the +cost of maintaining harness-specific integration and version compatibility. + +## Initial harness targets + +| Harness | ACP route | Evidence and qualification | +| --- | --- | --- | +| Codex | `codex-acp` adapter | [Adapter](https://github.com/agentclientprotocol/codex-acp) starts Codex App Server and translates requests and events. | +| Claude | `claude-agent-acp` adapter | [Adapter](https://github.com/agentclientprotocol/claude-agent-acp) uses Claude Agent SDK; this is an SDK integration, not control of the interactive terminal UI. | +| OpenCode | Native `opencode acp` | [Documentation](https://opencode.ai/docs/acp/) describes an ACP subprocess over stdio. | +| Pi | `pi-acp` adapter | [Adapter](https://github.com/svkozak/pi-acp) starts `pi --mode rpc`; it documents incomplete capabilities, including no forwarding of ACP-provided MCP configuration into Pi. | + +The compatibility trial pins harness and adapter versions. Presence in a +registry does not establish support for permissions, persistence, concurrency, +or every native feature. Record supported operations and limitations per +tested combination rather than exposing one global ACP-support flag. + +Start with ACP v1 and negotiated capabilities. The +[ACP v2 announcement](https://agentclientprotocol.com/announcements/acp-v2-draft) +currently labels v2 a draft and changes prompt and session lifecycle semantics. +Keep version mappings inside the driver; do not make a v1 prompt response the +universal definition of run completion. + +## Current repository boundary + +- [StoredEvent](../../crates/engine/src/session/stored.rs) already has a generic + `kind`, `version`, and `payload` envelope. The session store supports ordered + appends with an expected head and retains referenced content. +- [Session storage projections](../../crates/engine/src/storage/session.rs) + recognize native core lifecycle and run events. Generic storage does not yet + make sessions independent of the engine. +- [API projections](../../crates/api-projection/src/lib.rs) consume native core + state and events. Current-state reads, transcript views, and run output need + explicit external-backend support. +- [Session event reads](../../crates/api/src/sessions.rs) already support a + cursor and `waitMs`; the web client already tails this API. +- The [environment process protocol](../../crates/environment-protocol/src/data/process.rs) + closes ordinary pipe stdin after initial input; later input requires a PTY. + ACP requires persistent bidirectional pipes, so it cannot simply use the + existing interactive process surface unchanged. +- The [environment-job workflow](../../crates/temporal-workflow/src/workflows/environment_job.rs) + is a precedent for supervising machine work outside the engine. Its job + polling and completion semantics are not an ACP implementation. + +## Proposed design + +### 1. Separate sessions, harnesses, and processes + +Make the session backend explicit. The following is conceptual, not a final +Rust or public-wire definition: + +```text +SessionBackend = Lightspeed | ExternalHarness + +ExternalHarness configuration: + harness identity and pinned installation/configuration revision + driver kind (initially ACP) + environment identity + working directory + credential binding references + admitted execution policy + +ExternalHarness runtime binding: + harness instance ID + external session ID + journal generation and committed ingestion cursor + negotiated protocol version and capabilities +``` + +Model three distinct resources: + +| Resource | Meaning | +| --- | --- | +| Harness installation | Executable/adapter, version, and launch configuration available on an environment | +| Harness instance | Running endpoint, connection, and owned process tree | +| Harness session | Logical conversation addressed through that endpoint | + +The session backend is fixed at creation initially. Changing harnesses or +drivers for an existing conversation requires a deliberate compatibility or +transfer operation; a shared transcript is not transferable native context. +External configuration must not require a fictitious Lightspeed model route +or apply Lightspeed's native model default to an unrelated harness. + +ACP permits multiple sessions on a connection. Process allocation is an +implementation detail: an adapter may multiplex in one runtime or spawn a +process per session. [ACP architecture](https://agentclientprotocol.com/get-started/architecture) +and [session setup](https://agentclientprotocol.com/protocol/v1/session-setup) +describe the distinction between a connection and a conversation. + +Initially allocate one ACP instance per active Lightspeed session. An instance +may contain an adapter and several child processes. Keep the schema capable +of sharing an instance later, but do not introduce pooling before testing +concurrency, cancellation, authentication scope, and failure isolation. +Separate processes do not isolate shared machine files or credentials. + +### 2. envd owns the local bridge + +Extend `environment-protocol` with an optional harness-supervision capability +and a distinct method family. Reuse existing environment connections, +authentication, and gateway routing; do not introduce a second +Lightspeed-to-envd protocol or service. ACP is the envd-to-harness boundary. +Daemons that only support files and processes need not implement this capability. + +The initial operation families cover instance start/stop, conversation +open/restore/close, input submission, interaction responses, cancellation, +repeatable event reads, and acknowledgements. Use stable operation identities +and typed supervision facts, with versioned driver-specific observation +payloads where necessary. Avoid an unrestricted JSON-RPC forwarding method. +Settle exact method names and payloads in the environment-only validation slice. + +Add a harness supervisor to envd, with ACP as its first driver. It launches +the configured executable, maintains the connection, manages the process tree, +and journals observations. Reuse process-management primitives where suitable, +but provide persistent stdin/stdout pipes and separate stderr; do not carry +ACP through a PTY or a lossy terminal-output buffer. + +The driver performs initialization, capability negotiation, session creation +or restoration, prompt delivery, and response correlation. It continuously +reads ACP notifications and requests as they arrive. envd does not poll the +harness for generated text. This follows +[ACP's bidirectional stdio transport](https://agentclientprotocol.com/protocol/v1/transports). + +envd remains an environment service: no Lightspeed database access, Temporal +dependency, bot routing, or native engine dependency. The environment protocol +defines the supervision requests and observations. The Incus provider continues +to depend only on that protocol boundary. + +Launch definitions and credential bindings come from admitted configuration. +Do not let model-authored arguments choose arbitrary harness executables or +secret sources. Resolve credentials outside durable session state and keep +them out of protocol transcripts and process diagnostics. + +#### Module placement and Rust SDK + +Add no new workspace crates for this work. Keep responsibilities in focused +modules within the crates that already own each boundary: + +| Existing crate | Responsibility | +| --- | --- | +| `environment-protocol` | Harness capability, request/response types, observation envelopes, and typed errors | +| `environment-client` | Typed calls over existing connections and an executable validation example | +| `environment-daemon` | Harness supervisor, ACP driver, local journal, and process cleanup | +| `engine` | Shared deterministic session/log types in existing session/storage modules, separate from the native core reducer | +| `temporal-workflow` | External-session state machine and workflow, added after bridge validation | +| `temporal-server` | Runtime activities, admission, and backend dispatch, added after bridge validation | +| `store-pg` / `store-fs` | Existing session-log persistence and backend-aware projections | +| `api` / `api-projection` | Public session capabilities and views for both backends | + +Use the official +[`agent-client-protocol` Rust SDK](https://docs.rs/agent-client-protocol/2.2.0/agent_client_protocol/) +inside the daemon's ACP driver. Its client, typed handlers, session helpers, +and byte-stream transports supply the protocol machinery. Version 2.2.0 was +reviewed for this proposal; pin the chosen compatible release during the trial. +The SDK's version is distinct from the wire protocol: use stable ACP v1 +initially and leave draft protocol-v2 features disabled. + +Adding this upstream dependency does not introduce a new Lightspeed workspace +crate. Keep SDK types out of `environment-protocol` and public session types. +envd owns supervision, journaling, and cancellation policy even if SDK helpers +are used for transport or launching. Prefer supplying supervisor-owned pipes +when SDK process ownership would conflict with envd cleanup or restart rules. +The validation slice must exercise that lifetime boundary rather than only +demonstrate the SDK's one-shot client example. + +### 3. Long polling connects two event-driven boundaries + +```mermaid +flowchart TD + Harness[External harness] <-->|ACP over stdio| Daemon[envd supervisor and journal] + Daemon <-->|Commands and cursor-based long polling| Activities[Runtime activities] + Activities <--> Workflow[External session workflow] + Activities --> Log[Session event log and CAS] + Log -->|Existing session event long polling| Clients[Web and CLI] +``` + +Use a repeatable read operation conceptually shaped as: + +```text +read_events(instance, external_session, generation, after_seq, wait_ms, limit) +``` + +Return available observations immediately. When caught up, wait until new +observations arrive or the bounded timeout expires. A small coalescing window +may batch streaming output; permission requests and terminal outcomes should +not wait behind a large text buffer. This differs from a periodic poll that +only checks every few seconds, and from existing process reads that collect +output until their wait budget expires. + +The runtime activity reads envd; product clients independently read the committed +Lightspeed log. Product clients do not connect directly to envd. The initial +validation example deliberately exercises envd through `environment-client` +before this runtime path exists. Keep page and byte limits, +timeouts, and cancellation aligned through direct and gateway-routed paths. +The transport must allow cancellation or permission replies while a read is +outstanding; one blocked request must not monopolize the connection. + +Use bounded normal activities for waiting I/O, with heartbeat/cancellation +support where needed. Batch observations and put payloads in CAS so Temporal +history carries references and decision facts. Do not create a workflow signal +or activity result per token. Continue-as-new preserves pending command IDs, +binding identity, and committed cursors. + +### 4. A separate workflow supervises the external session + +Introduce an external-session workflow with a small deterministic state +machine. It admits inputs, queues runs, schedules bridge commands and reads, +handles interactions, records outcomes, and owns cleanup. It does not invoke +the native engine's model/tool planning loop or reconstruct native context. + +Initially allow one active foreground run per external session and queue later +inputs. Steering and background work are explicit capabilities, not assumed +equivalents across harnesses. A protocol update may belong to the session +without belonging to the current run. Map completion from documented driver +semantics and preserve the original stop reason alongside the public outcome. + +Use the existing runtime binary and sessions role initially. Other controllers +communicate through workflow starts and signals. For delegation from a native +Lightspeed agent, use the generic workflow-tool protocol to start or address +an external-session controller and return its result. ACP transport must not +become a feature-specific branch in the stable native session worker. + +### 5. Reuse the session log and extract shared session concepts + +Separate shared session identity, lifecycle, storage contracts, and public +facts from native core execution within the existing session/storage modules +in `engine`. Do not create a new session-domain crate. The external workflow +may reuse those deterministic types without driving the native core reducer +or requiring a populated `CoreAgentState`; it has its own reducer and +checkpoint format. envd remains independent of `engine` and session storage. + +Retain one ordered session log with backend-aware event families: + +| Family | Examples | +| --- | --- | +| Shared lifecycle | Session created/closed, backend binding established | +| Shared work | Input admitted, run started, completed, cancelled, interrupted | +| Shared interaction | Permission requested, decision recorded, interaction expired | +| Native execution | Generation, context, and native tool-execution facts | +| External execution | Imported update batch, command delivery state, connection/recovery facts | + +The lifecycle and run facts have one authoritative representation per backend; +do not emit independently mutable duplicate histories. Both backends project +into common public message, tool, interaction, and run views. + +Retain protocol-native external update batches in CAS with driver/schema +version, source identity, and sequence range. Use typed, bounded control facts +for workflow decisions and derive detailed transcript views from retained +payloads. Preserve unfamiliar updates for later inspection without pretending +they have known control semantics. Register content roots and nested retention +edges explicitly, including media referenced by external payloads. + +An external tool update reports work owned by the harness; it never schedules +the corresponding Lightspeed native tool. Likewise, a permission response +authorizes the pending external request rather than resolving a native MCP +continuation. Generalize interaction subjects and decision options instead of +forcing ACP requests into the current MCP-only approval representation. + +Keep one logical workflow owner of appends per session. Ingestion, input +admission, and interaction decisions converge through that owner and the +existing expected-head checks. Catalog activity, current-state reads, run +outputs, retention, and checkpoints must all recognize the selected backend. + +Unsupported operations return typed capability errors. In particular, native +context edits, compaction, fork/clone, provider changes, and tool configuration +cannot silently acquire invented external semantics. Expose capabilities to +clients so unavailable controls can be explained or omitted. + +### 6. Journal delivery and command recovery are separate guarantees + +envd keeps a local persisted delivery journal. Lightspeed's session log is the +authoritative retained product history. Import follows this order: + +1. envd journals an observation with a stable source sequence number. +2. Lightspeed reads after its committed source cursor. +3. A store operation atomically appends the imported observations and their + cursor advancement, deduplicating retries by source identity and range. +4. Only after commit does Lightspeed acknowledge the cursor to envd. +5. envd may prune acknowledged entries under its retention policy. + +Use an instance/journal generation to distinguish sequence spaces. On a lost +append response, recover the committed import before retrying or acknowledging. +A repeated read must not advance a destructive server-side cursor. Lost or +expired ranges produce a typed gap; never silently truncate protocol history. +Bound disk use and define backpressure or explicit failure when unacknowledged +data cannot be retained. A machine lost before import can still lose its +unreplicated observations; local journaling does not provide replication. + +Commands carry stable operation IDs and immutable input fingerprints. Record +admission before dispatch and recover the existing operation on retry. An ACP +JSON-RPC request ID only correlates messages; it is not an exactly-once promise. +There remains a crash window between sending a prompt and recording evidence +of its execution. Inspect or restore existing work where supported; otherwise +record an uncertain/interrupted outcome instead of automatically repeating it. + +Distinguish connection loss, daemon restart, harness exit, environment loss, +and session-history restoration. Restoring a conversation does not prove an +interrupted command resumed. Replayed harness history must be reconciled as +history, not appended as duplicate new output. Missing message identities or +ambiguous replay require a defined driver policy and explicit diagnostics. + +Cancellation records intent, sends the driver's cancellation operation, and +observes the result. Apply a declared escalation policy for owned processes +when graceful cancellation fails. Late success cannot revive a terminal run; +later cleanup observations remain recordable. Pending permission replies are +correlated to the instance generation and request, so a stale decision cannot +authorize a replacement process's unrelated request. + +### 7. Preserve environment and resource boundaries + +Bind execution to a concrete environment; changing the native session's +active-machine selection is not migration of an external harness. Advertise +ACP filesystem or terminal callbacks only when implemented against that +environment and its admitted access policy. The harness may also execute +directly on the machine, so callbacks alone are not a sandbox boundary. + +VFS attachments do not become machine files automatically. Transfers remain +explicit. Working directory and persistent harness state must survive for +restoration to work; retain their locations separately from the transcript. +Account for harness-owned work in environment idle/power policy and define +session close, inactivity, and environment deletion cleanup explicitly. + +Commands and observations use the existing environment authorization and +namespace boundary. The bridge can operate without a Lightspeed session row, +database, or workflow. When integrated into the runtime, bind bridge resources +to the admitted universe/environment/session relationship. Product-client +access continues through existing session visibility and runtime authorization. +envd receives the authority it needs to supervise the instance, not general +access to session records. + +## Delivery plan and validation + +- [x] Review current storage, projections, process transport, and upstream + integration choices; record this proposal. +- [ ] Extend `environment-protocol` and `environment-client` with the minimum + coherent harness capability and operations; add envd modules for ACP + supervision, journal reads/acknowledgements, command identity, and cleanup. +- [ ] Drive one pinned real harness end to end through `environment-client` + and envd using a small executable example in an existing crate. Validate + the environment-protocol boundary before wiring it into Lightspeed sessions. +- [ ] Extend that same validation path to pinned Codex, Claude, OpenCode, and + Pi combinations. Record capabilities, process topology, headless + authentication, limitations, and an ACP/native-driver decision per target. + Fix protocol or daemon gaps exposed by these trials before broader wiring. +- [ ] Extract shared session contracts and backend dispatch. Define native + event compatibility or an explicit greenfield reset/migration; do not + silently reinterpret existing stored events or checkpoints. +- [ ] Implement the external workflow, atomic ingestion, interaction handling, + cancellation, recovery, and generic workflow-tool integration. +- [ ] Extend public API projections and web/CLI controls, then publish a tested + capability matrix and operational guidance. + +### First implementation milestone: drive a harness through envd + +The first milestone is a working environment-only path: + +```text +environment-client example + -> existing environment transport and new harness methods + -> envd harness supervisor + -> ACP driver + -> real harness +``` + +Run this without Temporal, PostgreSQL, a Lightspeed session, Platform, or a +new public session API. The example uses the same typed environment calls +that later runtime activities will use; it must not talk ACP directly or +invoke private daemon methods. Harness credentials are supplied through the +chosen test environment's explicit setup. + +It must open a conversation, submit a task, receive incremental observations +through long polling, answer an interaction when requested, observe a terminal +outcome, and submit a second task to the same conversation. It must also +exercise cancellation while a read is pending, reconnect and reread from a +cursor without duplicate command execution, acknowledge consumed journal +entries, and close the instance with observable cleanup. Capture capability +limits and distinguish live reconnection from restoration after process loss. + +Use a scripted peer for deterministic failure and permission paths that the +selected real harness cannot reliably trigger. A small local cursor sink in +the example can exercise acknowledgement ordering; it does not validate the +later atomic session-store ingestion or become a second production store. + +Exit this milestone with reproducible instructions, pinned versions, captured +outcomes, and resolved transport/lifetime issues. Follow with the remaining +harness trials on the same environment-protocol path. Session model changes, +Temporal integration, and client UI work come after this validation, so a +failed integration assumption is corrected at the smallest boundary. + +### Compatibility and regression coverage + +The compatibility trial covers startup/authentication, a streamed task with +tool activity, permission grant/denial where supported, a second prompt, +cancellation during work, two concurrent conversations, disconnect/reconnect, +and restoration after process loss. Distinguish unsupported behavior from an +adapter defect. A native driver is justified by a required capability or +reliability gap; it is not an automatic parallel deliverable. + +Offline tests use scripted ACP peers and temporary daemon state. Cover partial +frames, stderr separation, interleaved updates and requests, repeated pages, +duplicate commands, uncertain dispatch, append-response loss, commit-before- +ack crashes, journal gaps, stale generations, cancellation races, pending +permissions during reads, and explicit teardown. Replay tests verify external +state and transcript projections without rerunning the harness. Preserve +native session lifecycle, transcript, retention, and checkpoint coverage. + +Live and credentialed trials are opt-in and must not silently skip missing +prerequisites. Follow the repository's serialized Temporal live-test rules. +No such trials have run as part of this document. + +Regenerate public API consumers and the workflow contract when their types +change. If schema migrations are added, align the required schema revision +and release metadata. Runtime and user documentation updates accompany the +implemented behavior after user review; this proposal does not claim it ships. + +## Questions to settle during the trial + +- Which required capabilities fail through ACP for the pinned target versions, + and does a native interface actually close those gaps? +- What journal limits, flush/coalescing intervals, and idle instance policy + give acceptable latency without excessive workflow history or machine cost? +- Which harness state directories and credential arrangements permit reliable + headless restoration within an environment? +- How should uncertain external execution appear in shared run status and + recovery controls without implying that a retry is safe? +- Which configuration and interaction options can use shared public types, + and which need explicitly driver-specific extensions? + +## Out of scope for the first delivery + +Harness pooling, automatic cross-harness context migration, automatic replay +of ambiguous prompts, universal feature parity, A2A federation, an inbound +editor ACP server, and AHP client/server compatibility. These can build on the +same session boundary without changing the initial protocol decision into a +requirement for every future backend. diff --git a/package-lock.json b/package-lock.json index 91350e537..4e325df0e 100644 --- a/package-lock.json +++ b/package-lock.json @@ -7350,6 +7350,19 @@ "node": ">=10.12.0" } }, + "node_modules/automation-events": { + "version": "7.1.19", + "resolved": "https://registry.npmjs.org/automation-events/-/automation-events-7.1.19.tgz", + "integrity": "sha512-cD+TLhJTI0q4AI3ktd353lrGZiVa9AchowSDzQzzGjSoYe22js4vlS32VUtWuaulghi1Yq0KYNWKk9wWuGymPA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.2", + "tslib": "^2.8.1" + }, + "engines": { + "node": ">=18.2.0" + } + }, "node_modules/axobject-query": { "version": "4.1.0", "resolved": "https://registry.npmjs.org/axobject-query/-/axobject-query-4.1.0.tgz", @@ -7640,6 +7653,18 @@ "node": ">=8" } }, + "node_modules/broker-factory": { + "version": "3.1.15", + "resolved": "https://registry.npmjs.org/broker-factory/-/broker-factory-3.1.15.tgz", + "integrity": "sha512-ko+aWvgNuP49meGrdjUu7rC+Y+Wai3cCPxP3xWwHsHfehFjOh5ZQM2yC4gEB2UddeZ/YXhm0K1eG/L6fxym2Og==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "fast-unique-numbers": "^9.0.27", + "tslib": "^2.8.1", + "worker-factory": "^7.0.50" + } + }, "node_modules/browserslist": { "version": "4.28.8", "funding": [ @@ -10331,6 +10356,44 @@ "version": "3.0.2", "license": "MIT" }, + "node_modules/extendable-media-recorder": { + "version": "9.2.40", + "resolved": "https://registry.npmjs.org/extendable-media-recorder/-/extendable-media-recorder-9.2.40.tgz", + "integrity": "sha512-7jU3W9vDzcb8V9wwy9TsZxq37NrhH8Ru0K5EfY6AGFmeohBYXsgmeDqK13F3tAGGf3uCYo48jfK4MdXTIvEwxA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "media-encoder-host": "^9.0.31", + "multi-buffer-data-view": "^6.0.27", + "recorder-audio-worklet": "^6.0.60", + "standardized-audio-context": "^25.3.77", + "subscribable-things": "^2.1.60", + "tslib": "^2.8.1" + } + }, + "node_modules/extendable-media-recorder-wav-encoder-broker": { + "version": "7.0.128", + "resolved": "https://registry.npmjs.org/extendable-media-recorder-wav-encoder-broker/-/extendable-media-recorder-wav-encoder-broker-7.0.128.tgz", + "integrity": "sha512-4pP+IBbU8n9aP6teDvewleI+QtYCQPaRiSDW12PuPkhjxc89QZkM56w1zXhm0M4niJiqzh8SoAs/ndlWVuQo9A==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "broker-factory": "^3.1.15", + "extendable-media-recorder-wav-encoder-worker": "^8.0.123", + "tslib": "^2.8.1" + } + }, + "node_modules/extendable-media-recorder-wav-encoder-worker": { + "version": "8.0.123", + "resolved": "https://registry.npmjs.org/extendable-media-recorder-wav-encoder-worker/-/extendable-media-recorder-wav-encoder-worker-8.0.123.tgz", + "integrity": "sha512-WRKybAbfSICvwCzkfvWIaHBClf5G9njUq1jyX8pFonzSv3koo5FqwKHv7e5jyGsd7mpDLAtOSZcPZOkM10SaRQ==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "tslib": "^2.8.1", + "worker-factory": "^7.0.50" + } + }, "node_modules/fast-deep-equal": { "version": "3.1.3", "license": "MIT" @@ -10364,6 +10427,19 @@ "fast-string-truncated-width": "^3.0.2" } }, + "node_modules/fast-unique-numbers": { + "version": "9.0.27", + "resolved": "https://registry.npmjs.org/fast-unique-numbers/-/fast-unique-numbers-9.0.27.tgz", + "integrity": "sha512-nDA9ADeINN8SA2u2wCtU+siWFTTDqQR37XvgPIDDmboWQeExz7X0mImxuaN+kJddliIqy2FpVRmnvRZ+j8i1/A==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.2", + "tslib": "^2.8.1" + }, + "engines": { + "node": ">=18.2.0" + } + }, "node_modules/fast-uri": { "version": "3.1.5", "funding": [ @@ -12597,6 +12673,43 @@ "integrity": "sha512-9Yubnt3e8A0OKwxYSXyhLymGW4sCufcLG6VdiDdUGVkPhpqLxlvP5vl1983gQjJl3tqbrM731mjaZaP68AgosQ==", "license": "CC0-1.0" }, + "node_modules/media-encoder-host": { + "version": "9.0.31", + "resolved": "https://registry.npmjs.org/media-encoder-host/-/media-encoder-host-9.0.31.tgz", + "integrity": "sha512-PTgBe18JxIAntMnxiiPU8bnWZi75ZVHp+ed9Zmm+ogUmtbEbCNkuuDu5T86FlEa2ekiQ0G7Lx0F7xA9EB+FKCA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "media-encoder-host-broker": "^8.0.29", + "media-encoder-host-worker": "^10.0.29", + "tslib": "^2.8.1" + } + }, + "node_modules/media-encoder-host-broker": { + "version": "8.0.29", + "resolved": "https://registry.npmjs.org/media-encoder-host-broker/-/media-encoder-host-broker-8.0.29.tgz", + "integrity": "sha512-ATJ1GRUPwoIiBeRXSBvfF96CIRBYjHKaRTg861CrcS+6pALgMMjTvn/XDIalkqjY2pXCnAWiXn5szRWDEFOflg==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "broker-factory": "^3.1.15", + "fast-unique-numbers": "^9.0.27", + "media-encoder-host-worker": "^10.0.29", + "tslib": "^2.8.1" + } + }, + "node_modules/media-encoder-host-worker": { + "version": "10.0.29", + "resolved": "https://registry.npmjs.org/media-encoder-host-worker/-/media-encoder-host-worker-10.0.29.tgz", + "integrity": "sha512-AT1X6mIp6W++OxMimGMm622GFkAARzny82T1k8jxZWK0rIZhVoro9FaqIx+xERdGSMPZioN4qj0MKAgY04XENA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "extendable-media-recorder-wav-encoder-broker": "^7.0.128", + "tslib": "^2.8.1", + "worker-factory": "^7.0.50" + } + }, "node_modules/media-typer": { "version": "2.0.0", "license": "MIT", @@ -13536,6 +13649,19 @@ "dev": true, "license": "MIT" }, + "node_modules/multi-buffer-data-view": { + "version": "6.0.27", + "resolved": "https://registry.npmjs.org/multi-buffer-data-view/-/multi-buffer-data-view-6.0.27.tgz", + "integrity": "sha512-hqbVIRFskEUX3PmHxGPGFOcLU93va4yfYkBBg4c1U0wK1Z/pDfIbYDexbIIWI+vtrXWkQAStCQ4drcQ1ooakiA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.2", + "tslib": "^2.8.1" + }, + "engines": { + "node": ">=18.2.0" + } + }, "node_modules/music-metadata": { "version": "11.14.0", "funding": [ @@ -14865,6 +14991,32 @@ "url": "https://opencollective.com/unified" } }, + "node_modules/recorder-audio-worklet": { + "version": "6.0.60", + "resolved": "https://registry.npmjs.org/recorder-audio-worklet/-/recorder-audio-worklet-6.0.60.tgz", + "integrity": "sha512-ZIDm5URZxsPH4bqr2BgGkSC5QLqtAKMSvovM9KeDsI2pCvZVpy+PSWZFrlpgJ/QeuAF/f9cbHX/s9GjFxm7Pjw==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "broker-factory": "^3.1.15", + "fast-unique-numbers": "^9.0.27", + "recorder-audio-worklet-processor": "^5.0.41", + "standardized-audio-context": "^25.3.77", + "subscribable-things": "^2.1.60", + "tslib": "^2.8.1", + "worker-factory": "^7.0.50" + } + }, + "node_modules/recorder-audio-worklet-processor": { + "version": "5.0.41", + "resolved": "https://registry.npmjs.org/recorder-audio-worklet-processor/-/recorder-audio-worklet-processor-5.0.41.tgz", + "integrity": "sha512-TXaP51ZloO7mDyi71pF6BVU1MenHTMbEOFCbY4KfQGBjOsPuvvvhJzj25b/9RjchaF857x6MGg2CT8Ai8XbtXg==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "tslib": "^2.8.1" + } + }, "node_modules/regex": { "version": "6.1.0", "resolved": "https://registry.npmjs.org/regex/-/regex-6.1.0.tgz", @@ -15350,6 +15502,12 @@ "tslib": "^2.1.0" } }, + "node_modules/rxjs-interop": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/rxjs-interop/-/rxjs-interop-2.0.0.tgz", + "integrity": "sha512-ASEq9atUw7lualXB+knvgtvwkCEvGWV2gDD/8qnASzBkzEARZck9JAyxmY8OS6Nc1pCPEgDTKNcx+YqqYfzArw==", + "license": "MIT" + }, "node_modules/safe-stable-stringify": { "version": "2.5.0", "license": "MIT", @@ -15867,6 +16025,17 @@ "devOptional": true, "license": "MIT" }, + "node_modules/standardized-audio-context": { + "version": "25.3.77", + "resolved": "https://registry.npmjs.org/standardized-audio-context/-/standardized-audio-context-25.3.77.tgz", + "integrity": "sha512-Ki9zNz6pKcC5Pi+QPjPyVsD9GwJIJWgryji0XL9cAJXMGyn+dPOf6Qik1AHei0+UNVcc4BOCa0hWLBzlwqsW/A==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.25.6", + "automation-events": "^7.0.9", + "tslib": "^2.7.0" + } + }, "node_modules/statuses": { "version": "2.0.2", "license": "MIT", @@ -16047,6 +16216,17 @@ "integrity": "sha512-5Z9ZpRzfuH6l/UAvCPAPUo3665Nk2wLaZU3x+TLHKVzIz33+sbJqbtrYoC3KD4/uVOr2Zp+L0LySezP9OHV9yA==", "license": "MIT" }, + "node_modules/subscribable-things": { + "version": "2.1.60", + "resolved": "https://registry.npmjs.org/subscribable-things/-/subscribable-things-2.1.60.tgz", + "integrity": "sha512-6gamStlTGrZAHmIMKUxnrZGW1y0b2N3ofQMX/ZrWdYCcNytk/wRyiRB8eGE4Jg+bYFLypoQAb7UpmWjqpiWCKA==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "rxjs-interop": "^2.0.0", + "tslib": "^2.8.1" + } + }, "node_modules/supports-color": { "version": "8.1.1", "license": "MIT", @@ -17665,6 +17845,17 @@ "version": "0.2.1", "license": "MIT" }, + "node_modules/worker-factory": { + "version": "7.0.50", + "resolved": "https://registry.npmjs.org/worker-factory/-/worker-factory-7.0.50.tgz", + "integrity": "sha512-hhwc0G+sFwM4qBuhJIUBn2p1Jf8v/FwmLUANBf/Q+Lt2uI8mfIZQhXaZQACodQD4R7Zp6cn/6702bIvNn2puJQ==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.29.7", + "fast-unique-numbers": "^9.0.27", + "tslib": "^2.8.1" + } + }, "node_modules/wrap-ansi": { "version": "7.0.0", "license": "MIT", @@ -18103,6 +18294,7 @@ "better-auth": "1.7.6", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", + "extendable-media-recorder": "^9.2.40", "hono": "^4.7.0", "lucide-react": "^1.23.0", "react": "^19.1.0", diff --git a/platform/README.md b/platform/README.md index d27af8cbc..810ddfa3c 100644 --- a/platform/README.md +++ b/platform/README.md @@ -179,8 +179,9 @@ under **API keys**, choosing the method groups each key may call from presets actors, and credential leasing and channel delivery are opt-in. Platform admins see and mint every key under **Platform admin → API keys**, choosing what a key reaches (the deployment or one universe), the method groups it may call, and whether it -speaks for people. A secret is shown once; keys never change, so revoke and mint -instead. The Configurator template asks which key its MCP server acts with: +speaks for people. A secret is shown once. Rotate it to replace it immediately +while preserving authority; revoke and mint a replacement to change permissions. +The Configurator template asks which key its MCP server acts with: the current one, a new key with chosen groups (starting from the configuration groups), or an existing universe key whose secret the Admin pastes; it revokes only keys it minted. The server lists only the tools that key may call, and diff --git a/platform/configurator-mcp/src/generated/tools.ts b/platform/configurator-mcp/src/generated/tools.ts index 91f5b5063..323a5d682 100644 --- a/platform/configurator-mcp/src/generated/tools.ts +++ b/platform/configurator-mcp/src/generated/tools.ts @@ -772,7 +772,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -1232,7 +1232,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "method": "session/config/put", "group": "session", "summary": "Replace session configuration", - "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked and an identical document is a no-op.", + "description": "Replaces the complete sparse config while the session is idle. Use the current config revision for safe read-modify-write; omitted features are revoked, an omitted model preserves the current model, and an identical document is a no-op.", "paramsType": "SessionConfigPutParams", "resultType": "AgentApiOutcome", "inputSchema": { @@ -1756,7 +1756,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -2354,7 +2354,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "method": "session/context/append", "group": "session", "summary": "Append keyed session context", - "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; media preprocessing can fail one entry without discarding successful entries.", + "description": "Admits a batch of context entries with per-entry results. Stable keys make same-content retries no-ops; invalid input can fail one entry without discarding successful entries.", "paramsType": "ContextAppendParams", "resultType": "AgentApiOutcome", "inputSchema": { @@ -2403,6 +2403,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -2429,6 +2436,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -2678,6 +2692,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -2704,6 +2725,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -3158,6 +3186,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "text": { "type": "string" }, @@ -3184,6 +3219,13 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "null" ] }, + "provenanceRef": { + "description": "Optional source blob in this universe, retained with the session.\nProvenance is metadata, not model input or an authorization identity.", + "type": [ + "string", + "null" + ] + }, "type": { "const": "textRef", "type": "string" @@ -3971,7 +4013,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -5033,6 +5075,237 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "object" } }, + { + "name": "lightspeed_transcriptions_start", + "method": "transcriptions/start", + "group": "transcriptions", + "summary": "Start transcription", + "description": "Admit or rejoin a requester-scoped audio transcription. The resolved model is immutable. No session or run is created.", + "paramsType": "TranscriptionStartParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "properties": { + "audio": { + "$ref": "#/definitions/TranscriptionAudio" + }, + "idempotencyKey": { + "description": "Scoped to the requester. Matching retries rejoin the original job,\nincluding after defaults change; changed requests conflict. Identity is\nretained for the Temporal namespace's workflow-history retention period.", + "type": "string" + }, + "language": { + "type": [ + "string", + "null" + ] + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "idempotencyKey", + "audio" + ], + "type": "object", + "definitions": { + "ModelConfig": { + "properties": { + "apiKind": { + "type": "string" + }, + "model": { + "type": "string" + }, + "providerId": { + "type": "string" + } + }, + "required": [ + "providerId", + "apiKind", + "model" + ], + "type": "object" + }, + "TranscriptionAudio": { + "additionalProperties": { + "not": {} + }, + "description": "Immutable audio input in this universe's content store.", + "properties": { + "blobRef": { + "type": "string" + }, + "mime": { + "type": "string" + }, + "name": { + "type": "string" + } + }, + "required": [ + "blobRef", + "mime", + "name" + ], + "type": "object" + } + } + } + }, + { + "name": "lightspeed_transcriptions_read", + "method": "transcriptions/read", + "group": "transcriptions", + "summary": "Read transcription", + "description": "Read status and transcript text. An asserted actor may only read their own drafts; direct universe keys retain method-group authority. Unsubmitted CAS results may expire.", + "paramsType": "TranscriptionReadParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + } + }, + { + "name": "lightspeed_transcriptions_cancel", + "method": "transcriptions/cancel", + "group": "transcriptions", + "summary": "Cancel transcription", + "description": "Cancel unfinished transcription. Repeated cancellation is safe; completed results remain unchanged. An asserted actor may only cancel their own drafts; direct universe keys retain method-group authority.", + "paramsType": "TranscriptionCancelParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "properties": { + "transcriptionId": { + "type": "string" + } + }, + "required": [ + "transcriptionId" + ], + "type": "object" + } + }, + { + "name": "lightspeed_models_defaults_read", + "method": "models/defaults/read", + "group": "models", + "summary": "Read universe model defaults", + "description": "Returns the revision and independent agentRun and speechToText selections. Revision zero means no update has been made. Does not contact model providers.", + "paramsType": "ModelDefaultsReadParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "type": "object" + } + }, + { + "name": "lightspeed_models_defaults_put", + "method": "models/defaults/put", + "group": "models", + "summary": "Set a universe model default", + "description": "Sets or explicitly clears one purpose slot using its current expected revision. Existing sessions and admitted work keep their model. Validates the purpose and protocol without contacting provider discovery.", + "paramsType": "ModelDefaultsPutParams", + "resultType": "AgentApiOutcome", + "inputSchema": { + "$schema": "http://json-schema.org/draft-07/schema#", + "additionalProperties": { + "not": {} + }, + "properties": { + "expectedRevision": { + "description": "Revision returned by read/put; zero for a universe with no updates.", + "format": "uint64", + "minimum": 0, + "type": "integer" + }, + "model": { + "anyOf": [ + { + "$ref": "#/definitions/ModelConfig" + }, + { + "type": "null" + } + ], + "description": "Complete selection, or explicit null to clear this slot. Required." + }, + "slot": { + "$ref": "#/definitions/ModelDefaultSlot" + } + }, + "required": [ + "slot", + "model", + "expectedRevision" + ], + "type": "object", + "definitions": { + "ModelConfig": { + "properties": { + "apiKind": { + "type": "string" + }, + "model": { + "type": "string" + }, + "providerId": { + "type": "string" + } + }, + "required": [ + "providerId", + "apiKind", + "model" + ], + "type": "object" + }, + "ModelDefaultSlot": { + "description": "A universe's model selection for a particular use. Protocol and purpose\nare separate: several purposes may use the same provider API.", + "enum": [ + "agentRun", + "speechToText" + ], + "type": "string" + } + } + } + }, { "name": "lightspeed_models_list", "method": "models/list", @@ -5046,7 +5319,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "description": "Direct provider model discovery. Results may be served from a brief\nprocess-local cache; clients refresh by calling this method.", "properties": { "selectableOnly": { - "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nremoves OpenAI model-id families that are clearly not text-generation\nroutes (embeddings, moderation, image/video, speech, and realtime).\nIt is an ID policy, not a provider capability claim.", + "description": "Apply Lightspeed's small, conservative selectable-model policy. It\nkeeps supported file-transcription routes and filters clearly unrelated\nOpenAI families from agent suggestions (embeddings, moderation, image/video,\nspeech synthesis, and realtime). Agent suggestions also have an age limit.\nClients select routes by API kind for their intended use. This is an ID\npolicy, not a provider capability claim.", "type": "boolean" } }, @@ -5685,7 +5958,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" @@ -6703,7 +6976,7 @@ export const GENERATED_TOOLS: readonly GeneratedToolDescriptor[] = [ "type": "null" } ], - "description": "Absent on input means the deployment default model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." + "description": "At creation, omission uses the profile model or universe agentRun\ndefault. On configuration replacement or profile application to an\nexisting session, omission preserves its current model. Documents read\nback from a session always carry the model. Provider identity and API\nkind are fixed for the session lifetime; the model name may change." } }, "type": "object" diff --git a/platform/server/src/auth.integration.test.ts b/platform/server/src/auth.integration.test.ts index 5e5f92e4c..610f058d8 100644 --- a/platform/server/src/auth.integration.test.ts +++ b/platform/server/src/auth.integration.test.ts @@ -85,6 +85,23 @@ async function companyUser() { return (await fixture.db.select().from(schema.user).where(eq(schema.user.identitySource, "company")))[0]!; } +it.each([ + ["enabled", identityEnv.oidc, "emergency.sign_in"], + ["disabled", null, "password.sign_in"], +] as const)("audits bootstrap admin password sign-in with SSO %s", async (_name, oidc, action) => { + const auth = createAuth(fixture.db, { ...identityEnv, oidc }); + const response = await auth.handler(new Request(`${identityEnv.baseUrl}/api/auth/sign-in/email`, { + method: "POST", headers: { "content-type": "application/json", origin: identityEnv.baseUrl }, + body: JSON.stringify({ email: identityEnv.adminEmail, password: identityEnv.adminPassword }), + })); + expect(response.status).toBe(200); + const admin = (await fixture.db.select().from(schema.user).where(eq(schema.user.email, identityEnv.adminEmail!)))[0]!; + expect(admin.emergencyAdmin).toBe(true); + expect(await fixture.db.select().from(schema.identityAudit)).toEqual([ + expect.objectContaining({ actorId: admin.id, targetId: admin.id, action, outcome: "success" }), + ]); +}); + it.each([ [["lightspeed-users"], "user"], [["lightspeed-admins"], "admin"], [["lightspeed-users", "lightspeed-admins"], "admin"], ])("admits %j with local platform role %s and no universe membership", async (groups, role) => { diff --git a/platform/server/src/auth.ts b/platform/server/src/auth.ts index da09cb1ca..19671168a 100644 --- a/platform/server/src/auth.ts +++ b/platform/server/src/auth.ts @@ -84,8 +84,8 @@ export function createAuth(db: Db, env: ServerEnv) { }, after: async (session) => { const [user] = await db.select().from(schema.user).where(eq(schema.user.id, session.userId)); - if (user?.emergencyAdmin && requests.getStore()?.identity?.source === "password") { - await auditIdentity(db, { actorId: user.id, action: "emergency.sign_in", targetId: user.id }); + if (user && requests.getStore()?.identity?.source === "password") { + await auditIdentity(db, { actorId: user.id, action: oidc && user.emergencyAdmin ? "emergency.sign_in" : "password.sign_in", targetId: user.id }); } }, } }, diff --git a/platform/server/src/routes/api-keys-admin.test.ts b/platform/server/src/routes/api-keys-admin.test.ts index 2f8e57658..087d5f17f 100644 --- a/platform/server/src/routes/api-keys-admin.test.ts +++ b/platform/server/src/routes/api-keys-admin.test.ts @@ -16,13 +16,14 @@ function setup(platformRole: string | undefined) { const rpc = JSON.parse(String(init.body)); calls.push({ method: rpc.method, params: rpc.params, headers: new Headers(init.headers) }); const apiKey = { keyPrefix: "lsk_abcdefgh" }; - const result = rpc.method === "deployment/api-keys/create" ? { apiKey, secret: "lsk_secret" } + const result = ["deployment/api-keys/create", "deployment/api-keys/rotate"].includes(rpc.method) ? { apiKey, secret: "lsk_secret" } : rpc.method === "deployment/api-keys/list" ? { apiKeys: [apiKey] } : { apiKey }; return Response.json({ id: rpc.id, result: { result, notifications: [] } }); })); const query = { from: () => query, where: () => query, limit: async () => [universe] }; + const audit = vi.fn(async () => undefined); const ctx = { - db: { insert: () => ({ values: async () => undefined }), select: (fields: unknown) => { expect(fields).toHaveProperty("lightspeedUniverseId", schema.universes.lightspeedUniverseId); return query; } }, + db: { insert: () => ({ values: audit }), select: (fields: unknown) => { expect(fields).toHaveProperty("lightspeedUniverseId", schema.universes.lightspeedUniverseId); return query; } }, env: { lightspeedApiUrl: "https://core.example/rpc", lightspeedApiKey: "lsk_platform" }, } as unknown as AppContext; const app = new Hono<{ Variables: ApiVariables }>(); @@ -34,7 +35,7 @@ function setup(platformRole: string | undefined) { const request = (method: string, path: string, body?: unknown) => app.request(path, { method, headers: { "content-type": "application/json" }, ...(body ? { body: JSON.stringify(body) } : {}), }); - return { request, calls }; + return { request, calls, audit }; } it("is for platform admins only", async () => { @@ -42,6 +43,7 @@ it("is for platform admins only", async () => { expect((await request("GET", "/api-keys")).status).toBe(403); expect((await request("POST", "/api-keys", { displayName: "x", scope: { kind: "deployment" } })).status).toBe(403); expect((await request("DELETE", "/api-keys/lsk_abcdefgh")).status).toBe(403); + expect((await request("POST", "/api-keys/lsk_abcdefgh/rotate")).status).toBe(403); expect(calls).toEqual([]); }); @@ -71,3 +73,12 @@ it("mints a deployment key that may assert actors, and lists and revokes keys", expect((await request("DELETE", "/api-keys/lsk_abcdefgh")).status).toBe(200); expect(calls.map((call) => call.method)).toEqual(["deployment/api-keys/create", "deployment/api-keys/list", "deployment/api-keys/revoke"]); }); + +it("rotates an admin key and audits only its identifiers", async () => { + const { request, calls, audit } = setup("admin"); + const response = await request("POST", "/api-keys/lsk_previous/rotate"); + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({ apiKey: { keyPrefix: "lsk_abcdefgh" }, secret: "lsk_secret" }); + expect(calls).toEqual([expect.objectContaining({ method: "deployment/api-keys/rotate", params: { keyPrefix: "lsk_previous" } })]); + expect(audit).toHaveBeenCalledWith({ actorId: "admin", action: "key.rotate", targetId: "lsk_previous", details: { newKeyPrefix: "lsk_abcdefgh" }, outcome: "success" }); +}); diff --git a/platform/server/src/routes/api-keys-admin.ts b/platform/server/src/routes/api-keys-admin.ts index 793fe47c0..1e1b887f8 100644 --- a/platform/server/src/routes/api-keys-admin.ts +++ b/platform/server/src/routes/api-keys-admin.ts @@ -22,8 +22,7 @@ const createSchema = z.object({ }); /// Every core key of the deployment, for platform admins: what each reaches -/// and may call, and whether it may assert actors. Keys are immutable; a -/// change is revoke and mint. The secret exists only in the create response. +/// and may call, and whether it may assert actors. Key authority is immutable. Secrets are returned only on create or rotation. export function apiKeyAdminRoutes(ctx: AppContext) { const app = new Hono<{ Variables: ApiVariables }>(); @@ -69,6 +68,13 @@ export function apiKeyAdminRoutes(ctx: AppContext) { }); }); + app.post("/api-keys/:keyPrefix/rotate", (c) => withGateway(c, async () => { + const keyPrefix = c.req.param("keyPrefix"); + const response = await deploymentClientFor(ctx).call("deployment/api-keys/rotate", { keyPrefix }); + await auditIdentity(ctx.db, { actorId: c.get("session").user.id, action: "key.rotate", targetId: keyPrefix, details: { newKeyPrefix: response.result.apiKey.keyPrefix } }); + return c.json(response.result); + })); + app.delete("/api-keys/:keyPrefix", (c) => withGateway(c, async () => { const response = await deploymentClientFor(ctx).call("deployment/api-keys/revoke", { keyPrefix: c.req.param("keyPrefix"), diff --git a/platform/server/src/routes/gateway.ts b/platform/server/src/routes/gateway.ts index c797b8a52..2a6c9c4f7 100644 --- a/platform/server/src/routes/gateway.ts +++ b/platform/server/src/routes/gateway.ts @@ -2,6 +2,7 @@ import { UniverseSlugCacheConflict } from "../universe-slugs.js"; import { deploymentClient, GatewayUnconfigured, GateRefusal, memberClient } from "../runtime-client.js"; import { Hono } from "hono"; import { z } from "zod"; +import { bodyLimit } from "hono/body-limit"; import { LightspeedClient, LightspeedRpcError, @@ -25,7 +26,21 @@ import { type SessionConfig, } from "@lightspeed-ai/agent-client"; import { schema } from "@lightspeed/platform-db"; -import { roleAtLeast, slugify, workspaceCreateSchema } from "@lightspeed/platform-shared"; +import { + MAX_ATTACHMENT_BYTES, + MAX_DICTATION_AUDIO_BYTES, + attachmentUploadSchema, + messageInputItems, + messageRunConfig, + modelDefaultsPutSchema, + roleAtLeast, + sessionMessageSchema, + sessionSteerSchema, + slugify, + transcriptionStartSchema, + transcriptionUploadSchema, + workspaceCreateSchema, +} from "@lightspeed/platform-shared"; import type { AppContext, ApiVariables } from "../context.js"; import { parseBody } from "../http.js"; import { @@ -209,7 +224,7 @@ const modelEndpointSchema = z.object({ baseUrl: z.string().trim().url(), headers: z.record(z.string(), z.string()).optional(), apiKinds: z - .array(z.enum(["openai:responses", "openai:completions"])) + .array(z.enum(["openai:responses", "openai:completions", "openai:audio-transcriptions"])) .min(1) .refine((kinds) => new Set(kinds).size === kinds.length, "API kinds must be unique"), }); @@ -302,15 +317,6 @@ export function environmentSecretGrantParams( /// One text message = one run. `submissionId` is client-minted so a /// retried POST (network flake) returns the original run instead of /// starting a duplicate. -const sessionMessageSchema = z.object({ - text: z.string().min(1).max(100_000), - submissionId: z.string().min(1).max(200), -}); - -const sessionSteerSchema = z.object({ - text: z.string().min(1).max(100_000), -}); - const sessionApprovalDecideSchema = z.object({ decisions: z .array( @@ -773,11 +779,12 @@ export function gatewayRoutes(ctx: AppContext) { }); }); - /// One user message → one run from input items. Returns the accepted - /// run immediately (`running`, or `queued` behind an active run) — - /// replies land in the event log, which the web follows via the long-poll - /// tail. No server-side await: runs can take minutes and an HTTP request - /// must not. + /// One user message → one run from input items: its attachments as media + /// items, then its text. Per-message model options ride the run as + /// overrides. Returns the accepted run immediately (`running`, or `queued` + /// behind an active run) — replies land in the event log, which the web + /// follows via the long-poll tail. No server-side await: runs can take + /// minutes and an HTTP request must not. app.post("/:id/sessions/:sessionId/messages", async (c) => { const access = await universeForSession(ctx, c, c.req.param("id")); if (!access) { @@ -790,17 +797,15 @@ export function gatewayRoutes(ctx: AppContext) { const input = body.data; return withGateway(c, async () => { const client = engineClientFor(ctx, access); + const config = messageRunConfig(input.options); const response = await client.call("session/runs/start", { sessionId: c.req.param("sessionId"), source: { type: "input" as const, - items: [{ - type: "text" as const, - text: input.text, - origin: `user:${c.get("session").user.id}`, - }], + items: messageInputItems(input.text, input.attachments, `user:${c.get("session").user.id}`), }, submissionId: input.submissionId, + ...(config ? { config } : {}), }); const run = response.result.run; return c.json({ run: { id: run.id, status: run.status } }); @@ -826,9 +831,9 @@ export function gatewayRoutes(ctx: AppContext) { }); }); - /// Steer the active run: the text is admitted into the run and reaches - /// the model at its next turn boundary without interrupting the in-flight - /// turn. Rejected for queued, cancelling, or finished runs. + /// Steer the active run: the text and attachments are admitted into the + /// run and reach the model at its next turn boundary without interrupting + /// the in-flight turn. Rejected for queued, cancelling, or finished runs. app.post("/:id/sessions/:sessionId/runs/:runId/steer", async (c) => { const access = await universeForSession(ctx, c, c.req.param("id")); if (!access) { @@ -843,11 +848,7 @@ export function gatewayRoutes(ctx: AppContext) { const response = await client.call("session/runs/steer", { sessionId: c.req.param("sessionId"), runId: c.req.param("runId"), - items: [{ - type: "text" as const, - text: body.data.text, - origin: `user:${c.get("session").user.id}`, - }], + items: messageInputItems(body.data.text, body.data.attachments, `user:${c.get("session").user.id}`), }); const run = response.result.run; return c.json({ @@ -939,7 +940,85 @@ export function gatewayRoutes(ctx: AppContext) { }); }); - /// Provider-discovered model routes for the session-config model picker. + /// Composer attachments upload as content-addressed blobs before the + /// message is sent, so sending is instant; an unsent upload is collected by + /// the ordinary blob grace period. Type and per-API limits are checked at + /// pick time in the browser and again by the runtime on admission. + app.post("/:id/attachments", bodyLimit({ maxSize: 16 * 1024 * 1024 }), async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + const body = await parseBody(c, attachmentUploadSchema); + if (!body.ok) return body.response; + if (Buffer.from(body.data.bytesBase64, "base64").length > MAX_ATTACHMENT_BYTES) { + return c.json({ error: "Attachment exceeds the 10 MiB limit." }, 413); + } + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("blobs/put", { blobs: [body.data] }); + return c.json(response.result); + }); + }); + + app.post("/:id/transcriptions/audio", bodyLimit({ maxSize: 36 * 1024 * 1024 }), async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + const body = await parseBody(c, transcriptionUploadSchema); + if (!body.ok) return body.response; + if (Buffer.from(body.data.bytesBase64, "base64").length > MAX_DICTATION_AUDIO_BYTES) { + return c.json({ error: "Recording exceeds the 25 MiB limit." }, 413); + } + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("blobs/put", { blobs: [body.data] }); + return c.json(response.result); + }); + }); + app.post("/:id/transcriptions", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + const body = await parseBody(c, transcriptionStartSchema); + if (!body.ok) return body.response; + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("transcriptions/start", body.data); + return c.json(response.result.transcription); + }); + }); + app.get("/:id/transcriptions/:transcriptionId", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("transcriptions/read", { transcriptionId: c.req.param("transcriptionId") }); + return c.json(response.result.transcription); + }); + }); + app.post("/:id/transcriptions/:transcriptionId/cancel", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("transcriptions/cancel", { transcriptionId: c.req.param("transcriptionId") }); + return c.json(response.result.transcription); + }); + }); + + app.get("/:id/models/defaults", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("models/defaults/read", {}); + return c.json(response.result.defaults); + }); + }); + + app.put("/:id/models/defaults", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access) return c.json({ error: "not found" }, 404); + const body = await parseBody(c, modelDefaultsPutSchema); + if (!body.ok) return body.response; + return withGateway(c, async () => { + const response = await engineClientFor(ctx, access).call("models/defaults/put", body.data); + return c.json(response.result.defaults); + }); + }); + + /// Provider-discovered routes for agent and speech model pickers. /// Lightspeed owns credential injection and sanitizes per-provider errors. app.get("/:id/models", async (c) => { const access = await universeForSession(ctx, c, c.req.param("id")); @@ -2103,6 +2182,7 @@ export async function withGateway( const code = (error as { cause?: { code?: string } } | null)?.cause?.code; if (code === "23505") return c.json({ error: "record already exists" }, 409); if (error instanceof LightspeedRpcError) { + if (error.kind === "model_default_unset") return c.json({ error: error.message, kind: error.kind, modelDefaultSlot: error.data?.modelDefaultSlot }, 400); if (error.kind === "invalid_request") return c.json({ error: error.message }, 400); // Core refusing the Platform means its own key or gate is wrong: the // member was already admitted here. diff --git a/platform/server/src/routes/messages.test.ts b/platform/server/src/routes/messages.test.ts new file mode 100644 index 000000000..4bd084c1a --- /dev/null +++ b/platform/server/src/routes/messages.test.ts @@ -0,0 +1,106 @@ +import { Hono } from "hono"; +import { afterEach, beforeEach, expect, it, vi } from "vitest"; +import type { ApiVariables, AppContext } from "../context.js"; +import { gatewayRoutes } from "./gateway.js"; + +const auth = vi.hoisted(() => ({ role: "contributor" })); +vi.mock("./universes.js", () => ({ universeForSession: vi.fn(async (_ctx, _c, id: string) => ({ + universe: { lightspeedUniverseId: id, gatewayUrl: "https://engine.example/rpc" }, + slug: "test", role: auth.role, member: { userId: "member", role: auth.role }, +})) })); +beforeEach(() => { auth.role = "contributor"; }); +afterEach(() => vi.unstubAllGlobals()); + +const image = { blobRef: `sha256:${"b".repeat(64)}`, mime: "image/png", kind: "image", name: "screen.png" }; +const pdf = { blobRef: `sha256:${"c".repeat(64)}`, mime: "application/pdf", kind: "document", name: "offer.pdf" }; +const route = { providerId: "openai", apiKind: "openai:responses", model: "gpt-5.5-mini" }; + +function fixture() { + const requests: { method: string; params: Record }[] = []; + const fetch = vi.fn(async (_url: unknown, init: RequestInit) => { + const rpc = JSON.parse(String(init.body)); + // The member gate reads the session before a session-targeted call. + if (rpc.method === "session/read") { + return Response.json({ id: rpc.id, result: { result: { session: { access: { visibility: "universe" } } }, notifications: [] } }); + } + requests.push(rpc); + const result = rpc.method === "blobs/put" + ? { blobs: [{ blobRef: image.blobRef, bytes: 3 }] } + : rpc.method === "session/runs/steer" + ? { steeringId: "steer_1", run: { id: "run_1", status: "running" } } + : { run: { id: "run_1", status: "running" } }; + return Response.json({ id: rpc.id, result: { result, notifications: [] } }); + }); + vi.stubGlobal("fetch", fetch); + const app = new Hono<{ Variables: ApiVariables }>(); + app.use("*", async (c, next) => { c.set("session", { user: { id: "member" } } as ApiVariables["session"]); await next(); }); + app.route("/", gatewayRoutes({ env: { lightspeedApiUrl: "https://engine.example/rpc", lightspeedApiKey: "lsk_fixture" } } as AppContext)); + const call = (path: string, body: unknown) => app.request(`/universe${path}`, { + method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify(body), + }); + return { call, requests, fetch }; +} + +it("sends attachments as media items before the text, with per-message run options", async () => { + const f = fixture(); + const response = await f.call("/sessions/s1/messages", { + text: "Compare these", submissionId: "sub", attachments: [image, pdf], + options: { model: route, reasoningEffort: "high" }, + }); + expect(response.status).toBe(200); + expect(f.requests[0]).toMatchObject({ + method: "session/runs/start", + params: { + sessionId: "s1", + submissionId: "sub", + source: { type: "input", items: [ + { type: "media", origin: "user:member", ...image }, + { type: "media", origin: "user:member", ...pdf }, + { type: "text", origin: "user:member", text: "Compare these" }, + ] }, + config: { model: route, generation: { reasoningEffort: "high" } }, + }, + }); +}); + +it("accepts an attachment-only message and sends no config without options", async () => { + const f = fixture(); + expect((await f.call("/sessions/s1/messages", { submissionId: "sub", attachments: [image] })).status).toBe(200); + const params = f.requests[0]!.params as { source: { items: unknown[] }; config?: unknown }; + expect(params.source.items).toEqual([{ type: "media", origin: "user:member", ...image }]); + expect(params.config).toBeUndefined(); +}); + +it("steers with attachments", async () => { + const f = fixture(); + const response = await f.call("/sessions/s1/runs/run_1/steer", { text: "Also this", attachments: [pdf] }); + expect(await response.json()).toMatchObject({ steeringId: "steer_1" }); + expect(f.requests[0]!.params).toMatchObject({ items: [ + { type: "media", origin: "user:member", ...pdf }, + { type: "text", origin: "user:member", text: "Also this" }, + ] }); +}); + +it.each([ + ["an empty message", { submissionId: "sub" }], + ["whitespace with no attachments", { text: " ", submissionId: "sub" }], + ["an unsupported type", { submissionId: "sub", attachments: [{ ...pdf, mime: "application/msword" }] }], + ["a kind that contradicts the type", { submissionId: "sub", attachments: [{ ...image, kind: "document" }] }], + ["too many attachments", { submissionId: "sub", attachments: Array.from({ length: 9 }, () => image) }], + ["an unknown option", { text: "hi", submissionId: "sub", options: { temperature: 1 } }], + ["a processing tier, which belongs to the session config", { text: "hi", submissionId: "sub", options: { processingTier: "flex" } }], +])("rejects %s before reaching the runtime", async (_name, body) => { + const f = fixture(); + expect((await f.call("/sessions/s1/messages", body)).status).toBe(400); + expect(f.fetch).not.toHaveBeenCalled(); +}); + +it("uploads attachments as blobs for contributors only", async () => { + const f = fixture(); + expect(await (await f.call("/attachments", { bytesBase64: btoa("png") })).json()).toMatchObject({ blobs: [{ blobRef: image.blobRef }] }); + expect(f.requests[0]).toMatchObject({ method: "blobs/put", params: { blobs: [{ bytesBase64: btoa("png") }] } }); + auth.role = "viewer"; + const viewer = fixture(); + expect((await viewer.call("/attachments", { bytesBase64: btoa("png") })).status).toBe(403); + expect(viewer.fetch).not.toHaveBeenCalled(); +}); diff --git a/platform/server/src/routes/method-roles.ts b/platform/server/src/routes/method-roles.ts index 422587325..94f73dd42 100644 --- a/platform/server/src/routes/method-roles.ts +++ b/platform/server/src/routes/method-roles.ts @@ -78,6 +78,8 @@ export const METHOD_ROLES: Readonly> = { "mcp/servers/put": "operator", "mcp/servers/read": "viewer", "mcp/servers/tools/discover": "operator", + "models/defaults/put": "operator", + "models/defaults/read": "viewer", "models/list": "viewer", "profiles/create": "operator", "profiles/delete": "operator", @@ -109,6 +111,9 @@ export const METHOD_ROLES: Readonly> = { "session/share": "contributor", "session/skills/list": "viewer", "session/start": "contributor", + "transcriptions/cancel": "contributor", + "transcriptions/read": "viewer", + "transcriptions/start": "contributor", "vfs/snapshots/commit": "contributor", "vfs/snapshots/read": "viewer", "vfs/workspaces/create": "contributor", @@ -154,6 +159,7 @@ export const UNMEMBERED_METHODS: ReadonlySet = new Set([ "deployment/api-keys/create", "deployment/api-keys/list", "deployment/api-keys/revoke", + "deployment/api-keys/rotate", "deployment/channels/accounts/list", "deployment/environment-provider-bindings/list", "deployment/environment-providers/bindings/delete", diff --git a/platform/server/src/routes/model-defaults.test.ts b/platform/server/src/routes/model-defaults.test.ts new file mode 100644 index 000000000..c3da70a12 --- /dev/null +++ b/platform/server/src/routes/model-defaults.test.ts @@ -0,0 +1,75 @@ +import { Hono } from "hono"; +import { afterEach, beforeEach, expect, it, vi } from "vitest"; +import type { ModelDefaults } from "@lightspeed-ai/agent-client"; +import type { ApiVariables, AppContext } from "../context.js"; +import { gatewayRoutes, withGateway } from "./gateway.js"; +import { LightspeedRpcError } from "@lightspeed-ai/agent-client"; + +const auth = vi.hoisted(() => ({ role: "operator" })); +vi.mock("./universes.js", () => ({ universeForSession: vi.fn(async (_ctx, _c, id: string) => ({ + universe: { lightspeedUniverseId: id, gatewayUrl: "https://engine.example/rpc" }, + slug: "test", role: auth.role, member: { userId: "member", role: auth.role }, +})) })); +beforeEach(() => { auth.role = "operator"; }); +afterEach(() => vi.unstubAllGlobals()); + +function fixture() { + const requests: { method: string; params: Record }[] = []; + let defaults: ModelDefaults = { revision: 2, agentRun: null, speechToText: null }; + const fetch = vi.fn(async (_url: unknown, init: RequestInit) => { + expect(new Headers(init.headers).get("x-lightspeed-universe")).toBe("universe"); + expect(new Headers(init.headers).get("x-lightspeed-actor")).toBe("member"); + const rpc = JSON.parse(String(init.body)); + requests.push(rpc); + if (rpc.method === "models/defaults/put") { + if (rpc.params.expectedRevision !== defaults.revision) return Response.json({ id: rpc.id, + error: { code: -32009, message: "defaults changed", data: { kind: "conflict", message: "defaults changed" } } }); + defaults = { ...defaults, [rpc.params.slot]: rpc.params.model, revision: defaults.revision + 1 }; + } + return Response.json({ id: rpc.id, result: { result: { defaults }, notifications: [] } }); + }); + vi.stubGlobal("fetch", fetch); + const app = new Hono<{ Variables: ApiVariables }>(); + app.use("*", async (c, next) => { c.set("session", { user: { id: "member" } } as ApiVariables["session"]); await next(); }); + app.route("/", gatewayRoutes({ env: { lightspeedApiUrl: "https://engine.example/rpc", lightspeedApiKey: "lsk_fixture" } } as AppContext)); + const call = (method = "GET", body?: unknown) => app.request("/universe/models/defaults", { + method, headers: { "content-type": "application/json" }, ...(body === undefined ? {} : { body: JSON.stringify(body) }), + }); + return { app, call, requests, fetch }; +} + +it("reads and writes defaults through the member-scoped runtime client", async () => { + const f = fixture(); + expect(await (await f.call()).json()).toMatchObject({ revision: 2, agentRun: null }); + const model = { providerId: "private", apiKind: "openai:completions", model: "unlisted" }; + expect(await (await f.call("PUT", { slot: "agentRun", model, expectedRevision: 2 })).json()).toMatchObject({ revision: 3, agentRun: model }); + expect((await f.call("PUT", { slot: "agentRun", model: null, expectedRevision: 2 })).status).toBe(409); + expect(await (await f.call("PUT", { slot: "agentRun", model: null, expectedRevision: 3 })).json()).toMatchObject({ revision: 4, agentRun: null }); + expect(f.requests.map((request) => request.method)).toEqual(["models/defaults/read", "models/defaults/put", "models/defaults/put", "models/defaults/put"]); +}); + +it.each(["viewer", "contributor"])("allows %s to inspect but not configure defaults", async (role) => { + auth.role = role; + const f = fixture(); + expect((await f.call()).status).toBe(200); + expect((await f.call("PUT", { slot: "agentRun", model: null, expectedRevision: 2 })).status).toBe(403); + expect(f.fetch).toHaveBeenCalledTimes(1); +}); + +it.each([ + { slot: "agentRun", expectedRevision: 2 }, + { slot: "agentRun", model: null }, + { slot: "agentRun", model: { providerId: "openai", apiKind: "openai:audio-transcriptions", model: "speech" }, expectedRevision: 2 }, +])("rejects incomplete or incompatible writes before sending them upstream", async (body) => { + const f = fixture(); + expect((await f.call("PUT", body)).status).toBe(400); + expect(f.fetch).not.toHaveBeenCalled(); +}); + +it("preserves the typed missing-default error for web clients", async () => { + const app = new Hono(); + app.get("/", (c) => withGateway(c, async () => { throw new LightspeedRpcError({ code: -32014, message: "Choose a model", data: { kind: "model_default_unset", message: "Choose a model", modelDefaultSlot: "agentRun" } }); })); + const response = await app.request("/"); + expect(response.status).toBe(400); + expect(await response.json()).toMatchObject({ kind: "model_default_unset", modelDefaultSlot: "agentRun" }); +}); diff --git a/platform/server/src/routes/transcriptions.test.ts b/platform/server/src/routes/transcriptions.test.ts new file mode 100644 index 000000000..515de4bdc --- /dev/null +++ b/platform/server/src/routes/transcriptions.test.ts @@ -0,0 +1,67 @@ +import { Hono } from "hono"; +import { afterEach, beforeEach, expect, it, vi } from "vitest"; +import type { ApiVariables, AppContext } from "../context.js"; +import { gatewayRoutes } from "./gateway.js"; +const auth = vi.hoisted(() => ({ role: "contributor" })); +vi.mock("./universes.js", () => ({ universeForSession: vi.fn(async (_ctx, _c, id: string) => ({ + universe: { lightspeedUniverseId: id, gatewayUrl: "https://engine.example/rpc" }, + slug: "test", role: auth.role, member: { userId: "member", role: auth.role }, +})) })); +beforeEach(() => { auth.role = "contributor"; }); +afterEach(() => vi.unstubAllGlobals()); +const audio = { blobRef: `sha256:${"a".repeat(64)}`, mime: "audio/webm", name: "dictation.webm" }; +function fixture() { + const requests: { method: string; params: Record }[] = []; + const fetch = vi.fn(async (_url: unknown, init: RequestInit) => { + expect(new Headers(init.headers).get("x-lightspeed-universe")).toBe("universe"); + expect(new Headers(init.headers).get("x-lightspeed-actor")).toBe("member"); + const rpc = JSON.parse(String(init.body)); + requests.push(rpc); + return Response.json({ id: rpc.id, result: { result: rpc.method === "blobs/put" ? { blobs: [{ blobRef: audio.blobRef, bytes: 5 }] } : { transcription: { transcriptionId: "job", status: "running" } }, notifications: [] } }); + }); + vi.stubGlobal("fetch", fetch); + const app = new Hono<{ Variables: ApiVariables }>(); + app.use("*", async (c, next) => { c.set("session", { user: { id: "member" } } as ApiVariables["session"]); await next(); }); + app.route("/", gatewayRoutes({ env: { lightspeedApiUrl: "https://engine.example/rpc", lightspeedApiKey: "lsk_fixture" } } as AppContext)); + const call = (method: string, path = "", body?: unknown) => app.request(`/universe/transcriptions${path}`, { + method, headers: { "content-type": "application/json" }, ...(body === undefined ? {} : { body: JSON.stringify(body) }), + }); + return { call, requests, fetch }; +} +it("lets contributors upload, transcribe, inspect, and cancel under their own identity", async () => { + const f = fixture(); + expect(await (await f.call("POST", "/audio", { bytesBase64: btoa("audio") })).json()).toMatchObject({ blobs: [{ blobRef: audio.blobRef }] }); + expect(await (await f.call("POST", "", { audio, idempotencyKey: "request" })).json()).toMatchObject({ transcriptionId: "job" }); + expect((await f.call("GET", "/job")).status).toBe(200); + expect((await f.call("POST", "/job/cancel")).status).toBe(200); + expect(f.requests.map(({ method, params }) => ({ method, params }))).toEqual([ + { method: "blobs/put", params: { blobs: [{ bytesBase64: btoa("audio") }] } }, + { method: "transcriptions/start", params: { audio, idempotencyKey: "request" } }, + { method: "transcriptions/read", params: { transcriptionId: "job" } }, + { method: "transcriptions/cancel", params: { transcriptionId: "job" } }, + ]); +}); +it("rejects viewer mutations before reaching the runtime", async () => { + auth.role = "viewer"; + const f = fixture(); + expect((await f.call("POST", "/audio", { bytesBase64: btoa("audio") })).status).toBe(403); + expect((await f.call("POST", "", { audio, idempotencyKey: "request" })).status).toBe(403); + expect((await f.call("POST", "/job/cancel")).status).toBe(403); + expect(f.fetch).not.toHaveBeenCalled(); +}); +it("rejects explicit model overrides and malformed audio before reaching the runtime", async () => { + const f = fixture(); + expect((await f.call("POST", "", { audio, idempotencyKey: "request", model: { providerId: "other", apiKind: "openai:audio-transcriptions", model: "override" } })).status).toBe(400); + expect((await f.call("POST", "/audio", { bytesBase64: "!!!" })).status).toBe(400); + expect(f.fetch).not.toHaveBeenCalled(); +}); +it("preserves the missing speech default error", async () => { + const f = fixture(); + f.fetch.mockImplementation(async (_url, init) => { + const rpc = JSON.parse(String(init.body)); + return Response.json({ id: rpc.id, error: { code: -32014, message: "Choose a speech model", data: { kind: "model_default_unset", message: "Choose a speech model", modelDefaultSlot: "speechToText" } } }); + }); + const response = await f.call("POST", "", { audio, idempotencyKey: "request" }); + expect(response.status).toBe(400); + expect(await response.json()).toMatchObject({ kind: "model_default_unset", modelDefaultSlot: "speechToText" }); +}); diff --git a/platform/server/src/routes/universe-api-keys.test.ts b/platform/server/src/routes/universe-api-keys.test.ts index 855bf438a..4a8d60968 100644 --- a/platform/server/src/routes/universe-api-keys.test.ts +++ b/platform/server/src/routes/universe-api-keys.test.ts @@ -9,20 +9,21 @@ const universe = { id: "platform-universe", organizationId: "org", lightspeedUni /// Universe key routes for a caller with `role`, over a core that records /// each call and a database that answers the universe and membership reads. -function setup(role: string) { +function setup(role: string, keys = [{ keyPrefix: "lsk_abcdefgh", assertActor: false, revokedAtMs: null as number | null }]) { const calls: { method: string; params: Record }[] = []; vi.stubGlobal("fetch", vi.fn(async (_url: unknown, init: RequestInit) => { const rpc = JSON.parse(String(init.body)); calls.push({ method: rpc.method, params: rpc.params }); const apiKey = { keyPrefix: "lsk_abcdefgh", groups: rpc.params.groups }; - return Response.json({ id: rpc.id, result: { result: { apiKey, secret: "lsk_secret" }, notifications: [] } }); + return Response.json({ id: rpc.id, result: { result: rpc.method === "deployment/api-keys/list" ? { apiKeys: keys } : { apiKey, secret: "lsk_secret" }, notifications: [] } }); })); const queue: unknown[][] = [[{ universe, slug: "test" }], [{ role }]]; const chain: Record = {}; for (const step of ["from", "innerJoin", "where"]) chain[step] = () => chain; chain.limit = async () => queue.shift() ?? []; + const audit = vi.fn(async () => undefined); const ctx = { - db: { insert: () => ({ values: async () => undefined }), select: () => chain }, + db: { insert: () => ({ values: audit }), select: () => chain }, env: { lightspeedApiUrl: "https://core.example/rpc", lightspeedApiKey: "lsk_platform" }, } as unknown as AppContext; const app = new Hono<{ Variables: ApiVariables }>(); @@ -34,7 +35,8 @@ function setup(role: string) { const create = (body: unknown) => app.request("/platform-universe/api-keys", { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify(body), }); - return { create, calls }; + const rotate = (prefix = "lsk_abcdefgh") => app.request(`/platform-universe/api-keys/${prefix}/rotate`, { method: "POST" }); + return { create, rotate, calls, audit }; } it("mints a universe key with only the groups the admin chose", async () => { @@ -64,3 +66,32 @@ it("keeps universe keys with admins", async () => { expect((await create({ displayName: "Agent", groups: ["session"] })).status).toBe(404); expect(calls).toEqual([]); }); + +it("rotates only a key belonging to the administered universe", async () => { + const { rotate, calls, audit } = setup("admin"); + const response = await rotate(); + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({ secret: "lsk_secret" }); + expect(calls).toEqual([ + { method: "deployment/api-keys/list", params: { scope: { kind: "universe", universeId: universe.lightspeedUniverseId } } }, + { method: "deployment/api-keys/rotate", params: { keyPrefix: "lsk_abcdefgh" } }, + ]); + expect(audit).toHaveBeenCalledWith(expect.objectContaining({ actorId: "caller", action: "key.rotate", targetId: "lsk_abcdefgh", universeId: universe.id })); +}); + +it.each(["operator", "contributor", "viewer"])("refuses rotation by a %s", async (role) => { + const { rotate, calls } = setup(role); + expect((await rotate()).status).toBe(404); + expect(calls).toEqual([]); +}); + +it.each([ + ["another universe's key", [], 404], + ["revoked key", [{ keyPrefix: "lsk_abcdefgh", assertActor: false, revokedAtMs: 10 }], 404], + ["actor-asserting key", [{ keyPrefix: "lsk_abcdefgh", assertActor: true, revokedAtMs: null }], 403], +] as const)("refuses rotation of %s", async (_name, keys, status) => { + const { rotate, calls, audit } = setup("admin", [...keys]); + expect((await rotate()).status).toBe(status); + expect(calls.map((call) => call.method)).toEqual(["deployment/api-keys/list"]); + expect(audit).not.toHaveBeenCalled(); +}); diff --git a/platform/server/src/routes/universes.ts b/platform/server/src/routes/universes.ts index 6f63fcd96..4fc30115d 100644 --- a/platform/server/src/routes/universes.ts +++ b/platform/server/src/routes/universes.ts @@ -302,7 +302,7 @@ export function universeRoutes(ctx: AppContext) { /// Universe API keys, for a universe admin. Keys are minted with the /// Platform's deployment key, scoped to this universe, with the groups the /// admin chose and no actor assertion. The plaintext secret exists only in the - /// create response and is never persisted by the platform. + /// create or rotation response and is never persisted by the platform. app.get("/:id/api-keys", async (c) => { const access = await universeForSession(ctx, c, c.req.param("id")); if (!access || access.role !== "admin") { @@ -337,6 +337,28 @@ export function universeRoutes(ctx: AppContext) { }); }); + app.post("/:id/api-keys/:keyPrefix/rotate", async (c) => { + const access = await universeForSession(ctx, c, c.req.param("id")); + if (!access || access.role !== "admin") { + return c.json({ error: "not found" }, 404); + } + return withGateway(c, async () => { + const client = deploymentClientFor(ctx, access.universe.gatewayUrl); + const keys = await client.call("deployment/api-keys/list", { + scope: { kind: "universe", universeId: access.universe.lightspeedUniverseId }, + }); + const keyPrefix = c.req.param("keyPrefix"); + const key = keys.result.apiKeys?.find((key) => key.keyPrefix === keyPrefix); + if (!key || key.revokedAtMs != null) return c.json({ error: "not found" }, 404); + // Rotation reveals a credential: universe admins cannot acquire actor + // assertion authority that only platform admins may grant. + if (key.assertActor) return c.json({ error: "Rotate this key from Platform admin" }, 403); + const response = await client.call("deployment/api-keys/rotate", { keyPrefix }); + await auditIdentity(ctx.db, { actorId: c.get("session").user.id, action: "key.rotate", targetId: keyPrefix, universeId: access.universe.id, details: { newKeyPrefix: response.result.apiKey.keyPrefix } }); + return c.json(response.result); + }); + }); + app.delete("/:id/api-keys/:keyPrefix", async (c) => { const access = await universeForSession(ctx, c, c.req.param("id")); if (!access || access.role !== "admin") { diff --git a/platform/shared/src/index.ts b/platform/shared/src/index.ts index bc3c6c3de..1a25d6bfc 100644 --- a/platform/shared/src/index.ts +++ b/platform/shared/src/index.ts @@ -136,3 +136,27 @@ export function mergeFeatureOverrides(stored: unknown, changes: FeatureOverrides } return merged; } +export { AGENT_MODEL_API_KINDS, modelDefaultsPutSchema } from "./model-defaults.js"; +export { MAX_DICTATION_AUDIO_BYTES, MAX_DICTATION_SECONDS, transcriptionStartSchema, transcriptionUploadSchema } from "./transcriptions.js"; +export { + ATTACHMENT_ACCEPT, + ATTACHMENT_MIMES, + ATTACHMENT_SUMMARY, + MAX_ANTHROPIC_IMAGE_BYTES, + MAX_ATTACHMENT_BYTES, + MAX_MESSAGE_ATTACHMENTS, + attachmentLimit, + attachmentType, + attachmentUploadSchema, + messageAttachmentSchema, + messageInputItems, + messageRunConfig, + messageRunOptionsSchema, + reasoningEffortTiers, + sessionMessageSchema, + sessionSteerSchema, + type AttachmentKind, + type AttachmentType, + type MessageAttachment, + type MessageRunOptions, +} from "./messages.js"; diff --git a/platform/shared/src/messages.ts b/platform/shared/src/messages.ts new file mode 100644 index 000000000..d307c42f9 --- /dev/null +++ b/platform/shared/src/messages.ts @@ -0,0 +1,147 @@ +import { z } from "zod"; + +/// Attachments a composer message may carry. The runtime accepts these media +/// types as run input and every agent adapter hands them to the model +/// natively: images and PDFs as media, the text types inlined as text. The +/// size limits mirror the runtime's admission limits so a refusal happens +/// at pick time instead of after the upload. +const MiB = 1024 * 1024; +export const MAX_MESSAGE_ATTACHMENTS = 8; +export const MAX_ATTACHMENT_BYTES = 10 * MiB; +/// Anthropic's Messages API rejects larger images than the runtime does. +export const MAX_ANTHROPIC_IMAGE_BYTES = 5 * MiB; + +export type AttachmentKind = "image" | "document"; +export interface AttachmentType { + mime: string; + kind: AttachmentKind; + maxBytes: number; + /// Short type name for chips and refusals, e.g. "PNG" or "PDF". + label: string; +} + +const TYPES: readonly AttachmentType[] = [ + { mime: "image/png", kind: "image", maxBytes: MAX_ATTACHMENT_BYTES, label: "PNG" }, + { mime: "image/jpeg", kind: "image", maxBytes: MAX_ATTACHMENT_BYTES, label: "JPEG" }, + { mime: "image/webp", kind: "image", maxBytes: MAX_ATTACHMENT_BYTES, label: "WebP" }, + { mime: "image/gif", kind: "image", maxBytes: MAX_ATTACHMENT_BYTES, label: "GIF" }, + { mime: "application/pdf", kind: "document", maxBytes: MAX_ATTACHMENT_BYTES, label: "PDF" }, + { mime: "text/plain", kind: "document", maxBytes: MiB, label: "Text" }, + { mime: "text/markdown", kind: "document", maxBytes: MiB, label: "Markdown" }, + { mime: "text/csv", kind: "document", maxBytes: MiB, label: "CSV" }, + { mime: "application/json", kind: "document", maxBytes: MiB, label: "JSON" }, +]; +const BY_EXTENSION: Readonly> = { + png: "image/png", jpg: "image/jpeg", jpeg: "image/jpeg", webp: "image/webp", gif: "image/gif", + pdf: "application/pdf", txt: "text/plain", text: "text/plain", log: "text/plain", + md: "text/markdown", markdown: "text/markdown", csv: "text/csv", json: "application/json", +}; +export const ATTACHMENT_MIMES = TYPES.map((type) => type.mime); +/// For a file input's `accept`: MIME types plus extensions browsers often +/// report without one (Markdown in particular). +export const ATTACHMENT_ACCEPT = [...ATTACHMENT_MIMES, ...Object.keys(BY_EXTENSION).map((ext) => `.${ext}`)].join(","); +/// A one-line list for tooltips and refusals. +export const ATTACHMENT_SUMMARY = "Images (PNG, JPEG, WebP, GIF), PDFs, and text files (TXT, Markdown, CSV, JSON)"; + +/// The accepted type for a picked file, or null. Browsers report an empty or +/// vendor MIME type for some text formats, so the extension decides when the +/// reported type is not one the model accepts. +export function attachmentType(mime: string, name: string): AttachmentType | null { + const reported = mime.split(";")[0]!.trim().toLowerCase(); + const direct = TYPES.find((type) => type.mime === reported); + if (direct) return direct; + const ext = name.includes(".") ? name.slice(name.lastIndexOf(".") + 1).toLowerCase() : ""; + const inferred = BY_EXTENSION[ext]; + return TYPES.find((type) => type.mime === inferred) ?? null; +} + +/// The effective byte limit for a type on the session's agent API. +export function attachmentLimit(type: AttachmentType, apiKind: string | undefined): number { + return type.kind === "image" && apiKind === "anthropic:messages" + ? Math.min(type.maxBytes, MAX_ANTHROPIC_IMAGE_BYTES) + : type.maxBytes; +} + +export const attachmentUploadSchema = z.object({ + bytesBase64: z.string().min(4).max(4 * Math.ceil(MAX_ATTACHMENT_BYTES / 3)) + .regex(/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/), +}).strict(); + +export const messageAttachmentSchema = z.object({ + blobRef: z.string().regex(/^sha256:[a-f0-9]{64}$/), + mime: z.enum(ATTACHMENT_MIMES as [string, ...string[]]), + kind: z.enum(["image", "document"]), + name: z.string().trim().min(1).max(256), +}).strict().refine( + ({ mime, kind }) => TYPES.find((type) => type.mime === mime)?.kind === kind, + { message: "The attachment kind does not match its type.", path: ["kind"] }, +); +export type MessageAttachment = z.infer; + +const routeName = z.string().min(1).max(512).refine((value) => value.trim() === value, "Remove surrounding whitespace."); + +/// Per-message model choices. They ride the run as overrides and never +/// change the stored session configuration. Provider and API kind must match +/// the session's pinned route; the runtime validates that and the effort tier. +export const messageRunOptionsSchema = z.object({ + model: z.object({ providerId: routeName, apiKind: routeName, model: routeName }).strict().optional(), + reasoningEffort: z.string().trim().min(1).max(32).optional(), +}).strict(); +export type MessageRunOptions = z.infer; + +const hasContent = ({ text, attachments }: { text: string; attachments: unknown[] }) => + text.trim().length > 0 || attachments.length > 0; +const contentMessage = { message: "Write a message or attach a file.", path: ["text"] }; + +export const sessionMessageSchema = z.object({ + text: z.string().max(100_000).default(""), + submissionId: z.string().min(1).max(200), + attachments: z.array(messageAttachmentSchema).max(MAX_MESSAGE_ATTACHMENTS).default([]), + options: messageRunOptionsSchema.optional(), +}).refine(hasContent, contentMessage); + +/// Steering has no run options: it joins a run that already chose them. +/// Unknown top-level keys are dropped, so a client cannot supply its own +/// origin; the route stamps the signed-in user. +export const sessionSteerSchema = z.object({ + text: z.string().max(100_000).default(""), + attachments: z.array(messageAttachmentSchema).max(MAX_MESSAGE_ATTACHMENTS).default([]), +}).refine(hasContent, contentMessage); + +/// Effort tiers each agent API accepts, used when provider discovery reports +/// none for a model. Keep aligned with the runtime's validation lists. +const EFFORT_TIERS: Readonly> = { + "openai:responses": ["none", "minimal", "low", "medium", "high", "xhigh"], + "openai:completions": ["none", "minimal", "low", "medium", "high", "xhigh", "max"], + "anthropic:messages": ["none", "low", "medium", "high", "xhigh", "max"], +}; +export function reasoningEffortTiers(apiKind: string): readonly string[] { + return EFFORT_TIERS[apiKind] ?? []; +} + +/// Run input items for a composer message: media first, so the model reads +/// the files before the words that refer to them. +export function messageInputItems(text: string, attachments: readonly MessageAttachment[], origin: string) { + return [ + ...attachments.map((attachment) => ({ + type: "media" as const, + origin, + blobRef: attachment.blobRef, + mime: attachment.mime, + kind: attachment.kind, + name: attachment.name, + })), + ...(text.trim() ? [{ type: "text" as const, text, origin }] : []), + ]; +} + +/// The run-start config for per-message options, or undefined when none. +export function messageRunConfig(options: MessageRunOptions | undefined) { + if (!options) return undefined; + const generation = options.reasoningEffort ? { reasoningEffort: options.reasoningEffort } : {}; + const config = { + ...(options.model ? { model: options.model } : {}), + ...(Object.keys(generation).length ? { generation } : {}), + }; + return Object.keys(config).length ? config : undefined; +} diff --git a/platform/shared/src/model-defaults.ts b/platform/shared/src/model-defaults.ts new file mode 100644 index 000000000..09e8a2981 --- /dev/null +++ b/platform/shared/src/model-defaults.ts @@ -0,0 +1,17 @@ +import { z } from "zod"; + +export const AGENT_MODEL_API_KINDS = ["openai:responses", "openai:completions", "anthropic:messages"] as const; +const routeName = z.string().min(1).refine( + (value) => value.trim() === value && new TextEncoder().encode(value).length <= 512, + "Use 1–512 bytes without surrounding whitespace.", +); + +export const modelDefaultsPutSchema = z.object({ + slot: z.enum(["agentRun", "speechToText"]), + model: z.object({ providerId: routeName, apiKind: z.string(), model: routeName }).strict().nullable(), + expectedRevision: z.number().int().min(0).max(Number.MAX_SAFE_INTEGER), +}).strict().refine(({ slot, model }) => !model || (slot === "agentRun" + ? (AGENT_MODEL_API_KINDS as readonly string[]).includes(model.apiKind) + : model.apiKind === "openai:audio-transcriptions"), { + message: "The selected API does not support this use.", path: ["model", "apiKind"], +}); diff --git a/platform/shared/src/transcriptions.ts b/platform/shared/src/transcriptions.ts new file mode 100644 index 000000000..eb0ca81ff --- /dev/null +++ b/platform/shared/src/transcriptions.ts @@ -0,0 +1,18 @@ +import { z } from "zod"; + +export const MAX_DICTATION_AUDIO_BYTES = 25 * 1024 * 1024; +export const MAX_DICTATION_SECONDS = 10 * 60; +export const transcriptionUploadSchema = z.object({ + bytesBase64: z.string().min(4).max(4 * Math.ceil(MAX_DICTATION_AUDIO_BYTES / 3)).regex(/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/), +}).strict(); + +// Web dictation always uses the universe default. Model overrides belong to +// the core API, not this browser route. +export const transcriptionStartSchema = z.object({ + idempotencyKey: z.string().min(1).max(200), + audio: z.object({ + blobRef: z.string().regex(/^sha256:[a-f0-9]{64}$/), + mime: z.string().min(1).max(128), + name: z.string().min(1).max(256), + }).strict(), +}).strict(); diff --git a/platform/web/package.json b/platform/web/package.json index d8783b64a..3cf23ca30 100644 --- a/platform/web/package.json +++ b/platform/web/package.json @@ -23,6 +23,7 @@ "better-auth": "1.7.6", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", + "extendable-media-recorder": "^9.2.40", "hono": "^4.7.0", "lucide-react": "^1.23.0", "react": "^19.1.0", diff --git a/platform/web/src/App.tsx b/platform/web/src/App.tsx index a77cb4afb..2a0de2a60 100644 --- a/platform/web/src/App.tsx +++ b/platform/web/src/App.tsx @@ -29,6 +29,7 @@ import { ProfilesPage } from "@/pages/ProfilesPage"; import { SessionsPage } from "@/pages/SessionsPage"; import { SetupsPage } from "@/pages/SetupsPage"; import { WorkspacesPage } from "@/pages/WorkspacesPage"; +import { BlobPage } from "@/pages/BlobPage"; import { UserPreferencesProvider } from "@/lib/user-preferences"; import { FeatureGate } from "@/components/feature-gate"; import { finishAutomaticSignIn } from "@/lib/automatic-sign-in"; @@ -124,6 +125,7 @@ export function App() { path="u/:slug/bots/:botId/activity" element={} /> + } /> } /> (method: string, path: string, body?: unknown): Promise { +export async function api(method: string, path: string, body?: unknown, signal?: AbortSignal): Promise { const res = await fetch(path, { + signal, method, headers: body !== undefined ? { "content-type": "application/json" } : undefined, body: body !== undefined ? JSON.stringify(body) : undefined, @@ -277,7 +279,7 @@ export interface SecretProvider { export interface ModelEndpointConfig { baseUrl: string; headers?: Record; - apiKinds: Array<"openai:responses" | "openai:completions">; + apiKinds: Array<"openai:responses" | "openai:completions" | "openai:audio-transcriptions">; } export interface SecretsInventory { diff --git a/platform/web/src/components/api-keys/rotate-key-dialog.tsx b/platform/web/src/components/api-keys/rotate-key-dialog.tsx new file mode 100644 index 000000000..0af422d0e --- /dev/null +++ b/platform/web/src/components/api-keys/rotate-key-dialog.tsx @@ -0,0 +1,43 @@ +import { useMutation } from "@tanstack/react-query"; +import type { DeploymentApiKeyCreateResponse, DeploymentApiKeyView } from "@lightspeed-ai/agent-client"; +import { api } from "@/api"; +import { ApiKeySecret } from "./secret-once"; +import { Button } from "@/components/ui/button"; +import { Dialog, DialogContent, DialogDescription, DialogFooter, DialogHeader, DialogTitle } from "@/components/ui/dialog"; + +export function RotateKeyDialog({ apiKey, basePath, onClose, onRotated }: { + apiKey: DeploymentApiKeyView; + basePath: string; + onClose: () => void; + onRotated: () => void; +}) { + const rotate = useMutation({ + mutationFn: () => api("POST", `${basePath}/${encodeURIComponent(apiKey.keyPrefix)}/rotate`), + retry: false, + gcTime: 0, + onSuccess: onRotated, + }); + const close = () => { + if (rotate.isPending) return; + rotate.reset(); + onClose(); + }; + return { if (!open) close(); }}> + + {rotate.data ? : <> + + Rotate this API key? + + The current secret for {apiKey.displayName ?? apiKey.keyPrefix} will stop working immediately. + Update every client using this key with the new secret. Its permissions stay the same. + + + {rotate.error &&

{rotate.error.message}

} + + + + + } +
+
; +} diff --git a/platform/web/src/components/models/add-model-provider-dialog.tsx b/platform/web/src/components/models/add-model-provider-dialog.tsx index cce485aca..25cceab72 100644 --- a/platform/web/src/components/models/add-model-provider-dialog.tsx +++ b/platform/web/src/components/models/add-model-provider-dialog.tsx @@ -32,6 +32,7 @@ export function AddModelProviderDialog({ initialKind = null, onOpenChange, onAdded, + onChooseDefault, }: { universeId: string; open: boolean; @@ -40,6 +41,7 @@ export function AddModelProviderDialog({ initialKind?: ModelProviderKind | null; onOpenChange: (open: boolean) => void; onAdded: () => void; + onChooseDefault?: () => void; }) { const [selected, setSelected] = useState(initialKind); useEffect(() => { @@ -167,6 +169,7 @@ export function AddModelProviderDialog({ )} + {done.type === "modelKey" && onChooseDefault && } diff --git a/platform/web/src/components/models/model-api-key.tsx b/platform/web/src/components/models/model-api-key.tsx index b6ed66fa9..2a27239f5 100644 --- a/platform/web/src/components/models/model-api-key.tsx +++ b/platform/web/src/components/models/model-api-key.tsx @@ -366,6 +366,7 @@ export function OpenAiCompatibleForm({ const [responses, setResponses] = useState( initialEndpoint?.apiKinds.includes("openai:responses") ?? false, ); + const [transcriptions, setTranscriptions] = useState(initialEndpoint?.apiKinds.includes("openai:audio-transcriptions") ?? false); const [completions, setCompletions] = useState( initialEndpoint?.apiKinds.includes("openai:completions") ?? true, ); @@ -386,12 +387,14 @@ export function OpenAiCompatibleForm({ setBaseUrl(preset.baseUrl); setResponses(false); setCompletions(true); + setTranscriptions(false); }; const save = useMutation({ mutationFn: () => { const apiKinds = [ ...(responses ? (["openai:responses"] as const) : []), ...(completions ? (["openai:completions"] as const) : []), + ...(transcriptions ? (["openai:audio-transcriptions"] as const) : []), ]; if (!apiKinds.length) throw new Error("select at least one API kind"); return api( @@ -525,6 +528,10 @@ export function OpenAiCompatibleForm({ />{" "} Responses + Extra headers diff --git a/platform/web/src/components/models/model-defaults.tsx b/platform/web/src/components/models/model-defaults.tsx new file mode 100644 index 000000000..53a95bea8 --- /dev/null +++ b/platform/web/src/components/models/model-defaults.tsx @@ -0,0 +1,152 @@ +import { useId, useState } from "react"; +import { useMutation, useQueryClient } from "@tanstack/react-query"; +import { AGENT_MODEL_API_KINDS, modelDefaultsPutSchema } from "@lightspeed/platform-shared"; +import { api, ApiError, type ModelConfig, type ModelDefaults, type ModelDefaultsPutParams } from "@/api"; +import { modelDefaultsKey, modelLabel, useModelDefaults, useModelDiscovery } from "@/lib/model-defaults"; +import { summarizeProviderReadiness } from "@/lib/provider-readiness"; +import { Button } from "@/components/ui/button"; +import { Input } from "@/components/ui/input"; +import { Field, FieldLabel, FieldDescription } from "@/components/ui/field"; +import { Dialog, DialogContent, DialogDescription, DialogFooter, DialogHeader, DialogTitle } from "@/components/ui/dialog"; +import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select"; +import { Combobox, ComboboxContent, ComboboxEmpty, ComboboxInput, ComboboxItem, ComboboxList } from "@/components/ui/combobox"; + +const apiLabels: Record = { + "openai:responses": "OpenAI Responses", + "openai:completions": "OpenAI Chat Completions", + "anthropic:messages": "Anthropic Messages", + "openai:audio-transcriptions": "OpenAI Audio Transcriptions", +}; + +export type DefaultSlot = ModelDefaultsPutParams["slot"]; +const slots = [ + { slot: "agentRun", label: "Agent runs", description: "Used when a new session and its profile leave the model unset. Existing sessions keep their model." }, + { slot: "speechToText", label: "Speech-to-text", description: "Used for audio transcription and web dictation. Dictation is disabled until a default is selected." }, +] as const; + +export function ModelDefaultsSection({ universeId, writable, onEdit }: { universeId: string; writable: boolean; onEdit: (defaults: ModelDefaults, slot: DefaultSlot) => void }) { + const defaults = useModelDefaults(universeId); + const discovery = useModelDiscovery(universeId); + const queryClient = useQueryClient(); + const clear = useMutation({ + mutationFn: ({ revision, slot }: { revision: number; slot: DefaultSlot }) => api("PUT", `/api/v1/universes/${universeId}/models/defaults`, { + slot, model: null, expectedRevision: revision, + } satisfies ModelDefaultsPutParams), + onSuccess: (value) => queryClient.setQueryData(modelDefaultsKey(universeId), value), + onError: () => { void queryClient.invalidateQueries({ queryKey: modelDefaultsKey(universeId) }); }, + }); + + return ( +
+
+

Defaults

+
+
+ {slots.map(({ slot, label, description }) => { + const model = defaults.data?.[slot]; + const readiness = summarizeProviderReadiness(model, discovery.error ? undefined : discovery.data?.providers, slot); + return ( +
+
+

{label}

+

{description}

+

{defaults.isLoading ? "Loading…" : defaults.error ? "Default unavailable" : model ? modelLabel(model) : "No default selected"}

+ {model &&

{apiLabels[model.apiKind] ?? model.apiKind}

} + {model &&

+ {discovery.isLoading ? "Checking provider…" : readiness.message} +

} +
+ {writable && defaults.data && !defaults.error &&
+ + {model && } +
} +
+ ); + })} + {defaults.error &&
+ Could not load defaults: {defaults.error.message} + +
} + {clear.error &&

{clear.error instanceof ApiError && clear.error.status === 409 ? "Defaults changed elsewhere. Review the current selection before clearing it." : clear.error.message}

} +
+
+ ); +} + +export function DefaultModelDialog({ universeId, initial, slot, onClose }: { universeId: string; initial: ModelDefaults; slot: DefaultSlot; onClose: () => void }) { + const apiKinds = slot === "agentRun" ? AGENT_MODEL_API_KINDS : ["openai:audio-transcriptions"] as const; + const emptyModel = { providerId: "", apiKind: apiKinds[0], model: "" }; + const [revision, setRevision] = useState(initial.revision); + const [model, setModel] = useState(initial[slot] ?? emptyModel); + const [reloadError, setReloadError] = useState(null); + const [reloading, setReloading] = useState(false); + const discovery = useModelDiscovery(universeId); + const queryClient = useQueryClient(); + const id = useId(); + const [search, setSearch] = useState(""); + const routes = (discovery.data?.models ?? []).filter((route) => (apiKinds as readonly string[]).includes(route.apiKind)); + const choices = routes.map((route) => JSON.stringify([route.providerId, route.apiKind, route.model])); + const routeFor = (key: string) => routes[choices.indexOf(key)]; + const clean = { providerId: model.providerId.trim(), apiKind: model.apiKind, model: model.model.trim() }; + const params: ModelDefaultsPutParams = { slot, model: clean, expectedRevision: revision }; + const valid = modelDefaultsPutSchema.safeParse(params).success; + const save = useMutation({ + mutationFn: () => api("PUT", `/api/v1/universes/${universeId}/models/defaults`, params), + onSuccess: (value) => { queryClient.setQueryData(modelDefaultsKey(universeId), value); onClose(); }, + }); + const conflict = save.error instanceof ApiError && save.error.status === 409; + const reload = async () => { + setReloading(true); + setReloadError(null); + try { + const latest = await api("GET", `/api/v1/universes/${universeId}/models/defaults`); + queryClient.setQueryData(modelDefaultsKey(universeId), latest); + setRevision(latest.revision); + setModel(latest[slot] ?? emptyModel); + save.reset(); + } catch (error) { setReloadError(error instanceof Error ? error.message : "Could not reload defaults."); } + finally { setReloading(false); } + }; + return { if (!open && !save.isPending) onClose(); }}> + + + Default model for {slot === "agentRun" ? "agent runs" : "speech-to-text"} + Choose a discovered model or enter your provider and model below. + +
{ event.preventDefault(); if (valid && !conflict && !save.isPending) save.mutate(); }}> +
+ + Find a model + items={choices} value={null} inputValue={search} onInputValueChange={setSearch} + itemToStringLabel={(key) => { const route = routeFor(key); return route ? `${modelLabel(route)} · ${apiLabels[route.apiKind]}` : key; }} + filter={(key, query) => key.toLowerCase().includes(query.toLowerCase())} + onValueChange={(key) => { const route = key ? routeFor(key) : undefined; if (route) { setModel({ providerId: route.providerId, apiKind: route.apiKind, model: route.model }); setSearch(""); } }}> + + No matching models. Enter a model below.{(key: string) => { + const route = routeFor(key); + return {route ? modelLabel(route) : key}{route && apiLabels[route.apiKind]}; + }} + + {discovery.isLoading ? "Loading suggestions…" : discovery.error ? "Suggestions are unavailable. You can still enter and save a model." : "Manual models do not need to appear in discovery."} + +
+ Provider setModel({ ...model, providerId: event.target.value })} /> + {discovery.data?.providers?.map((provider) => + API +
+ Model setModel({ ...model, model: event.target.value })} /> +
+ {save.error &&

{conflict ? "Defaults changed elsewhere. Reload and review the saved default before making another change." : save.error.message}

} + {reloadError &&

{reloadError}

} + + + {conflict ? + : } + +
+
+
; +} diff --git a/platform/web/src/components/provider-readiness-banner.tsx b/platform/web/src/components/provider-readiness-banner.tsx index 0a5638210..49371f7a0 100644 --- a/platform/web/src/components/provider-readiness-banner.tsx +++ b/platform/web/src/components/provider-readiness-banner.tsx @@ -1,38 +1,26 @@ import { useActionPermissions } from "@/lib/permissions"; import { Link } from "react-router-dom"; -import { KeyRound } from "lucide-react"; +import { Info } from "lucide-react"; +import type { ModelConfig } from "@/api"; import { Button } from "@/components/ui/button"; -import { addModelProviderHref, useProviderReadiness } from "@/lib/provider-readiness"; +import { useProviderReadiness } from "@/lib/provider-readiness"; -/// Nudges toward Models when no model provider has a usable credential. -/// Renders nothing while loading or when at least one provider is ready. -export function ProviderReadinessBanner({ - universeId, - slug, - className, -}: { +export function ProviderReadinessBanner({ universeId, slug, model, enabled = true, className }: { universeId: string; slug: string; + /// Omitted means universe policy; an existing session supplies its stored model. + model?: ModelConfig | null; + enabled?: boolean; className?: string; }) { - const readiness = useProviderReadiness(universeId); + const readiness = useProviderReadiness(universeId, model, enabled); const permissions = useActionPermissions(universeId); - if (readiness.isLoading || readiness.ready) return null; - const invalidOnly = readiness.missing.length === 0 && readiness.invalid.length > 0; + if (!enabled || readiness.isLoading || readiness.state === "configured") return null; return ( -
- - - {invalidOnly - ? "The configured model provider key was rejected. Sessions cannot run until a valid model provider API key is set." - : "No model provider is configured for this universe. Sessions cannot run until a model provider API key is added."} - - {permissions.can("configure_resource") && } +
+ + {readiness.message} + {permissions.can("configure_resource") && }
); } diff --git a/platform/web/src/components/session/composer-attachments.tsx b/platform/web/src/components/session/composer-attachments.tsx new file mode 100644 index 000000000..f7e5f31ab --- /dev/null +++ b/platform/web/src/components/session/composer-attachments.tsx @@ -0,0 +1,74 @@ +import { FileText, LoaderCircle, RotateCw, TriangleAlert, X } from "lucide-react"; +import { attachmentLabel, formatBytes, type ComposerAttachment } from "@/lib/composer-attachments"; +import { Button } from "@/components/ui/button"; +import { cn } from "@/lib/utils"; + +/// Attached files above the message field. Each chip owns its state: +/// uploading shows a spinner over the thumbnail, a failed upload offers a +/// retry, and every chip can be removed. +export function AttachmentStrip({ items, disabled, onRemove, onRetry }: { + items: ComposerAttachment[]; + disabled?: boolean; + onRemove: (id: string) => void; + onRetry: (id: string) => void; +}) { + if (!items.length) return null; + return ( +
    + {items.map((item) => )} +
+ ); +} + +function AttachmentChip({ item, disabled, onRemove, onRetry }: { + item: ComposerAttachment; + disabled?: boolean; + onRemove: (id: string) => void; + onRetry: (id: string) => void; +}) { + const failed = item.status === "failed"; + const uploading = item.status === "uploading"; + const detail = failed + ? "Upload failed" + : uploading + ? "Uploading…" + : `${attachmentLabel(item)} · ${formatBytes(item.size)}`; + return ( +
  • + + {item.previewUrl + ? + : failed ? : } + {uploading && ( + + + + )} + + + {item.name} + + {detail} + + + + {failed && ( + + )} + + +
  • + ); +} diff --git a/platform/web/src/components/session/composer-model-picker.tsx b/platform/web/src/components/session/composer-model-picker.tsx new file mode 100644 index 000000000..89750a20a --- /dev/null +++ b/platform/web/src/components/session/composer-model-picker.tsx @@ -0,0 +1,154 @@ +import { useState } from "react"; +import { Box, Check, ChevronDown, Search } from "lucide-react"; +import { + effortLabel, + effortShortLabel, + type ComposerModelChoice, + type RunChoice, +} from "@/lib/composer-model"; +import { Button } from "@/components/ui/button"; +import { Popover, PopoverContent, PopoverTrigger } from "@/components/ui/popover"; +import { cn } from "@/lib/utils"; + +/// The model pill: the model and reasoning effort the next message runs +/// with. Choices override the session for each message sent from here and +/// never change the stored configuration unless saved as its default. +export function ModelPicker({ choice, stored, runActive, canSaveDefault, onChange, onSaveDefault }: { + choice: ComposerModelChoice; + stored: RunChoice; + runActive: boolean; + canSaveDefault: boolean; + onChange: (next: RunChoice) => void; + onSaveDefault: () => Promise; +}) { + const [open, setOpen] = useState(false); + const [query, setQuery] = useState(""); + const [saving, setSaving] = useState(false); + const [saveError, setSaveError] = useState(); + const overridden = choice.options !== undefined; + const search = query.trim().toLowerCase(); + const models = choice.models.filter((option) => + !search || option.model.toLowerCase().includes(search) || option.displayName.toLowerCase().includes(search)); + const exact = choice.models.some((option) => option.model === query.trim()); + const effortOptions = [ + ...(choice.session.reasoningEffort ? [] : [undefined]), + ...choice.efforts, + ]; + const pick = (patch: RunChoice) => { + setSaveError(undefined); + onChange({ ...stored, ...patch }); + }; + const save = async () => { + setSaving(true); + setSaveError(undefined); + try { + await onSaveDefault(); + } catch (cause) { + setSaveError(cause instanceof Error ? cause.message : "Could not save the session default."); + } finally { + setSaving(false); + } + }; + const summary = `${choice.modelLabel}, ${effortLabel(choice.reasoningEffort).toLowerCase()} effort`; + + return ( + { setOpen(next); if (!next) setQuery(""); }}> + + }> + + {choice.modelLabel} + · + {effortShortLabel(choice.reasoningEffort)} + + {overridden && } + + +
    +
    + +
    + {models.map((option) => ( + + ))} + {query.trim() && !exact && ( + + )} + {!models.length && !query.trim() &&

    No models were discovered.

    } +
    +
    +
    +

    Reasoning effort

    +
    + {effortOptions.map((effort) => ( + + ))} +
    +
    +
    +
    +

    + {saveError + ? {saveError} + : overridden + ? "Applies to each message you send from here. Steering joins the running run as it is." + : "Matches the session configuration."} +

    + {overridden && ( + + )} + {canSaveDefault && ( + + )} +
    +
    +
    + ); +} + +function Option({ checked, onSelect, children }: { checked: boolean; onSelect: () => void; children: React.ReactNode }) { + return ( + + ); +} diff --git a/platform/web/src/components/session/composer-voice.tsx b/platform/web/src/components/session/composer-voice.tsx new file mode 100644 index 000000000..88591ffdf --- /dev/null +++ b/platform/web/src/components/session/composer-voice.tsx @@ -0,0 +1,152 @@ +import { useEffect, useRef } from "react"; +import { Link } from "react-router-dom"; +import { LoaderCircle, Mic, MicOff, RotateCw, Square, X } from "lucide-react"; +import { MAX_DICTATION_SECONDS } from "@lightspeed/platform-shared"; +import type { useDictation } from "@/lib/use-dictation"; +import { Button } from "@/components/ui/button"; +import { Popover, PopoverContent, PopoverTrigger } from "@/components/ui/popover"; +import { cn } from "@/lib/utils"; + +type Voice = ReturnType; + +const clock = (seconds: number) => `${Math.floor(seconds / 60)}:${String(seconds % 60).padStart(2, "0")}`; +/// The timer turns amber in the last half minute before the automatic stop. +const WARN_AT = MAX_DICTATION_SECONDS - 30; +const BARS = [0.55, 0.85, 1, 0.75, 0.5]; + +/// Dictation lives in one control. Idle, it is a microphone icon; while +/// recording, transcribing, or failed, the icon grows into a pill that +/// carries that state and its actions. Nothing renders outside it. +export function VoiceControl({ voice, unavailableReason, settingsHref, demo, onStart }: { + voice: Voice; + /// Why dictation cannot start (no speech default, insecure origin, …). + unavailableReason?: string; + settingsHref?: string; + demo?: boolean; + onStart: () => void; +}) { + if (voice.phase === "requesting") { + return ( + + + Allow the microphone… + + + ); + } + if (voice.phase === "recording") { + const warn = voice.seconds >= WARN_AT; + return ( + + + + {clock(voice.seconds)} + + + + void voice.stop()}> + + + + ); + } + if (voice.phase === "transcribing") { + return ( + + + Transcribing… + + + ); + } + if (voice.phase === "error") { + return ( + + + {voice.error ?? "Dictation failed."} + {voice.canRetry + ? + : } + + + ); + } + if (unavailableReason) { + return ( + + + }> + + + +

    Dictation is unavailable

    +

    {unavailableReason}

    + {settingsHref && ( + Open Models + )} +
    +
    + ); + } + return ( + + ); +} + +function Pill({ tone, label, children }: { tone: "muted" | "recording" | "error"; label: string; children: React.ReactNode }) { + return ( +
    + {children} +
    + ); +} + +function PillButton({ label, title, onClick, children }: { label: string; title?: string; onClick: () => void; children: React.ReactNode }) { + return ( + + ); +} + +/// Five bars that follow the input level. Updated per animation frame +/// through refs, so a recording does not re-render the composer. +function LevelMeter({ level }: { level: () => number }) { + const bars = useRef>([]); + useEffect(() => { + if (typeof requestAnimationFrame !== "function") return; + let frame = 0; + let smoothed = 0; + const tick = () => { + smoothed = smoothed * 0.6 + level() * 0.4; + bars.current.forEach((bar, index) => { + if (bar) bar.style.transform = `scaleY(${Math.max(0.18, Math.min(1, smoothed * BARS[index]! * 1.6))})`; + }); + frame = requestAnimationFrame(tick); + }; + frame = requestAnimationFrame(tick); + return () => cancelAnimationFrame(frame); + }, [level]); + return ( + + {BARS.map((_, index) => ( + { bars.current[index] = node; }} style={{ transform: "scaleY(0.18)" }} + className="h-full w-0.5 origin-center rounded-full bg-current transition-transform duration-75" /> + ))} + + ); +} diff --git a/platform/web/src/components/session/composer.dictation.test.tsx b/platform/web/src/components/session/composer.dictation.test.tsx new file mode 100644 index 000000000..daaf351b2 --- /dev/null +++ b/platform/web/src/components/session/composer.dictation.test.tsx @@ -0,0 +1,151 @@ +// @vitest-environment jsdom +import { act } from "react"; +import { createRoot, type Root } from "react-dom/client"; +import { afterEach, beforeEach, expect, it, vi } from "vitest"; +import { SessionComposer } from "./composer"; + +const mocks = vi.hoisted(() => ({ capture: vi.fn(), transcribe: vi.fn(), cancel: vi.fn() })); +vi.mock("@/lib/audio-capture", () => ({ startAudioCapture: mocks.capture, isDemoDictation: false })); +vi.mock("@/lib/dictation", () => ({ transcribeRecording: mocks.transcribe, cancelRecording: mocks.cancel })); +vi.mock("@/components/ui/popover", () => import("@/components/ui/popover.test-double")); +let root: Root; +let container: HTMLDivElement; +let finish: (text: string) => void; +let stop: ReturnType; +let onSend: ReturnType; +beforeEach(() => { + vi.stubGlobal("IS_REACT_ACT_ENVIRONMENT", true); + vi.clearAllMocks(); + stop = vi.fn(); + mocks.capture.mockResolvedValue({ stop, name: "dictation.webm", result: Promise.resolve(new Blob(["audio"], { type: "audio/webm" })) }); + mocks.transcribe.mockImplementation(() => new Promise((resolve) => { finish = resolve; })); + onSend = vi.fn(); + container = document.createElement("div"); + document.body.append(container); + root = createRoot(container); +}); +afterEach(async () => { + await act(async () => root.unmount()); + container.remove(); + localStorage.clear(); + vi.unstubAllGlobals(); +}); +async function show(disabledReason?: string, disabled = false) { + await act(async () => root.render()); +} +async function click(label: string) { + const button = [...container.querySelectorAll("button")].find((node) => node.getAttribute("aria-label") === label || node.textContent === label)!; + await act(async () => button.click()); +} +async function type(text: string) { + const input = container.querySelector("textarea")!; + await act(async () => { + Object.getOwnPropertyDescriptor(HTMLTextAreaElement.prototype, "value")!.set!.call(input, text); + input.dispatchEvent(new Event("input", { bubbles: true })); + }); +} +async function record() { await click("Dictate message"); await click("Stop recording"); } +it("appends to the latest edited draft and waits for an explicit send", async () => { + await show(); + await type("Before"); + await record(); + await type("Edited while transcribing."); + await act(async () => finish("Spoken text.")); + expect(container.querySelector("textarea")!.value).toBe("Edited while transcribing. Spoken text."); + expect(localStorage.getItem("voice-test")).toBe("Edited while transcribing. Spoken text."); + expect(container.textContent).toContain("Review and edit before sending"); + expect(onSend).not.toHaveBeenCalled(); + expect(stop).toHaveBeenCalled(); + await click("Send message"); + expect(onSend).toHaveBeenCalledWith({ text: "Edited while transcribing. Spoken text.", attachments: [] }, null); +}); +it("inserts at the caret the field last had, when the text is unchanged", async () => { + await show(); + await type("Hello world"); + const input = container.querySelector("textarea")!; + await act(async () => { + input.focus(); + input.setSelectionRange(5, 5); + input.blur(); + }); + await record(); + await act(async () => finish("big")); + expect(input.value).toBe("Hello big world"); + expect(input.selectionStart).toBe(9); +}); +it("stops and transcribes on Enter while recording instead of sending", async () => { + await show(); + await type("Draft"); + await click("Dictate message"); + expect(container.querySelector('[aria-label^="Recording"]')).not.toBeNull(); + const input = container.querySelector("textarea")!; + await act(async () => input.dispatchEvent(new KeyboardEvent("keydown", { key: "Enter", bubbles: true }))); + expect(stop).toHaveBeenCalled(); + expect(onSend).not.toHaveBeenCalled(); + await act(async () => finish("Spoken.")); + expect(input.value).toBe("Draft Spoken."); +}); +it("discards the recording on Escape", async () => { + await show(); + await click("Dictate message"); + const input = container.querySelector("textarea")!; + await act(async () => input.dispatchEvent(new KeyboardEvent("keydown", { key: "Escape", bubbles: true }))); + expect(stop).toHaveBeenCalled(); + expect(container.querySelector('[aria-label="Dictate message"]')).not.toBeNull(); + expect(mocks.transcribe).not.toHaveBeenCalled(); +}); +it.each(["Cancel dictation", "Send message"])("ignores late completion after %s", async (action) => { + await show(); + await type("Keep this"); + await record(); + await click(action); + expect(mocks.transcribe.mock.calls[0]![2].aborted).toBe(true); + await act(async () => finish("Late text")); + expect(container.querySelector("textarea")!.value).toBe(action === "Send message" ? "" : "Keep this"); + expect(onSend).toHaveBeenCalledTimes(action === "Send message" ? 1 : 0); +}); +it("cancels on navigation without modifying the saved draft", async () => { + await show(); + await type("Saved draft"); + await record(); + await act(async () => root.render(null)); + await act(async () => finish("Late text")); + expect(mocks.transcribe.mock.calls[0]![2].aborted).toBe(true); + expect(localStorage.getItem("voice-test")).toBe("Saved draft"); +}); +it("explains a missing speech default instead of recording", async () => { + await show("Set a speech-to-text default in Models to enable dictation."); + const button = container.querySelector('[aria-label="Dictate message"]')!; + expect(button.getAttribute("aria-disabled")).toBe("true"); + await click("Dictate message"); + expect(mocks.capture).not.toHaveBeenCalled(); + expect(container.textContent).toContain("Set a speech-to-text default in Models"); +}); +it("offers no dictation when the composer is disabled", async () => { + await show(undefined, true); + expect(container.querySelector('[aria-label="Dictate message"]')).toBeNull(); + expect(mocks.capture).not.toHaveBeenCalled(); +}); +it("retains the draft and recording for a transcription retry", async () => { + mocks.transcribe.mockRejectedValueOnce(new Error("Service unavailable")); + await show(); + await type("Draft"); + await record(); + expect(container.textContent).toContain("Service unavailable"); + expect(container.querySelector("textarea")!.value).toBe("Draft"); + await click("Retry transcription"); + expect(mocks.capture).toHaveBeenCalledTimes(1); + expect(mocks.transcribe.mock.calls[1]![1]).toBe(mocks.transcribe.mock.calls[0]![1]); + await act(async () => finish("Retry worked")); + expect(container.querySelector("textarea")!.value).toBe("Draft Retry worked"); +}); +it("explains permission denial without losing text", async () => { + mocks.capture.mockRejectedValueOnce(new DOMException("Denied", "NotAllowedError")); + await show(); + await type("Draft"); + await click("Dictate message"); + expect(container.textContent).toContain("Microphone access was denied"); + expect(container.querySelector("textarea")!.value).toBe("Draft"); + expect(mocks.transcribe).not.toHaveBeenCalled(); +}); diff --git a/platform/web/src/components/session/composer.test.tsx b/platform/web/src/components/session/composer.test.tsx index 2fa0f69e7..ea3d3f404 100644 --- a/platform/web/src/components/session/composer.test.tsx +++ b/platform/web/src/components/session/composer.test.tsx @@ -1,41 +1,177 @@ // @vitest-environment jsdom import { act } from "react"; import { createRoot } from "react-dom/client"; -import { afterEach, expect, it, vi } from "vitest"; +import { afterEach, beforeEach, expect, it, vi } from "vitest"; +import type { ModelOption } from "@/api"; import { SessionComposer } from "./composer"; +const mocks = vi.hoisted(() => ({ api: vi.fn() })); +vi.mock("@/api", async (original) => ({ ...(await original()), api: mocks.api })); +vi.mock("@/components/ui/popover", () => import("@/components/ui/popover.test-double")); + Object.assign(globalThis, { IS_REACT_ACT_ENVIRONMENT: true }); let root: ReturnType; let container: HTMLDivElement; +let uploads: Array<(blobRef: string) => void>; +beforeEach(() => { + uploads = []; + mocks.api.mockReset(); + mocks.api.mockImplementation(() => new Promise((resolve) => { + uploads.push((blobRef) => resolve({ blobs: [{ blobRef, bytes: 3 }] })); + })); +}); afterEach(async () => { await act(async () => root?.unmount()); container?.remove(); localStorage.clear(); }); -async function setup(runActive = true, canSteer = true, disabled = false, text = " Change direction ") { + +const config = { + model: { providerId: "openai", apiKind: "openai:responses", model: "gpt-5.5" }, + generation: { reasoningEffort: "high", maxOutputTokens: 4000 }, +}; +const models: ModelOption[] = ["gpt-5.5", "gpt-5.5-mini"].map((model) => ({ + providerId: "openai", apiKind: "openai:responses", model, displayName: model, + capabilities: { reasoningEfforts: ["low", "medium", "high"] }, source: "provider", fetchedAtMs: 0, +})); + +async function setup({ runActive = true, canSteer = true, text = " Change direction ", onSaveDefault = vi.fn(async () => {}) } = {}) { localStorage.setItem("composer-test", text); container = document.createElement("div"); document.body.append(container); root = createRoot(container); const onSend = vi.fn(); await act(async () => root.render()); - return onSend; + canSteer={canSteer} error={null} onSend={onSend} onStop={vi.fn()} + attachments={{ universeId: "u1", apiKind: "anthropic:messages" }} + model={{ config, models, canSaveDefault: true, onSaveDefault }} />)); + return { onSend, onSaveDefault }; } +const button = (label: string) => + [...document.querySelectorAll("button")].find((node) => node.getAttribute("aria-label") === label || node.textContent === label)!; +const textarea = () => container.querySelector("textarea")!; +async function press(key: string, init: KeyboardEventInit = {}) { + await act(async () => textarea().dispatchEvent(new KeyboardEvent("keydown", { key, bubbles: true, ...init }))); +} +async function attach(...files: File[]) { + const input = container.querySelector('input[type="file"]')!; + Object.defineProperty(input, "files", { value: files, configurable: true }); + await act(async () => input.dispatchEvent(new Event("change", { bubbles: true }))); + // The base64 read is asynchronous; let it reach the upload call. + await act(async () => { await new Promise((resolve) => setTimeout(resolve, 20)); }); +} +const png = (name = "screen.png", size = 3) => new File([new Uint8Array(size)], name, { type: "image/png" }); + it.each([[true, true, "Queue message", "queue"], [false, false, "Send message", null]] as const)( "keeps the primary send action for runActive=%s", async (active, steerable, label, mode) => { - const onSend = await setup(active, steerable); - await act(async () => container.querySelector(`[aria-label="${label}"]`)!.click()); - expect(onSend).toHaveBeenCalledWith("Change direction", mode); - expect(container.querySelector("textarea")!.value).toBe(""); + const { onSend } = await setup({ runActive: active, canSteer: steerable }); + await act(async () => button(label).click()); + expect(onSend).toHaveBeenCalledWith({ text: "Change direction", attachments: [] }, mode); + expect(textarea().value).toBe(""); expect(localStorage.getItem("composer-test")).toBeNull(); }, ); + it("preserves keyboard steering and does not send while composing text", async () => { - const onSend = await setup(); - const input = container.querySelector("textarea")!; - await act(async () => input.dispatchEvent(new KeyboardEvent("keydown", { key: "Enter", isComposing: true, bubbles: true }))); + const { onSend } = await setup(); + await press("Enter", { isComposing: true }); + expect(onSend).not.toHaveBeenCalled(); + await press("Enter", { ctrlKey: true }); + expect(onSend).toHaveBeenCalledWith({ text: "Change direction", attachments: [] }, "steer"); +}); + +it("keeps the draft and explains when nothing can be steered", async () => { + const { onSend } = await setup({ canSteer: false }); + await press("Enter", { metaKey: true }); expect(onSend).not.toHaveBeenCalled(); - await act(async () => input.dispatchEvent(new KeyboardEvent("keydown", { key: "Enter", ctrlKey: true, bubbles: true }))); - expect(onSend).toHaveBeenCalledWith("Change direction", "steer"); + expect(textarea().value).toBe(" Change direction "); + expect(container.textContent).toContain("There is no run to steer right now"); +}); + +it("uploads attachments on pick and sends them with the message", async () => { + const { onSend } = await setup({ runActive: false }); + await attach(png()); + expect(mocks.api).toHaveBeenCalledWith("POST", "/api/v1/universes/u1/attachments", { bytesBase64: btoa("\0\0\0") }, expect.any(AbortSignal)); + expect(container.textContent).toContain("Uploading…"); + await act(async () => uploads[0]!(`sha256:${"a".repeat(64)}`)); + expect(container.textContent).toContain("PNG · 3 B"); + await act(async () => button("Send message").click()); + expect(onSend).toHaveBeenCalledWith({ + text: "Change direction", + attachments: [{ blobRef: `sha256:${"a".repeat(64)}`, mime: "image/png", kind: "image", name: "screen.png", size: 3 }], + }, null); + expect(container.querySelector('[aria-label="Attachments"]')).toBeNull(); +}); + +it("waits for uploads when Enter is pressed early, then sends", async () => { + const { onSend } = await setup({ runActive: false, text: "" }); + await attach(png()); + await press("Enter"); + expect(onSend).not.toHaveBeenCalled(); + await act(async () => uploads[0]!(`sha256:${"b".repeat(64)}`)); + expect(onSend).toHaveBeenCalledWith(expect.objectContaining({ text: "", attachments: [expect.objectContaining({ name: "screen.png" })] }), null); +}); + +it("refuses unsupported and oversized files at pick time", async () => { + await setup({ runActive: false }); + await attach(new File(["x"], "notes.docx", { type: "application/msword" }), png("huge.png", 6 * 1024 * 1024)); + expect(mocks.api).not.toHaveBeenCalled(); + expect(container.textContent).toContain("notes.docx is not a type the model can read."); + expect(container.textContent).toContain("1 more file was not attached."); + await act(async () => button("Dismiss").click()); + await attach(png("huge.png", 6 * 1024 * 1024)); + // Anthropic's per-image limit is lower than the runtime's. + expect(container.textContent).toContain("huge.png is 6.0 MB. PNG files can be up to 5.0 MB."); +}); + +it("blocks sending while an upload has failed until it is removed", async () => { + mocks.api.mockRejectedValueOnce(new Error("Gateway unavailable")); + const { onSend } = await setup({ runActive: false }); + await attach(png()); + expect(container.textContent).toContain("Upload failed"); + await act(async () => button("Send message").click()); + expect(onSend).not.toHaveBeenCalled(); + expect(container.textContent).toContain("Remove or retry the attachments that failed to upload."); + await act(async () => button("Remove screen.png").click()); + await act(async () => button("Send message").click()); + expect(onSend).toHaveBeenCalledWith({ text: "Change direction", attachments: [] }, null); +}); + +it("restores ready attachments with the draft", async () => { + localStorage.setItem("composer-test:attachments", JSON.stringify([ + { id: "a", name: "offer.pdf", mime: "application/pdf", kind: "document", size: 2048, blobRef: `sha256:${"c".repeat(64)}` }, + ])); + await setup({ runActive: false }); + expect(container.textContent).toContain("offer.pdf"); + expect(container.textContent).toContain("PDF · 2 KB"); +}); + +it("sends model choices as run options and saves them as the session default", async () => { + const { onSend, onSaveDefault } = await setup({ runActive: false }); + const pill = [...container.querySelectorAll("button")].find((node) => node.getAttribute("aria-label")?.startsWith("Model:"))!; + expect(pill.textContent).toContain("gpt-5.5"); + expect(pill.textContent).toContain("High"); + await act(async () => button("gpt-5.5-mini").click()); + await act(async () => button("Low").click()); + expect(document.body.textContent).toContain("Applies to each message you send from here."); + await act(async () => button("Save as session default").click()); + expect(onSaveDefault).toHaveBeenCalledWith({ + model: { providerId: "openai", apiKind: "openai:responses", model: "gpt-5.5-mini" }, + generation: { reasoningEffort: "low", maxOutputTokens: 4000 }, + }); + // Saving clears the override; choose again and send with it. + await act(async () => button("gpt-5.5-mini").click()); + await act(async () => button("Send message").click()); + expect(onSend).toHaveBeenCalledWith({ + text: "Change direction", + attachments: [], + options: { model: { providerId: "openai", apiKind: "openai:responses", model: "gpt-5.5-mini" } }, + }, null); +}); + +it("does not send run options with steering", async () => { + localStorage.setItem("composer-test:run", JSON.stringify({ reasoningEffort: "low" })); + const { onSend } = await setup(); + await press("Enter", { ctrlKey: true }); + expect(onSend).toHaveBeenCalledWith({ text: "Change direction", attachments: [] }, "steer"); }); diff --git a/platform/web/src/components/session/composer.tsx b/platform/web/src/components/session/composer.tsx index a0f9c7287..45866b518 100644 --- a/platform/web/src/components/session/composer.tsx +++ b/platform/web/src/components/session/composer.tsx @@ -1,7 +1,25 @@ -import { useState, type KeyboardEvent, type ReactNode } from "react"; -import { ArrowUp, LoaderCircle, Square } from "lucide-react"; +import { useEffect, useRef, useState, type DragEvent, type KeyboardEvent, type ReactNode } from "react"; +import { ArrowUp, ChevronDown, CornerDownRight, ListPlus, LoaderCircle, Lock, Plus, Square, TriangleAlert, X } from "lucide-react"; +import { ATTACHMENT_ACCEPT, ATTACHMENT_SUMMARY, type MessageRunOptions } from "@lightspeed/platform-shared"; +import type { ModelOption } from "@/api"; +import { useDictation } from "@/lib/use-dictation"; +import { isDemoDictation } from "@/lib/audio-capture"; +import { useComposerAttachments, type SentAttachment } from "@/lib/composer-attachments"; +import { + composerModelChoice, + configWithRunChoice, + normalizeRunChoice, + readRunChoice, + writeRunChoice, + type RunChoice, +} from "@/lib/composer-model"; import { Button } from "@/components/ui/button"; +import { DropdownMenu, DropdownMenuContent, DropdownMenuItem, DropdownMenuShortcut, DropdownMenuTrigger } from "@/components/ui/dropdown-menu"; import { readSessionDraft, writeSessionDraft } from "@/lib/sessions/draft"; +import { cn } from "@/lib/utils"; +import { AttachmentStrip } from "./composer-attachments"; +import { ModelPicker } from "./composer-model-picker"; +import { VoiceControl } from "./composer-voice"; /// How a message sent while a run is in progress is delivered. /// - `queue`: starts the next run once the active one (and anything @@ -10,16 +28,31 @@ import { readSessionDraft, writeSessionDraft } from "@/lib/sessions/draft"; /// turn boundary without interrupting the in-flight turn. ⌘/Ctrl+Enter. export type ComposerMode = "steer" | "queue"; +/// What one send carries. Options are per-message model overrides; steering +/// never carries them because it joins a run that already chose. +export interface ComposerMessage { + text: string; + attachments: SentAttachment[]; + options?: MessageRunOptions; +} + const isMac = typeof navigator !== "undefined" && /Mac|iPhone|iPad/.test(navigator.platform); const steerKeyLabel = isMac ? "⌘↵" : "Ctrl+↵"; +const VOICE_BUSY = new Set(["requesting", "recording", "transcribing", "error"]); -/// Chat input pinned under the transcript. Enter sends, Shift+Enter adds a -/// newline. While a run is in progress the input stays live: Enter queues -/// the message as the next run, ⌘/Ctrl+Enter steers it into the current -/// run, and a Stop button cancels the active run. Closed sessions render -/// the composer read-only for transcript inspection. +/// Chat input pinned under the transcript: one capsule that holds the +/// attachments, the message, and a toolbar with attach, the model pill, +/// dictation, and send. Every status the composer has (uploads, recording, +/// transcription, errors) is shown inside the capsule on the control that +/// owns it. Enter sends, Shift+Enter adds a newline. While a run is in +/// progress Enter queues the message as the next run, ⌘/Ctrl+Enter steers +/// it into the current run, and Stop cancels the active run. Closed +/// sessions render the composer read-only for transcript inspection. export function SessionComposer({ draftKey, + dictation, + attachments: attachmentOptions, + model, runActive, canSteer, canStop = false, @@ -28,11 +61,24 @@ export function SessionComposer({ disabledReason, banner, error, + onDismissError, onSend, onStop, }: { /// Stable universe + session storage key; also used as the React key. draftKey: string; + dictation?: { universeId: string; disabledReason?: string; settingsHref?: string }; + /// Enables images and documents. `apiKind` is the session's agent API, + /// which decides some size limits. + attachments?: { universeId: string; apiKind?: string }; + /// Enables the model pill for the session's configured route. + model?: { + config: unknown; + models?: readonly ModelOption[]; + canSaveDefault: boolean; + /// Stores a complete session config with the pill's choice folded in. + onSaveDefault: (config: Record) => Promise; + }; /// A run is running, cancelling, or queued: Enter queues, ⌘/Ctrl+Enter /// steers. runActive: boolean; @@ -45,36 +91,139 @@ export function SessionComposer({ stopping?: boolean; disabled?: boolean; disabledReason?: string; - /// Rendered above the input, inside the composer block (e.g. the - /// managed-session direct-input override). + /// Rendered above the capsule (e.g. the managed-session direct-input + /// override). banner?: ReactNode; error: string | null; - onSend: (text: string, mode: ComposerMode | null) => void; + onDismissError?: () => void; + onSend: (message: ComposerMessage, mode: ComposerMode | null) => void; onStop: () => void; }) { const [text, setText] = useState(() => readSessionDraft(draftKey)); + const textRef = useRef(text); + const textarea = useRef(null); + const fileInput = useRef(null); + /// Where the caret was when the field last lost focus, so dictation lands + /// there if the text has not changed since. + const selection = useRef<{ start: number; end: number; text: string } | null>(null); + const caretAfterInsert = useRef(null); + const [announcement, setAnnouncement] = useState(); + const [notice, setNotice] = useState(); + const [flash, setFlash] = useState(false); + const [dragging, setDragging] = useState(false); + const dragDepth = useRef(0); + const [pendingSubmit, setPendingSubmit] = useState<"send" | "steer" | null>(null); + const updateText = (value: string) => { + textRef.current = value; + setAnnouncement(undefined); setText(value); // Write on input rather than on unmount, so navigation cannot lose edits. writeSessionDraft(draftKey, value); }; + const files = useComposerAttachments(attachmentOptions?.universeId, attachmentOptions?.apiKind, `${draftKey}:attachments`); + const canAttach = Boolean(attachmentOptions) && !disabled; + + const runKey = `${draftKey}:run`; + const [storedChoice, setStoredChoice] = useState(() => readRunChoice(runKey)); + const modelChoice = model ? composerModelChoice(model.config, model.models, storedChoice) : null; + const changeChoice = (next: RunChoice) => { + const normalized = modelChoice ? normalizeRunChoice(next, modelChoice.session) : next; + setStoredChoice(normalized); + writeRunChoice(runKey, normalized); + }; + + const voice = useDictation(dictation?.universeId, Boolean(dictation && !dictation.disabledReason && !disabled), (transcript) => { + const current = textRef.current; + const at = selection.current; + let next: string; + let caret: number; + if (at && at.text === current && at.start < current.length) { + const before = current.slice(0, at.start); + const after = current.slice(at.end); + const lead = before && !/\s$/.test(before) ? " " : ""; + const trail = after && !/^\s/.test(after) ? " " : ""; + next = `${before}${lead}${transcript}${trail}${after}`; + caret = before.length + lead.length + transcript.length; + } else { + next = `${current}${current && !/\s$/.test(current) ? " " : ""}${transcript}`; + caret = next.length; + } + updateText(next); + selection.current = null; + caretAfterInsert.current = caret; + setAnnouncement("Dictation added. Review and edit before sending."); + setFlash(true); + }); + + useEffect(() => { + const caret = caretAfterInsert.current; + if (caret === null) return; + caretAfterInsert.current = null; + const input = textarea.current; + input?.focus(); + input?.setSelectionRange(caret, caret); + }, [text]); + useEffect(() => { + if (!flash) return; + const timer = setTimeout(() => setFlash(false), 1400); + return () => clearTimeout(timer); + }, [flash]); + + const hasContent = text.trim().length > 0 || files.items.length > 0; + const submit = (steer: boolean) => { - const trimmed = text.trim(); - if (!trimmed || disabled) { + if (disabled) return; + // Enter while recording finishes the recording; the transcript still + // needs a review before anything is sent. + if (voice.phase === "recording") { + void voice.stop(); + return; + } + if (!hasContent) return; + if (files.failed) { + setNotice("Remove or retry the attachments that failed to upload."); + return; + } + if (files.uploading) { + setPendingSubmit(steer ? "steer" : "send"); return; } - // ⌘/Ctrl+Enter while nothing can be steered is reported by the page - // (the text stays in the box) rather than silently queued — the - // difference matters to the reader. const mode: ComposerMode | null = !runActive ? null : steer ? "steer" : "queue"; - onSend(trimmed, mode); - if (mode !== "steer" || canSteer) { - updateText(""); + if (mode === "steer" && !canSteer) { + setNotice(`There is no run to steer right now. Press Enter to queue the message instead.`); + return; } + voice.cancel(); + setPendingSubmit(null); + setNotice(undefined); + const attachments = files.take(); + onSend({ + text: text.trim(), + attachments, + ...(mode !== "steer" && modelChoice?.options ? { options: modelChoice.options } : {}), + }, mode); + updateText(""); }; + // A send asked for while uploads were running goes out once they finish. + useEffect(() => { + if (!pendingSubmit || files.uploading) return; + if (!hasContent) { + setPendingSubmit(null); + return; + } + submit(pendingSubmit === "steer"); + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [pendingSubmit, files.uploading]); + const onKeyDown = (event: KeyboardEvent) => { + if (event.key === "Escape" && VOICE_BUSY.has(voice.phase)) { + event.preventDefault(); + voice.cancel(); + return; + } if (event.key !== "Enter" || event.shiftKey || event.nativeEvent.isComposing) { return; } @@ -82,58 +231,212 @@ export function SessionComposer({ submit(event.metaKey || event.ctrlKey); }; + const startVoice = () => { + setAnnouncement(undefined); + void voice.start(); + }; + + const hasFiles = (event: DragEvent) => canAttach && [...event.dataTransfer.types].includes("Files"); + const dropHandlers = { + onDragEnter: (event: DragEvent) => { + if (!hasFiles(event)) return; + dragDepth.current += 1; + setDragging(true); + }, + onDragOver: (event: DragEvent) => { + if (!hasFiles(event)) return; + event.preventDefault(); + event.dataTransfer.dropEffect = "copy"; + }, + onDragLeave: (event: DragEvent) => { + if (!hasFiles(event)) return; + dragDepth.current = Math.max(0, dragDepth.current - 1); + if (dragDepth.current === 0) setDragging(false); + }, + onDrop: (event: DragEvent) => { + if (!hasFiles(event)) return; + event.preventDefault(); + dragDepth.current = 0; + setDragging(false); + files.add([...event.dataTransfer.files]); + textarea.current?.focus(); + }, + }; + + // A disabled composer states its reason as text in the capsule, where the + // toolbar would be; the banner states it instead when there is one. + const reasonLine = disabled && disabledReason && !banner ? disabledReason : undefined; const placeholder = disabled - ? disabledReason ?? "This session is closed" + ? reasonLine ? "" : disabledReason ?? "This session is closed" : runActive ? canSteer - ? `Enter queues the next message · ${steerKeyLabel} steers the current run…` - : "Enter queues the next message…" + ? `Enter queues a follow-up · ${steerKeyLabel} steers the current run…` + : "Enter queues a follow-up…" : "Message the agent…"; + const sendLabel = runActive ? "Queue message" : "Send message"; + const sendTitle = voice.phase === "recording" + ? "Enter stops the recording; review the transcript before sending" + : pendingSubmit + ? "Sends when the uploads finish" + : runActive + ? canSteer ? `Queue as the next run (↵) · steer with ${steerKeyLabel}` : "Queue as the next run (↵)" + : "Send (↵)"; + const notices = [ + error ? { key: "error", tone: "error" as const, text: error, dismiss: onDismissError } : null, + notice ? { key: "notice", tone: "warn" as const, text: notice, dismiss: () => setNotice(undefined) } : null, + files.notice ? { key: "files", tone: "warn" as const, text: files.notice, dismiss: files.dismissNotice } : null, + ].filter((value) => value !== null); + const showToolbar = !disabled || (runActive && canStop); return ( -
    +
    {banner} - {error &&

    {error}

    } - {disabled && disabledReason && !banner && ( -

    {disabledReason}

    - )} -
    -