diff --git a/docs/migrating-from-openai-chat.md b/docs/migrating-from-openai-chat.md index ec1298e8..902ecfc6 100644 --- a/docs/migrating-from-openai-chat.md +++ b/docs/migrating-from-openai-chat.md @@ -501,7 +501,7 @@ A key with no verdict at all is refused. Malformed input — a wrong JSON type, - The first row, when `system` or `developer`, becomes `Request.system`; a later `system` / `developer` row becomes a `developer` message at that position. - Consecutive `tool` rows become **one** tool message with one `ToolResultPart` per row (`tool_call_id` → `id`, `name` → `name`). - An assistant row's parts come out in a fixed order: `reasoning_content` as a `ThinkingPart`, then `content`, then `refusal`, then `tool_calls` with `arguments` parsed. `content: null` with nothing else is one empty text part (a message is never empty). -- `text` → `TextPart`; `image_url` → `ImagePart` (a data URI becomes inline data; a URL stays a URL); `input_audio` → `AudioPart`; `file` → `DocumentPart`; `refusal` → `RefusalPart`. +- `text` → `TextPart`; `image_url` → `ImagePart` (a data URI becomes inline data; a URL stays a URL); `input_audio` → `AudioPart` (its `format` read as the media type: `wav`, `mp3`/`mpeg`, `ogg`, `opus`, `flac`, `aac`, `aiff`, `webm`; the send raises where the wire cannot carry audio); `file` → `DocumentPart`; `refusal` → `RefusalPart`. - A `prompt_cache_breakpoint` on the system row is `prefix="stable"`; on the last text block of message *N* it is `prefix_until_index=N`. ### Where the round trip is not exact diff --git a/lm15/providers/openai_chat.py b/lm15/providers/openai_chat.py index f5a6eab2..35f32e77 100644 --- a/lm15/providers/openai_chat.py +++ b/lm15/providers/openai_chat.py @@ -246,7 +246,13 @@ def _response_format_to_chat(format_config: dict[str, Any]) -> dict[str, Any]: _INGEST_GROQ_BUILTIN_INVERSE: dict[str, str] = {wire: name for name, wire in _GROQ_BUILTIN_MAP.items()} -_INGEST_AUDIO_MEDIA_TYPES: dict[str, str] = {"wav": "audio/wav", "mp3": "audio/mpeg"} +# MAP-12 rule 4: OpenAI's server takes wav and mp3, Gemini's any audio type, +# and DSPy writes the MIME subtype (mpeg for .mp3). Each format reads as its +# true media type; a builder with no audio slot raises at send (MAP-10). +_INGEST_AUDIO_MEDIA_TYPES: dict[str, str] = { + "wav": "audio/wav", "mp3": "audio/mpeg", "mpeg": "audio/mpeg", "ogg": "audio/ogg", "opus": "audio/opus", + "flac": "audio/flac", "aac": "audio/aac", "aiff": "audio/aiff", "webm": "audio/webm", +} def _ingest_unsupported(provider: str, what: str, why: str) -> UnsupportedFeatureError: diff --git a/tests/test_openai_chat_ingest.py b/tests/test_openai_chat_ingest.py index 8b71be04..8094917c 100644 --- a/tests/test_openai_chat_ingest.py +++ b/tests/test_openai_chat_ingest.py @@ -19,9 +19,10 @@ from lm15 import serde from lm15.compat import OpenAIChatCompat from lm15.errors import UnsupportedFeatureError -from lm15.providers import OpenAIChatLM +from lm15.providers import GeminiLM, OpenAIChatLM from lm15.providers.openai_chat import request_from_openai_chat from lm15.types import ( + AudioPart, CacheConfig, Config, FunctionTool, @@ -241,6 +242,7 @@ def test_breakpoints_map_to_cache_config() -> None: ({"model": "m", "messages": USER, "top_logprobs": 3}, ValueError), ({"model": "m", "messages": USER, "tool_choice": {"type": "function", "function": {"name": "ghost"}}}, ValueError), # INV-031 ({"model": "m", "messages": [{"role": "user", "content": [{"type": "image_url", "image_url": {"url": "data:image/png,notbase64"}}]}]}, ValueError), + ({"model": "m", "messages": [{"role": "user", "content": [{"type": "input_audio", "input_audio": {"data": "QUJD", "format": "midi"}}]}]}, ValueError), ({"model": "m", "messages": [{"role": "user", "content": [{"type": "text", "text": "a", "prompt_cache_breakpoint": {"mode": "explicit"}}, {"type": "text", "text": "b"}]}]}, ValueError), # not last ([], TypeError), ]) @@ -258,6 +260,33 @@ def test_vet_op_refuses_a_non_chat_provider() -> None: assert out == {"canonical_request": {"model": "gpt-5-mini", "messages": [{"role": "user", "parts": [{"type": "text", "text": "Hi"}]}]}} +# ─── input_audio formats (rule 4) ──────────────────────────────────── + +def _audio_body(fmt: str) -> dict: + return {"model": "gemini-3.8-flash", "messages": [{"role": "user", "content": [ + {"type": "text", "text": "Transcribe."}, + {"type": "input_audio", "input_audio": {"data": "T2dnUw==", "format": fmt}}, + ]}]} + + +@pytest.mark.parametrize("fmt, media_type", [ + ("wav", "audio/wav"), ("mp3", "audio/mpeg"), ("mpeg", "audio/mpeg"), ("ogg", "audio/ogg"), + ("opus", "audio/opus"), ("flac", "audio/flac"), ("aac", "audio/aac"), ("aiff", "audio/aiff"), ("webm", "audio/webm"), +]) +def test_input_audio_reads_its_true_media_type(fmt, media_type) -> None: + assert request_from_openai_chat(_audio_body(fmt)) == Request(model="gemini-3.8-flash", messages=( + Message.user((TextPart("Transcribe."), AudioPart(media_type=media_type, data="T2dnUw=="))), + )) + + +def test_ogg_audio_reaches_gemini_inline_and_the_chat_wire_refuses_it() -> None: + req = request_from_openai_chat(_audio_body("ogg")) + sent = json.loads(GeminiLM(api_key="k").build_request(req, stream=False).body) + assert sent["contents"][0]["parts"][1] == {"inlineData": {"mimeType": "audio/ogg", "data": "T2dnUw=="}} + with pytest.raises(UnsupportedFeatureError): + OpenAIChatLM(api_key="k").build_request(req, stream=False) + + # ─── message objects dumped back into history (MAP-12 addendum) ────── SDK_MESSAGE = {"content": "Ok! How can I help?", "refusal": None, "role": "assistant", "annotations": [], "audio": None, "function_call": None, "tool_calls": None}