diff --git a/docs/docs/models/providers/copilot.md b/docs/docs/models/providers/copilot.md index 71f38a45a..a73515af4 100644 --- a/docs/docs/models/providers/copilot.md +++ b/docs/docs/models/providers/copilot.md @@ -266,3 +266,16 @@ uv run examples/copilot/image_upload_probe.py It checks image recognition and URL reuse through Messages/SSE and Responses/SSE and WebSocket, without printing credentials or attachment URLs. + +## Document attachments + +Inline PDFs are supported on Copilot Responses (SSE and WebSocket) and Claude +Messages. Local PDF attachments, including `attach_media` results, are sent +inline without a provider file upload. Responses wraps base64 file bytes in a +MIME data URL; Claude uses its native inline document source. Original history +is retained unchanged. Claude inline text/content document sources are also +allowed. + +This does **not** enable the OpenAI or Anthropic Files APIs. Hosted file IDs and +document URL references remain rejected; attach a local copy instead. Gateway +model, file-size and context limits still apply. diff --git a/publish/fast-agent-acp/pyproject.toml b/publish/fast-agent-acp/pyproject.toml index 3fea8cd1b..e7470d420 100644 --- a/publish/fast-agent-acp/pyproject.toml +++ b/publish/fast-agent-acp/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "fast-agent-acp" -version = "0.6.10" +version = "0.10.41" description = "Convenience launcher that pulls in fast-agent-mcp and exposes the ACP CLI entrypoint." readme = "README.md" license = { text = "Apache-2.0" } @@ -18,7 +18,7 @@ classifiers = [ ] requires-python = ">=3.12,<3.15" dependencies = [ - "fast-agent-mcp==0.6.10", + "fast-agent-mcp==0.10.41", ] [project.urls] diff --git a/publish/hf-inference-acp/pyproject.toml b/publish/hf-inference-acp/pyproject.toml index d36a83902..d03eb51c6 100644 --- a/publish/hf-inference-acp/pyproject.toml +++ b/publish/hf-inference-acp/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "hf-inference-acp" -version = "0.6.10" +version = "0.10.41" description = "Hugging Face inference agent with ACP support, powered by fast-agent-mcp" readme = "README.md" license = { text = "Apache-2.0" } @@ -18,7 +18,7 @@ classifiers = [ ] requires-python = ">=3.12,<3.15" dependencies = [ - "fast-agent-mcp==0.6.10", + "fast-agent-mcp==0.10.41", "huggingface_hub>=1.11.0", ] diff --git a/pyproject.toml b/pyproject.toml index d6ff6d6a1..093ff2d4b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "fast-agent-mcp" -version = "0.10.40" +version = "0.10.41" description = "Code, Build and Evaluate agents - excellent Model and Skills/MCP/ACP/A2A Support" readme = "README.md" license = { file = "LICENSE" } diff --git a/src/fast_agent/llm/provider/copilot/policy.py b/src/fast_agent/llm/provider/copilot/policy.py index aab67be0f..cb844a34f 100644 --- a/src/fast_agent/llm/provider/copilot/policy.py +++ b/src/fast_agent/llm/provider/copilot/policy.py @@ -45,11 +45,26 @@ def contains_type(value: object, types: set[str]) -> bool: def reject_files(value: object) -> None: + """Reject hosted file references, not inline document content.""" if isinstance(value, Mapping): kind = value.get("type") - if kind in ("input_file", "file", "document") or ( - kind == "input_image" and value.get("file_id") + if kind == "input_file" and ( + not value.get("file_data") or value.get("file_id") or value.get("file_url") ): + raise ValueError( + "Copilot files require inline file_data; file IDs/URLs are unsupported." + ) + if kind == "document": + source = value.get("source") + if not isinstance(source, Mapping) or source.get("type") not in ( + "base64", + "text", + "content", + ): + raise ValueError( + "Copilot documents require inline content; file IDs/URLs are unsupported." + ) + if kind == "file" or (kind == "input_image" and value.get("file_id")): raise ValueError("Copilot provider file APIs are not supported.") # Traverse wire content, not arbitrary local tool arguments (which may # legitimately contain a field named file_id or type="file"). diff --git a/src/fast_agent/llm/provider/copilot/responses.py b/src/fast_agent/llm/provider/copilot/responses.py index 4d7432e9f..e4833e05b 100644 --- a/src/fast_agent/llm/provider/copilot/responses.py +++ b/src/fast_agent/llm/provider/copilot/responses.py @@ -32,6 +32,7 @@ resolve_responses_ws_url, ) from fast_agent.llm.provider_types import Provider +from fast_agent.mcp.mime_utils import guess_mime_type if TYPE_CHECKING: from fast_agent.llm.provider.copilot.broker import CopilotEndpoint @@ -158,10 +159,37 @@ async def _acquire_responses_ws_attempt( ) return await super()._acquire_responses_ws_attempt(attempt=attempt, context=context) + async def _normalize_input_part( + self, client: AsyncOpenAI, part: dict[str, Any] + ) -> tuple[dict[str, Any], bool]: + # Keep images with the Copilot attachment normalizer, never the Files API. + if part.get("type") != "input_file": + return part, False + data = part.get("file_data") + if not isinstance(data, str) or data.startswith("data:"): + return part, False + filename = part.get("filename") + mime_type = guess_mime_type(filename) if isinstance(filename, str) else None + return { + **part, + "file_data": f"data:{mime_type or 'application/octet-stream'};base64,{data}", + }, True + async def _normalize_input_files( self, client: AsyncOpenAI, input_items: list[dict[str, Any]] ) -> list[dict[str, Any]]: reject_files(input_items) + # Responses accepts both shorthand strings and lists of content blocks. + # Normalize only block lists and leave the caller's history untouched. + normalized_items: list[dict[str, Any]] = [] + for item in input_items: + updated = dict(item) + for key in ("content", "output"): + content = item.get(key) + if isinstance(content, list): + updated[key], _ = await self._normalize_content_parts(client, content) + normalized_items.append(updated) + input_items = normalized_items endpoint = self._copilot_endpoint.get() display_model = ( f"{endpoint.model_id} [ws]" if endpoint.transport == "websocket" else endpoint.model_id diff --git a/tests/unit/llm/test_copilot_adapters.py b/tests/unit/llm/test_copilot_adapters.py index a392fbfb4..e6dfc799f 100644 --- a/tests/unit/llm/test_copilot_adapters.py +++ b/tests/unit/llm/test_copilot_adapters.py @@ -636,7 +636,7 @@ def test_headers_ignore_arbitrary_tool_inputs() -> None: ) == {"x-initiator": "agent", "copilot-vision-request": "true"} -def test_messages_document_uploads_unsupported(context: Context) -> None: +def test_messages_files_api_uploads_unsupported(context: Context) -> None: llm = CopilotMessagesLLM(context=context, model="claude-sonnet-5") assert not llm.supports_files_api() assert not llm.supports_document_uploads() @@ -1549,3 +1549,181 @@ def client_factory(**kwargs: Any) -> AsyncAnthropic: thinking = next(block for block in assistant["content"] if block["type"] == "thinking") assert thinking["thinking"] == summary assert thinking["signature"] == signature + + +@pytest.mark.parametrize( + "block", + [ + { + "type": "input_file", + "filename": "report.pdf", + "file_data": "data:application/pdf;base64,AA==", + }, + { + "type": "document", + "source": {"type": "base64", "media_type": "application/pdf", "data": "AA=="}, + }, + { + "type": "document", + "source": {"type": "text", "media_type": "text/plain", "data": "report"}, + }, + { + "type": "document", + "source": {"type": "content", "content": [{"type": "text", "text": "report"}]}, + }, + ], +) +def test_inline_documents_do_not_require_files_api(block: dict[str, Any]) -> None: + from fast_agent.llm.provider.copilot.policy import reject_files + + reject_files([{"role": "user", "content": [block]}]) + reject_files([{"role": "user", "content": [{"type": "tool_result", "content": [block]}]}]) + reject_files([{"type": "function_call_output", "output": [block]}]) + + +@pytest.mark.parametrize( + "block", + [ + {"type": "input_file", "file_id": "hosted"}, + {"type": "input_file", "file_url": "https://example.com/report.pdf"}, + {"type": "input_file", "file_data": "AA==", "file_id": "hosted"}, + {"type": "input_file", "file_data": "AA==", "file_url": "https://example.com/report.pdf"}, + {"type": "input_file"}, + {"type": "document", "source": {"type": "file", "file_id": "hosted"}}, + {"type": "document", "source": {"type": "url", "url": "https://example.com/report.pdf"}}, + {"type": "document"}, + {"type": "input_image", "file_id": "hosted"}, + {"type": "file", "file_id": "hosted"}, + { + "type": "document", + "source": {"type": "content", "content": [{"type": "file", "file_id": "hosted"}]}, + }, + ], +) +def test_hosted_documents_still_rejected(block: dict[str, Any]) -> None: + from fast_agent.llm.provider.copilot.policy import reject_files + + with pytest.raises(ValueError, match="file"): + reject_files([{"role": "user", "content": [{"type": "tool_result", "content": [block]}]}]) + + +@pytest.mark.asyncio +@pytest.mark.parametrize("wire", ["messages", "responses"]) +@pytest.mark.parametrize("from_tool", [False, True]) +async def test_inline_pdf_through_parent_loop( + wire: str, + from_tool: bool, + broker: FakeBroker, + context: Context, + monkeypatch: pytest.MonkeyPatch, +) -> None: + import base64 + + from mcp.types import ( + BlobResourceContents, + CallToolRequest, + CallToolRequestParams, + CallToolResult, + EmbeddedResource, + ) + + from fast_agent.types import PromptMessageExtended + + encoded = base64.b64encode(b"%PDF-1.4 synthetic").decode() + resource = EmbeddedResource( + type="resource", + resource=BlobResourceContents( + uri="file:///report.pdf", mime_type="application/pdf", blob=encoded + ), + ) + bodies: list[dict[str, Any]] = [] + + async def respond(request: httpx2.Request) -> httpx2.Response: + # There must be no request to a provider Files API. + assert request.url.path.endswith("/messages" if wire == "messages" else "/responses") + bodies.append(json.loads(await request.aread())) + return httpx2.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=anthropic_events() if wire == "messages" else responses_events(), + ) + + def client_factory(**kwargs: Any) -> AsyncAnthropic | AsyncOpenAI: + kwargs["http_client"] = httpx2.AsyncClient(transport=httpx2.MockTransport(respond)) + return AsyncAnthropic(**kwargs) if wire == "messages" else AsyncOpenAI(**kwargs) + + monkeypatch.setattr( + f"fast_agent.llm.provider.copilot.{wire}.{'AsyncAnthropic' if wire == 'messages' else 'AsyncOpenAI'}", + client_factory, + ) + llm = ( + CopilotMessagesLLM(context=context, model="claude-sonnet-5") + if wire == "messages" + else CopilotResponsesLLM(context=context, model="gpt-6-astra", transport="sse") + ) + if from_tool: + messages = [ + Prompt.user("Read report.pdf"), + Prompt.assistant( + tool_calls={ + "call": CallToolRequest( + method="tools/call", + params=CallToolRequestParams( + name="attach_media", arguments={"source": "report.pdf"} + ), + ) + } + ), + PromptMessageExtended( + role="user", + content=[], + tool_results={ + "call": CallToolResult(content=[resource]), + }, + ), + ] + else: + messages = [Prompt.user("Read this PDF", resource)] + before = [message.model_dump() for message in messages] + result = await llm.generate(messages) + assert result.all_text() == "OK" + assert len(bodies) == 1 + body = json.dumps(bodies[0]) + assert encoded in body + if wire == "responses": + assert f"data:application/pdf;base64,{encoded}" in body + assert '"document"' in body if wire == "messages" else '"input_file"' in body + assert '"file_id"' not in body + assert [message.model_dump() for message in messages] == before + + +@pytest.mark.asyncio +@pytest.mark.parametrize("slot", ["content", "output"]) +@pytest.mark.parametrize("prefixed", [False, True]) +async def test_responses_inline_file_normalization_is_idempotent( + slot: str, prefixed: bool, broker: FakeBroker, context: Context +) -> None: + llm = CopilotResponsesLLM(context=context, model="gpt-6-astra", transport="sse") + await llm._prepare_responses_client("gpt-6-astra", "sse") + data_url = "data:application/pdf;base64,AA==" + original = [ + { + "type": "message" if slot == "content" else "function_call_output", + slot: [ + { + "type": "input_file", + "filename": "report.pdf", + "file_data": data_url if prefixed else "AA==", + }, + {"type": "input_image", "image_url": "data:image/png;base64,AA=="}, + ], + } + ] + before = deepcopy(original) + client = llm._responses_client() + once = await llm._normalize_input_files(client, original) + twice = await llm._normalize_input_files(client, once) + assert once[0][slot][0]["file_data"] == data_url + assert once[0][slot][1] == before[0][slot][1] + assert twice == once + assert original == before diff --git a/uv.lock b/uv.lock index 984f7dcdf..57105899a 100644 --- a/uv.lock +++ b/uv.lock @@ -900,7 +900,7 @@ wheels = [ [[package]] name = "fast-agent-acp" -version = "0.6.10" +version = "0.10.41" source = { editable = "publish/fast-agent-acp" } dependencies = [ { name = "fast-agent-mcp" }, @@ -911,7 +911,7 @@ requires-dist = [{ name = "fast-agent-mcp", editable = "." }] [[package]] name = "fast-agent-mcp" -version = "0.10.40" +version = "0.10.41" source = { editable = "." } dependencies = [ { name = "a2a-sdk" }, @@ -1372,7 +1372,7 @@ wheels = [ [[package]] name = "hf-inference-acp" -version = "0.6.10" +version = "0.10.41" source = { editable = "publish/hf-inference-acp" } dependencies = [ { name = "fast-agent-mcp" },