From 4e2032af8646def1f1ed8b204907a6c2ae10622a Mon Sep 17 00:00:00 2001 From: Peter Wilson Date: Tue, 22 Sep 2026 14:48:43 +0100 Subject: [PATCH 1/6] test(files): specify the Anthropic GA Files API contract The Anthropic flavor of /v1/files serves the files-api-2025-04-14 beta shape, which Anthropic's GA Files API has replaced. These tests state the GA contract: every file object carries expires_at, a listing is {data, next_page} paged by page or read by ids[], after_id and before_id are refused, and a request carrying the beta header gets a 400 on every route. One test drives Anthropic's own SDK through the app: client.files.list() must page past the first page and retrieve_metadata() must return expires_at. The ascending-order check moves to the OpenAI flavor, which it only reached through Anthropic's headers before. The new tests fail until the following commits land. --- tests/integration/test_files_endpoint.py | 192 +++++++++++++++++++++-- 1 file changed, 181 insertions(+), 11 deletions(-) diff --git a/tests/integration/test_files_endpoint.py b/tests/integration/test_files_endpoint.py index 19dbfc2c3b..094f05f1ca 100644 --- a/tests/integration/test_files_endpoint.py +++ b/tests/integration/test_files_endpoint.py @@ -14,16 +14,19 @@ import base64 import json +from collections.abc import Generator from datetime import UTC, datetime, timedelta from pathlib import Path from typing import Any, cast from unittest.mock import patch +import httpx import pytest +from anthropic import Anthropic, BadRequestError from fastapi.testclient import TestClient from sqlalchemy.orm import Session -from gateway.core.config import API_ROOT +from gateway.core.config import API_ROOT, API_VERSION from gateway.models.tools import FileObject from gateway.services.file_extractors import ExtractionResult from gateway.services.file_store import LocalDirFileStore @@ -506,6 +509,8 @@ def test_anthropic_sdk_headers_get_anthropic_shapes( assert meta["mime_type"] == "application/pdf" assert meta["downloadable"] is True assert meta["created_at"].endswith("Z") + assert "expires_at" in meta + assert meta["expires_at"] is None assert "object" not in meta and "bytes" not in meta got = client.get(f"{API_ROOT}/files/{meta['id']}", headers=headers) @@ -515,9 +520,9 @@ def test_anthropic_sdk_headers_get_anthropic_shapes( listed = client.get(f"{API_ROOT}/files", headers=headers) assert listed.status_code == 200 page = listed.json() - assert "object" not in page - assert page["has_more"] is False - assert page["first_id"] == page["last_id"] == meta["id"] + assert set(page) == {"data", "next_page"} + assert page["next_page"] is None + assert [f["id"] for f in page["data"]] == [meta["id"]] # The same file, read with OpenAI's headers, is the OpenAI object. assert client.get(f"{API_ROOT}/files/{meta['id']}", headers=api_key_header).json()["object"] == "file" @@ -551,15 +556,10 @@ def test_list_is_cursor_paged(client: TestClient, api_key_header: dict[str, str] assert sorted(seen) == sorted(ids) assert len(set(seen)) == 3 - # Anthropic's cursor name, ascending, walks the same set the other way. - asc = client.get( - f"{API_ROOT}/files", headers={**api_key_header, **_ANTHROPIC}, params={"limit": 3, "order": "asc"} - ).json() + asc = client.get(f"{API_ROOT}/files", headers=api_key_header, params={"limit": 3, "order": "asc"}).json() assert [f["id"] for f in asc["data"]] == list(reversed(seen)) tail = client.get( - f"{API_ROOT}/files", - headers={**api_key_header, **_ANTHROPIC}, - params={"after_id": asc["data"][0]["id"], "order": "asc"}, + f"{API_ROOT}/files", headers=api_key_header, params={"after": asc["data"][0]["id"], "order": "asc"} ).json() assert [f["id"] for f in tail["data"]] == [f["id"] for f in asc["data"][1:]] @@ -578,6 +578,176 @@ def test_list_is_cursor_paged(client: TestClient, api_key_header: dict[str, str] assert client.get(f"{API_ROOT}/files", headers=api_key_header, params={"limit": 0}).status_code == 422 +def _upload_text_files(client: TestClient, headers: dict[str, str], count: int) -> list[str]: + return [ + client.post(f"{API_ROOT}/files", headers=headers, files={"file": (f"{n}.txt", b"x", "text/plain")}).json()["id"] + for n in range(count) + ] + + +def test_anthropic_listing_pages_with_next_page( + client: TestClient, + api_key_header: dict[str, str], + master_key_header: dict[str, str], + tmp_file_store: None, +) -> None: + headers = {**api_key_header, **_ANTHROPIC} + ids = _upload_text_files(client, headers, 3) + + first = client.get(f"{API_ROOT}/files", headers=headers, params={"limit": 2}).json() + assert set(first) == {"data", "next_page"} + assert len(first["data"]) == 2 + assert first["next_page"].startswith("page_") + + second = client.get(f"{API_ROOT}/files", headers=headers, params={"limit": 2, "page": first["next_page"]}).json() + assert second["next_page"] is None + assert sorted(f["id"] for f in first["data"] + second["data"]) == sorted(ids) + + # A token is opaque, so a file ID or a token this gateway never issued is a bad request. + # page_AA decodes to a NUL character, which must not reach the database. + for page in ("page_nope", "page_é", "page_AA", ids[0]): + refused = client.get(f"{API_ROOT}/files", headers=headers, params={"page": page}) + assert refused.status_code == 400, refused.text + + # A token issued to another user names a file this caller cannot see. + other = client.post( + f"{API_ROOT}/keys", json={"key_name": "other", "user_id": "other-user"}, headers=master_key_header + ) + other_headers = {next(iter(api_key_header)): f"Bearer {other.json()['key']}", **_ANTHROPIC} + _upload_text_files(client, other_headers, 2) + foreign = client.get(f"{API_ROOT}/files", headers=other_headers, params={"limit": 1}).json()["next_page"] + assert client.get(f"{API_ROOT}/files", headers=headers, params={"page": foreign}).status_code == 400 + + +def test_anthropic_listing_reads_named_ids_in_one_page( + client: TestClient, api_key_header: dict[str, str], tmp_file_store: None +) -> None: + headers = {**api_key_header, **_ANTHROPIC} + first, _, third = _upload_text_files(client, headers, 3) + + named = client.get( + f"{API_ROOT}/files", headers=headers, params={"ids[]": [first, third, "file-missing", first]} + ).json() + assert named["next_page"] is None + assert sorted(f["id"] for f in named["data"]) == sorted([first, third]) + + # The cap counts distinct IDs. + repeated = client.get(f"{API_ROOT}/files", headers=headers, params={"ids[]": [first] * 150}) + assert repeated.status_code == 200, repeated.text + assert [f["id"] for f in repeated.json()["data"]] == [first] + + # An ID with a NUL names no file, so it is left out like any other miss. + with_nul = client.get(f"{API_ROOT}/files", headers=headers, params={"ids[]": [first, "file-\x00"]}) + assert with_nul.status_code == 200, with_nul.text + assert [f["id"] for f in with_nul.json()["data"]] == [first] + + for extra in ({"limit": 1}, {"page": "page_x"}): + mixed = client.get(f"{API_ROOT}/files", headers=headers, params={"ids[]": [first], **extra}) + assert mixed.status_code == 400, mixed.text + + too_many = client.get(f"{API_ROOT}/files", headers=headers, params={"ids[]": [f"file-{n}" for n in range(101)]}) + assert too_many.status_code == 400, too_many.text + + +@pytest.mark.parametrize("cursor", ["after_id", "before_id"]) +def test_anthropic_listing_refuses_the_beta_cursors( + client: TestClient, api_key_header: dict[str, str], tmp_file_store: None, cursor: str +) -> None: + (file_id,) = _upload_text_files(client, api_key_header, 1) + refused = client.get(f"{API_ROOT}/files", headers={**api_key_header, **_ANTHROPIC}, params={cursor: file_id}) + assert refused.status_code == 400 + assert "page" in refused.json()["detail"] + + +_FILES_BETA = "files-api-2025-04-14" + + +@pytest.mark.parametrize( + ("method", "path"), + [ + ("POST", "/files"), + ("GET", "/files"), + ("GET", "/files/file-any"), + ("GET", "/files/file-any/content"), + ("DELETE", "/files/file-any"), + ], +) +@pytest.mark.parametrize("anthropic_version", [True, False]) +def test_the_files_beta_header_is_refused( + client: TestClient, + api_key_header: dict[str, str], + tmp_file_store: None, + method: str, + path: str, + anthropic_version: bool, +) -> None: + headers = {**api_key_header, "anthropic-beta": f"code-execution-2025-08-25, {_FILES_BETA}"} + if anthropic_version: + headers.update(_ANTHROPIC) + upload = {"file": ("a.txt", b"x", "text/plain")} if method == "POST" else None + + refused = client.request(method, f"{API_ROOT}{path}", headers=headers, files=upload) + + assert refused.status_code == 400 + assert _FILES_BETA in refused.json()["detail"] + + +def test_other_anthropic_betas_are_served( + client: TestClient, api_key_header: dict[str, str], tmp_file_store: None +) -> None: + headers = {**api_key_header, **_ANTHROPIC, "anthropic-beta": "code-execution-2025-08-25"} + listed = client.get(f"{API_ROOT}/files", headers=headers) + assert listed.status_code == 200, listed.text + assert set(listed.json()) == {"data", "next_page"} + + +@pytest.fixture +def anthropic_sdk(client: TestClient, api_key_obj: dict[str, Any]) -> Generator[Anthropic]: + """Anthropic's SDK, sending its requests through the test app. + + Gotcha: the SDK accepts only an ``httpx.Client``, and ``TestClient`` is built on ``httpx2``. + """ + + def forward(request: httpx.Request) -> httpx.Response: + sent = client.request( + request.method, str(request.url), headers=request.headers.multi_items(), content=request.read() + ) + return httpx.Response(sent.status_code, headers=sent.headers.multi_items(), content=sent.content) + + with Anthropic( + base_url=f"{client.base_url}{API_ROOT.removesuffix(f'/{API_VERSION}')}", + api_key=api_key_obj["key"], + http_client=httpx.Client(transport=httpx.MockTransport(forward)), + max_retries=0, + ) as sdk: + yield sdk + + +def test_anthropic_sdk_files_client_pages_and_reads_expires_at( + anthropic_sdk: Anthropic, tmp_file_store: None, db_session: Session +) -> None: + uploaded = [anthropic_sdk.files.upload(file=(f"{n}.txt", b"x", "text/plain")).id for n in range(3)] + + first = anthropic_sdk.files.list(limit=2) + assert len(first.data) == 2 + assert first.has_next_page() + assert sorted(f.id for f in anthropic_sdk.files.list(limit=2)) == sorted(uploaded) + + expires_at = datetime.now(UTC).replace(microsecond=0) + timedelta(days=1) + record = db_session.get(FileObject, uploaded[0]) + assert record is not None + record.expires_at = expires_at + db_session.commit() + assert anthropic_sdk.files.retrieve_metadata(uploaded[0]).expires_at == expires_at + assert anthropic_sdk.files.retrieve_metadata(uploaded[1]).expires_at is None + + assert anthropic_sdk.files.download(uploaded[1]).read() == b"x" + assert anthropic_sdk.files.delete(uploaded[1]).type == "file_deleted" + + with pytest.raises(BadRequestError, match=_FILES_BETA): + anthropic_sdk.beta.files.list(betas=[_FILES_BETA]) + + def test_sweep_reclaims_expired_and_deleted_files( client: TestClient, api_key_header: dict[str, str], From 166bc1c3a1b15b4d899583365e4d291f9283ddc2 Mon Sep 17 00:00:00 2001 From: Peter Wilson Date: Tue, 22 Sep 2026 14:49:30 +0100 Subject: [PATCH 2/6] feat(files): return expires_at on Anthropic file objects Anthropic's GA Files API puts expires_at on every file object, null when the file does not expire, and its SDK reads it from there. The beta shape the gateway served left it out. Both timestamps now go through one RFC 3339 helper, which keeps reading a naive stored value as UTC. --- src/gateway/models/tools.py | 24 ++++++++++++++---------- 1 file changed, 14 insertions(+), 10 deletions(-) diff --git a/src/gateway/models/tools.py b/src/gateway/models/tools.py index 684e0c9e36..f5a0306480 100644 --- a/src/gateway/models/tools.py +++ b/src/gateway/models/tools.py @@ -37,6 +37,15 @@ def _epoch_seconds(value: datetime | None) -> int | None: return int(value.timestamp()) +def _rfc3339(value: datetime | None) -> str | None: + """Return an RFC 3339 timestamp from a stored datetime, reading a naive value as UTC.""" + if value is None: + return None + if value.tzinfo is None: + value = value.replace(tzinfo=UTC) + return value.isoformat().replace("+00:00", "Z") + + class SearchToolCredential(Base): """A ``POST /v1/search`` tool configured at runtime through the dashboard. @@ -161,24 +170,19 @@ def to_dict(self) -> dict[str, Any]: } def to_anthropic_dict(self) -> dict[str, Any]: - """Convert to the Anthropic Files API ``FileMetadata`` shape. + """Convert to the ``FileMetadata`` shape of Anthropic's GA Files API. - Anthropic's SDK reads ``size_bytes`` and ``mime_type`` where OpenAI's - reads ``bytes`` and nothing, and takes ``created_at`` as an RFC 3339 - string rather than an epoch. ``downloadable`` is always true here: the - gateway serves every stored file's bytes back, unlike Anthropic, which - withholds user uploads. + ``expires_at`` is always present and ``None`` for a file kept indefinitely. + ``downloadable`` is always true, because the gateway serves every stored file's bytes back. """ - created_at = self.created_at - if created_at.tzinfo is None: - created_at = created_at.replace(tzinfo=UTC) return { "id": self.id, "type": "file", "filename": self.filename, "mime_type": self.mime_type, "size_bytes": self.bytes, - "created_at": created_at.isoformat().replace("+00:00", "Z"), + "created_at": _rfc3339(self.created_at), + "expires_at": _rfc3339(self.expires_at), "downloadable": True, } From f5159d63df4b3b722ca5ff32450ae91ad454c1d0 Mon Sep 17 00:00:00 2001 From: Peter Wilson Date: Tue, 22 Sep 2026 14:52:48 +0100 Subject: [PATCH 3/6] feat(files): page the Anthropic listing with next_page and ids[] Anthropic's GA Files API lists files as {data, next_page}: the caller passes next_page back as page, or names up to 100 files with ids[] to read them in one page. The gateway served the beta's {data, has_more, first_id, last_id} paged by after_id, so client.files.list() in Anthropic's SDK stopped after the first page. The Anthropic flavor now answers in the GA shape. Its page token is the last file's ID behind a page_ prefix, encoded so that callers treat it as opaque; a token the gateway did not issue, including one naming another user's file, is a 400. ids[] cannot be combined with page or limit, and after_id and before_id get a 400 as they do on Anthropic without the beta header. The OpenAI flavor keeps its shape and its after cursor. The public artifacts are regenerated for the changed list parameters. --- docs/public/openapi.json | 39 +++++++--- docs/public/otari.postman_collection.json | 14 +++- src/gateway/api/routes/files.py | 89 ++++++++++++++++++++--- web/src/client/schema.ts | 12 +-- 4 files changed, 123 insertions(+), 31 deletions(-) diff --git a/docs/public/openapi.json b/docs/public/openapi.json index ff0af312f8..226c6dcfda 100644 --- a/docs/public/openapi.json +++ b/docs/public/openapi.json @@ -20923,7 +20923,7 @@ }, "/api/v1/files": { "get": { - "description": "List the authenticated user's uploaded files in the request's workspace.\n\n``workspace_id`` narrows a master-key listing to one workspace; a keyed\nrequest is already confined to its key's own and cannot widen or move it.\n\nPages are cursor-based: ``after`` (OpenAI) or ``after_id`` (Anthropic) names\nthe last file of the previous page, and ``has_more`` says whether to ask\nagain. A cursor that has since been deleted or has expired is still a\nposition; one the caller never owned is a 404.", + "description": "List the authenticated user's uploaded files in the request's workspace.\n\n``workspace_id`` narrows a master-key listing to one workspace; a keyed\nrequest is already confined to its key's own and cannot widen or move it.\n\nEach flavor pages with its own cursor.\nOpenAI's ``after`` names the last file of the previous page, and ``has_more`` says whether to ask again.\nAnthropic's ``next_page`` is passed back as ``page``, and ``ids[]`` reads up to 100 named files in one page.\nA cursor whose file has since been deleted or has expired is still a position.\nAn ``after`` the caller never owned is a 404, and a ``page`` token this gateway did not issue is a 400.", "operationId": "files-list_files", "parameters": [ { @@ -21005,7 +21005,21 @@ }, { "in": "query", - "name": "after_id", + "name": "order", + "required": false, + "schema": { + "default": "desc", + "enum": [ + "asc", + "desc" + ], + "title": "Order", + "type": "string" + } + }, + { + "in": "query", + "name": "page", "required": false, "schema": { "anyOf": [ @@ -21016,21 +21030,26 @@ "type": "null" } ], - "title": "After Id" + "title": "Page" } }, { "in": "query", - "name": "order", + "name": "ids[]", "required": false, "schema": { - "default": "desc", - "enum": [ - "asc", - "desc" + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } ], - "title": "Order", - "type": "string" + "title": "Ids[]" } } ], diff --git a/docs/public/otari.postman_collection.json b/docs/public/otari.postman_collection.json index e54dd7ca97..f7b0612634 100644 --- a/docs/public/otari.postman_collection.json +++ b/docs/public/otari.postman_collection.json @@ -1763,7 +1763,7 @@ { "name": "List Files", "request": { - "description": "List the authenticated user's uploaded files in the request's workspace.\n\n``workspace_id`` narrows a master-key listing to one workspace; a keyed\nrequest is already confined to its key's own and cannot widen or move it.\n\nPages are cursor-based: ``after`` (OpenAI) or ``after_id`` (Anthropic) names\nthe last file of the previous page, and ``has_more`` says whether to ask\nagain. A cursor that has since been deleted or has expired is still a\nposition; one the caller never owned is a 404.", + "description": "List the authenticated user's uploaded files in the request's workspace.\n\n``workspace_id`` narrows a master-key listing to one workspace; a keyed\nrequest is already confined to its key's own and cannot widen or move it.\n\nEach flavor pages with its own cursor.\nOpenAI's ``after`` names the last file of the previous page, and ``has_more`` says whether to ask again.\nAnthropic's ``next_page`` is passed back as ``page``, and ``ids[]`` reads up to 100 named files in one page.\nA cursor whose file has since been deleted or has expired is still a position.\nAn ``after`` the caller never owned is a 404, and a ``page`` token this gateway did not issue is a 400.", "header": [], "method": "GET", "url": { @@ -1809,17 +1809,23 @@ { "description": "", "disabled": true, - "key": "after_id", + "key": "order", "value": "" }, { "description": "", "disabled": true, - "key": "order", + "key": "page", + "value": "" + }, + { + "description": "", + "disabled": true, + "key": "ids[]", "value": "" } ], - "raw": "{{baseUrl}}/api/v1/files?user=&purpose=&workspace_id=&limit=&after=&after_id=&order=" + "raw": "{{baseUrl}}/api/v1/files?user=&purpose=&workspace_id=&limit=&after=&order=&page=&ids[]=" } } }, diff --git a/src/gateway/api/routes/files.py b/src/gateway/api/routes/files.py index 7da3c704dd..535488a464 100644 --- a/src/gateway/api/routes/files.py +++ b/src/gateway/api/routes/files.py @@ -20,6 +20,7 @@ else gets the OpenAI file object. """ +import base64 import uuid from collections.abc import AsyncGenerator, AsyncIterator from datetime import UTC, datetime @@ -54,6 +55,9 @@ # OpenAI's 10000 because a page is one query and one JSON body. _DEFAULT_LIST_LIMIT = 100 _MAX_LIST_LIMIT = 1000 +_MAX_LIST_IDS = 100 + +_PAGE_TOKEN_PREFIX = "page_" def _anthropic_shape(raw_request: Request) -> bool: @@ -63,6 +67,54 @@ def _anthropic_shape(raw_request: Request) -> bool: ) +def _page_token(file_id: str) -> str: + """The opaque Anthropic ``next_page`` token that resumes a listing after ``file_id``.""" + return _PAGE_TOKEN_PREFIX + base64.urlsafe_b64encode(file_id.encode()).decode().rstrip("=") + + +def _could_name_a_file(value: str) -> bool: + # Every file ID is printable ASCII, and PostgreSQL rejects a NUL in a text parameter. + return value.isascii() and value.isprintable() + + +def _invalid_page_token() -> HTTPException: + return HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail="Invalid page token") + + +def _page_token_file_id(token: str) -> str: + """The file ID a ``page`` token resumes after, raising a 400 for a token this gateway did not issue.""" + if not token.startswith(_PAGE_TOKEN_PREFIX): + raise _invalid_page_token() + encoded = token.removeprefix(_PAGE_TOKEN_PREFIX) + try: + file_id = base64.urlsafe_b64decode(encoded + "=" * (-len(encoded) % 4)).decode() + except ValueError as exc: + raise _invalid_page_token() from exc + if not _could_name_a_file(file_id): + raise _invalid_page_token() + return file_id + + +def _check_anthropic_list_params(raw_request: Request, page: str | None, ids: list[str] | None) -> None: + """Refuse the parameter combinations Anthropic's GA listing refuses, given de-duplicated ``ids``.""" + params = raw_request.query_params + if "after_id" in params or "before_id" in params: + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail="after_id and before_id are not supported: pass next_page back as page instead", + ) + if ids is None: + return + if page is not None or "limit" in params: + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, detail="ids[] cannot be combined with page or limit" + ) + if len(ids) > _MAX_LIST_IDS: + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, detail=f"ids[] takes at most {_MAX_LIST_IDS} file IDs" + ) + + def _serialize(record: FileObject, raw_request: Request) -> dict[str, Any]: return record.to_anthropic_dict() if _anthropic_shape(raw_request) else record.to_dict() @@ -255,22 +307,31 @@ async def list_files( workspace_id: uuid.UUID | None = None, limit: Annotated[int, Query(ge=1, le=_MAX_LIST_LIMIT)] = _DEFAULT_LIST_LIMIT, after: str | None = None, - after_id: str | None = None, order: Literal["asc", "desc"] = "desc", + page: str | None = None, + ids: Annotated[list[str] | None, Query(alias="ids[]")] = None, ) -> dict[str, Any]: """List the authenticated user's uploaded files in the request's workspace. ``workspace_id`` narrows a master-key listing to one workspace; a keyed request is already confined to its key's own and cannot widen or move it. - Pages are cursor-based: ``after`` (OpenAI) or ``after_id`` (Anthropic) names - the last file of the previous page, and ``has_more`` says whether to ask - again. A cursor that has since been deleted or has expired is still a - position; one the caller never owned is a 404. + Each flavor pages with its own cursor. + OpenAI's ``after`` names the last file of the previous page, and ``has_more`` says whether to ask again. + Anthropic's ``next_page`` is passed back as ``page``, and ``ids[]`` reads up to 100 named files in one page. + A cursor whose file has since been deleted or has expired is still a position. + An ``after`` the caller never owned is a 404, and a ``page`` token this gateway did not issue is a 400. """ if not config.files_enabled: raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="File uploads are disabled") + anthropic = _anthropic_shape(raw_request) + if anthropic: + named_ids = None if ids is None else list(dict.fromkeys(ids)) + _check_anthropic_list_params(raw_request, page, named_ids) + cursor_id = _page_token_file_id(page) if page is not None else None + else: + named_ids, cursor_id = None, after user_id = _resolve_user(auth_result, user, config) # The key's own workspace wins over anything the caller sent, rather than # 400ing on a mismatch: the parameter is a master-key narrowing, and a keyed @@ -289,17 +350,20 @@ async def list_files( stmt = stmt.where(FileObject.workspace_id == scope) if purpose is not None: stmt = stmt.where(FileObject.purpose == purpose) + if named_ids is not None: + stmt = stmt.where(FileObject.id.in_([file_id for file_id in named_ids if _could_name_a_file(file_id)])) - cursor_id = after or after_id if cursor_id is not None: # A position, not a file: the row is read with the tenant predicates # only, so a cursor that was deleted or expired between two pages (the # usual "list, delete each, list again" loop) still says where the next - # page starts. Another user's id stays a 404. + # page starts. Another user's ID is refused. cursor_conditions = [FileObject.id == cursor_id, FileObject.user_id == user_id] if scope is not None: cursor_conditions.append(FileObject.workspace_id == scope) cursor = (await db.execute(select(FileObject).where(*cursor_conditions))).scalar_one_or_none() + if cursor is None and anthropic: + raise _invalid_page_token() if cursor is None: raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="File not found") # (created_at, id) is the sort key, so the page after the cursor is @@ -325,16 +389,17 @@ async def list_files( records = list((await db.execute(stmt.limit(limit + 1))).scalars().all()) has_more = len(records) > limit records = records[:limit] + data = [_serialize(r, raw_request) for r in records] - page: dict[str, Any] = { - "data": [_serialize(r, raw_request) for r in records], + if anthropic: + return {"data": data, "next_page": _page_token(records[-1].id) if has_more else None} + return { + "object": "list", + "data": data, "has_more": has_more, "first_id": records[0].id if records else None, "last_id": records[-1].id if records else None, } - if not _anthropic_shape(raw_request): - page = {"object": "list", **page} - return page @router.get("/files/{file_id}") diff --git a/web/src/client/schema.ts b/web/src/client/schema.ts index fa77beaa88..5643a7cb92 100644 --- a/web/src/client/schema.ts +++ b/web/src/client/schema.ts @@ -1102,10 +1102,11 @@ export interface paths { * ``workspace_id`` narrows a master-key listing to one workspace; a keyed * request is already confined to its key's own and cannot widen or move it. * - * Pages are cursor-based: ``after`` (OpenAI) or ``after_id`` (Anthropic) names - * the last file of the previous page, and ``has_more`` says whether to ask - * again. A cursor that has since been deleted or has expired is still a - * position; one the caller never owned is a 404. + * Each flavor pages with its own cursor. + * OpenAI's ``after`` names the last file of the previous page, and ``has_more`` says whether to ask again. + * Anthropic's ``next_page`` is passed back as ``page``, and ``ids[]`` reads up to 100 named files in one page. + * A cursor whose file has since been deleted or has expired is still a position. + * An ``after`` the caller never owned is a 404, and a ``page`` token this gateway did not issue is a 400. */ get: operations["files-list_files"]; put?: never; @@ -14918,8 +14919,9 @@ export interface operations { workspace_id?: string | null; limit?: number; after?: string | null; - after_id?: string | null; order?: "asc" | "desc"; + page?: string | null; + "ids[]"?: string[] | null; }; header?: never; path?: never; From 0ddf003fbe1683068ec6e3b57ccaeef2f2f3ffa9 Mon Sep 17 00:00:00 2001 From: Peter Wilson Date: Tue, 22 Sep 2026 14:53:53 +0100 Subject: [PATCH 4/6] feat(files): refuse the files-api-2025-04-14 beta header Anthropic still honors the files-api-2025-04-14 beta and answers it in the beta shapes. The gateway serves one Anthropic contract, the GA one, so a files request whose anthropic-beta header names that beta now gets a 400 that says to send the request without it. Anthropic's Python SDK before 1.2.0, including the locked 0.125.0, sends the header from client.beta.files; client.files does not. The check is a dependency on the router, so a route added to it inherits the refusal. The Anthropic flavor is selected by anthropic-version alone, which Anthropic's SDK sends on every call, rather than also by an anthropic-beta value that starts with files-api. --- src/gateway/api/routes/files.py | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/src/gateway/api/routes/files.py b/src/gateway/api/routes/files.py index 535488a464..2fe3cf5a33 100644 --- a/src/gateway/api/routes/files.py +++ b/src/gateway/api/routes/files.py @@ -18,6 +18,7 @@ shape follows the caller: a request carrying Anthropic's ``anthropic-version`` header (which its SDK sends on every call) gets ``FileMetadata``, everything else gets the OpenAI file object. +The Anthropic flavor is its GA shape only, so a request for the Files API beta is a 400. """ import base64 @@ -45,7 +46,20 @@ from gateway.services.files.provider_files import stream_provider_file from gateway.services.workspace_scope import default_workspace_id -router = APIRouter(tags=["files"]) +_FILES_BETA = "files-api-2025-04-14" + + +async def _refuse_files_beta(raw_request: Request) -> None: + """Refuse Anthropic's Files API beta, whose shapes differ from the GA shapes served here.""" + for header_value in raw_request.headers.getlist("anthropic-beta"): + if _FILES_BETA in (beta.strip() for beta in header_value.split(",")): + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail=f"The {_FILES_BETA} beta is not supported: send Files API requests without it in anthropic-beta", + ) + + +router = APIRouter(tags=["files"], dependencies=[Depends(_refuse_files_beta)]) # OpenAI's documented file purposes plus a generic default. We don't enforce the # enum (forward-compat), but normalise the empty case to "user_data". @@ -62,9 +76,7 @@ def _anthropic_shape(raw_request: Request) -> bool: """Whether the caller speaks Anthropic's Files API rather than OpenAI's.""" - return "anthropic-version" in raw_request.headers or any( - beta.strip().startswith("files-api") for beta in raw_request.headers.get("anthropic-beta", "").split(",") - ) + return "anthropic-version" in raw_request.headers def _page_token(file_id: str) -> str: From 3d13cdb8f82b90aa7bb9ff1c40c48cf3c76c3722 Mon Sep 17 00:00:00 2001 From: Peter Wilson Date: Tue, 22 Sep 2026 14:54:12 +0100 Subject: [PATCH 5/6] chore(demo): read the plot through client.files The Anthropic SDK demo read the produced chart through client.beta.files, which sends the files-api-2025-04-14 beta header that the gateway now refuses. client.files speaks the GA Files API the gateway serves. --- demo/code-exec/plot_with_anthropic_sdk.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/demo/code-exec/plot_with_anthropic_sdk.py b/demo/code-exec/plot_with_anthropic_sdk.py index edbb1ba4c8..98ef161556 100644 --- a/demo/code-exec/plot_with_anthropic_sdk.py +++ b/demo/code-exec/plot_with_anthropic_sdk.py @@ -66,8 +66,8 @@ def produced_file_ids(message: Any) -> list[str]: def download(file_id: str, model: str) -> pathlib.Path: """Fetch a produced file from Otari, wherever its bytes actually live.""" - meta = otari.beta.files.retrieve_metadata(file_id) - body = otari.beta.files.download(file_id) + meta = otari.files.retrieve_metadata(file_id) + body = otari.files.download(file_id) OUT_DIR.mkdir(parents=True, exist_ok=True) path = OUT_DIR / f"{model.replace(':', '-').replace('/', '-')}-{meta.filename}" From f40ef1c88cf138d9fb05e47627b65ae24805d6c9 Mon Sep 17 00:00:00 2001 From: Peter Wilson Date: Tue, 22 Sep 2026 14:54:42 +0100 Subject: [PATCH 6/6] docs(files): describe the Anthropic GA shape and the beta refusal The Files API guide described the beta shape and showed client.beta.files. It now names the GA fields, including expires_at, the 400 for the files-api-2025-04-14 header and which SDK releases send it, and how each flavor pages its listing. --- docs/files.md | 34 +++++++++++++++++++++++++--------- 1 file changed, 25 insertions(+), 9 deletions(-) diff --git a/docs/files.md b/docs/files.md index d282944fc8..376305e17a 100644 --- a/docs/files.md +++ b/docs/files.md @@ -65,10 +65,18 @@ The five routes (`POST`/`GET /v1/files`, `GET`/`DELETE /v1/files/{id}`, `GET /v1/files/{id}/content`) share their paths and verbs with both vendors' Files APIs, so either official SDK works against Otari with only its base URL changed. The response shape follows the caller: a request carrying Anthropic's -`anthropic-version` header, which its SDK sends on every call, gets Anthropic's -`FileMetadata` (`type`, `size_bytes`, `mime_type`, `downloadable`, an RFC 3339 -`created_at`); everything else gets the OpenAI file object (`object`, `bytes`, -`purpose`, an epoch `created_at`). +`anthropic-version` header, which its SDK sends on every call, gets the +`FileMetadata` of Anthropic's GA Files API (`type`, `size_bytes`, `mime_type`, +`downloadable`, and RFC 3339 `created_at` and `expires_at`, with `expires_at` +`null` for a file kept indefinitely); everything else gets the OpenAI file +object (`object`, `bytes`, `purpose`, an epoch `created_at`). + +Otari serves Anthropic's GA shapes only. A request whose `anthropic-beta` +header includes `files-api-2025-04-14` gets a 400, because that beta answers in +different shapes. Anthropic's Python SDK before 1.2.0, and earlier releases of +its other SDKs, send that header from `client.beta.files`, so call +`client.files` instead (see Anthropic's +[migration notes](https://platform.claude.com/docs/en/build-with-claude/files#migrate-from-files-api-2025-04-14)). Mind the base URL: Anthropic's SDK appends `/v1` itself, so it takes `http://localhost:8000/api`, while an OpenAI-compatible client takes @@ -78,13 +86,21 @@ Mind the base URL: Anthropic's SDK appends `/v1` itself, so it takes ```python from anthropic import Anthropic client = Anthropic(base_url="http://localhost:8000/api", api_key="") -meta = client.beta.files.upload(file=("report.pdf", open("report.pdf", "rb"), "application/pdf")) -client.beta.files.download(meta.id) # Otari serves every stored file's bytes back +meta = client.files.upload(file=("report.pdf", open("report.pdf", "rb"), "application/pdf")) +client.files.download(meta.id) # Otari serves every stored file's bytes back ``` -Listings are cursor-paged: `limit` (default 100, at most 1000), `after` -(OpenAI) or `after_id` (Anthropic) naming the last file of the previous page, -`order` (`desc` by default), and `has_more`, `first_id`, `last_id` on the page. +Listings are cursor-paged, and each flavor pages with its own vendor's cursor. +Both take `limit` (default 100, at most 1000). + +- OpenAI: `after` names the last file of the previous page, `order` is `desc` + by default, and the page carries `has_more`, `first_id` and `last_id`. +- Anthropic: the page is `{data, next_page}`, and `next_page` goes back as + `page` to get the next one. To read up to 100 known files in one page, name + them with `ids[]`, which cannot be combined with `page` or `limit`; a file you + cannot see is left out. `after_id` and `before_id` get a 400. + +Unlike Anthropic, Otari leaves an expired file out of a listing. ## Files and code execution