From a6a32b4e6e0ce69326609666252446a4abe82f36 Mon Sep 17 00:00:00 2001 From: Minh Nguyen Cong Date: Mon, 5 Oct 2026 16:08:44 +0200 Subject: [PATCH 1/7] feat(boxsdk): Replace requests-toolbelt with streaming MultipartStream encoder Back the legacy boxsdk MultipartStream with the streaming encoder from box_sdk_gen.networking.multipart_stream, keeping its (data, files) constructor and data-before-files ordering. Seekable file streams are now sent with an exact Content-Length. This removes the requests-toolbelt dependency from the combined SDK. Co-Authored-By: Claude Opus 5.5 (1M context) --- boxsdk/util/multipart_stream.py | 41 +++++++++++------ docs/boxsdk/usage/files.md | 2 +- setup.py | 1 - test/boxsdk/unit/session/test_session.py | 30 +++++++------ .../boxsdk/unit/util/test_multipart_stream.py | 45 +++++++++++++++++-- 5 files changed, 88 insertions(+), 31 deletions(-) diff --git a/boxsdk/util/multipart_stream.py b/boxsdk/util/multipart_stream.py index 10c53bf24..1fe5e23d4 100644 --- a/boxsdk/util/multipart_stream.py +++ b/boxsdk/util/multipart_stream.py @@ -1,19 +1,34 @@ -from collections import OrderedDict +from io import BytesIO +from typing import Any -from requests_toolbelt.multipart.encoder import MultipartEncoder +from box_sdk_gen.networking.multipart_stream import ( + MultipartField, + MultipartStream as _MultipartStream, +) -class MultipartStream(MultipartEncoder): +class MultipartStream(_MultipartStream): """ - Subclass of the requests_toolbelt's :class:`MultipartEncoder` that ensures that data - is encoded before files. This allows a server to process information in the data before - receiving the file bytes. + Streaming multipart/form-data body that ensures that data is encoded before files. + This allows a server to process information in the data before receiving the file bytes. + File streams are read lazily, so uploads are sent without loading whole files in memory. + + Field values are either a string, bytes or a binary stream, or a + (file_name, value) or (file_name, value, content_type) tuple. """ - def __init__(self, data, files): - fields = OrderedDict() - for k in data: - fields[k] = data[k] - for k in files: - fields[k] = files[k] - super().__init__(fields) + def __init__(self, data: dict, files: dict): + super().__init__( + [self._to_field(name, value) for name, value in data.items()] + + [self._to_field(name, value) for name, value in files.items()] + ) + + @staticmethod + def _to_field(name: str, value: Any) -> MultipartField: + file_name, content_type = None, None + if isinstance(value, tuple): + file_name, value, *rest = value + content_type = rest[0] if rest else None + if isinstance(value, bytes): + value = BytesIO(value) + return name, file_name, value, content_type diff --git a/docs/boxsdk/usage/files.md b/docs/boxsdk/usage/files.md index ea96c3eff..7a5f89ff3 100644 --- a/docs/boxsdk/usage/files.md +++ b/docs/boxsdk/usage/files.md @@ -197,7 +197,7 @@ which controls how the file content is uploaded. If you are uploading a large file, you may want to stream the request to avoid excessive memory usage. According to `requests'` library [docs][request_docs], by default, the `requests` library does not support streaming uploads, and all the data must be read into memory before being sent to the server. -However, the `requests-toolbelt` package includes a `MultipartEncoder` class, which enables file uploads without +However, the Box Python SDK includes a streaming multipart encoder, which enables file uploads without loading the entire file into memory. This approach is the default in the Box Python SDK. That said, handling 307 Temporary Redirects presents a challenge with streamed file uploads. diff --git a/setup.py b/setup.py index 88cc8843d..d5313e66b 100644 --- a/setup.py +++ b/setup.py @@ -56,7 +56,6 @@ def main(): 'urllib3', 'dataclasses', 'requests<3', - 'requests-toolbelt<2', 'python-dateutil', ] redis_requires = ['redis>=2.10.3'] diff --git a/test/boxsdk/unit/session/test_session.py b/test/boxsdk/unit/session/test_session.py index 932b42650..d54ef53d9 100644 --- a/test/boxsdk/unit/session/test_session.py +++ b/test/boxsdk/unit/session/test_session.py @@ -1,5 +1,5 @@ from functools import partial -from io import IOBase, BytesIO +from io import IOBase, BytesIO, SEEK_END from numbers import Number import os from unittest.mock import MagicMock, Mock, PropertyMock, call, patch, ANY @@ -8,8 +8,6 @@ SSLError, ConnectionError as RequestsConnectionError, ) -from requests_toolbelt import MultipartEncoder - import pytest from boxsdk import CCGAuth @@ -19,6 +17,7 @@ from boxsdk.network.default_network import DefaultNetwork, DefaultNetworkResponse from boxsdk.session.box_response import BoxResponse from boxsdk.session.session import Session, Translator, AuthorizedSession +from boxsdk.util.multipart_stream import MultipartStream @pytest.fixture(scope='function', params=[False, True]) @@ -271,12 +270,10 @@ def test_box_session_seeks_file_after_retry( assert box_response.ok == generic_successful_response.ok mock_file_1.tell.assert_called_with() mock_file_2.tell.assert_called_with() - mock_file_1.seek.assert_called_with(0) - assert mock_file_1.seek.call_count == 2 - mock_file_1.seek.assert_has_calls([call(0), call(0)]) - mock_file_2.seek.assert_called_with(3) - assert mock_file_2.seek.call_count == 2 - mock_file_2.seek.assert_has_calls([call(3), call(3)]) + # before each attempt the session rewinds the stream, then the multipart + # encoder measures its size and restores the position + assert mock_file_1.seek.call_args_list == [call(0), call(0, SEEK_END), call(0)] * 2 + assert mock_file_2.seek.call_args_list == [call(3), call(0, SEEK_END), call(3)] * 2 def test_box_session_raises_for_non_json_response( @@ -645,7 +642,14 @@ def test_multipart_request_with_enabled_streaming_file_content( assert call_args[1] == test_url assert call_kwargs['access_token'] == 'fake_access_token' assert call_kwargs['log_response_content'] is True - assert isinstance(call_kwargs['data'], MultipartEncoder) - assert call_kwargs['data'].fields['attributes'] == '{"name": "test_file"}' - assert call_kwargs['data'].fields['file'][0] == 'unused' - assert isinstance(call_kwargs['data'].fields['file'][1], BytesIO) + multipart_stream = call_kwargs['data'] + assert isinstance(multipart_stream, MultipartStream) + assert call_kwargs['headers']['Content-Type'] == multipart_stream.content_type + body = multipart_stream.read() + assert ( + b'name="attributes"\r\n\r\n{"name": "test_file"}\r\n' + + f'--{multipart_stream.boundary}\r\n'.encode() + + b'Content-Disposition: form-data; name="file"; filename="unused"\r\n\r\n' + + file_bytes + + b'\r\n' + ) in body diff --git a/test/boxsdk/unit/util/test_multipart_stream.py b/test/boxsdk/unit/util/test_multipart_stream.py index d872fb1c5..83be5de99 100644 --- a/test/boxsdk/unit/util/test_multipart_stream.py +++ b/test/boxsdk/unit/util/test_multipart_stream.py @@ -1,3 +1,5 @@ +from io import BytesIO + import pytest from boxsdk.util.multipart_stream import MultipartStream @@ -17,10 +19,8 @@ def test_multipart_stream_orders_data_before_files( multipart_stream_data, multipart_stream_files ): # pylint:disable=redefined-outer-name - if not multipart_stream_data and not multipart_stream_files: - pytest.xfail('Encoder does not support empty fields.') stream = MultipartStream(multipart_stream_data, multipart_stream_files) - encoded_stream = stream.to_string() + encoded_stream = stream.read() data_indices = [ encoded_stream.find(value) for value in multipart_stream_data.values() ] @@ -30,3 +30,42 @@ def test_multipart_stream_orders_data_before_files( assert -1 not in data_indices assert -1 not in file_indices assert all(all(data_index < f for f in file_indices) for data_index in data_indices) + assert len(encoded_stream) == stream.len + + +def test_multipart_stream_encodes_data_and_file_tuples(): + stream = MultipartStream( + {'attributes': '{"name": "test_file"}'}, + { + 'file': ('unused', BytesIO(b'file content')), + 'pic': ('avatar.png', BytesIO(b'png bytes'), 'image/png'), + }, + ) + + assert stream.content_type == f'multipart/form-data; boundary={stream.boundary}' + assert ( + stream.read() + == ( + f'--{stream.boundary}\r\n' + 'Content-Disposition: form-data; name="attributes"\r\n\r\n' + '{"name": "test_file"}\r\n' + f'--{stream.boundary}\r\n' + 'Content-Disposition: form-data; name="file"; filename="unused"\r\n\r\n' + 'file content\r\n' + f'--{stream.boundary}\r\n' + 'Content-Disposition: form-data; name="pic"; filename="avatar.png"\r\n' + 'Content-Type: image/png\r\n\r\n' + 'png bytes\r\n' + f'--{stream.boundary}--\r\n' + ).encode() + ) + + +def test_multipart_stream_reads_file_lazily(): + file_stream = BytesIO(b'file content') + file_stream.read(5) + + stream = MultipartStream({}, {'file': ('unused', file_stream)}) + + assert file_stream.tell() == 5 + assert b'\r\n\r\ncontent\r\n' in stream.read() From aba3039b5810dadb4eba420e714137503a24472c Mon Sep 17 00:00:00 2001 From: Minh Nguyen Cong Date: Mon, 5 Oct 2026 16:36:25 +0200 Subject: [PATCH 2/7] style: Apply black 26.10 formatting to test_jwt_auth.py The format-check job installs the latest black, which now wraps this yield statement. Co-Authored-By: Claude Opus 5.5 (1M context) --- test/boxsdk/unit/auth/test_jwt_auth.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/test/boxsdk/unit/auth/test_jwt_auth.py b/test/boxsdk/unit/auth/test_jwt_auth.py index e22d8b8b5..720df5776 100644 --- a/test/boxsdk/unit/auth/test_jwt_auth.py +++ b/test/boxsdk/unit/auth/test_jwt_auth.py @@ -208,7 +208,12 @@ def _jwt_auth_init_mocks(**kwargs): backend=default_backend(), ) - yield oauth, assertion, fake_client_id, load_pem_private_key.return_value + yield ( + oauth, + assertion, + fake_client_id, + load_pem_private_key.return_value, + ) if assert_authed: mock_box_session.request.assert_called_once_with( From c3cfa447efbbe7ec7639d79f3ca45998f2aebe56 Mon Sep 17 00:00:00 2001 From: Minh Nguyen Cong Date: Mon, 5 Oct 2026 16:48:08 +0200 Subject: [PATCH 3/7] fix(boxsdkgen): Encode text streams and prefer stream-reported length in MultipartStream Sync the generated encoder and its tests with box/box-codegen#995: - text streams such as io.StringIO are encoded as UTF-8 and sent chunked - a length the stream reports through __len__ or len is used before seeking to its end, like requests and requests-toolbelt The legacy functional tests upload through a mocked file that reports a length but can't seek, which was measured as empty and sent no content. Co-Authored-By: Claude Opus 5.5 (1M context) --- box_sdk_gen/networking/multipart_stream.py | 22 +++++++--- test/box_sdk_gen/test/box_network_client.py | 47 ++++++++++++++++++++- 2 files changed, 62 insertions(+), 7 deletions(-) diff --git a/box_sdk_gen/networking/multipart_stream.py b/box_sdk_gen/networking/multipart_stream.py index 22c240cd1..ae4bcc24d 100644 --- a/box_sdk_gen/networking/multipart_stream.py +++ b/box_sdk_gen/networking/multipart_stream.py @@ -1,4 +1,4 @@ -from io import SEEK_END +from io import SEEK_END, TextIOBase from typing import Iterator, List, Optional, Tuple, Union from urllib3.fields import RequestField @@ -17,7 +17,8 @@ class MultipartStream: so uploads are sent without buffering whole files in memory. Fields are (name, file_name, value, content_type) tuples, where value is - either a string or a binary stream read from its current position. + either a string or a stream read from its current position. Text streams + are encoded as UTF-8. """ def __init__(self, fields: List[MultipartField]): @@ -51,12 +52,19 @@ def __init__(self, fields: List[MultipartField]): @staticmethod def _stream_size(stream: ByteStream) -> Optional[int]: try: - if not stream.seekable(): + # text stream positions count characters, not encoded bytes + if isinstance(stream, TextIOBase) or not stream.seekable(): return None position = stream.tell() - stream.seek(0, SEEK_END) - end = stream.tell() - stream.seek(position) + # like requests and requests-toolbelt, prefer a length the stream reports + if hasattr(stream, '__len__'): + end = len(stream) + elif getattr(stream, 'len', None) is not None: + end = stream.len + else: + stream.seek(0, SEEK_END) + end = stream.tell() + stream.seek(position) # a stream positioned past its end has nothing left to send return max(0, end - position) except (OSError, AttributeError, TypeError): @@ -103,6 +111,8 @@ def read(self, size: Optional[int] = -1) -> bytes: raise self.size_error self._index += 1 continue + if isinstance(chunk, str): + chunk = chunk.encode('utf-8') if remaining is not None: self._remaining[self._index] = remaining - len(chunk) chunks.append(chunk) diff --git a/test/box_sdk_gen/test/box_network_client.py b/test/box_sdk_gen/test/box_network_client.py index 0af0c65c2..e998b1c3b 100644 --- a/test/box_sdk_gen/test/box_network_client.py +++ b/test/box_sdk_gen/test/box_network_client.py @@ -2,7 +2,7 @@ import json import threading from http.server import BaseHTTPRequestHandler, HTTPServer -from io import BytesIO, RawIOBase, UnsupportedOperation, SEEK_END, SEEK_SET +from io import BytesIO, RawIOBase, StringIO, UnsupportedOperation, SEEK_END, SEEK_SET from unittest import mock from unittest.mock import Mock, patch from requests import Session, Response, RequestException @@ -1513,3 +1513,48 @@ def seek(self, *args): assert len(body) == multipart_stream.len assert b"\r\n\r\n23456789\r\n" in body + + +def test_multipart_upload_text_stream_is_encoded_as_utf8(multipart_server): + url, received, _ = multipart_server + + BoxNetworkClient().fetch(_upload_options(url, StringIO("héllo wörld"))) + + assert len(received) == 1 + headers, body = received[0] + assert headers["Transfer-Encoding"] == "chunked" + assert "Content-Length" not in headers + assert "héllo wörld\r\n".encode("utf-8") in body + + +def test_multipart_stream_encodes_text_file_as_utf8(tmp_path): + path = tmp_path / "file.txt" + path.write_text("héllo wörld\n" * 10000, encoding="utf-8") + + with open(path, "r", encoding="utf-8") as text_file: + multipart_stream = MultipartStream([("file", "file.txt", text_file, None)]) + body = multipart_stream.read() + + assert multipart_stream.len is None + assert body.endswith( + b"\r\n\r\n" + + ("héllo wörld\n" * 10000).encode("utf-8") + + f"\r\n--{multipart_stream.boundary}--\r\n".encode() + ) + + +def test_multipart_stream_uses_length_reported_by_stream(): + class ReportedLength(BytesIO): + len = 10 + + def seek(self, *args): + raise AssertionError("size should come from the reported length") + + stream = ReportedLength(b"0123456789") + stream.read(2) + multipart_stream = MultipartStream([("file", "f", stream, None)]) + + body = multipart_stream.read() + + assert len(body) == multipart_stream.len + assert b"\r\n\r\n23456789\r\n" in body From b76b7860a3aeca775a63a6459db58c39d71570f8 Mon Sep 17 00:00:00 2001 From: Minh Nguyen Cong Date: Mon, 5 Oct 2026 16:57:51 +0200 Subject: [PATCH 4/7] Revert "fix(boxsdkgen): Encode text streams and prefer stream-reported length in MultipartStream" This reverts commit c3cfa44. The generated files come to combined-sdk through the box/box-codegen#995 sync PR instead. Co-Authored-By: Claude Opus 5.5 (1M context) --- box_sdk_gen/networking/multipart_stream.py | 22 +++------- test/box_sdk_gen/test/box_network_client.py | 47 +-------------------- 2 files changed, 7 insertions(+), 62 deletions(-) diff --git a/box_sdk_gen/networking/multipart_stream.py b/box_sdk_gen/networking/multipart_stream.py index ae4bcc24d..22c240cd1 100644 --- a/box_sdk_gen/networking/multipart_stream.py +++ b/box_sdk_gen/networking/multipart_stream.py @@ -1,4 +1,4 @@ -from io import SEEK_END, TextIOBase +from io import SEEK_END from typing import Iterator, List, Optional, Tuple, Union from urllib3.fields import RequestField @@ -17,8 +17,7 @@ class MultipartStream: so uploads are sent without buffering whole files in memory. Fields are (name, file_name, value, content_type) tuples, where value is - either a string or a stream read from its current position. Text streams - are encoded as UTF-8. + either a string or a binary stream read from its current position. """ def __init__(self, fields: List[MultipartField]): @@ -52,19 +51,12 @@ def __init__(self, fields: List[MultipartField]): @staticmethod def _stream_size(stream: ByteStream) -> Optional[int]: try: - # text stream positions count characters, not encoded bytes - if isinstance(stream, TextIOBase) or not stream.seekable(): + if not stream.seekable(): return None position = stream.tell() - # like requests and requests-toolbelt, prefer a length the stream reports - if hasattr(stream, '__len__'): - end = len(stream) - elif getattr(stream, 'len', None) is not None: - end = stream.len - else: - stream.seek(0, SEEK_END) - end = stream.tell() - stream.seek(position) + stream.seek(0, SEEK_END) + end = stream.tell() + stream.seek(position) # a stream positioned past its end has nothing left to send return max(0, end - position) except (OSError, AttributeError, TypeError): @@ -111,8 +103,6 @@ def read(self, size: Optional[int] = -1) -> bytes: raise self.size_error self._index += 1 continue - if isinstance(chunk, str): - chunk = chunk.encode('utf-8') if remaining is not None: self._remaining[self._index] = remaining - len(chunk) chunks.append(chunk) diff --git a/test/box_sdk_gen/test/box_network_client.py b/test/box_sdk_gen/test/box_network_client.py index e998b1c3b..0af0c65c2 100644 --- a/test/box_sdk_gen/test/box_network_client.py +++ b/test/box_sdk_gen/test/box_network_client.py @@ -2,7 +2,7 @@ import json import threading from http.server import BaseHTTPRequestHandler, HTTPServer -from io import BytesIO, RawIOBase, StringIO, UnsupportedOperation, SEEK_END, SEEK_SET +from io import BytesIO, RawIOBase, UnsupportedOperation, SEEK_END, SEEK_SET from unittest import mock from unittest.mock import Mock, patch from requests import Session, Response, RequestException @@ -1513,48 +1513,3 @@ def seek(self, *args): assert len(body) == multipart_stream.len assert b"\r\n\r\n23456789\r\n" in body - - -def test_multipart_upload_text_stream_is_encoded_as_utf8(multipart_server): - url, received, _ = multipart_server - - BoxNetworkClient().fetch(_upload_options(url, StringIO("héllo wörld"))) - - assert len(received) == 1 - headers, body = received[0] - assert headers["Transfer-Encoding"] == "chunked" - assert "Content-Length" not in headers - assert "héllo wörld\r\n".encode("utf-8") in body - - -def test_multipart_stream_encodes_text_file_as_utf8(tmp_path): - path = tmp_path / "file.txt" - path.write_text("héllo wörld\n" * 10000, encoding="utf-8") - - with open(path, "r", encoding="utf-8") as text_file: - multipart_stream = MultipartStream([("file", "file.txt", text_file, None)]) - body = multipart_stream.read() - - assert multipart_stream.len is None - assert body.endswith( - b"\r\n\r\n" - + ("héllo wörld\n" * 10000).encode("utf-8") - + f"\r\n--{multipart_stream.boundary}--\r\n".encode() - ) - - -def test_multipart_stream_uses_length_reported_by_stream(): - class ReportedLength(BytesIO): - len = 10 - - def seek(self, *args): - raise AssertionError("size should come from the reported length") - - stream = ReportedLength(b"0123456789") - stream.read(2) - multipart_stream = MultipartStream([("file", "f", stream, None)]) - - body = multipart_stream.read() - - assert len(body) == multipart_stream.len - assert b"\r\n\r\n23456789\r\n" in body From 33c7da263c4150a78034384c36e9e656cafa9dfb Mon Sep 17 00:00:00 2001 From: Minh Nguyen Cong Date: Mon, 5 Oct 2026 17:58:43 +0200 Subject: [PATCH 5/7] test: Count only session seeks in retry seek test Patch out MultipartStream, which sizes file streams itself, so the test checks that the session rewinds file streams before each attempt without depending on how the generated encoder measures them. Co-Authored-By: Claude Opus 5.5 (1M context) --- test/boxsdk/unit/session/test_session.py | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/test/boxsdk/unit/session/test_session.py b/test/boxsdk/unit/session/test_session.py index d54ef53d9..c865ac36c 100644 --- a/test/boxsdk/unit/session/test_session.py +++ b/test/boxsdk/unit/session/test_session.py @@ -1,5 +1,5 @@ from functools import partial -from io import IOBase, BytesIO, SEEK_END +from io import IOBase, BytesIO from numbers import Number import os from unittest.mock import MagicMock, Mock, PropertyMock, call, patch, ANY @@ -264,16 +264,20 @@ def test_box_session_seeks_file_after_retry( mock_file_2.tell.return_value = 3 files = {'file': ('unused', mock_file_1), 'f2': ('unused', mock_file_2)} - box_response = box_session.post(url=test_url, files=files) + # the multipart encoder sizes the streams itself, so only count the session's seeks + with patch('boxsdk.session.session.MultipartStream'): + box_response = box_session.post(url=test_url, files=files) assert box_response.status_code == 200 assert box_response.json() == generic_successful_response.json() assert box_response.ok == generic_successful_response.ok mock_file_1.tell.assert_called_with() mock_file_2.tell.assert_called_with() - # before each attempt the session rewinds the stream, then the multipart - # encoder measures its size and restores the position - assert mock_file_1.seek.call_args_list == [call(0), call(0, SEEK_END), call(0)] * 2 - assert mock_file_2.seek.call_args_list == [call(3), call(0, SEEK_END), call(3)] * 2 + mock_file_1.seek.assert_called_with(0) + assert mock_file_1.seek.call_count == 2 + mock_file_1.seek.assert_has_calls([call(0), call(0)]) + mock_file_2.seek.assert_called_with(3) + assert mock_file_2.seek.call_count == 2 + mock_file_2.seek.assert_has_calls([call(3), call(3)]) def test_box_session_raises_for_non_json_response( From a891485a6e5217730b1d3e01035d17bcdbc275a3 Mon Sep 17 00:00:00 2001 From: Minh Nguyen Cong Date: Mon, 5 Oct 2026 19:49:58 +0200 Subject: [PATCH 6/7] test(boxsdkgen): Temporarily sync MultipartStream from box/box-codegen#995 Sync the generated encoder and its tests with box/box-codegen#995 (029716b5) to run CI against them. To be reverted once CI passes; the generated files come to combined-sdk through the #995 sync PR. Co-Authored-By: Claude Opus 5.5 (1M context) --- box_sdk_gen/networking/multipart_stream.py | 47 ++++++-- test/box_sdk_gen/test/box_network_client.py | 118 +++++++++++++++++++- 2 files changed, 153 insertions(+), 12 deletions(-) diff --git a/box_sdk_gen/networking/multipart_stream.py b/box_sdk_gen/networking/multipart_stream.py index 22c240cd1..78be03017 100644 --- a/box_sdk_gen/networking/multipart_stream.py +++ b/box_sdk_gen/networking/multipart_stream.py @@ -1,4 +1,4 @@ -from io import SEEK_END +from io import SEEK_END, TextIOBase from typing import Iterator, List, Optional, Tuple, Union from urllib3.fields import RequestField @@ -8,7 +8,9 @@ CHUNK_SIZE = 64 * 1024 -MultipartField = Tuple[str, Optional[str], Union[str, ByteStream], Optional[str]] +PartStream = Union[ByteStream, TextIOBase] + +MultipartField = Tuple[str, Optional[str], Union[str, PartStream], Optional[str]] class MultipartStream: @@ -17,13 +19,14 @@ class MultipartStream: so uploads are sent without buffering whole files in memory. Fields are (name, file_name, value, content_type) tuples, where value is - either a string or a binary stream read from its current position. + either a string or a stream read from its current position. Text streams + are encoded as UTF-8. """ def __init__(self, fields: List[MultipartField]): self.boundary = choose_boundary() self.content_type = f'multipart/form-data; boundary={self.boundary}' - self._segments: List[Union[bytes, ByteStream]] = [] + self._segments: List[Union[bytes, PartStream]] = [] for name, file_name, value, content_type in fields: field = RequestField(name=name, data=b'', filename=file_name) field.make_multipart(content_type=content_type) @@ -37,6 +40,8 @@ def __init__(self, fields: List[MultipartField]): self._segments.append(f'--{self.boundary}--\r\n'.encode('utf-8')) self._index = 0 self._offset = 0 + # encoded text stream bytes that didn't fit in the last read + self._pending = b'' # set when a part stream ends before its declared size; retrying won't help self.size_error: Optional[IOError] = None # bytes still to send per stream segment, None when the size is unknown @@ -49,17 +54,31 @@ def __init__(self, fields: List[MultipartField]): self.len = self._compute_length() @staticmethod - def _stream_size(stream: ByteStream) -> Optional[int]: + def _stream_size(stream: PartStream) -> Optional[int]: try: - if not stream.seekable(): + # text stream positions count characters, not encoded bytes + if isinstance(stream, TextIOBase) or not stream.seekable(): return None position = stream.tell() - stream.seek(0, SEEK_END) - end = stream.tell() - stream.seek(position) + # read(0) also catches text-like streams that aren't a TextIOBase + is_text = isinstance(stream.read(0), str) + if stream.tell() != position: + # read(0) shouldn't move a stream, but don't skip data if it does + stream.seek(position) + if is_text: + return None + # like requests and requests-toolbelt, prefer a length the stream reports + if hasattr(stream, '__len__'): + end = len(stream) + elif getattr(stream, 'len', None) is not None: + end = stream.len + else: + stream.seek(0, SEEK_END) + end = stream.tell() + stream.seek(position) # a stream positioned past its end has nothing left to send return max(0, end - position) - except (OSError, AttributeError, TypeError): + except (OSError, AttributeError, TypeError, ValueError): return None def _compute_length(self) -> Optional[int]: @@ -80,7 +99,9 @@ def read(self, size: Optional[int] = -1) -> bytes: chunks = [] while size > 0 and self._index < len(self._segments): segment = self._segments[self._index] - if isinstance(segment, bytes): + if self._pending: + chunk, self._pending = self._pending[:size], self._pending[size:] + elif isinstance(segment, bytes): chunk = segment[self._offset : self._offset + size] self._offset += len(chunk) if self._offset >= len(segment): @@ -103,8 +124,12 @@ def read(self, size: Optional[int] = -1) -> bytes: raise self.size_error self._index += 1 continue + if isinstance(chunk, str): + chunk = chunk.encode('utf-8') if remaining is not None: self._remaining[self._index] = remaining - len(chunk) + # a character can encode to several bytes, so keep what doesn't fit + chunk, self._pending = chunk[:size], chunk[size:] chunks.append(chunk) size -= len(chunk) return b''.join(chunks) diff --git a/test/box_sdk_gen/test/box_network_client.py b/test/box_sdk_gen/test/box_network_client.py index 0af0c65c2..21abdd5b6 100644 --- a/test/box_sdk_gen/test/box_network_client.py +++ b/test/box_sdk_gen/test/box_network_client.py @@ -2,7 +2,7 @@ import json import threading from http.server import BaseHTTPRequestHandler, HTTPServer -from io import BytesIO, RawIOBase, UnsupportedOperation, SEEK_END, SEEK_SET +from io import BytesIO, RawIOBase, StringIO, UnsupportedOperation, SEEK_END, SEEK_SET from unittest import mock from unittest.mock import Mock, patch from requests import Session, Response, RequestException @@ -1513,3 +1513,119 @@ def seek(self, *args): assert len(body) == multipart_stream.len assert b"\r\n\r\n23456789\r\n" in body + + +def test_multipart_upload_text_stream_is_encoded_as_utf8(multipart_server): + url, received, _ = multipart_server + + BoxNetworkClient().fetch(_upload_options(url, StringIO("héllo wörld"))) + + assert len(received) == 1 + headers, body = received[0] + assert headers["Transfer-Encoding"] == "chunked" + assert "Content-Length" not in headers + assert "héllo wörld\r\n".encode("utf-8") in body + + +def test_multipart_stream_encodes_text_file_as_utf8(tmp_path): + path = tmp_path / "file.txt" + path.write_text("héllo wörld\n" * 10000, encoding="utf-8") + + with open(path, "r", encoding="utf-8") as text_file: + multipart_stream = MultipartStream([("file", "file.txt", text_file, None)]) + body = multipart_stream.read() + + assert multipart_stream.len is None + assert body.endswith( + b"\r\n\r\n" + + ("héllo wörld\n" * 10000).encode("utf-8") + + f"\r\n--{multipart_stream.boundary}--\r\n".encode() + ) + + +def test_multipart_stream_uses_length_reported_by_stream(): + class ReportedLength(BytesIO): + len = 10 + + def seek(self, *args): + raise AssertionError("size should come from the reported length") + + stream = ReportedLength(b"0123456789") + stream.read(2) + multipart_stream = MultipartStream([("file", "f", stream, None)]) + + body = multipart_stream.read() + + assert len(body) == multipart_stream.len + assert b"\r\n\r\n23456789\r\n" in body + + +def test_multipart_stream_read_returns_at_most_size_bytes_for_text_stream(): + text = "é" * 100 + "€" * 50 + multipart_stream = MultipartStream([("file", "f", StringIO(text), None)]) + + chunks = list(iter(lambda: multipart_stream.read(3), b"")) + + assert all(len(chunk) <= 3 for chunk in chunks) + assert b"\r\n\r\n" + text.encode("utf-8") + b"\r\n" in b"".join(chunks) + + +def test_multipart_stream_sends_text_like_stream_with_unknown_length(): + class TextLike: + def __init__(self, text): + self._stream = StringIO(text) + + def seekable(self): + return True + + def tell(self): + return self._stream.tell() + + def seek(self, *args): + return self._stream.seek(*args) + + def read(self, size=-1): + return self._stream.read(size) + + text = "héllo wörld " * 100 + multipart_stream = MultipartStream([("file", "f", TextLike(text), None)]) + + body = multipart_stream.read() + + assert multipart_stream.len is None + assert b"\r\n\r\n" + text.encode("utf-8") + b"\r\n" in body + + +def test_multipart_stream_sends_stream_failing_read_zero_with_unknown_length(): + class FailsOnEmptyRead(BytesIO): + def read(self, size=-1): + if size == 0: + raise ValueError("read(0) is not supported") + return super().read(size) + + multipart_stream = MultipartStream( + [("file", "f", FailsOnEmptyRead(b"0123456789"), None)] + ) + + body = multipart_stream.read() + + assert multipart_stream.len is None + assert b"\r\n\r\n0123456789\r\n" in body + + +def test_multipart_stream_restores_position_moved_by_read_zero(): + class MovesOnEmptyRead(BytesIO): + def read(self, size=-1): + if size == 0: + self.seek(self.tell() + 2) + return b"" + return super().read(size) + + multipart_stream = MultipartStream( + [("file", "f", MovesOnEmptyRead(b"0123456789"), None)] + ) + + body = multipart_stream.read() + + assert len(body) == multipart_stream.len + assert b"\r\n\r\n0123456789\r\n" in body From d8fc6b9161b8d1e09653ba10e255e607de2287e2 Mon Sep 17 00:00:00 2001 From: Minh Nguyen Cong Date: Mon, 5 Oct 2026 19:56:32 +0200 Subject: [PATCH 7/7] Revert "test(boxsdkgen): Temporarily sync MultipartStream from box/box-codegen#995" This reverts commit a891485 now that CI passed with it. The generated files come to combined-sdk through the box/box-codegen#995 sync PR. Co-Authored-By: Claude Opus 5.5 (1M context) --- box_sdk_gen/networking/multipart_stream.py | 47 ++------ test/box_sdk_gen/test/box_network_client.py | 118 +------------------- 2 files changed, 12 insertions(+), 153 deletions(-) diff --git a/box_sdk_gen/networking/multipart_stream.py b/box_sdk_gen/networking/multipart_stream.py index 78be03017..22c240cd1 100644 --- a/box_sdk_gen/networking/multipart_stream.py +++ b/box_sdk_gen/networking/multipart_stream.py @@ -1,4 +1,4 @@ -from io import SEEK_END, TextIOBase +from io import SEEK_END from typing import Iterator, List, Optional, Tuple, Union from urllib3.fields import RequestField @@ -8,9 +8,7 @@ CHUNK_SIZE = 64 * 1024 -PartStream = Union[ByteStream, TextIOBase] - -MultipartField = Tuple[str, Optional[str], Union[str, PartStream], Optional[str]] +MultipartField = Tuple[str, Optional[str], Union[str, ByteStream], Optional[str]] class MultipartStream: @@ -19,14 +17,13 @@ class MultipartStream: so uploads are sent without buffering whole files in memory. Fields are (name, file_name, value, content_type) tuples, where value is - either a string or a stream read from its current position. Text streams - are encoded as UTF-8. + either a string or a binary stream read from its current position. """ def __init__(self, fields: List[MultipartField]): self.boundary = choose_boundary() self.content_type = f'multipart/form-data; boundary={self.boundary}' - self._segments: List[Union[bytes, PartStream]] = [] + self._segments: List[Union[bytes, ByteStream]] = [] for name, file_name, value, content_type in fields: field = RequestField(name=name, data=b'', filename=file_name) field.make_multipart(content_type=content_type) @@ -40,8 +37,6 @@ def __init__(self, fields: List[MultipartField]): self._segments.append(f'--{self.boundary}--\r\n'.encode('utf-8')) self._index = 0 self._offset = 0 - # encoded text stream bytes that didn't fit in the last read - self._pending = b'' # set when a part stream ends before its declared size; retrying won't help self.size_error: Optional[IOError] = None # bytes still to send per stream segment, None when the size is unknown @@ -54,31 +49,17 @@ def __init__(self, fields: List[MultipartField]): self.len = self._compute_length() @staticmethod - def _stream_size(stream: PartStream) -> Optional[int]: + def _stream_size(stream: ByteStream) -> Optional[int]: try: - # text stream positions count characters, not encoded bytes - if isinstance(stream, TextIOBase) or not stream.seekable(): + if not stream.seekable(): return None position = stream.tell() - # read(0) also catches text-like streams that aren't a TextIOBase - is_text = isinstance(stream.read(0), str) - if stream.tell() != position: - # read(0) shouldn't move a stream, but don't skip data if it does - stream.seek(position) - if is_text: - return None - # like requests and requests-toolbelt, prefer a length the stream reports - if hasattr(stream, '__len__'): - end = len(stream) - elif getattr(stream, 'len', None) is not None: - end = stream.len - else: - stream.seek(0, SEEK_END) - end = stream.tell() - stream.seek(position) + stream.seek(0, SEEK_END) + end = stream.tell() + stream.seek(position) # a stream positioned past its end has nothing left to send return max(0, end - position) - except (OSError, AttributeError, TypeError, ValueError): + except (OSError, AttributeError, TypeError): return None def _compute_length(self) -> Optional[int]: @@ -99,9 +80,7 @@ def read(self, size: Optional[int] = -1) -> bytes: chunks = [] while size > 0 and self._index < len(self._segments): segment = self._segments[self._index] - if self._pending: - chunk, self._pending = self._pending[:size], self._pending[size:] - elif isinstance(segment, bytes): + if isinstance(segment, bytes): chunk = segment[self._offset : self._offset + size] self._offset += len(chunk) if self._offset >= len(segment): @@ -124,12 +103,8 @@ def read(self, size: Optional[int] = -1) -> bytes: raise self.size_error self._index += 1 continue - if isinstance(chunk, str): - chunk = chunk.encode('utf-8') if remaining is not None: self._remaining[self._index] = remaining - len(chunk) - # a character can encode to several bytes, so keep what doesn't fit - chunk, self._pending = chunk[:size], chunk[size:] chunks.append(chunk) size -= len(chunk) return b''.join(chunks) diff --git a/test/box_sdk_gen/test/box_network_client.py b/test/box_sdk_gen/test/box_network_client.py index 21abdd5b6..0af0c65c2 100644 --- a/test/box_sdk_gen/test/box_network_client.py +++ b/test/box_sdk_gen/test/box_network_client.py @@ -2,7 +2,7 @@ import json import threading from http.server import BaseHTTPRequestHandler, HTTPServer -from io import BytesIO, RawIOBase, StringIO, UnsupportedOperation, SEEK_END, SEEK_SET +from io import BytesIO, RawIOBase, UnsupportedOperation, SEEK_END, SEEK_SET from unittest import mock from unittest.mock import Mock, patch from requests import Session, Response, RequestException @@ -1513,119 +1513,3 @@ def seek(self, *args): assert len(body) == multipart_stream.len assert b"\r\n\r\n23456789\r\n" in body - - -def test_multipart_upload_text_stream_is_encoded_as_utf8(multipart_server): - url, received, _ = multipart_server - - BoxNetworkClient().fetch(_upload_options(url, StringIO("héllo wörld"))) - - assert len(received) == 1 - headers, body = received[0] - assert headers["Transfer-Encoding"] == "chunked" - assert "Content-Length" not in headers - assert "héllo wörld\r\n".encode("utf-8") in body - - -def test_multipart_stream_encodes_text_file_as_utf8(tmp_path): - path = tmp_path / "file.txt" - path.write_text("héllo wörld\n" * 10000, encoding="utf-8") - - with open(path, "r", encoding="utf-8") as text_file: - multipart_stream = MultipartStream([("file", "file.txt", text_file, None)]) - body = multipart_stream.read() - - assert multipart_stream.len is None - assert body.endswith( - b"\r\n\r\n" - + ("héllo wörld\n" * 10000).encode("utf-8") - + f"\r\n--{multipart_stream.boundary}--\r\n".encode() - ) - - -def test_multipart_stream_uses_length_reported_by_stream(): - class ReportedLength(BytesIO): - len = 10 - - def seek(self, *args): - raise AssertionError("size should come from the reported length") - - stream = ReportedLength(b"0123456789") - stream.read(2) - multipart_stream = MultipartStream([("file", "f", stream, None)]) - - body = multipart_stream.read() - - assert len(body) == multipart_stream.len - assert b"\r\n\r\n23456789\r\n" in body - - -def test_multipart_stream_read_returns_at_most_size_bytes_for_text_stream(): - text = "é" * 100 + "€" * 50 - multipart_stream = MultipartStream([("file", "f", StringIO(text), None)]) - - chunks = list(iter(lambda: multipart_stream.read(3), b"")) - - assert all(len(chunk) <= 3 for chunk in chunks) - assert b"\r\n\r\n" + text.encode("utf-8") + b"\r\n" in b"".join(chunks) - - -def test_multipart_stream_sends_text_like_stream_with_unknown_length(): - class TextLike: - def __init__(self, text): - self._stream = StringIO(text) - - def seekable(self): - return True - - def tell(self): - return self._stream.tell() - - def seek(self, *args): - return self._stream.seek(*args) - - def read(self, size=-1): - return self._stream.read(size) - - text = "héllo wörld " * 100 - multipart_stream = MultipartStream([("file", "f", TextLike(text), None)]) - - body = multipart_stream.read() - - assert multipart_stream.len is None - assert b"\r\n\r\n" + text.encode("utf-8") + b"\r\n" in body - - -def test_multipart_stream_sends_stream_failing_read_zero_with_unknown_length(): - class FailsOnEmptyRead(BytesIO): - def read(self, size=-1): - if size == 0: - raise ValueError("read(0) is not supported") - return super().read(size) - - multipart_stream = MultipartStream( - [("file", "f", FailsOnEmptyRead(b"0123456789"), None)] - ) - - body = multipart_stream.read() - - assert multipart_stream.len is None - assert b"\r\n\r\n0123456789\r\n" in body - - -def test_multipart_stream_restores_position_moved_by_read_zero(): - class MovesOnEmptyRead(BytesIO): - def read(self, size=-1): - if size == 0: - self.seek(self.tell() + 2) - return b"" - return super().read(size) - - multipart_stream = MultipartStream( - [("file", "f", MovesOnEmptyRead(b"0123456789"), None)] - ) - - body = multipart_stream.read() - - assert len(body) == multipart_stream.len - assert b"\r\n\r\n0123456789\r\n" in body