From a17ea61af478baa73c6921da9f62fd7627fbaec9 Mon Sep 17 00:00:00 2001 From: "Peter A. Jonsson" Date: Mon, 21 Sep 2026 15:08:23 +0200 Subject: [PATCH] download_file: do not HEAD S3 URIs Running HEAD on a pre-signed S3 URI gives a 403 response because the URI is signed for GET. Use a GET with a Range 0-0 request instead of HEAD to figure out if the server accepts range requests. Fixes #939 --- CHANGELOG.md | 1 + openeo/rest/_connection.py | 10 +++++++--- tests/rest/test_job.py | 9 +++++++-- 3 files changed, 15 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9a7f59c8b..dde6b00d8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -13,6 +13,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Convert setup.py to pyproject.toml ([#920](https://github.com/Open-EO/openeo-python-client/issues/920)) - Make `_DerivedFrom._from_url` more resilient against unresolvable/unparsable `derived_from` links ([#928](https://github.com/Open-EO/openeo-python-client/issues/928), eu-cdse/openeo-cdse-infra#1338) +- Use GET with a 0-0 range request instead of HEAD when downloading ([#939](https://github.com/Open-EO/openeo-python-client/issues/939)) ### Removed diff --git a/openeo/rest/_connection.py b/openeo/rest/_connection.py index 42167ba74..e0a14c0a2 100644 --- a/openeo/rest/_connection.py +++ b/openeo/rest/_connection.py @@ -305,9 +305,13 @@ def download_url( chunk_size: int = DEFAULT_DOWNLOAD_CHUNK_SIZE, range_size: int = DEFAULT_DOWNLOAD_RANGE_SIZE, ) -> None: - head = self.head(url, stream=True) - if head.ok and head.headers.get("Accept-Ranges") == "bytes" and "Content-Length" in head.headers: - file_size = int(head.headers["Content-Length"]) + # URL might be pre-signed S3 URL, so use a GET with a 0-0 range request + # to figure out if the server supports range requests. + # Trying to GET a 0-byte file will give a 416-response, so accept that + # since the download will succeed. + head = self.get(url, headers={"Range": "bytes=0-0"}, expected_status=[200, 206, 416], stream=True) + if head.ok and head.headers.get("Accept-Ranges") == "bytes" and "Content-Range" in head.headers: + file_size = int(head.headers["Content-Range"].split("/")[1]) self._download_ranged( url=url, target=target, file_size=file_size, chunk_size=chunk_size, range_size=range_size ) diff --git a/tests/rest/test_job.py b/tests/rest/test_job.py index ed0784334..6efa21508 100644 --- a/tests/rest/test_job.py +++ b/tests/rest/test_job.py @@ -734,8 +734,13 @@ def handle_content(request, context): assert search from_bytes = int(search.group(1)) to_bytes = int(search.group(2)) - assert from_bytes < to_bytes - return TIFF_CONTENT[from_bytes : to_bytes + 1] + assert from_bytes <= to_bytes + sliced_content = TIFF_CONTENT[from_bytes : to_bytes + 1] + context.status_code = 206 + context.headers["Accept-Ranges"] = "bytes" + context.headers["Content-Range"] = f"bytes {from_bytes}-{to_bytes}/{len(TIFF_CONTENT)}" + context.headers["Content-Length"] = str(len(sliced_content)) + return sliced_content requests_mock.get( API_URL + "/jobs/jj1/results",