From 64dd65cebbfdf7e18e14aaf8b9ae7240f501a401 Mon Sep 17 00:00:00 2001 From: Matthew Larson Date: Fri, 10 Jul 2026 16:44:25 -0500 Subject: [PATCH 1/6] Detect filtered chunk size overflow with small size of sizes --- release_docs/CHANGELOG.md | 4 + src/H5Dchunk.c | 27 ++- test/dsets.c | 423 ++++++++++++++++++++++++++++++++++++++ 3 files changed, 448 insertions(+), 6 deletions(-) diff --git a/release_docs/CHANGELOG.md b/release_docs/CHANGELOG.md index 020296be11d..2aa1aee4bea 100644 --- a/release_docs/CHANGELOG.md +++ b/release_docs/CHANGELOG.md @@ -222,6 +222,10 @@ The `h5repack` tool now obtains its default low and high library version bounds `H5S_select_deserialize()` and the per-selection-type deserialize callbacks (all, hyperslab, none, and point) previously computed the pointer to the last valid buffer byte as `buffer + size - 1` without first checking the buffer size. A buffer shorter than the 4-byte selection-type header, or a zero-length selection-info buffer, would underflow this computation and produce an out-of-bounds end pointer, defeating subsequent overflow checks. The deserialize routines now reject a buffer that is too small to hold the selection type, and they reject an empty selection-info buffer before deriving the end pointer. Hyperslab decoding additionally now rejects a serialized rank of 0 or greater than `H5S_MAX_RANK`. As a companion fix, `H5S__hyper_serialize()` now returns an error when asked to serialize a hyperslab selection on a rank-0 (scalar or null) dataspace, a state that can arise when a dataspace extent is collapsed to a scalar after a hyperslab selection has already been made. +### Fixed silent truncation of filtered chunk sizes with a small file "size of sizes" + + For all chunk index types in version-5 chunk layout messages, and for the single chunk index in any version of the chunk layout message, the on-disk size of a filtered chunk is encoded in a fixed-width field equal to the file's "size of sizes" (set via `H5Pset_sizes()`). The encode check assumed this field was always 8 bytes, so when the size of sizes was set to 2 or 4 an oversized chunk had its encoded size silently truncated (e.g. 160000 encoded in 2 bytes became 160000 & 0xFFFF = 28928), silently corrupting the chunk. The library now verifies that a filtered chunk's size fits in the file's size of sizes and reports an error at write time instead. + ## Java Library ### Fixed a datatype ID leak when reading or writing array/vlen datatypes diff --git a/src/H5Dchunk.c b/src/H5Dchunk.c index 9e1d20cc52e..a5d7d123edf 100644 --- a/src/H5Dchunk.c +++ b/src/H5Dchunk.c @@ -90,11 +90,13 @@ } while (0) /* Macro to Check for chunk size being too big to encode. Early versions were simply limited to 32 bits. - * Version 4, except for the single chunk index, was limited using a formula described below. Version 5 uses - * 64 bits, as does the single chunk index (with all versions). We additionally impose the restriction that - * version 4 cannot encode more than 32 bits, even though it is not precluded by the file format, because - * those versions of the library cannot handle chunks larger than 32 bits internally. */ -#define H5D_CHUNK_ENCODE_SIZE_CHECK(layout, length, err) \ + * Version 4, except for the single chunk index, was limited using a formula described below. Version 5, and + * the single chunk index with any version, encode the filtered chunk size in a fixed-width field whose size + * is the file's "size of sizes" (see H5Pset_sizes()); the chunk size must therefore fit in that many bytes. + * We additionally impose the restriction that version 4 cannot encode more than 32 bits, even though it is + * not precluded by the file format, because those versions of the library cannot handle chunks larger than + * 32 bits internally. */ +#define H5D_CHUNK_ENCODE_SIZE_CHECK(f, layout, length, err) \ do { \ if ((layout)->version <= H5O_LAYOUT_VERSION_4) { \ if ((length) > UINT32_MAX) \ @@ -125,6 +127,19 @@ "filter increased chunk size by too much and it cannot be encoded with " \ "this file format version - see H5Pset_libver_bounds()"); \ } \ + } \ + \ + /* For version 5 (all chunk index types) and for the single chunk index (any version), the */ \ + /* filtered chunk size is encoded in a fixed-width field equal to the file's "size of sizes". */ \ + /* Make sure the chunk size fits in that many bytes to avoid silently truncating it. */ \ + if ((layout)->version > H5O_LAYOUT_VERSION_4 || \ + (layout)->storage.u.chunk.idx_type == H5D_CHUNK_IDX_SINGLE) { \ + unsigned size_of_sizes = (unsigned)H5F_SIZEOF_SIZE(f); \ + \ + if (size_of_sizes < 8 && (length) > (((uint64_t)1 << (8 * size_of_sizes)) - 1)) \ + HGOTO_ERROR(H5E_DATASET, H5E_BADRANGE, err, \ + "filtered chunk size is too large to be encoded with the file's size of sizes " \ + "- see H5Pset_sizes()"); \ } \ } while (0) @@ -7670,7 +7685,7 @@ H5D__chunk_file_alloc(const H5D_chk_idx_info_t *idx_info, const H5F_block_t *old /* Check for chunk size overflowing format limitations */ /* Only needed for filtered datasets because the unfiltered chunk size * was already checked in H5D__chunk_construct() */ - H5D_CHUNK_ENCODE_SIZE_CHECK(idx_info->layout, new_chunk->length, FAIL); + H5D_CHUNK_ENCODE_SIZE_CHECK(idx_info->f, idx_info->layout, new_chunk->length, FAIL); if (old_chunk && H5_addr_defined(old_chunk->offset)) { /* Sanity check */ diff --git a/test/dsets.c b/test/dsets.c index fd2feb6c8ce..5ae5115b3a6 100644 --- a/test/dsets.c +++ b/test/dsets.c @@ -82,6 +82,7 @@ static const char *FILENAME[] = {"dataset", /* 0 */ "chunk_expand2", /* 30 */ "scalar_datasets", /* 31 */ "read_only_vlen_fill", /* 32 */ + "size_of_sizes", /* 33 */ NULL}; #define OHMIN_FILENAME_A "ohdr_min_a" @@ -9354,6 +9355,426 @@ test_deprec(hid_t file) } /* end test_deprec() */ #endif /* H5_NO_DEPRECATED_SYMBOLS */ +/*------------------------------------------------------------------------- + * Function: test_sos_roundtrip_dset + * + * Purpose: Helper for test_chunk_size_of_sizes(). Creates a filtered + * chunked dataset with the given shape, writes data to it, + * closes and reopens it, reads the data back, and verifies both + * the round-tripped data and the chunk index type. + * + * The no-op 'bogus' filter is used so that per-chunk sizes are + * stored in the file (and therefore encoded using the file's + * "size of sizes") while keeping the on-disk chunk size equal to + * the in-memory chunk size and thus predictable. The caller is + * responsible for registering the bogus filter and for choosing a + * chunk small enough to be encoded by the file's size of sizes. + * + * Return: Success: 0 + * Failure: -1 + * + *------------------------------------------------------------------------- + */ +static herr_t +test_sos_roundtrip_dset(hid_t fid, const char *dset_name, int rank, const hsize_t *dims, + const hsize_t *maxdims, const hsize_t *chunk, H5D_chunk_index_t expected_idx) +{ + hid_t sid = H5I_INVALID_HID; /* Dataspace ID */ + hid_t did = H5I_INVALID_HID; /* Dataset ID */ + hid_t dcpl = H5I_INVALID_HID; /* Dataset creation property list ID */ + int *wbuf = NULL; /* Write buffer */ + int *rbuf = NULL; /* Read buffer */ + H5D_chunk_index_t idx_type; /* Actual chunk index type */ + hsize_t nelmts = 1; /* Number of elements in the dataset */ + size_t u; /* Local index variable */ + + /* Compute the number of elements and fill the write buffer */ + for (u = 0; u < (size_t)rank; u++) + nelmts *= dims[u]; + if (NULL == (wbuf = malloc((size_t)nelmts * sizeof(int)))) + TEST_ERROR; + if (NULL == (rbuf = malloc((size_t)nelmts * sizeof(int)))) + TEST_ERROR; + for (u = 0; u < (size_t)nelmts; u++) + wbuf[u] = (int)u; + + /* Create the dataspace and a chunked, filtered DCPL */ + if ((sid = H5Screate_simple(rank, dims, maxdims)) < 0) + TEST_ERROR; + if ((dcpl = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_chunk(dcpl, rank, chunk) < 0) + TEST_ERROR; + if (H5Pset_filter(dcpl, H5Z_FILTER_BOGUS, 0, (size_t)0, NULL) < 0) + TEST_ERROR; + + /* Create the dataset */ + if ((did = H5Dcreate2(fid, dset_name, H5T_NATIVE_INT, sid, H5P_DEFAULT, dcpl, H5P_DEFAULT)) < 0) + TEST_ERROR; + + /* Verify that the expected chunk index type is in use */ + if (H5D__layout_idx_type_test(did, &idx_type) < 0) + TEST_ERROR; + if (idx_type != expected_idx) + FAIL_PUTS_ERROR(" unexpected chunk index type"); + + /* Write the data */ + if (H5Dwrite(did, H5T_NATIVE_INT, H5S_ALL, H5S_ALL, H5P_DEFAULT, wbuf) < 0) + TEST_ERROR; + + /* Close everything so the data must be read back from disk */ + if (H5Dclose(did) < 0) + TEST_ERROR; + did = H5I_INVALID_HID; + if (H5Pclose(dcpl) < 0) + TEST_ERROR; + dcpl = H5I_INVALID_HID; + if (H5Sclose(sid) < 0) + TEST_ERROR; + sid = H5I_INVALID_HID; + + /* Reopen and read the data back */ + if ((did = H5Dopen2(fid, dset_name, H5P_DEFAULT)) < 0) + TEST_ERROR; + if (H5Dread(did, H5T_NATIVE_INT, H5S_ALL, H5S_ALL, H5P_DEFAULT, rbuf) < 0) + TEST_ERROR; + for (u = 0; u < (size_t)nelmts; u++) + if (wbuf[u] != rbuf[u]) + FAIL_PUTS_ERROR(" data read back differs from data written"); + if (H5Dclose(did) < 0) + TEST_ERROR; + + free(wbuf); + free(rbuf); + + return SUCCEED; + +error: + H5E_BEGIN_TRY + { + H5Pclose(dcpl); + H5Dclose(did); + H5Sclose(sid); + } + H5E_END_TRY + free(wbuf); + free(rbuf); + return FAIL; +} /* end test_sos_roundtrip_dset() */ + +/*------------------------------------------------------------------------- + * Function: test_chunk_size_of_sizes + * + * Purpose: Tests that filtered chunked datasets can be created, written, + * and read back correctly when the file's "size of sizes" + * (H5Pset_sizes) is set to a value smaller than the default of 8. + * + * For version-5 chunk layout messages (the 2.0 file format), the + * on-disk size of each filtered chunk is encoded using the file's + * size of sizes. This exercises that encoding path for every + * chunk index type that stores per-chunk sizes (single chunk, + * fixed array, extensible array, and version-2 B-tree) with a + * size of sizes of 2, 4, and 8. The chunks are kept small enough + * to be encoded by all three settings. + * + * Return: Success: 0 + * Failure: -1 + * + *------------------------------------------------------------------------- + */ +static herr_t +test_chunk_size_of_sizes(hid_t fapl) +{ + hid_t my_fapl = H5I_INVALID_HID; /* Copy of the file access property list */ + hid_t fcpl = H5I_INVALID_HID; /* File creation property list ID */ + hid_t fid = H5I_INVALID_HID; /* File ID */ + char filename[FILENAME_BUF_SIZE]; + size_t sizes[] = {2, 4, 8}; /* "size of sizes" values to test (all <= 8) */ + bool registered = false; /* Whether the bogus filter is registered */ + unsigned s; /* Local index variable */ + + TESTING("chunked datasets with size of sizes < 8"); + + /* Copy the FAPL and force the latest format so that version-5 chunk layout + * messages (which encode filtered chunk sizes using the file's size of + * sizes) are used for all chunk index types. */ + if ((my_fapl = H5Pcopy(fapl)) < 0) + TEST_ERROR; + if (H5Pset_libver_bounds(my_fapl, H5F_LIBVER_V200, H5F_LIBVER_LATEST) < 0) + TEST_ERROR; + + h5_fixname(FILENAME[33], my_fapl, filename, sizeof filename); + + /* Register the no-op 'bogus' filter so that per-chunk sizes are stored in + * the file without changing the on-disk chunk size. */ + if (H5Zregister(H5Z_BOGUS) < 0) + TEST_ERROR; + registered = true; + + for (s = 0; s < sizeof(sizes) / sizeof(sizes[0]); s++) { + /* Single chunk index: cur dims == max dims == chunk dims */ + hsize_t single_dims[2] = {10, 10}; + hsize_t single_max[2] = {10, 10}; + /* Fixed array index: fixed max dims, multiple chunks */ + hsize_t farray_dims[2] = {40, 40}; + hsize_t farray_max[2] = {40, 40}; + /* Extensible array index: exactly one unlimited dimension */ + hsize_t earray_dims[2] = {40, 40}; + hsize_t earray_max[2] = {H5S_UNLIMITED, 40}; + /* Version 2 B-tree index: more than one unlimited dimension */ + hsize_t bt2_dims[2] = {40, 40}; + hsize_t bt2_max[2] = {H5S_UNLIMITED, H5S_UNLIMITED}; + hsize_t chunk[2] = {10, 10}; /* 10 * 10 * 4 = 400 bytes on disk */ + + /* Create a file whose size of sizes is the value under test. The size + * of addresses is left at its default. */ + if ((fcpl = H5Pcreate(H5P_FILE_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_sizes(fcpl, (size_t)0, sizes[s]) < 0) + TEST_ERROR; + if ((fid = H5Fcreate(filename, H5F_ACC_TRUNC, fcpl, my_fapl)) < 0) + TEST_ERROR; + + /* Exercise each chunk index type that stores per-chunk sizes */ + if (test_sos_roundtrip_dset(fid, "single", 2, single_dims, single_max, chunk, + H5D_CHUNK_IDX_SINGLE) < 0) + goto error; + if (test_sos_roundtrip_dset(fid, "farray", 2, farray_dims, farray_max, chunk, + H5D_CHUNK_IDX_FARRAY) < 0) + goto error; + if (test_sos_roundtrip_dset(fid, "earray", 2, earray_dims, earray_max, chunk, + H5D_CHUNK_IDX_EARRAY) < 0) + goto error; + if (test_sos_roundtrip_dset(fid, "bt2", 2, bt2_dims, bt2_max, chunk, H5D_CHUNK_IDX_BT2) < 0) + goto error; + + if (H5Fclose(fid) < 0) + TEST_ERROR; + fid = H5I_INVALID_HID; + if (H5Pclose(fcpl) < 0) + TEST_ERROR; + fcpl = H5I_INVALID_HID; + } + + if (H5Zunregister(H5Z_FILTER_BOGUS) < 0) + TEST_ERROR; + if (H5Pclose(my_fapl) < 0) + TEST_ERROR; + + PASSED(); + return SUCCEED; + +error: + H5E_BEGIN_TRY + { + H5Fclose(fid); + H5Pclose(fcpl); + H5Pclose(my_fapl); + if (registered) + H5Zunregister(H5Z_FILTER_BOGUS); + } + H5E_END_TRY + return FAIL; +} /* end test_chunk_size_of_sizes() */ + +/*------------------------------------------------------------------------- + * Function: test_sos_overflow_dset + * + * Purpose: Helper for test_chunk_size_of_sizes_overflow(). Creates a + * filtered chunked dataset whose on-disk chunk size is too large + * to be encoded by the file's "size of sizes", then verifies that + * the library reports an error when the oversized chunk is written + * to disk rather than silently truncating the encoded size (which + * would corrupt the chunk). + * + * Dataset creation is expected to succeed (the overflow is only + * detectable once the chunk's on-disk size is known at write + * time), so the expected chunk index type is verified first. + * + * Return: Success: 0 (the library correctly reported an error) + * Failure: -1 + * + *------------------------------------------------------------------------- + */ +static herr_t +test_sos_overflow_dset(hid_t fid, const char *dset_name, int rank, const hsize_t *dims, + const hsize_t *maxdims, const hsize_t *chunk, H5D_chunk_index_t expected_idx) +{ + hid_t sid = H5I_INVALID_HID; /* Dataspace ID */ + hid_t did = H5I_INVALID_HID; /* Dataset ID */ + hid_t dcpl = H5I_INVALID_HID; /* Dataset creation property list ID */ + int *wbuf = NULL; /* Write buffer */ + H5D_chunk_index_t idx_type; /* Actual chunk index type */ + hsize_t nelmts = 1; /* Number of elements in the dataset */ + herr_t status = SUCCEED; /* Return value of the write/flush */ + size_t u; /* Local index variable */ + + for (u = 0; u < (size_t)rank; u++) + nelmts *= dims[u]; + if (NULL == (wbuf = calloc((size_t)nelmts, sizeof(int)))) + TEST_ERROR; + + if ((sid = H5Screate_simple(rank, dims, maxdims)) < 0) + TEST_ERROR; + if ((dcpl = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_chunk(dcpl, rank, chunk) < 0) + TEST_ERROR; + if (H5Pset_filter(dcpl, H5Z_FILTER_BOGUS, 0, (size_t)0, NULL) < 0) + TEST_ERROR; + + /* Creating the dataset should succeed */ + if ((did = H5Dcreate2(fid, dset_name, H5T_NATIVE_INT, sid, H5P_DEFAULT, dcpl, H5P_DEFAULT)) < 0) + TEST_ERROR; + + /* Verify that the expected chunk index type is in use */ + if (H5D__layout_idx_type_test(did, &idx_type) < 0) + TEST_ERROR; + if (idx_type != expected_idx) + FAIL_PUTS_ERROR(" unexpected chunk index type"); + + /* Writing the oversized chunk must fail (either when the chunk is flushed + * during the write, or when it is flushed at flush/close time). */ + H5E_BEGIN_TRY + { + status = H5Dwrite(did, H5T_NATIVE_INT, H5S_ALL, H5S_ALL, H5P_DEFAULT, wbuf); + if (status >= 0) + status = H5Fflush(fid, H5F_SCOPE_LOCAL); + } + H5E_END_TRY + + if (status >= 0) + FAIL_PUTS_ERROR(" write of a chunk too large for the file's size of sizes should have failed"); + + /* Clean up. The dataset was left in an error state, so ignore errors. */ + H5E_BEGIN_TRY + { + H5Dclose(did); + } + H5E_END_TRY + did = H5I_INVALID_HID; + if (H5Pclose(dcpl) < 0) + TEST_ERROR; + if (H5Sclose(sid) < 0) + TEST_ERROR; + + free(wbuf); + return SUCCEED; + +error: + H5E_BEGIN_TRY + { + H5Pclose(dcpl); + H5Dclose(did); + H5Sclose(sid); + } + H5E_END_TRY + free(wbuf); + return FAIL; +} /* end test_sos_overflow_dset() */ + +/*------------------------------------------------------------------------- + * Function: test_chunk_size_of_sizes_overflow + * + * Purpose: Tests that the library detects and reports overflow when a + * filtered chunk's on-disk size is too large to be encoded by the + * file's "size of sizes" (rather than silently truncating the + * encoded size and corrupting the chunk). + * + * A size of sizes of 2 can encode chunk sizes up to 65535 bytes. + * Using the no-op 'bogus' filter, a chunk of 200 x 200 4-byte + * integers is 160000 bytes on disk, which cannot be encoded. This + * is exercised for each chunk index type that encodes per-chunk + * sizes using the file's size of sizes. + * + * Return: Success: 0 + * Failure: -1 + * + *------------------------------------------------------------------------- + */ +static herr_t +test_chunk_size_of_sizes_overflow(hid_t fapl) +{ + hid_t my_fapl = H5I_INVALID_HID; /* Copy of the file access property list */ + hid_t fcpl = H5I_INVALID_HID; /* File creation property list ID */ + hid_t fid = H5I_INVALID_HID; /* File ID */ + char filename[FILENAME_BUF_SIZE]; + bool registered = false; /* Whether the bogus filter is registered */ + + /* Single chunk index: cur dims == max dims == chunk dims (one 160000-byte chunk) */ + hsize_t single_dims[2] = {200, 200}; + hsize_t single_max[2] = {200, 200}; + hsize_t single_chunk[2] = {200, 200}; + /* Fixed array index: fixed max dims, more than one chunk */ + hsize_t farray_dims[2] = {400, 200}; + hsize_t farray_max[2] = {400, 200}; + /* Extensible array index: exactly one unlimited dimension */ + hsize_t earray_dims[2] = {200, 200}; + hsize_t earray_max[2] = {H5S_UNLIMITED, 200}; + /* Version 2 B-tree index: more than one unlimited dimension */ + hsize_t bt2_dims[2] = {200, 200}; + hsize_t bt2_max[2] = {H5S_UNLIMITED, H5S_UNLIMITED}; + hsize_t chunk[2] = {200, 200}; /* 200 * 200 * 4 = 160000 bytes on disk */ + + TESTING("overflow detection for chunked datasets with size of sizes < 8"); + + if ((my_fapl = H5Pcopy(fapl)) < 0) + TEST_ERROR; + if (H5Pset_libver_bounds(my_fapl, H5F_LIBVER_V200, H5F_LIBVER_LATEST) < 0) + TEST_ERROR; + + h5_fixname(FILENAME[33], my_fapl, filename, sizeof filename); + + if (H5Zregister(H5Z_BOGUS) < 0) + TEST_ERROR; + registered = true; + + /* Use the smallest legal size of sizes (2 bytes, max encodable 65535) */ + if ((fcpl = H5Pcreate(H5P_FILE_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_sizes(fcpl, (size_t)0, (size_t)2) < 0) + TEST_ERROR; + if ((fid = H5Fcreate(filename, H5F_ACC_TRUNC, fcpl, my_fapl)) < 0) + TEST_ERROR; + + if (test_sos_overflow_dset(fid, "single", 2, single_dims, single_max, single_chunk, + H5D_CHUNK_IDX_SINGLE) < 0) + goto error; + if (test_sos_overflow_dset(fid, "farray", 2, farray_dims, farray_max, chunk, H5D_CHUNK_IDX_FARRAY) < 0) + goto error; + if (test_sos_overflow_dset(fid, "earray", 2, earray_dims, earray_max, chunk, H5D_CHUNK_IDX_EARRAY) < 0) + goto error; + if (test_sos_overflow_dset(fid, "bt2", 2, bt2_dims, bt2_max, chunk, H5D_CHUNK_IDX_BT2) < 0) + goto error; + + if (H5Fclose(fid) < 0) + TEST_ERROR; + fid = H5I_INVALID_HID; + if (H5Pclose(fcpl) < 0) + TEST_ERROR; + fcpl = H5I_INVALID_HID; + + if (H5Zunregister(H5Z_FILTER_BOGUS) < 0) + TEST_ERROR; + if (H5Pclose(my_fapl) < 0) + TEST_ERROR; + + PASSED(); + return SUCCEED; + +error: + H5E_BEGIN_TRY + { + H5Fclose(fid); + H5Pclose(fcpl); + H5Pclose(my_fapl); + if (registered) + H5Zunregister(H5Z_FILTER_BOGUS); + } + H5E_END_TRY + return FAIL; +} /* end test_chunk_size_of_sizes_overflow() */ + /*------------------------------------------------------------------------- * Function: test_huge_chunks * @@ -19556,6 +19977,8 @@ main(void) #endif /* H5_NO_DEPRECATED_SYMBOLS */ nerrors += (test_huge_chunks(fapl, low) < 0 ? 1 : 0); + nerrors += (test_chunk_size_of_sizes(fapl) < 0 ? 1 : 0); + nerrors += (test_chunk_size_of_sizes_overflow(fapl) < 0 ? 1 : 0); nerrors += (test_chunk_cache(fapl) < 0 ? 1 : 0); nerrors += (test_big_chunks_bypass_cache(fapl) < 0 ? 1 : 0); nerrors += (test_chunk_fast(driver_name, fapl) < 0 ? 1 : 0); From 20494ad4591ab0510f0a9bb09c6b93d8326e5ff7 Mon Sep 17 00:00:00 2001 From: Matt L <124107509+mattjala@users.noreply.github.com> Date: Fri, 10 Jul 2026 17:07:55 -0500 Subject: [PATCH 2/6] Potential fix for pull request finding Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- test/dsets.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/test/dsets.c b/test/dsets.c index 5ae5115b3a6..2d3de2fcaf6 100644 --- a/test/dsets.c +++ b/test/dsets.c @@ -9493,7 +9493,7 @@ test_chunk_size_of_sizes(hid_t fapl) bool registered = false; /* Whether the bogus filter is registered */ unsigned s; /* Local index variable */ - TESTING("chunked datasets with size of sizes < 8"); + TESTING("chunked datasets with size of sizes <= 8"); /* Copy the FAPL and force the latest format so that version-5 chunk layout * messages (which encode filtered chunk sizes using the file's size of From dbb05829de861b988fc6b3ef48cbafde8e65e8eb Mon Sep 17 00:00:00 2001 From: Matthew Larson Date: Fri, 10 Jul 2026 17:08:52 -0500 Subject: [PATCH 3/6] Clang-format --- src/H5Dchunk.c | 4 +-- test/dsets.c | 82 +++++++++++++++++++++++++------------------------- 2 files changed, 43 insertions(+), 43 deletions(-) diff --git a/src/H5Dchunk.c b/src/H5Dchunk.c index a5d7d123edf..5a8b964d0c5 100644 --- a/src/H5Dchunk.c +++ b/src/H5Dchunk.c @@ -129,14 +129,14 @@ } \ } \ \ - /* For version 5 (all chunk index types) and for the single chunk index (any version), the */ \ + /* For version 5 (all chunk index types) and for the single chunk index (any version), the */ \ /* filtered chunk size is encoded in a fixed-width field equal to the file's "size of sizes". */ \ /* Make sure the chunk size fits in that many bytes to avoid silently truncating it. */ \ if ((layout)->version > H5O_LAYOUT_VERSION_4 || \ (layout)->storage.u.chunk.idx_type == H5D_CHUNK_IDX_SINGLE) { \ unsigned size_of_sizes = (unsigned)H5F_SIZEOF_SIZE(f); \ \ - if (size_of_sizes < 8 && (length) > (((uint64_t)1 << (8 * size_of_sizes)) - 1)) \ + if (size_of_sizes < 8 && (length) > (((uint64_t)1 << (8 * size_of_sizes)) - 1)) \ HGOTO_ERROR(H5E_DATASET, H5E_BADRANGE, err, \ "filtered chunk size is too large to be encoded with the file's size of sizes " \ "- see H5Pset_sizes()"); \ diff --git a/test/dsets.c b/test/dsets.c index 2d3de2fcaf6..f2450c10b03 100644 --- a/test/dsets.c +++ b/test/dsets.c @@ -9379,14 +9379,14 @@ static herr_t test_sos_roundtrip_dset(hid_t fid, const char *dset_name, int rank, const hsize_t *dims, const hsize_t *maxdims, const hsize_t *chunk, H5D_chunk_index_t expected_idx) { - hid_t sid = H5I_INVALID_HID; /* Dataspace ID */ - hid_t did = H5I_INVALID_HID; /* Dataset ID */ - hid_t dcpl = H5I_INVALID_HID; /* Dataset creation property list ID */ - int *wbuf = NULL; /* Write buffer */ - int *rbuf = NULL; /* Read buffer */ - H5D_chunk_index_t idx_type; /* Actual chunk index type */ - hsize_t nelmts = 1; /* Number of elements in the dataset */ - size_t u; /* Local index variable */ + hid_t sid = H5I_INVALID_HID; /* Dataspace ID */ + hid_t did = H5I_INVALID_HID; /* Dataset ID */ + hid_t dcpl = H5I_INVALID_HID; /* Dataset creation property list ID */ + int *wbuf = NULL; /* Write buffer */ + int *rbuf = NULL; /* Read buffer */ + H5D_chunk_index_t idx_type; /* Actual chunk index type */ + hsize_t nelmts = 1; /* Number of elements in the dataset */ + size_t u; /* Local index variable */ /* Compute the number of elements and fill the write buffer */ for (u = 0; u < (size_t)rank; u++) @@ -9489,9 +9489,9 @@ test_chunk_size_of_sizes(hid_t fapl) hid_t fcpl = H5I_INVALID_HID; /* File creation property list ID */ hid_t fid = H5I_INVALID_HID; /* File ID */ char filename[FILENAME_BUF_SIZE]; - size_t sizes[] = {2, 4, 8}; /* "size of sizes" values to test (all <= 8) */ - bool registered = false; /* Whether the bogus filter is registered */ - unsigned s; /* Local index variable */ + size_t sizes[] = {2, 4, 8}; /* "size of sizes" values to test (all <= 8) */ + bool registered = false; /* Whether the bogus filter is registered */ + unsigned s; /* Local index variable */ TESTING("chunked datasets with size of sizes <= 8"); @@ -9513,18 +9513,18 @@ test_chunk_size_of_sizes(hid_t fapl) for (s = 0; s < sizeof(sizes) / sizeof(sizes[0]); s++) { /* Single chunk index: cur dims == max dims == chunk dims */ - hsize_t single_dims[2] = {10, 10}; - hsize_t single_max[2] = {10, 10}; + hsize_t single_dims[2] = {10, 10}; + hsize_t single_max[2] = {10, 10}; /* Fixed array index: fixed max dims, multiple chunks */ - hsize_t farray_dims[2] = {40, 40}; - hsize_t farray_max[2] = {40, 40}; + hsize_t farray_dims[2] = {40, 40}; + hsize_t farray_max[2] = {40, 40}; /* Extensible array index: exactly one unlimited dimension */ - hsize_t earray_dims[2] = {40, 40}; - hsize_t earray_max[2] = {H5S_UNLIMITED, 40}; + hsize_t earray_dims[2] = {40, 40}; + hsize_t earray_max[2] = {H5S_UNLIMITED, 40}; /* Version 2 B-tree index: more than one unlimited dimension */ - hsize_t bt2_dims[2] = {40, 40}; - hsize_t bt2_max[2] = {H5S_UNLIMITED, H5S_UNLIMITED}; - hsize_t chunk[2] = {10, 10}; /* 10 * 10 * 4 = 400 bytes on disk */ + hsize_t bt2_dims[2] = {40, 40}; + hsize_t bt2_max[2] = {H5S_UNLIMITED, H5S_UNLIMITED}; + hsize_t chunk[2] = {10, 10}; /* 10 * 10 * 4 = 400 bytes on disk */ /* Create a file whose size of sizes is the value under test. The size * of addresses is left at its default. */ @@ -9536,14 +9536,14 @@ test_chunk_size_of_sizes(hid_t fapl) TEST_ERROR; /* Exercise each chunk index type that stores per-chunk sizes */ - if (test_sos_roundtrip_dset(fid, "single", 2, single_dims, single_max, chunk, - H5D_CHUNK_IDX_SINGLE) < 0) + if (test_sos_roundtrip_dset(fid, "single", 2, single_dims, single_max, chunk, H5D_CHUNK_IDX_SINGLE) < + 0) goto error; - if (test_sos_roundtrip_dset(fid, "farray", 2, farray_dims, farray_max, chunk, - H5D_CHUNK_IDX_FARRAY) < 0) + if (test_sos_roundtrip_dset(fid, "farray", 2, farray_dims, farray_max, chunk, H5D_CHUNK_IDX_FARRAY) < + 0) goto error; - if (test_sos_roundtrip_dset(fid, "earray", 2, earray_dims, earray_max, chunk, - H5D_CHUNK_IDX_EARRAY) < 0) + if (test_sos_roundtrip_dset(fid, "earray", 2, earray_dims, earray_max, chunk, H5D_CHUNK_IDX_EARRAY) < + 0) goto error; if (test_sos_roundtrip_dset(fid, "bt2", 2, bt2_dims, bt2_max, chunk, H5D_CHUNK_IDX_BT2) < 0) goto error; @@ -9600,14 +9600,14 @@ static herr_t test_sos_overflow_dset(hid_t fid, const char *dset_name, int rank, const hsize_t *dims, const hsize_t *maxdims, const hsize_t *chunk, H5D_chunk_index_t expected_idx) { - hid_t sid = H5I_INVALID_HID; /* Dataspace ID */ - hid_t did = H5I_INVALID_HID; /* Dataset ID */ - hid_t dcpl = H5I_INVALID_HID; /* Dataset creation property list ID */ - int *wbuf = NULL; /* Write buffer */ - H5D_chunk_index_t idx_type; /* Actual chunk index type */ - hsize_t nelmts = 1; /* Number of elements in the dataset */ - herr_t status = SUCCEED; /* Return value of the write/flush */ - size_t u; /* Local index variable */ + hid_t sid = H5I_INVALID_HID; /* Dataspace ID */ + hid_t did = H5I_INVALID_HID; /* Dataset ID */ + hid_t dcpl = H5I_INVALID_HID; /* Dataset creation property list ID */ + int *wbuf = NULL; /* Write buffer */ + H5D_chunk_index_t idx_type; /* Actual chunk index type */ + hsize_t nelmts = 1; /* Number of elements in the dataset */ + herr_t status = SUCCEED; /* Return value of the write/flush */ + size_t u; /* Local index variable */ for (u = 0; u < (size_t)rank; u++) nelmts *= dims[u]; @@ -9695,15 +9695,15 @@ test_sos_overflow_dset(hid_t fid, const char *dset_name, int rank, const hsize_t static herr_t test_chunk_size_of_sizes_overflow(hid_t fapl) { - hid_t my_fapl = H5I_INVALID_HID; /* Copy of the file access property list */ - hid_t fcpl = H5I_INVALID_HID; /* File creation property list ID */ - hid_t fid = H5I_INVALID_HID; /* File ID */ - char filename[FILENAME_BUF_SIZE]; - bool registered = false; /* Whether the bogus filter is registered */ + hid_t my_fapl = H5I_INVALID_HID; /* Copy of the file access property list */ + hid_t fcpl = H5I_INVALID_HID; /* File creation property list ID */ + hid_t fid = H5I_INVALID_HID; /* File ID */ + char filename[FILENAME_BUF_SIZE]; + bool registered = false; /* Whether the bogus filter is registered */ /* Single chunk index: cur dims == max dims == chunk dims (one 160000-byte chunk) */ - hsize_t single_dims[2] = {200, 200}; - hsize_t single_max[2] = {200, 200}; + hsize_t single_dims[2] = {200, 200}; + hsize_t single_max[2] = {200, 200}; hsize_t single_chunk[2] = {200, 200}; /* Fixed array index: fixed max dims, more than one chunk */ hsize_t farray_dims[2] = {400, 200}; From c926b58ab53924cccabed7a5e8f77a3da082091d Mon Sep 17 00:00:00 2001 From: Matthew Larson Date: Wed, 22 Jul 2026 09:36:28 -0500 Subject: [PATCH 4/6] Modify CHANGELOG comment --- release_docs/CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/release_docs/CHANGELOG.md b/release_docs/CHANGELOG.md index 2aa1aee4bea..88cbf283348 100644 --- a/release_docs/CHANGELOG.md +++ b/release_docs/CHANGELOG.md @@ -224,7 +224,7 @@ The `h5repack` tool now obtains its default low and high library version bounds ### Fixed silent truncation of filtered chunk sizes with a small file "size of sizes" - For all chunk index types in version-5 chunk layout messages, and for the single chunk index in any version of the chunk layout message, the on-disk size of a filtered chunk is encoded in a fixed-width field equal to the file's "size of sizes" (set via `H5Pset_sizes()`). The encode check assumed this field was always 8 bytes, so when the size of sizes was set to 2 or 4 an oversized chunk had its encoded size silently truncated (e.g. 160000 encoded in 2 bytes became 160000 & 0xFFFF = 28928), silently corrupting the chunk. The library now verifies that a filtered chunk's size fits in the file's size of sizes and reports an error at write time instead. + For all chunk index types in version-5 chunk layout messages, and for the single chunk index in any version of the chunk layout message, the on-disk size of a filtered chunk is encoded in a fixed-width field equal to the file's "size of sizes" (set via `H5Pset_sizes()`). The encode check assumed this field was always 8 bytes, so when the size of sizes was set to 2 or 4 an oversized chunk had its encoded size silently truncated (e.g., 160000 encoded in 2 bytes became 160000 & 0xFFFF = 28928), silently corrupting the chunk. The library now verifies that a filtered chunk's size fits in the file's size of sizes and reports an error at write time instead. ## Java Library From fca37b9c9dd507441f3ec7aaf649479e49f5d7c4 Mon Sep 17 00:00:00 2001 From: Matt L <124107509+mattjala@users.noreply.github.com> Date: Wed, 5 Aug 2026 10:31:53 -0500 Subject: [PATCH 5/6] remove erroneously included changelog entries --- release_docs/CHANGELOG.md | 71 --------------------------------------- 1 file changed, 71 deletions(-) diff --git a/release_docs/CHANGELOG.md b/release_docs/CHANGELOG.md index a057094c91f..c2ca3e4313e 100644 --- a/release_docs/CHANGELOG.md +++ b/release_docs/CHANGELOG.md @@ -77,77 +77,6 @@ We would like to thank the many HDF5 community members who contributed to this r ## Library -### Fixed a possible heap leak in a utility function - - A couple of unnecessary allocations in h5str_convert were never freed, causing memory leaks. These are now removed. - - Fixes GitHub issue #6511 -### Fixed bug that prevented internal library filters from printing error messages - - Previously the error stack would be cleared when exiting a data filter, even an internal library filter, so the user could not see what caused the filter to fail. This has been fixed by not treating internal data filters like a user callback. Note that user-defined or third-party filters that use the default error stack will need to print that stack before returning from their callbacks. - -### Fixed error when reading variable-length chunked datasets in read-only mode - - Passing NULL for the callback function pointer to H5Aiterate2 and H5Aiterate_by_name was not detected, leading to a subsequent access of an uninitialized pointer. This is now fixed. - - Fixes CVE-2025-9274 - -### Fixed error when reading variable-length chunked datasets in read-only mode - - When reading from a chunked dataset with a variable-length type, a non-default fill value, and unwritten chunks, the library would internally try to write data to the file and fail due to writing to a read-only file. Reworked the I/O code to avoid these writes in this case. This may also improve performance and file space usage in similar cases with files open with write access. - -### Validate free space section type during decode - - When loading a free space section info block, the per-section type byte read from the file was used directly to index the free space manager's section class array and to call the class `deserialize` callback, guarded only by an assertion that is removed in release builds. A corrupted or fuzzed file could supply a type beyond the number of registered classes, causing an out-of-bounds read of the class array and an indirect call through a bogus function pointer. `H5FS__cache_sinfo_deserialize()` now rejects a section type that is not less than the number of section classes. - -### Fixed a heap buffer overflow when decoding a shared message list - - When reading a shared object header message (SOHM) list from the metadata cache, `H5SM__cache_list_deserialize()` allocated the message array for `list_max` entries but drove the decode loop with the `num_messages` count read from the on-disk index header. A corrupted or malicious file whose `num_messages` exceeds `list_max` caused writes past the end of the array and reads past the end of the input buffer. The count is now validated against `list_max` before the loop runs. - -### HTTP 403 errors in the ROS3 VFD for object keys with special characters - - The ROS3 VFD did not URI-encode the S3 object key when building the HTTP request path, so keys containing characters that AWS Signature Version 4 requires to be percent-encoded — such as the '=' in Hive-style `key=value` partition prefixes, '+', or spaces — produced a signed request whose signature did not match S3's server-side recomputation. S3 rejects such requests with `SignatureDoesNotMatch`, which surfaces as an HTTP 403 error (indistinguishable from a permissions error on a HEAD request), even though tools like the AWS CLI could access the same object. The object key is now percent-encoded exactly once when the request path is built, matching the behavior of other S3 clients. Note that URLs must now be passed to the ROS3 VFD with their object keys unencoded; a key that was pre-encoded as a workaround for this issue will now be double-encoded and fail to resolve. - -### Fixed file descriptor leaks in stdio VFD error paths - - Fixed multiple resource leaks in the H5FDstdio driver where file descriptors were not properly closed on error paths. The error handling code was incorrectly attempting to close a local variable instead of the file pointer stored in the file structure, leading to file descriptor leaks. This issue affected 5 error paths in `H5FD_stdio_open()` and could cause file descriptor exhaustion in long-running applications. - -### Added defensive NULL pointer checks in native VOL connector - - Added assertion checks for NULL pointer parameters in `H5VL_native_get_file_struct()` to catch programming errors earlier and improve code robustness. - -### Added checks for data filter behavior - - The library now verifies that the returned data size from a data filter's filter callback function can fit inside the returned data buffer size. The library also checks that, when data is filtered then unfiltered (filtered in reverse), the returned data size is exactly the same as the original data size. - -### Fixed bugs with chunk buffer handling - - Fixed a bug in the deflate filter that caused it to report the wrong buffer size. Fixed a bug in the chunk copy code that could cause a background buffer overflow. Fixed a bug in the chunk copy code that could cause a double free if the filter realloced the data buffer. - -### Fixed checking of data alignment requirements in direct I/O VFD - - The direct I/O VFD attempts to determine data alignment requirements for a file on file open to try and avoid extra work when data alignment isn't required. Depending on the file access flags used when opening a file, the VFD could incorrectly determine these requirements for either writes or reads, eventually leading to a possible EINVAL return value on write or read. This has been fixed by separately determining the requirements for writes and reads and being more conservative about trying to avoid data alignment requirements. - -### Fixed integer overflow in array datatype element count computation - - Fixed a bug in H5O__dtype_decode_helper() where the loop computing the total number of elements in an array datatype had no per-step overflow check. On 64-bit systems, large dimension sizes could cause the element count to wrap around, bypassing the post-loop overflow check and producing silently incorrect results in downstream type conversion and size calculations. - -### Fixed an issue with chunked datasets using the wrong index type with parallel HDF5 - - Fixed a bug in parallel HDF5 that would cause chunked datasets with fixed dimensions and without filters applied to use the "none" index type instead of the "fixed array" index type. - -### Fixed an issue with decoding metadata cache image superblock extension messages - - Fixed a bug where loading of a metadata cache image superblock extension message would fail when the image had an undefined address and size of 0. - -### Fixed an issue with an incorrect file format validation check when decoding metadata cache entries - - Fixed a bug where a flag in H5Cimage.c wasn't getting set correctly for release builds of HDF5, leading to incorrect error checking when reconstructing metadata cache entries. - -### Hardened decoding of serialized dataspace selections against malformed buffers - - `H5S_select_deserialize()` and the per-selection-type deserialize callbacks (all, hyperslab, none, and point) previously computed the pointer to the last valid buffer byte as `buffer + size - 1` without first checking the buffer size. A buffer shorter than the 4-byte selection-type header, or a zero-length selection-info buffer, would underflow this computation and produce an out-of-bounds end pointer, defeating subsequent overflow checks. The deserialize routines now reject a buffer that is too small to hold the selection type, and they reject an empty selection-info buffer before deriving the end pointer. Hyperslab decoding additionally now rejects a serialized rank of 0 or greater than `H5S_MAX_RANK`. As a companion fix, `H5S__hyper_serialize()` now returns an error when asked to serialize a hyperslab selection on a rank-0 (scalar or null) dataspace, a state that can arise when a dataspace extent is collapsed to a scalar after a hyperslab selection has already been made. - ### Fixed silent truncation of filtered chunk sizes with a small file "size of sizes" For all chunk index types in version-5 chunk layout messages, and for the single chunk index in any version of the chunk layout message, the on-disk size of a filtered chunk is encoded in a fixed-width field equal to the file's "size of sizes" (set via `H5Pset_sizes()`). The encode check assumed this field was always 8 bytes, so when the size of sizes was set to 2 or 4 an oversized chunk had its encoded size silently truncated (e.g., 160000 encoded in 2 bytes became 160000 & 0xFFFF = 28928), silently corrupting the chunk. The library now verifies that a filtered chunk's size fits in the file's size of sizes and reports an error at write time instead. From 4582d1f980f5c7dc6288ab09cea16a3e95b5ccbb Mon Sep 17 00:00:00 2001 From: Matthew Larson Date: Thu, 6 Aug 2026 15:33:22 -0500 Subject: [PATCH 6/6] Reference the fixed issue in the CHANGELOG entry Co-Authored-By: Claude Opus 5 --- release_docs/CHANGELOG.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/release_docs/CHANGELOG.md b/release_docs/CHANGELOG.md index c2ca3e4313e..6ddfe425cc0 100644 --- a/release_docs/CHANGELOG.md +++ b/release_docs/CHANGELOG.md @@ -81,6 +81,8 @@ We would like to thank the many HDF5 community members who contributed to this r For all chunk index types in version-5 chunk layout messages, and for the single chunk index in any version of the chunk layout message, the on-disk size of a filtered chunk is encoded in a fixed-width field equal to the file's "size of sizes" (set via `H5Pset_sizes()`). The encode check assumed this field was always 8 bytes, so when the size of sizes was set to 2 or 4 an oversized chunk had its encoded size silently truncated (e.g., 160000 encoded in 2 bytes became 160000 & 0xFFFF = 28928), silently corrupting the chunk. The library now verifies that a filtered chunk's size fits in the file's size of sizes and reports an error at write time instead. + Fixes GitHub issue #6023 + ## Java Library ## Configuration