From 393410babb8ad1586968db27725b5d002ca87d20 Mon Sep 17 00:00:00 2001 From: vchoi-hdfgroup <55293060+vchoi-hdfgroup@users.noreply.github.com> Date: Wed, 19 Aug 2026 00:22:44 -0700 Subject: [PATCH 1/9] Fix issue #6605: (#6615) --Fix overloaded N and inconsistent # indexing in V2 B-tree section --Clean up Global Heap section as well. --- docs/doxygen/dox/H5.format.4.0.dox | 122 ++++++++++++++++------------- 1 file changed, 69 insertions(+), 53 deletions(-) diff --git a/docs/doxygen/dox/H5.format.4.0.dox b/docs/doxygen/dox/H5.format.4.0.dox index b4558211976..8a404c8f149 100644 --- a/docs/doxygen/dox/H5.format.4.0.dox +++ b/docs/doxygen/dox/H5.format.4.0.dox @@ -1221,37 +1221,37 @@ the header). Version Type - Records 0, 1, 2...N-1 (variable size) + Record \#0, Record \#1, ... Record \#(R-1) (variable size) -
Child Node Pointer 0O

+
Child Node Pointer \#0O

-
Number of Records N0 for Child Node 0 (variable size) +
Number of Records for Child Node \#0 (variable size) -
Total Number of Records for Child Node 0 (optional, variable size) +
Total Number of Records for Child Node \#0 (optional, variable size) -
Child Node Pointer 1O

+
Child Node Pointer \#1O

-
Number of Records N1 for Child Node 1 (variable size) +
Number of Records for Child Node \#1 (variable size) -
Total Number of Records for Child Node 1 (optional, variable size) +
Total Number of Records for Child Node \#1 (optional, variable size) ... -
Child Node Pointer NO

+
Child Node Pointer \#RO

-
Number of Records Nn for Child Node N (variable size) +
Number of Records for Child Node \#R (variable size) -
Total Number of Records for Child Node N (optional, variable size) +
Total Number of Records for Child Node \#R (optional, variable size) Checksum @@ -1281,35 +1281,45 @@ the header). the header. - Records - The size of this field is determined by the number of records for this node and the record size - (from the header). The format of records depends on the type of B-tree. + Record \#i (i = 0 to (R-1)) + Let R be the number of records for this node: for the root node, R is given + by the B-tree Header's Number of Records in Root Node field; for any other node, + R is given by the Number of Records for Child Node \#j field in this node's + parent, specifically the pointer triplet whose Child Node Pointer \#j gave this + node's address. The size of this field is + determined by R and the Record Size (the latter always from the header); the + format of records depends on the type of B-tree. Because each internal node has one more child + pointer triplet than it has records, this node has R+1 child node pointers, indexed + \#0 through \#R (see below). - Child Node Pointer - This field is the address of the child node pointed to by the internal node. + Child Node Pointer \#j (j = 0 to R) + This field is the address of child \#j pointed to by this internal node. Together with the + following field, Number of Records for Child Node \#j, and (where present) the field + after that, Total Number of Records for Child Node \#j, this forms what the rest of + this table calls a pointer triplet — the repeating group of per-child fields + between Child Node Pointer \#j and Child Node Pointer \#(j+1). - Number of Records in Child Node - This is the number of records in the child node pointed to by the corresponding Node Pointer.
+ Number of Records for Child Node \#j (j = 0 to R) + This is the number of records in child \#j (see the Child Node Pointer field above).
The number of bytes used to store this field is determined by the maximum possible number of records able to be stored in the child node.
The maximum number of records in a child node is computed in the following way:
Note that leaf nodes do not encode any child pointer triplets, so the maximum number of records in a - leaf node is just the node size minus the leaf node overhead, divided by the record size.
- Also note that the first level of internal nodes above the leaf nodes do not encode the Total - Number of Records in Child Node value in the child pointer triplets (since it is the same as - the Number of Records in Child Node), so the maximum number of records in these nodes is - computed with the equation above, but using (Child Pointer, Number of Records in Child - Node) pairs instead of triplets.
+ leaf node is just the Node Size minus the leaf node overhead, divided by the Record Size.
+ Also note that the first level of internal nodes above the leaf nodes — the “twig” + internal nodes — do not encode a Total Number of Records for Child Node \#j value + (since it would be the same as Number of Records for Child Node \#j), so the maximum number + of records in these nodes is computed with the equation above, but using (Child Node Pointer + \#j, Number of Records for Child Node \#j) pairs instead of full triplets.
The number of bytes used to encode this field is the least number of bytes required to encode the maximum number of records in a child node value for the child nodes below this level in the B-tree.
For example, if the maximum number of child records is 123, one byte will be used to encode these @@ -1318,11 +1328,10 @@ the header). is 8 (in other words, an unsigned 64-bit integer). - Total Number of Records in Child Node - This is the total number of records for the node pointed to by the corresponding Node Pointer - and all its children. This field exists only in nodes whose depth in the B-tree node is greater than 1 - (in other words, the “twig” internal nodes, just above leaf nodes, do not store this field - in their child node pointers).
+ Total Number of Records for Child Node \#j (j = 0 to R) + This is the total number of records for child \#j and all its descendants. This field exists + only in nodes whose depth in the B-tree node is greater than 1; see the note about + “twig” internal nodes under Number of Records for Child Node \#j above.
The number of bytes used to store this field is determined by the maximum possible number of records able to be stored in the child node and its descendants.
The maximum possible number of records able to be stored in a child node and its descendants is @@ -1332,7 +1341,7 @@ the header). maximum possible number of records in child node pointers for the level of nodes two levels above leaf nodes. This process is continued up to any level in the B-tree.
The number of bytes used to encode this value is computed in the same way as for the Number - of Records in Child Node field. + of Records for Child Node \#j field. Checksum @@ -1354,7 +1363,7 @@ the header). Version Type - Record 0, 1, 2...N-1 (variable size) + Record \#0, Record \#1, ... Record \#(R-1) (variable size) Checksum @@ -1382,9 +1391,13 @@ the header). the header. - Records - The size of this field is determined by the number of records for this node and the record size - (from the header). The format of records depends on the type of B-tree. + Record \#i (i = 0 to (R-1)) + Let R be the number of records for this node: for the root node, R is given + by the B-tree Header's Number of Records in Root Node field; for any other node, + R is given by the Number of Records for Child Node \#j field in this leaf's + parent (always a “twig” internal node, so this is one of a pointer pair, not a full + triplet). The size of this field is determined by R and the Record Size (the + latter always from the header); the format of records depends on the type of B-tree. Checksum @@ -2386,10 +2399,10 @@ as an atomic object, addressing goal A. When a global heap object is deleted from a collection (which occurs when its reference count falls to zero), objects located after the deleted object in the collection are packed down toward the beginning of the -collection and the collection’s global heap object 0 is created (if possible) or its size is increased +collection and the collection’s global heap object \#0 is created (if possible) or its size is increased to account for the recently freed space. There are no gaps between objects in each collection, with the possible exception of the final space in the collection, if it is not large enough to hold the header for the -collection’s global heap object 0. These features address goal C. +collection’s global heap object \#0. These features address goal C. The HDF5 library creates global heap collections as needed, so there may be multiple collections throughout the file. The set of all of them is abstractly called the “global heap”, although they do not @@ -2417,19 +2430,19 @@ global heap ID. The format for global heap IDs is described at the end of this s
Collection SizeL

-
Global Heap Object 1

+
Global Heap Object \#1

-
Global Heap Object 2

+
Global Heap Object \#2


...

-
Global Heap Object N

+
Global Heap Object \#N

-
Global Heap Object 0 (free space)

+
Global Heap Object \#0 (free space)

\li Items marked with an ‘L’ in the above table are of the size specified in @@ -2459,16 +2472,19 @@ global heap ID. The format for global heap IDs is described at the end of this s about each heap object). - Global Heap Object 1 through N - The objects are stored in any order with no intervening unused space. + Global Heap Object \#1 through \#N + Let N be the number of real (non-free-space) objects in this collection. N is not + stored explicitly; a reader determines it by parsing objects sequentially, in the order stored, until + either the free-space object (Global Heap Object \#0, Heap Object Index 0) is reached or the + Collection Size is exhausted. The objects are stored in any order with no intervening unused space. - Global Heap Object 0 - Global Heap Object 0 (zero), when present, represents the free space in the collection. Free space always - appears at the end of the collection. If the free space is too small to store the header for Object 0 - (described below) then the header is implied and is not written.
- The field Object Size for Object 0 indicates the amount of possible free space in the collection - including the 16-byte header size of Object 0. + Global Heap Object \#0 + Global Heap Object \#0 (zero), when present, represents the free space in the collection. Free space + always appears at the end of the collection. If the free space is too small to store the header for + Object \#0 (described below) then the header is implied and is not written.
+ The field Object Size for Object \#0 indicates the amount of possible free space in the + collection including the 16-byte header size of Object \#0. @@ -2512,7 +2528,7 @@ global heap ID. The format for global heap IDs is described at the end of this s Reference Count All heap objects have a reference count field. An object which is referenced from some other part of the - file will have a positive reference count. The reference count for Object 0 is always zero. + file will have a positive reference count. The reference count for Object \#0 is always zero. Reserved From baf2733277226212507a57b1d7b90e228e154ec6 Mon Sep 17 00:00:00 2001 From: scivision <10931741+scivision@users.noreply.github.com> Date: Wed, 19 Aug 2026 19:29:00 -0400 Subject: [PATCH 2/9] cmake remove noop policies (#6586) These are made no-ops by the cmake_minimum_required(VERSION 3.26) * CMP0074 by CMake 3.12 * CMP0077 by CMake 3.13 * CMP0083 by CMake 3.14 * CMP0127 by CMake 3.22 * CMP0135 by CMake 3.24 --- CMakeLists.txt | 25 +------------------------ 1 file changed, 1 insertion(+), 24 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 52544283440..d4f534872df 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -30,33 +30,11 @@ cmake_minimum_required (VERSION 3.26) project (HDF5 C) -if (POLICY CMP0074) - # find_package() uses _ROOT variables. - cmake_policy (SET CMP0074 NEW) -endif () - if (POLICY CMP0144) - # is the upper-cased package name. + # CMake >= 3.27 is the upper-cased package name. cmake_policy (SET CMP0144 NEW) endif () -if (POLICY CMP0083) - # To control generation of Position Independent Executable (PIE) or not, - # some flags are required at link time. - cmake_policy (SET CMP0083 NEW) -endif () - -if (POLICY CMP0127) - # to evaluate each condition as if(), where is re-parsed - # as if literally written in a call to if(). - cmake_policy (SET CMP0127 NEW) -endif () - -# Avoid warning about DOWNLOAD_EXTRACT_TIMESTAMP in CMake 3.24: -if (CMAKE_VERSION VERSION_GREATER_EQUAL "3.24.0") - cmake_policy(SET CMP0135 NEW) -endif() - #----------------------------------------------------------------------------- # Instructions for use : Normal Build # @@ -1571,4 +1549,3 @@ include (CMakeInstallation.cmake) # Hide some miscellaneous CMake variables from the default GUI #----------------------------------------------------------------------------- mark_as_advanced (FETCHCONTENT_BASE_DIR FETCHCONTENT_FULLY_DISCONNECTED FETCHCONTENT_QUIET FETCHCONTENT_UPDATES_DISCONNECTED) - From 767ac04b21c786427772e5b7392039f76952a090 Mon Sep 17 00:00:00 2001 From: Matt L <124107509+mattjala@users.noreply.github.com> Date: Wed, 19 Aug 2026 19:31:41 -0500 Subject: [PATCH 3/9] Fix standalone example build issues (#6598) * Build the examples as C++11 to match the HDF5 C++ library The standalone examples build forces CMAKE_CXX_STANDARD 98, but H5public.h includes , which requires C++11. Any C++ translation unit that includes hdf5.h is therefore affected, not just users of the C++ API, and the HDF5 C++ library itself is built as C++11 (config/flags/HDFCompilerCXXFlags.cmake). The C++ examples do not compile as a result, against static or shared HDF5 alike. Only the standalone build is affected, which is why this is not visible in ordinary use. The C++98 setting lives in BASIC_SETTINGS, and HDF5Examples/CMakeLists.txt skips that whole block when EXAMPLES_EXTERNALLY_CONFIGURED is set -- which HDF5 does for its own in-tree example build (config/cmake/HDF5ExampleCache.cmake). Built in tree, the examples inherit HDF5's C++11 and compile normally, and that is the path the CI workflows exercise. The standalone path, where the C++98 setting does apply, is driven by the release scripts rather than by the workflows, and has the C++ examples off by default. * Select the examples' HL, Fortran and C++ libraries on the right variable When the examples are built standalone against an installed HDF5, the HL, Fortran and C++ branches choose between the shared and static libraries using BUILD_SHARED_LIBS, while the C branch just above them uses H5EXAMPLE_USE_SHARED_LIBS. H5EXAMPLE_USE_SHARED_LIBS is what decides whether the "shared" or the "static" component is requested from find_package, so only the matching HDF5___FOUND variables are ever set. BUILD_SHARED_LIBS cannot select a linkage on its own; it can only agree or fail to match. Of its four combinations with H5EXAMPLE_USE_SHARED_LIBS, three produce no observable difference. In the fourth, H5EXAMPLE_USE_SHARED_LIBS=ON with BUILD_SHARED_LIBS unset, the shared branch is not taken and the static branch cannot be, so the HL, Fortran and C++ examples are disabled with "libs not found" even though the libraries are installed and were found. Use H5EXAMPLE_USE_SHARED_LIBS, which is the declared option and is already what the C branch uses. A build driven through config/examples/CTestScript.cmake does not reach the broken combination, because it configures with HDF5Examples/config/cmake/cacheinit.cmake, which forces BUILD_SHARED_LIBS=ON. A direct cmake invocation without that cache file does. In either case the HL, Fortran and C++ examples are off by default, so this is only visible once they are enabled. BUILD_SHARED_LIBS remains documented as a user option in config/examples/HDF5_Examples_options.cmake but no longer influences library selection; that comment should be revisited separately. --- .../config/cmake/HDFExampleMacros.cmake | 11 +++++--- release_docs/CHANGELOG.md | 28 +++++++++++++++++++ 2 files changed, 35 insertions(+), 4 deletions(-) diff --git a/HDF5Examples/config/cmake/HDFExampleMacros.cmake b/HDF5Examples/config/cmake/HDFExampleMacros.cmake index 68a970d789f..9b89c81dedd 100644 --- a/HDF5Examples/config/cmake/HDFExampleMacros.cmake +++ b/HDF5Examples/config/cmake/HDFExampleMacros.cmake @@ -37,7 +37,10 @@ macro (BASIC_SETTINGS varname) if (H5EXAMPLE_BUILD_CXX) ENABLE_LANGUAGE (CXX) - set (CMAKE_CXX_STANDARD 98) + # Match the standard the HDF5 C++ library itself is built with. + # H5public.h includes , so any C++ translation unit + # with hdf5.h needs C++11 + set (CMAKE_CXX_STANDARD 11) set (CMAKE_CXX_STANDARD_REQUIRED TRUE) set (CMAKE_CXX_EXTENSIONS OFF) endif () @@ -246,7 +249,7 @@ macro (HDF5_SUPPORT) message (STATUS "HDF5 HL libs not found - disable build of HL examples") else () if (H5EXAMPLE_BUILD_HL AND ${HDF5_PROVIDES_HL_LIB}) - if (BUILD_SHARED_LIBS AND HDF5_shared_HL_FOUND) + if (H5EXAMPLE_USE_SHARED_LIBS AND HDF5_shared_HL_FOUND) set (H5EXAMPLE_HDF5_LINK_LIBS ${H5EXAMPLE_HDF5_LINK_LIBS} ${HDF5_HL_SHARED_LIBRARY}) elseif (HDF5_static_HL_FOUND) set (H5EXAMPLE_HDF5_LINK_LIBS ${H5EXAMPLE_HDF5_LINK_LIBS} ${HDF5_HL_STATIC_LIBRARY}) @@ -262,7 +265,7 @@ macro (HDF5_SUPPORT) message (STATUS "HDF5 Fortran libs not found - disable build of Fortran examples") else () if (H5EXAMPLE_BUILD_FORTRAN AND ${HDF5_PROVIDES_FORTRAN}) - if (BUILD_SHARED_LIBS AND HDF5_shared_Fortran_FOUND) + if (H5EXAMPLE_USE_SHARED_LIBS AND HDF5_shared_Fortran_FOUND) set (H5EXAMPLE_HDF5_LINK_LIBS ${H5EXAMPLE_HDF5_LINK_LIBS} ${HDF5_FORTRAN_SHARED_LIBRARY}) if (H5EXAMPLE_BUILD_HL AND ${HDF5_PROVIDES_HL_LIB}) set (H5EXAMPLE_HDF5_LINK_LIBS ${H5EXAMPLE_HDF5_LINK_LIBS} ${HDF5_FORTRAN_HL_SHARED_LIBRARY}) @@ -284,7 +287,7 @@ macro (HDF5_SUPPORT) message (STATUS "HDF5 CXX libs not found - disable build of CXX examples") else () if (H5EXAMPLE_BUILD_CXX AND ${HDF5_PROVIDES_CPP_LIB}) - if (BUILD_SHARED_LIBS AND HDF5_shared_CXX_FOUND) + if (H5EXAMPLE_USE_SHARED_LIBS AND HDF5_shared_CXX_FOUND) set (H5EXAMPLE_HDF5_LINK_LIBS ${H5EXAMPLE_HDF5_LINK_LIBS} ${HDF5_CXX_SHARED_LIBRARY}) if (H5EXAMPLE_BUILD_HL AND ${HDF5_PROVIDES_HL_LIB}) set (H5EXAMPLE_HDF5_LINK_LIBS ${H5EXAMPLE_HDF5_LINK_LIBS} ${HDF5_CXX_HL_SHARED_LIBRARY}) diff --git a/release_docs/CHANGELOG.md b/release_docs/CHANGELOG.md index 5a7b7093c8b..7d6ef588702 100644 --- a/release_docs/CHANGELOG.md +++ b/release_docs/CHANGELOG.md @@ -91,6 +91,34 @@ We would like to thank the many HDF5 community members who contributed to this r The installed CMake package version configuration file for the library previously used `SameMinorVersion` for the version compatibility logic, causing a `find_package(HDF5 X.Y.Z)` call to fail unless the version of a located HDF5 installation matched both `X` and `Y` of the version number exactly (i.e., releases with a greater minor version number weren't considered backward compatible). This reflected the version compatibility of HDF5 releases prior to version 2.0.0, but doesn't reflect the version compatibility of HDF5 version 2.0.0+ releases. The version compatibility logic now uses `SameMajorVersion`, so a `find_package(HDF5 X.Y.Z)` call will accept all versions of HDF5 where the major version matches `X` (i.e., only releases with a greater major version number will be rejected as not backward compatible). +### Fixed the C++ examples failing to compile when built standalone + + The standalone examples build used C++98, but `H5public.h` includes + ``, which requires C++11. This affected any C++ translation unit + including `hdf5.h`, and did not match the HDF5 C++ library itself, which is + built as C++11. The C++ examples did not compile, against either static or + shared HDF5. The examples are now built as C++11. + + Only the standalone build was affected. Examples built as part of the HDF5 + build inherit the library's own C++ standard. + +### Fixed the examples skipping the HL, Fortran and C++ programs in some configurations + + When built standalone against an installed HDF5, the examples chose between + the shared and static HL, Fortran and C++ libraries using `BUILD_SHARED_LIBS`, + while the C library used `H5EXAMPLE_USE_SHARED_LIBS`. Since + `H5EXAMPLE_USE_SHARED_LIBS` determines which component is requested from + `find_package`, and therefore which `HDF5___FOUND` variables + exist, `BUILD_SHARED_LIBS` could not select a linkage on its own. With + `H5EXAMPLE_USE_SHARED_LIBS` on and `BUILD_SHARED_LIBS` unset, those examples + were disabled with a "libs not found" message even though the libraries were + installed and had been found. The selection now uses + `H5EXAMPLE_USE_SHARED_LIBS`, matching the C library. + + Builds driven through `CTestScript.cmake` were not affected, since its cache + file forces `BUILD_SHARED_LIBS` on. This affected cases where the examples + were built directly without that cache file. + ## Tools ### Fixed an issue with quoting of data values in h5ls and h5dump when displaying as ASCII characters From f934da4bfe139cce8989e76b537766d9d3e78657 Mon Sep 17 00:00:00 2001 From: jhendersonHDF Date: Thu, 20 Aug 2026 09:42:57 -0500 Subject: [PATCH 4/9] Fix memory leaks and ID reference counting in H5E code (#6607) When pushing errors to an error stack that is already full, the library skipped the push operation but didn't inform calling code about what happened. This resulted in calling code leaking memory, leaving reference counts on IDs incremented and causing an infinite loop while closing the library. --- release_docs/CHANGELOG.md | 11 +++++++++ src/H5E.c | 43 +++++++++++++++++++++++++-------- src/H5Edeprec.c | 33 +++++++++++++++++++------ src/H5Eint.c | 18 ++++++++------ src/H5Epkg.h | 2 +- test/error_test.c | 51 +++++++++++++++++++++++++++++++++++++++ 6 files changed, 133 insertions(+), 25 deletions(-) diff --git a/release_docs/CHANGELOG.md b/release_docs/CHANGELOG.md index 7d6ef588702..fbf20d93143 100644 --- a/release_docs/CHANGELOG.md +++ b/release_docs/CHANGELOG.md @@ -77,6 +77,17 @@ We would like to thank the many HDF5 community members who contributed to this r ## Library +### Fixed memory leaks and ID reference count issues when pushing an error to an error stack that is full + + When an error is pushed to an error stack, the library may make a copy of the file + and function strings to ensure that they exist for the same duration as the error + stack entry. When an error stack is full, the library simply makes any further pushes + no-ops, but previously gave no information to calling code that this happened. This + caused calling code to assume that the duplicated strings were owned by an error stack + entry that was never pushed, leaking the duplicated strings. Additionally, IDs + associated with the error stack entry were left with incremented reference counts, + resulting in an infinite loop while closing the library. + ### Library shutdown no longer aborts on a detected infinite loop When the library detects that it cannot make progress closing itself (an "infinite loop closing library"), it no longer calls `abort()`. The abort behaved inconsistently, only firing when automatic error message display was enabled. Additionally, terminating the entire host process on a shutdown-time condition is undesirable for applications that embed HDF5. The library now reports the condition (when error display is enabled) and returns without aborting. diff --git a/src/H5E.c b/src/H5E.c index 915f4722127..936d80c771a 100644 --- a/src/H5E.c +++ b/src/H5E.c @@ -525,12 +525,16 @@ herr_t H5Epush2(hid_t err_stack, const char *file, const char *func, unsigned line, hid_t cls_id, hid_t maj_id, hid_t min_id, const char *fmt, ...) { - H5E_stack_t *estack; /* Pointer to error stack to modify */ - va_list ap; /* Varargs info */ - bool va_started = false; /* Whether the variable argument list is open */ - const char *tmp_file; /* Copy of the file name */ - const char *tmp_func; /* Copy of the function name */ - herr_t ret_value = SUCCEED; /* Return value */ + H5E_stack_t *estack; /* Pointer to error stack to modify */ + va_list ap; /* Varargs info */ + htri_t push_ret = true; /* Was an error stack entry actually pushed? */ + bool va_started = false; /* Whether the variable argument list is open */ + char *tmp_file = NULL; /* Copy of the file name */ + char *tmp_func = NULL; /* Copy of the function name */ + bool inc_cls_id = false; /* Incremented error class ID ref. count? */ + bool inc_maj_id = false; /* Incremented major error ID ref. count? */ + bool inc_min_id = false; /* Incremented minor error ID ref. count? */ + herr_t ret_value = SUCCEED; /* Return value */ /* Don't clear the error stack! :-) */ FUNC_ENTER_API_NOCLEAR(FAIL) @@ -562,18 +566,25 @@ H5Epush2(hid_t err_stack, const char *file, const char *func, unsigned line, hid HGOTO_ERROR(H5E_ERROR, H5E_CANTALLOC, FAIL, "can't duplicate function string"); /* Increment refcount on non-library IDs */ - if (cls_id != H5E_ERR_CLS_g) + if (cls_id != H5E_ERR_CLS_g) { if (H5I_inc_ref(cls_id, false) < 0) HGOTO_ERROR(H5E_ERROR, H5E_CANTINC, FAIL, "can't increment class ID"); - if (maj_id < H5E_first_maj_id_g || maj_id > H5E_last_maj_id_g) + inc_cls_id = true; + } + if (maj_id < H5E_first_maj_id_g || maj_id > H5E_last_maj_id_g) { if (H5I_inc_ref(maj_id, false) < 0) HGOTO_ERROR(H5E_ERROR, H5E_CANTINC, FAIL, "can't increment major error ID"); - if (min_id < H5E_first_min_id_g || min_id > H5E_last_min_id_g) + inc_maj_id = true; + } + if (min_id < H5E_first_min_id_g || min_id > H5E_last_min_id_g) { if (H5I_inc_ref(min_id, false) < 0) HGOTO_ERROR(H5E_ERROR, H5E_CANTINC, FAIL, "can't increment minor error ID"); + inc_min_id = true; + } /* Push the error on the stack */ - if (H5E__push_stack(estack, true, tmp_file, tmp_func, line, cls_id, maj_id, min_id, fmt, &ap) < 0) + push_ret = H5E__push_stack(estack, true, tmp_file, tmp_func, line, cls_id, maj_id, min_id, fmt, &ap); + if (push_ret < 0) HGOTO_ERROR(H5E_ERROR, H5E_CANTSET, FAIL, "can't push error on stack"); } @@ -581,6 +592,18 @@ done: if (va_started) va_end(ap); + if (ret_value < 0 || !push_ret) { + if (inc_cls_id && H5I_dec_ref(cls_id) < 0) + HDONE_ERROR(H5E_ERROR, H5E_CANTDEC, FAIL, "can't decrement class ID"); + if (inc_maj_id && H5I_dec_ref(maj_id) < 0) + HDONE_ERROR(H5E_ERROR, H5E_CANTDEC, FAIL, "can't decrement major error ID"); + if (inc_min_id && H5I_dec_ref(min_id) < 0) + HDONE_ERROR(H5E_ERROR, H5E_CANTDEC, FAIL, "can't decrement minor error ID"); + + free(tmp_func); + free(tmp_file); + } + FUNC_LEAVE_API(ret_value) } /* end H5Epush2() */ diff --git a/src/H5Edeprec.c b/src/H5Edeprec.c index 5f32ccd204f..8d57d7ad16f 100644 --- a/src/H5Edeprec.c +++ b/src/H5Edeprec.c @@ -182,10 +182,13 @@ done: herr_t H5Epush1(const char *file, const char *func, unsigned line, H5E_major_t maj, H5E_minor_t min, const char *str) { - H5E_stack_t *estack; /* Pointer to error stack to modify */ - const char *tmp_file; /* Copy of the file name */ - const char *tmp_func; /* Copy of the function name */ - herr_t ret_value = SUCCEED; /* Return value */ + H5E_stack_t *estack; /* Pointer to error stack to modify */ + htri_t push_ret = true; /* Was an error stack entry actually pushed? */ + char *tmp_file = NULL; /* Copy of the file name */ + char *tmp_func = NULL; /* Copy of the function name */ + bool inc_maj_id = false; /* Incremented major error ID ref. count? */ + bool inc_min_id = false; /* Incremented minor error ID ref. count? */ + herr_t ret_value = SUCCEED; /* Return value */ /* Don't clear the error stack! :-) */ FUNC_ENTER_API_NOCLEAR(FAIL) @@ -203,19 +206,35 @@ H5Epush1(const char *file, const char *func, unsigned line, H5E_major_t maj, H5E HGOTO_ERROR(H5E_ERROR, H5E_CANTALLOC, FAIL, "can't duplicate function string"); /* Increment refcount on non-library IDs */ - if (maj < H5E_first_maj_id_g || maj > H5E_last_maj_id_g) + if (maj < H5E_first_maj_id_g || maj > H5E_last_maj_id_g) { if (H5I_inc_ref(maj, false) < 0) HGOTO_ERROR(H5E_ERROR, H5E_CANTINC, FAIL, "can't increment major error ID"); - if (min < H5E_first_min_id_g || min > H5E_last_min_id_g) + inc_maj_id = true; + } + if (min < H5E_first_min_id_g || min > H5E_last_min_id_g) { if (H5I_inc_ref(min, false) < 0) HGOTO_ERROR(H5E_ERROR, H5E_CANTINC, FAIL, "can't increment minor error ID"); + inc_min_id = true; + } /* Push the error on the default error stack */ - if (H5E__push_stack(estack, true, tmp_file, tmp_func, line, H5E_ERR_CLS_g, maj, min, str, NULL) < 0) + push_ret = + H5E__push_stack(estack, true, tmp_file, tmp_func, line, H5E_ERR_CLS_g, maj, min, str, NULL); + if (push_ret < 0) HGOTO_ERROR(H5E_ERROR, H5E_CANTSET, FAIL, "can't push error on stack"); } done: + if (ret_value < 0 || !push_ret) { + if (inc_maj_id && H5I_dec_ref(maj) < 0) + HDONE_ERROR(H5E_ERROR, H5E_CANTDEC, FAIL, "can't decrement major error ID"); + if (inc_min_id && H5I_dec_ref(min) < 0) + HDONE_ERROR(H5E_ERROR, H5E_CANTDEC, FAIL, "can't decrement minor error ID"); + + free(tmp_func); + free(tmp_file); + } + FUNC_LEAVE_API(ret_value) } /* end H5Epush1() */ diff --git a/src/H5Eint.c b/src/H5Eint.c index cd7646fc2d2..4948fd7d987 100644 --- a/src/H5Eint.c +++ b/src/H5Eint.c @@ -1594,20 +1594,22 @@ done: * MIN_ID, the name of a function where the error was detected, * the name of the file where the error was detected, the * line within that file, and an error description string. The - * function name, file name, and error description strings must - * be statically allocated (the FUNC_ENTER() macro takes care of - * the function name and file name automatically, but the - * programmer is responsible for the description string). + * error description string must be statically allocated (the + * FUNC_ENTER() macro takes care of the function name and file + * name automatically, but the programmer is responsible for + * the description string). * - * Return: SUCCEED/FAIL + * Return: true if an error stack entry was pushed + * false if an error stack entry was not pushed + * FAIL on failure * *------------------------------------------------------------------------- */ -herr_t +htri_t H5E__push_stack(H5E_stack_t *estack, bool app_entry, const char *file, const char *func, unsigned line, hid_t cls_id, hid_t maj_id, hid_t min_id, const char *fmt, va_list *ap) { - herr_t ret_value = SUCCEED; /* Return value */ + htri_t ret_value = true; /* * WARNING: We cannot call HERROR() from within this function or else we @@ -1633,6 +1635,8 @@ H5E__push_stack(H5E_stack_t *estack, bool app_entry, const char *file, const cha HGOTO_DONE(FAIL); estack->nused++; } /* end if */ + else + HGOTO_DONE(false); done: FUNC_LEAVE_NOAPI(ret_value) diff --git a/src/H5Epkg.h b/src/H5Epkg.h index 849b292e02a..8580e7371e0 100644 --- a/src/H5Epkg.h +++ b/src/H5Epkg.h @@ -142,7 +142,7 @@ H5_DLL void H5E__set_default_auto(H5E_stack_t *stk); H5_DLL H5E_stack_t *H5E__get_current_stack(void); H5_DLL herr_t H5E__set_current_stack(H5E_stack_t *estack); H5_DLL ssize_t H5E__get_num(const H5E_stack_t *err_stack); -H5_DLL herr_t H5E__push_stack(H5E_stack_t *estack, bool app_entry, const char *file, const char *func, +H5_DLL htri_t H5E__push_stack(H5E_stack_t *estack, bool app_entry, const char *file, const char *func, unsigned line, hid_t cls_id, hid_t maj_id, hid_t min_id, const char *fmt, va_list *ap); H5_DLL herr_t H5E__print(const H5E_stack_t *estack, FILE *stream, bool bk_compat); diff --git a/test/error_test.c b/test/error_test.c index 050cdfe01e7..85fd649d406 100644 --- a/test/error_test.c +++ b/test/error_test.c @@ -16,6 +16,9 @@ #include "h5test.h" #include "H5srcdir.h" +#define H5E_FRIEND +#include "H5Epkg.h" /* For access to H5E-specific macros */ + #ifdef H5_USE_16_API int main(void) @@ -836,6 +839,50 @@ error: return -1; } /* end test_pause() */ +/*------------------------------------------------------------------------- + * Function: test_overflow_stack + * + * Purpose: Test pushing more than H5E_MAX_ENTRIES entries to an error + * stack. Once the stack is full, the next push is simply a + * no-op. + * + * Return: Success: 0 + * Failure: -1 + * + *------------------------------------------------------------------------- + */ +static herr_t +test_overflow_stack(void) +{ + ssize_t err_num; + char err_buf[32]; + + if (H5Eclear2(H5E_DEFAULT) < 0) + TEST_ERROR; + + err_num = H5Eget_num(H5E_DEFAULT); + if (err_num != 0) + TEST_ERROR; + + for (int i = 0; i < H5E_MAX_ENTRIES; i++) { + snprintf(err_buf, sizeof(err_buf), "error number %d", i + 1); + + if (H5Epush(H5E_DEFAULT, __FILE__, __func__, __LINE__, ERR_CLS, ERR_MAJ_TEST, ERR_MIN_SUBROUTINE, + "%s", err_buf) < 0) + TEST_ERROR; + } + + snprintf(err_buf, sizeof(err_buf), "error number %d", H5E_MAX_ENTRIES + 1); + if (H5Epush(H5E_DEFAULT, __FILE__, __func__, __LINE__, ERR_CLS, ERR_MAJ_TEST, ERR_MIN_SUBROUTINE, "%s", + err_buf) < 0) + TEST_ERROR; + + return 0; + +error: + return -1; +} + /*------------------------------------------------------------------------- * Function: close_error * @@ -1023,6 +1070,10 @@ main(void) if (test_pause() < 0) TEST_ERROR; + /* Test pushing more than H5E_MAX_ENTRIES entries to an error stack */ + if (test_overflow_stack() < 0) + TEST_ERROR; + /* Close error information */ if (close_error() < 0) TEST_ERROR; From 51b92ccee54ed3f14af3509c9ac5e4f3c82e58ce Mon Sep 17 00:00:00 2001 From: Larry Knox Date: Thu, 20 Aug 2026 14:33:48 -0500 Subject: [PATCH 5/9] Create tag with publish (#6606) * Move release tag creation in draft to publish of draft. * target_commitish needs the full sha for the release tag. --- .github/workflows/release-files.yml | 28 ++-------------------------- .github/workflows/tarball.yml | 4 ++-- 2 files changed, 4 insertions(+), 28 deletions(-) diff --git a/.github/workflows/release-files.yml b/.github/workflows/release-files.yml index 2f38593d2c0..b32ca8aef01 100644 --- a/.github/workflows/release-files.yml +++ b/.github/workflows/release-files.yml @@ -36,34 +36,8 @@ env: # Previous workflows must pass to get here so tag the commit that created the files jobs: - create-tag: - runs-on: ubuntu-latest - permissions: - contents: write # In order to allow tag creation - steps: - # Checks-out your repository under $GITHUB_WORKSPACE, so your job can access it - - name: Get Sources - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - with: - fetch-depth: 0 - ref: '${{ github.head_ref || github.ref_name }}' - - - uses: rickstaa/action-create-tag@a1c7777fcb2fee4f19b0f283ba888afa11678b72 # v1.7.2 - id: "tag_create" - with: - commit_sha: ${{ inputs.file_sha }} - tag: "${{ inputs.use_tag }}" - force_push_tag: false - tag_exists_error: false - message: "Latest snapshot" - - # Print result using the action output. - - run: | - echo "Tag already present: ${{ steps.tag_create.outputs.tag_exists }}" - PreRelease-getfiles: runs-on: ubuntu-latest - needs: create-tag environment: ${{ inputs.use_environ }} permissions: contents: write @@ -223,6 +197,7 @@ jobs: uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v3.0.2 with: tag_name: "${{ inputs.use_tag }}" + target_commitish: ${{ inputs.file_sha }} prerelease: true body_path: description.txt files: | @@ -264,6 +239,7 @@ jobs: uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v3.0.2 with: tag_name: "${{ steps.create_release_tag_name.outputs.TAG_BASE }}" + target_commitish: ${{ inputs.file_sha }} name: "HDF5 Release ${{ inputs.use_tag }}" draft: true prerelease: false diff --git a/.github/workflows/tarball.yml b/.github/workflows/tarball.yml index 73a0688290c..86711815fb5 100644 --- a/.github/workflows/tarball.yml +++ b/.github/workflows/tarball.yml @@ -33,8 +33,8 @@ on: description: "The branch used for the source tarballs" value: ${{ jobs.check_commits.outputs.branch_ref }} file_sha: - description: "The sha used for the source tarballs" - value: ${{ jobs.check_commits.outputs.branch_sha }} + description: "The full commit sha used as target_commitish for the release tag" + value: ${{ jobs.check_commits.outputs.branch_full_sha }} permissions: contents: read From b816e1b699856e453b8092948d4557d3739f70d0 Mon Sep 17 00:00:00 2001 From: vchoi-hdfgroup <55293060+vchoi-hdfgroup@users.noreply.github.com> Date: Tue, 25 Aug 2026 09:02:32 -0700 Subject: [PATCH 6/9] Fix for issue #6336: (#6634) Clarify object header chunk that is used in this section. --- docs/doxygen/dox/H5.format.4.0.dox | 41 ++++++++++++++++++++++-------- 1 file changed, 30 insertions(+), 11 deletions(-) diff --git a/docs/doxygen/dox/H5.format.4.0.dox b/docs/doxygen/dox/H5.format.4.0.dox index 8a404c8f149..f423daaea9a 100644 --- a/docs/doxygen/dox/H5.format.4.0.dox +++ b/docs/doxygen/dox/H5.format.4.0.dox @@ -5036,6 +5036,16 @@ Header messages are aligned on 8-byte boundaries for version 1 object headers. \subsubsection subsubsec_fmt4_dataobject_hdr_prefix_two IV.A.1.b Version 2 Data Object Header Prefix +\anchor FMT4OhdrChunkDef A version 2 object header’s messages are stored in one or more chunks. +Chunk \#0 is the chunk embedded directly in this prefix, below: it runs from the OHDR +signature through this chunk’s own Gap and Checksum fields. Additional chunks +(chunk \#1 through chunk \#n) are stored in continuation blocks, each located and sized +by an Object Header Continuation message in some earlier chunk (see +@ref subsubsec_fmt4_dataobject_hdr_msg_continuation). Each chunk, including chunk \#0, ends with its +own Gap and Checksum fields, computed only over that chunk’s bytes. This +use of “chunk” is unrelated to the raw data chunks of a chunked-storage dataset (see +@ref subsubsec_fmt4_dataobject_hdr_msg_layout and @ref sec_fmt4_appendixc). + Note that the “total number of messages” field has been dropped from the data object header prefix in this version. The number of messages in the data object header is just determined by the messages encountered in all the object header blocks. @@ -5220,9 +5230,10 @@ or padding bytes inserted - they are stored packed together. Size of Chunk \#0 - This unsigned value specifies the number of bytes of header message data following this field - that contain object header information. This value does not include the size of object header - continuation blocks for this object elsewhere in the file. The length of this field varies + This unsigned value specifies the number of bytes of header message data following this field, + within @ref FMT4OhdrChunkDef "chunk \#0" (the messages, and the Gap, if present, but not the + Checksum field). This value does not include the size of object header continuation blocks + (chunk \#1 and beyond) for this object elsewhere in the file. The length of this field varies depending on bits 0 and 1 of the flags field. @@ -5250,16 +5261,20 @@ or padding bytes inserted - they are stored packed together. Gap - A gap in an object header chunk is inferred by the end of the messages for the chunk before the - beginning of the chunk’s checksum. Gaps are always smaller than the size of an object header - message prefix (message type + message size + message flags).
+ A gap in @ref FMT4OhdrChunkDef "chunk \#0" is inferred by the end of the messages for the chunk + before the beginning of this chunk’s Checksum field. Gaps are always smaller than the size + of an object header message prefix (message type + message size + message flags).
Gaps are formed when a message (typically an attribute message) in an earlier chunk is deleted and a message from a later chunk that does not quite fit into the free space is moved into the earlier chunk. Checksum - This is the checksum for the object header chunk. + This is the checksum for chunk \#0 only (see @ref FMT4OhdrChunkDef "object header chunks"), + computed over every byte of this chunk from the OHDR signature through the end of + this chunk’s Gap field (if present), excluding this Checksum field itself. Each continuation + chunk, if any, carries its own separate checksum; see + @ref subsubsec_fmt4_dataobject_hdr_msg_continuation. @@ -10021,16 +10036,20 @@ described here (see also the description of @ref subsubsec_fmt4_dataobject_hdr_p Gap - A gap in an object header chunk is inferred by the end of the messages for the chunk before the - beginning of the chunk’s checksum. Gaps are always smaller than the size of an object header - message prefix (message type + message size + message flags).
+ A gap in this @ref FMT4OhdrChunkDef "continuation chunk" is inferred by the end of the messages for + the chunk before the beginning of this chunk’s Checksum field. Gaps are always smaller than the + size of an object header message prefix (message type + message size + message flags).
Gaps are formed when a message (typically an attribute message) in an earlier chunk is deleted and a message from a later chunk that does not quite fit into the free space is moved into the earlier chunk. Checksum - This is the checksum for the object header chunk. + This is the checksum for this continuation chunk only (see + @ref FMT4OhdrChunkDef "object header chunks"), computed over every byte of this chunk from the + OCHK signature through the end of this chunk’s Gap field (if present), excluding + this Checksum field itself. Chunk \#0, in the object header prefix, and every other continuation + chunk each carry their own separate checksum. From d8ec63966c44394029226b320416f7d5beb7b3a1 Mon Sep 17 00:00:00 2001 From: Scot Breitenfeld Date: Tue, 25 Aug 2026 12:07:04 -0500 Subject: [PATCH 7/9] Fix segfault in ph5_f90_filtered_writes_no_sel example: swap inverted mem_space_id (#6633) The Fortran example set sel_type = H5S_BLOCK_F on the branch where a rank calls H5Sselect_none (no data to contribute) and H5S_ALL_F on the branch where a rank sets a real hyperslab, which is backwards. The equivalent C example (ph5_filtered_writes_no_sel.c) does the opposite: H5Dwrite(..., no_selection ? H5S_ALL : H5S_BLOCK, ...); i.e. H5S_ALL when there is no selection (mirrors the file dataspace's own empty selection) and H5S_BLOCK when a hyperslab was set (memory buffer matches the block shape of the file selection). With the inversion, H5Dwrite_f feeds a bogus selection into the collective filtered I/O path, corrupting the element/byte-length accounting in H5D__mpio_collective_filtered_chunk_update, which segfaults inside H5D_select_io_mem -> H5VM_memcpyvv on 12 MPI ranks. Reproduces on current develop; the bug has been present since the example was added in #3916. --- HDF5Examples/FORTRAN/H5PAR/ph5_f90_filtered_writes_no_sel.F90 | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/HDF5Examples/FORTRAN/H5PAR/ph5_f90_filtered_writes_no_sel.F90 b/HDF5Examples/FORTRAN/H5PAR/ph5_f90_filtered_writes_no_sel.F90 index 4a34d1f2ed0..6928ce1f57a 100644 --- a/HDF5Examples/FORTRAN/H5PAR/ph5_f90_filtered_writes_no_sel.F90 +++ b/HDF5Examples/FORTRAN/H5PAR/ph5_f90_filtered_writes_no_sel.F90 @@ -186,7 +186,7 @@ CONTAINS ! passed to H5Dwrite. CALL H5Sselect_none_f(file_dataspace, status) - sel_type = H5S_BLOCK_F + sel_type = H5S_ALL_F ELSE ! ! Even MPI ranks contribute data to @@ -206,7 +206,7 @@ CONTAINS CALL H5Sselect_hyperslab_f(file_dataspace, H5S_SELECT_SET_F, start, count, status, stride=stride) - sel_type = H5S_ALL_F + sel_type = H5S_BLOCK_F ! ! -------------------------------------- ! Fill data buffer with MPI rank's rank From 6f427d993d990ecd6fbf6d40edf09edd6750bbfa Mon Sep 17 00:00:00 2001 From: "H. Joe Lee" Date: Fri, 28 Aug 2026 11:03:06 -0500 Subject: [PATCH 8/9] ci: update h5py installation to use hdf5@develop-2 (#6650) Sync with Spack change --- .github/workflows/h5py.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/h5py.yml b/.github/workflows/h5py.yml index c45961c665e..e6fb251e5ce 100644 --- a/.github/workflows/h5py.yml +++ b/.github/workflows/h5py.yml @@ -26,8 +26,8 @@ jobs: ./spack/bin/spack spec py-h5py@master+mpi sed -i 's/hdf5@1.10.7:1.14/hdf5@1.10.7:/g' \ /home/runner/.spack/package_repos/*/repos/spack_repo/builtin/packages/py_h5py/package.py - ./spack/bin/spack spec py-h5py@master+mpi ^hdf5@develop-2.0 - ./spack/bin/spack install py-h5py@master+mpi ^hdf5@develop-2.0 + ./spack/bin/spack spec py-h5py@master+mpi ^hdf5@develop-2 + ./spack/bin/spack install py-h5py@master+mpi ^hdf5@develop-2 ./spack/bin/spack install py-pytest ./spack/bin/spack install py-pytest-mpi spack load py-h5py From b7b85e7abf9aa9b1dd9693523defa35217684eb2 Mon Sep 17 00:00:00 2001 From: Matt L <124107509+mattjala@users.noreply.github.com> Date: Fri, 28 Aug 2026 13:49:02 -0500 Subject: [PATCH 9/9] Fix CVE-2026-19025 (Reject chunked datasets with mismatched chunk/dspace rank at open time) (#6508) * Reject chunked datasets with mismatched chunk/dspace rank H5D__chunk_construct() validates that the chunk layout dimensionality matches the dataspace rank, but that runs only at dataset creation time. When an existing dataset is opened, H5D__chunk_init() didn't repeat the check, so a file whose stored chunk rank disagreed with its dataspace rank was accepted. During chunk I/O the memory-selection rank (from the dataspace) and the file-selection rank (chunk ndims - 1) then differ, which produces a zero stride that causes a divide-by-zero in H5S__hyper_iter_get_seq_list(). H5D__chunk_init() now performs the same dimensionality check on open (the stored chunk rank includes the extra element-size dimension, so it must be exactly one greater than the dataspace rank) and rejects a mismatch with an error. Added test_chunk_dims_mismatch() as a regression test in test/dsets.c Fixes #6491 * Fix typo Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> * Clarify element-vs-byte wording * Validate chunk/dataspace rank at layout decode time Move the stored-chunk-rank vs dataspace-rank consistency check out of H5D__chunk_init() and into H5O__layout_decode(), so a malformed chunked layout is rejected as the message is decoded (mirroring the fill/datatype size check in the fill message decode). * Update release_docs/CHANGELOG.md Co-authored-by: Larry Knox * Update CHANGELOG * Pin format version bounds in bad chunk layout generator --------- Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> Co-authored-by: Larry Knox --- release_docs/CHANGELOG.md | 8 ++ src/H5Olayout.c | 32 ++++++ test/CMakeTests.cmake | 2 + test/dsets.c | 59 ++++++++++ test/gen_bad_chunk.c | 179 ++++++++++++++++++++++++++++++ test/testfiles/bad_chunk_ndims.h5 | Bin 0 -> 5048 bytes 6 files changed, 280 insertions(+) create mode 100644 test/gen_bad_chunk.c create mode 100644 test/testfiles/bad_chunk_ndims.h5 diff --git a/release_docs/CHANGELOG.md b/release_docs/CHANGELOG.md index fbf20d93143..615c766dfa8 100644 --- a/release_docs/CHANGELOG.md +++ b/release_docs/CHANGELOG.md @@ -94,6 +94,14 @@ We would like to thank the many HDF5 community members who contributed to this r Fixes GitHub issue #6531 +### Fixed a crash when reading a chunked dataset whose chunk rank does not match the dataspace rank + + The chunk layout's stored dimensionality was validated against the dataspace rank at creation time, but not at open time, so a file whose stored chunk rank disagreed with its dataspace rank was not caught. The resulting inconsistent selection ranks during chunk I/O caused a divide-by-zero in the hyperslab iterator. The chunk dimensionality is now also validated on open, and such a dataset is rejected with an error instead of crashing. + + Fixes GitHub issue #6491 + + Fixes CVE-2026-19025 + ## Java Library ## Configuration diff --git a/src/H5Olayout.c b/src/H5Olayout.c index d230feb9925..5dce35e916e 100644 --- a/src/H5Olayout.c +++ b/src/H5Olayout.c @@ -23,6 +23,7 @@ #include "H5FLprivate.h" /* Free Lists */ #include "H5MMprivate.h" /* Memory management */ #include "H5Opkg.h" /* Object headers */ +#include "H5Sprivate.h" /* Dataspaces */ /* Local macros */ @@ -561,6 +562,37 @@ H5O__layout_decode(H5F_t *f, H5O_t H5_ATTR_UNUSED *open_oh, unsigned H5_ATTR_UNU } } + /* For a chunked layout, the stored dimensionality includes an extra + * element-size dimension, so it must be exactly one greater than the + * dataspace rank. Validate that here + * to reject malformed files before the inconsistent + * ranks can cause problems during chunk I/O. + */ + if (mesg->type == H5D_CHUNKED && open_oh != NULL) { + htri_t space_exists; /* Whether the dataspace message exists */ + + if ((space_exists = H5O_msg_exists_oh(open_oh, H5O_SDSPACE_ID)) < 0) + HGOTO_ERROR(H5E_OHDR, H5E_CANTGET, NULL, "can't check for dataspace message"); + if (space_exists) { + H5S_extent_t *extent; /* Dataspace extent from the sibling message */ + int rank; /* Dataspace rank */ + + if (NULL == (extent = (H5S_extent_t *)H5O_msg_read_oh(f, open_oh, H5O_SDSPACE_ID, NULL))) + HGOTO_ERROR(H5E_OHDR, H5E_CANTGET, NULL, "can't read dataspace message"); + + rank = H5S_extent_get_dims(extent, NULL, NULL); + + /* Done with the sibling dataspace message */ + H5O_msg_free(H5O_SDSPACE_ID, extent); + + if (rank < 0) + HGOTO_ERROR(H5E_OHDR, H5E_CANTGET, NULL, "can't get dataspace rank"); + if (mesg->u.chunk.ndims != (unsigned)rank + 1) + HGOTO_ERROR(H5E_OHDR, H5E_BADVALUE, NULL, + "dimensionality of chunks doesn't match the dataspace"); + } + } + /* Set return value */ ret_value = mesg; diff --git a/test/CMakeTests.cmake b/test/CMakeTests.cmake index a41dbe746b6..81fce1c3dae 100644 --- a/test/CMakeTests.cmake +++ b/test/CMakeTests.cmake @@ -115,6 +115,7 @@ endforeach () # -------------------------------------------------------------------- set (HDF5_REFERENCE_TEST_FILES aggr.h5 + bad_chunk_ndims.h5 bad_compound.h5 bad_offset.h5 be_data.h5 @@ -1165,6 +1166,7 @@ endmacro () # generator executables set (H5_GENERATORS + gen_bad_chunk gen_bad_offset gen_bad_ohdr gen_bogus diff --git a/test/dsets.c b/test/dsets.c index fd2feb6c8ce..9ecb67c9f38 100644 --- a/test/dsets.c +++ b/test/dsets.c @@ -8088,6 +8088,64 @@ error: return FAIL; } /* end test_filters_endianess() */ +/*------------------------------------------------------------------------- + * Function: test_chunk_dims_mismatch + * + * Purpose: Test that a malformed file whose stored chunk layout + * dimensionality does not match the dataset's dataspace rank + * is properly rejected at open-time. + * + * Return: Success: 0 + * Failure: -1 + * + *------------------------------------------------------------------------- + */ +static herr_t +test_chunk_dims_mismatch(void) +{ + hid_t fid = H5I_INVALID_HID; + hid_t did = H5I_INVALID_HID; + const char *data_file = H5_get_srcdir_filename("bad_chunk_ndims.h5"); + + TESTING("rejection of chunk dimensionality that mismatches the dataspace"); + + if ((fid = H5Fopen(data_file, H5F_ACC_RDONLY, H5P_DEFAULT)) < 0) { + printf(" Could not open file %s. Try setting $srcdir to point at the " + "source directory of the test suite\n", + data_file); + goto error; + } + + /* Opening the dataset must fail cleanly */ + H5E_BEGIN_TRY + { + did = H5Dopen2(fid, "dset", H5P_DEFAULT); + } + H5E_END_TRY + + if (did >= 0) { + H5_FAILED(); + puts(" Opening a dataset with mismatched chunk/dataspace rank should have failed."); + goto error; + } + + if (H5Fclose(fid) < 0) + FAIL_STACK_ERROR; + + PASSED(); + + return SUCCEED; + +error: + H5E_BEGIN_TRY + { + H5Dclose(did); + H5Fclose(fid); + } + H5E_END_TRY + return FAIL; +} /* end test_chunk_dims_mismatch() */ + /*------------------------------------------------------------------------- * Function: test_zero_dims * @@ -19545,6 +19603,7 @@ main(void) if (driver_is_default_compatible) { nerrors += (test_filters_endianess() < 0 ? 1 : 0); + nerrors += (test_chunk_dims_mismatch() < 0 ? 1 : 0); } nerrors += (test_zero_dims(file) < 0 ? 1 : 0); diff --git a/test/gen_bad_chunk.c b/test/gen_bad_chunk.c new file mode 100644 index 00000000000..dc730467c43 --- /dev/null +++ b/test/gen_bad_chunk.c @@ -0,0 +1,179 @@ +/* * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * + * Copyright by The HDF Group. * + * All rights reserved. * + * * + * This file is part of HDF5. The full HDF5 copyright notice, including * + * terms governing use, modification, and redistribution, is contained in * + * the LICENSE file, which can be found at the root of the source code * + * distribution tree, or in https://www.hdfgroup.org/licenses. * + * If you do not have access to either file, you may request a copy from * + * help@hdfgroup.org. * + * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * * */ + +/* + * Generate an HDF5 file with a chunked dataset whose stored chunk layout + * dimensionality does not match the dataset's dataspace rank. + * + * A valid 3-D chunked dataset of native int is written, then the version-3 + * chunk layout message's "dimensionality" byte is patched from 4 down to 3. + * The stored chunk rank includes an extra element-size dimension, so a valid + * 3-D dataset stores 4. Patching it to 3 makes the layout describe a 2-D + * chunk over a 3-D dataspace. + * + */ + +#include "h5test.h" + +#define BAD_CHUNK_FILE "bad_chunk_ndims.h5" +#define BAD_CHUNK_DSET "dset" + +/* Read an entire file into a newly allocated buffer. */ +static unsigned char * +slurp(const char *name, size_t *len_out) +{ + FILE *f = fopen(name, "rb"); + long len; + unsigned char *buf; + + if (!f) + return NULL; + if (fseek(f, 0, SEEK_END) != 0 || (len = ftell(f)) < 0 || fseek(f, 0, SEEK_SET) != 0) { + fclose(f); + return NULL; + } + if (NULL == (buf = malloc((size_t)len))) { + fclose(f); + return NULL; + } + if (fread(buf, 1, (size_t)len, f) != (size_t)len) { + free(buf); + fclose(f); + return NULL; + } + fclose(f); + *len_out = (size_t)len; + return buf; +} + +/* Write a buffer back out to a file. */ +static int +spew(const char *name, const unsigned char *buf, size_t len) +{ + FILE *f = fopen(name, "wb"); + if (!f) + return -1; + if (fwrite(buf, 1, len, f) != len) { + fclose(f); + return -1; + } + return fclose(f) == 0 ? 0 : -1; +} + +/* Find the unique occurrence of pattern in buf; return offset or (size_t)-1. */ +static size_t +find_once(const unsigned char *buf, size_t len, const unsigned char *pat, size_t patlen) +{ + size_t i, found = (size_t)-1; + + if (patlen == 0 || len < patlen) + return (size_t)-1; + for (i = 0; i <= len - patlen; i++) { + if (memcmp(buf + i, pat, patlen) == 0) { + if (found != (size_t)-1) + return (size_t)-1; /* not unique */ + found = i; + } + } + return found; +} + +int +main(void) +{ + hid_t fapl = H5I_INVALID_HID, file = H5I_INVALID_HID, sid = H5I_INVALID_HID; + hid_t dcpl = H5I_INVALID_HID, dset = H5I_INVALID_HID; + hsize_t dims[3] = {3, 4, 5}; + hsize_t chunk[3] = {2, 2, 4}; /* chunk edge lengths, in elements */ + int data[3 * 4 * 5]; + unsigned char *buf = NULL; + size_t len, off; + int i; + + /* The version-3 chunk layout stores, after the 3-byte header (version, + * class, dimensionality) and the 8-byte b-tree address, one little-endian + * uint32 per stored dimension: the three chunk edge lengths (in elements) + * followed by the element size in bytes. For this dataset those on-disk + * values are literally {2, 2, 4, 4} -- the three chunk edge lengths and a + * trailing 4-byte element size; this byte pattern uniquely locates the + * layout message. */ + static const unsigned char layout_dims[] = {2, 0, 0, 0, 2, 0, 0, 0, 4, 0, 0, 0, 4, 0, 0, 0}; + + for (i = 0; i < 3 * 4 * 5; i++) + data[i] = i; + + if ((fapl = H5Pcreate(H5P_FILE_ACCESS)) < 0) + TEST_ERROR; + /* The high bound caps the layout message at version 3 to prevent + * drift if new chunk layout formats are introduced in the future. */ + if (H5Pset_libver_bounds(fapl, H5F_LIBVER_EARLIEST, H5F_LIBVER_V18) < 0) + TEST_ERROR; + if ((file = H5Fcreate(BAD_CHUNK_FILE, H5F_ACC_TRUNC, H5P_DEFAULT, fapl)) < 0) + TEST_ERROR; + if ((sid = H5Screate_simple(3, dims, NULL)) < 0) + TEST_ERROR; + + if ((dcpl = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_chunk(dcpl, 3, chunk) < 0) + TEST_ERROR; + + if ((dset = H5Dcreate2(file, BAD_CHUNK_DSET, H5T_NATIVE_INT, sid, H5P_DEFAULT, dcpl, H5P_DEFAULT)) < 0) + TEST_ERROR; + if (H5Dwrite(dset, H5T_NATIVE_INT, H5S_ALL, H5S_ALL, H5P_DEFAULT, data) < 0) + TEST_ERROR; + + if (H5Dclose(dset) < 0 || H5Pclose(dcpl) < 0 || H5Sclose(sid) < 0 || H5Fclose(file) < 0 || + H5Pclose(fapl) < 0) + TEST_ERROR; + + /* Patch the chunk layout's dimensionality byte from 4 to 3. */ + if (NULL == (buf = slurp(BAD_CHUNK_FILE, &len))) + TEST_ERROR; + if ((off = find_once(buf, len, layout_dims, sizeof(layout_dims))) == (size_t)-1) + TEST_ERROR; + /* Layout header is 3 bytes (version, class, ndims) + 8-byte address before + * the chunk sizes, so the ndims byte is 11 bytes before the sizes. + * + * Patching ndims from 4 to 3 makes the layout describe a 2-D chunk over the + * 3-D dataspace. The decoder then reads only the first three stored sizes, + * {2, 2, 4}, and treats the last of those (4) as the element-size + * dimension. Because the dataset's third chunk edge was chosen to be 4, + * that reinterpreted element size still matches the stored 4-byte element + * size, so the layout decodes consistently and the dataset opens instead + * of being rejected by the element-size check -- which is what let the + * original bug reach the I/O path and crash. */ + if (off < 11 || buf[off - 11] != 3 /* version */ || buf[off - 10] != 2 /* chunked */ || + buf[off - 9] != 4 /* ndims */) + TEST_ERROR; + buf[off - 9] = 3; + if (spew(BAD_CHUNK_FILE, buf, len) < 0) + TEST_ERROR; + + free(buf); + printf("Generated %s\n", BAD_CHUNK_FILE); + return EXIT_SUCCESS; + +error: + free(buf); + H5E_BEGIN_TRY + { + H5Dclose(dset); + H5Pclose(dcpl); + H5Sclose(sid); + H5Fclose(file); + H5Pclose(fapl); + } + H5E_END_TRY + fprintf(stderr, "failed to generate %s\n", BAD_CHUNK_FILE); + return EXIT_FAILURE; +} diff --git a/test/testfiles/bad_chunk_ndims.h5 b/test/testfiles/bad_chunk_ndims.h5 new file mode 100644 index 0000000000000000000000000000000000000000..83c5b0ded60a539806f0f0f6dad7e68f59f2f15b GIT binary patch literal 5048 zcmeI0$xZ@65QcjgmJvk(cieE_6?fDNCSW)iHR{2m2fcXJ1COBZ-~;#wJc5tJP5+v% zLegaH#-6$Pcl7t7sLA*yShIIR6w8K_o{&S}EeJZo8{Z@kVV8MhC2 zk5rfSSd|Muw2q4FSRBw%ML&|`YKO~ld3mZBw}bs@KVBLOS@+5NJz~o9{Si0f2lkZq zyAij>F=?g*b(7X7GDrwLod2!K%-I*(VOub0s5eEiu`|7;&DYM z){I|cz5Gnki50zA(OGLVU^8GdU^8GdU^8GdU^8Gd@LL(6Gav0w=?F$=#5hO*I-B)@ zelP$A!7#{y5iknOeQXDOtwr9~h;K|PZ0UtjC+Gqx&;#g9PR9(|s*i&SkOz}s8q9!M zFbBFpYcaOpYO3`j>hnf>0qsSw1eU=nSOe>z0K&?OA=oh5Va-hlw!k*n0egVn4D5pg O5LR{#!J*L(YkmN8yiZpE literal 0 HcmV?d00001