From bc9914d2f0e50d6dcf172293134b84845c867ec9 Mon Sep 17 00:00:00 2001 From: Neil Fortner Date: Wed, 16 Jul 2025 09:23:59 -0500 Subject: [PATCH] Optimizations for repeated VDS source file and dataset names (#5640) Add hash tables for detecting repeated source file and dataset names. Share these repeated strings between mappings in memory. Add new encoding format for VDS for shared names. Add tests and documentation for these changes. Co-authored-by: github-actions <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: Matthew Larson --- doxygen/aliases | 2 +- doxygen/dox/FileFormatDisc.dox | 2 +- doxygen/dox/FileFormatSpec.dox | 2 + doxygen/dox/H5.format.4.0.dox | 12706 +++++++++++++++++++++++++++++++ doxygen/dox/Specifications.dox | 1 + release_docs/RELEASE.txt | 8 + src/H5Dvirtual.c | 175 +- src/H5Olayout.c | 223 +- src/H5Oprivate.h | 25 +- src/H5Pdcpl.c | 108 +- src/uthash.h | 31 + test/dsets.c | 981 +++ 12 files changed, 14188 insertions(+), 76 deletions(-) create mode 100644 doxygen/dox/H5.format.4.0.dox diff --git a/doxygen/aliases b/doxygen/aliases index 1e0e5523359..20fdf5765d0 100644 --- a/doxygen/aliases +++ b/doxygen/aliases @@ -259,7 +259,7 @@ ALIASES += sa_metadata_ops="\sa \li H5Pget_all_coll_metadata_ops() \li H5Pget_co # Specifications ################################################################################ -ALIASES += ref_spec_fileformat="\ref FMT3" +ALIASES += ref_spec_fileformat="\ref FMT4" ALIASES += ref_spec_fileformat_btrees_v1="\ref subsubsec_fmt3_infra_btrees_v1" ################################################################################ diff --git a/doxygen/dox/FileFormatDisc.dox b/doxygen/dox/FileFormatDisc.dox index 50251eb1590..66c39107b62 100644 --- a/doxygen/dox/FileFormatDisc.dox +++ b/doxygen/dox/FileFormatDisc.dox @@ -10,7 +10,7 @@ Navigate back: \ref index "Main" / \ref TN
Current H5 library designers and knowledgeable external developers.
Background Reading:
-
\ref FMT3
This describes the current HDF5 file format.
+
\ref FMT4
This describes the current HDF5 file format.
\section sec_fmtdisc_intro Introduction diff --git a/doxygen/dox/FileFormatSpec.dox b/doxygen/dox/FileFormatSpec.dox index 308237c8d67..21711357040 100644 --- a/doxygen/dox/FileFormatSpec.dox +++ b/doxygen/dox/FileFormatSpec.dox @@ -1,3 +1,5 @@ +\ref FMT4 + \ref FMT3 \ref FMT2 diff --git a/doxygen/dox/H5.format.4.0.dox b/doxygen/dox/H5.format.4.0.dox new file mode 100644 index 00000000000..9c2a14b219a --- /dev/null +++ b/doxygen/dox/H5.format.4.0.dox @@ -0,0 +1,12706 @@ + +/** \page FMT4 HDF5 File Format Specification Version 4.0 + +Navigate back: \ref index "Main" / \ref SPEC +
+ + + +
+
    +
  1. @ref sec_fmt4_intro +
      +
    1. @ref subsec_fmt4_intro_doc
    2. +
    3. @ref subsec_fmt4_intro_20
    4. +
    5. @ref subsec_fmt4_intro_112
    6. +
    7. @ref subsec_fmt4_intro_110
    8. +
  2. +
  3. @ref sec_fmt4_meta +
      +
    1. @ref subsec_fmt4_boot_super
    2. +
    3. @ref subsec_fmt4_boot_driver
    4. +
    5. @ref subsec_fmt4_boot_supext
    6. +
  4. +
  5. @ref sec_fmt4_infra +
      +
    1. @ref subsec_fmt4_infra_btrees +
        +
      1. @ref subsubsec_fmt4_infra_btrees_v1
      2. +
      3. @ref subsubsec_fmt4_infra_btrees_v2
      4. +
    2. +
    3. @ref subsec_fmt4_infra_symboltable
    4. +
    5. @ref subsec_fmt4_infra_symboltableentry
    6. +
    7. @ref subsec_fmt4_infra_localheap
    8. +
    9. @ref subsec_fmt4_infra_globalheap
    10. +
    11. @ref subsec_fmt4_infra_globalheapvds
    12. +
    13. @ref subsec_fmt4_infra_fractalheap
    14. +
    15. @ref subsec_fmt4_infra_freespaceindex
    16. +
    17. @ref subsec_fmt4_infra_sohm
    18. +
  6. +
  7. @ref sec_fmt4_dataobject +
      +
    1. @ref subsec_fmt4_dataobject_hdr +
        +
      1. @ref subsec_fmt4_dataobject_hdr_prefix
      2. +
          +
        1. @ref subsubsec_fmt4_dataobject_hdr_prefix_one
        2. +
        3. @ref subsubsec_fmt4_dataobject_hdr_prefix_two
        4. +
        +
      3. @ref subsec_fmt4_dataobject_hdr_msg
      4. +
          +
        1. @ref subsubsec_fmt4_dataobject_hdr_msg_nil
        2. +
        3. @ref subsubsec_fmt4_dataobject_hdr_msg_simple
        4. +
        5. @ref subsubsec_fmt4_dataobject_hdr_msg_linkinfo
        6. +
        7. @ref subsubsec_fmt4_dataobject_hdr_msg_dtmessage
        8. +
        9. @ref subsubsec_fmt4_dataobject_hdr_msg_ofvmessage
        10. +
        11. @ref subsubsec_fmt4_dataobject_hdr_msg_fvmessage
        12. +
        13. @ref subsubsec_fmt4_dataobject_hdr_msg_link
        14. +
        15. @ref subsubsec_fmt4_dataobject_hdr_msg_external
        16. +
        17. @ref subsubsec_fmt4_dataobject_hdr_msg_layout
        18. +
        19. @ref subsubsec_fmt4_dataobject_hdr_msg_bogus
        20. +
        21. @ref subsubsec_fmt4_dataobject_hdr_msg_groupinfo
        22. +
        23. @ref subsubsec_fmt4_dataobject_hdr_msg_filter
        24. +
        25. @ref subsubsec_fmt4_dataobject_hdr_msg_attribute
        26. +
        27. @ref subsubsec_fmt4_dataobject_hdr_msg_comment
        28. +
        29. @ref subsubsec_fmt4_dataobject_hdr_msg_omodified
        30. +
        31. @ref subsubsec_fmt4_dataobject_hdr_msg_shared
        32. +
        33. @ref subsubsec_fmt4_dataobject_hdr_msg_continuation
        34. +
        35. @ref subsubsec_fmt4_dataobject_hdr_msg_stmgroup
        36. +
        37. @ref subsubsec_fmt4_dataobject_hdr_msg_mod
        38. +
        39. @ref subsubsec_fmt4_dataobject_hdr_msg_btreek
        40. +
        41. @ref subsubsec_fmt4_dataobject_hdr_msg_drvinfo
        42. +
        43. @ref subsubsec_fmt4_dataobject_hdr_msg_attrinfo
        44. +
        45. @ref subsubsec_fmt4_dataobject_hdr_msg_refcount
        46. +
        47. @ref subsubsec_fmt4_dataobject_hdr_msg_fsinfo
        48. +
        +
    2. +
    3. @ref subsec_fmt4_dataobject_storage
    4. +
    +
  8. +
  9. @ref sec_fmt4_appendixa +
  10. @ref sec_fmt4_appendixb +
  11. @ref sec_fmt4_appendixc +
      +
    1. @ref subsec_fmt4_appendixc_chunk +
    2. @ref subsec_fmt4_appendixc_implicit +
    3. @ref subsec_fmt4_appendixc_fixedarr +
    4. @ref subsec_fmt4_appendixc_extarr +
    5. @ref subsec_fmt4_appendixc_appv2btree +
    +
  12. +
  13. @ref sec_fmt4_appendixd +
      +
    1. @ref subsec_fmt4_appendixd_encode +
    2. @ref subsec_fmt4_appendixd_encoderv +
    3. @ref subsec_fmt4_appendixd_encodedp +
    +
  14. +
+
+ +\section sec_fmt4_intro I. Introduction + + + + + + + + + + + + + + +
Figure 1: Relationships among the HDF5 root group, other groups, and objects
\image html FF-IH_FileGroup.gif
Figure 2: HDF5 objects -- datasets, datatypes, or dataspaces
\image html FF-IH_FileObject.gif
+ +The format of an HDF5 file on disk encompasses several key ideas of the HDF4 and AIO file formats as well +as addressing some shortcomings therein. The new format is more self-describing than the HDF4 format and +is more uniformly applied to data objects in the file. + +An HDF5 file appears to the user as a directed graph. The nodes of this graph are the higher-level HDF5 +objects that are exposed by the HDF5 APIs: +\li Groups +\li Datasets +\li Committed (formerly Named) datatypes + +At the lowest level, as information is actually written to the disk, an HDF5 file is made up of the +following objects: +\li A superblock +\li B-tree nodes +\li Heap blocks +\li Object headers +\li Object data +\li Free space + +The HDF5 library uses these lower-level objects to represent the higher-level objects that are then +presented to the user or to applications through the APIs. For instance, a group is an object header that +contains a message that points to a local heap (for storing the links to objects in the group) and to a +B-tree (which indexes the links). A dataset is an object header that contains messages that describe +datatype, dataspace, layout, filters, external files, fill value, and other elements with the layout message +pointing to either a raw data chunk or to a B-tree that points to raw data chunks. + +\subsection subsec_fmt4_intro_doc I.A. This Document +This document describes the lower-level data objects; the higher-level objects and their properties are +described in the \ref UG. + +Three levels of information comprise the file format. Level 0 contains basic information for identifying +and defining information about the file. Level 1 information contains the information about the pieces of a +file shared by many objects in the file (such as a B-trees and heaps). Level 2 is the rest of the file and +contains all of the data objects with each object partitioned into header information, also known as +metadata, and data. + +The various components of the lower-level data objects are described in pairs of tables. The first table +shows the format layout, and the second table describes the fields. The titles of format layout tables +begin with “Layout”. The titles of the tables where the fields are described begin with +“Fields”. For example, the table that describes the format of the +@ref subsubsec_fmt4_infra_btrees_v2 has a title of “Layout: Version 2 B-tree Header”, and the +fields in the version 2 B-tree header are described in the table titled “Fields: Version 2 B-tree Header”. + +The sizes of various fields in the following layout tables are determined by looking at the number of +columns the field spans in the table. There are exceptions: +\li The size may be overridden by specifying a size in parentheses +\li The size of addresses is determined by the @ref FMT4SizeOfOffsetsV0 "Size of Offsets" field +in the superblock and is indicated in this document with a superscripted ‘O’ +\li The size of length fields is determined by the @ref FMT4SizeOfLengthsV0 "Size of Lengths" field +in the superblock and is indicated in this document with a superscripted ‘L’. + +Values for all fields in this document should be treated as unsigned integers, unless otherwise noted in +the description of a field. Additionally, all metadata fields are stored in little-endian byte order. + +All checksums used in the format are computed with the +Jenkins’ lookup3 algorithm. + +Whenever a bit flag or field is mentioned for an entry, bits are numbered from the lowest bit position +in the entry. + +Various format tables in this document have cells with “This space inserted only to align table nicely”. +These entries in the table are just to make the table presentation nicer and do not represent any values +or padding in the file. + +\subsection subsec_fmt4_intro_20 I.B. Changes for HDF5 2.0 +The following sections have been changed or added for the 2.0 release: +\li Under @ref subsubsec_fmt4_dataobject_hdr_msg_dtmessage, in the Description for + “Fields:Datatype Message”, version 5 was added, as well as the new Complex class (11). + +\subsection subsec_fmt4_intro_112 I.C. Changes for HDF5 1.12 +The following sections have been changed or added for the 1.12 release: +\li Under @ref subsubsec_fmt4_dataobject_hdr_msg_dtmessage, in the Description for + “Fields:Datatype Message”, version 4 was added and Reference class (7) of the + datatype was updated to describe version 4. +\li @ref sec_fmt4_appendixd was added. + +\subsection subsec_fmt4_intro_110 I.D. Changes for HDF5 1.10 +The following sections have been changed or added for the 1.10 release: +\li In the @ref subsec_fmt4_boot_super section, version 3 of the superblock was added. +\li In the @ref subsec_fmt4_boot_supext section, a link to the Data Storage message was added. +\li In the @ref subsubsec_fmt4_infra_btrees_v2 section, additional B-tree types were added. + Tables that describe the @ref FMT4V2BtType10"type 10" and @ref FMT4V2BtType11"11" record + layouts were added at the end of the section. +\li The @ref subsec_fmt4_infra_globalheapvds was added. +\li @ref subsubsec_fmt4_dataobject_hdr_msg_layout section was changed. The name was changed, + and @ref FMT4DataLayoutV4"version 4" of the data layout message was added for the virtual type. +\li The @ref subsubsec_fmt4_dataobject_hdr_msg_fsinfo header message type was added. +\li @ref sec_fmt4_appendixc was added. Five indexing types were added. + +\section sec_fmt4_meta II. Disk Format: Level 0 - File Metadata + +\subsection subsec_fmt4_boot_super II.A. Disk Format: Level 0A - Format Signature and Superblock +The superblock may begin at certain predefined offsets within the HDF5 file, allowing a block of +unspecified content for users to place additional information at the beginning (and end) of the HDF5 file +without limiting the HDF5 library’s ability to manage the objects within the file itself. This feature +was designed to accommodate wrapping an HDF5 file in another file format or adding descriptive information +to an HDF5 file without requiring the modification of the actual file’s information. The superblock +is located by searching for the HDF5 file signature at byte offset 0, byte offset 512 and at successive +locations in the file, each a multiple of two of the previous location, in other words, at these byte +offsets: 0, 512, 1024, 2048, and so on. + +The superblock is composed of the format signature, followed by a superblock version number and information +that is specific to each version of the superblock. + +Currently, there are four versions of the superblock format: +\li Version 0 is the default format. +\li Version 1 is the same as version 0 but with the “Indexed Storage Internal Node K” + field for storing non-default B-tree ‘K’ value. +\li Version 2 has some fields eliminated and compressed from superblock format versions 0 and 1. It has + added checksum support and superblock extension to store additional superblock metadata. +\li Version 3 is the same as version 2 except that the field “File Consistency Flags” + is used for file locking. This format version will enable support for the latest version. + +Version 0 and 1 of the superblock are described below: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Superblock (Versions 0 and 1)
bytebytebytebyte

Format Signature (8 bytes)

Version \# of SuperblockVersion \# of File’s Free Space StorageVersion \# of Root Group Symbol Table EntryReserved (zero)
Version \# of Shared Header Message FormatSize of OffsetsSize of LengthsReserved (zero)
Group Leaf Node KGroup Internal Node K
File Consistency Flags
Indexed Storage Internal Node K1Reserved (zero)1
Base AddressO
Address of File Free Space InfoO
End of File AddressO
Driver Information Block AddressO
Root Group Symbol Table Entry
+\li Items marked with an ‘1’ in the above table are new in version 1 of the superblock. +\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Superblock (Versions 0 and 1)
Field NameDescription
Format SignatureThis field contains a constant value and can be used to quickly identify a file as being an HDF5 + file. The constant value is designed to allow easy identification of an HDF5 file and to allow + certain types of data corruption to be detected. The file signature of an HDF5 file always + contains the following values: +

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Decimal:13772687013102610
Hexadecimal:894844460d0a1a0a
ASCII C Notation:\211HDF\\r\\n\032\\n
+
+ This signature both identifies the file as an HDF5 file and provides for immediate detection of common + file-transfer problems. The first two bytes distinguish HDF5 files on systems that expect the first two + bytes to identify the file type uniquely. The first byte is chosen as a non-ASCII value to reduce the + probability that a text file may be misrecognized as an HDF5 file; also, it catches bad file transfers + that clear bit 7. Bytes two through four name the format. The CR-LF sequence catches bad file transfers + that alter newline sequences. The control-Z character stops file display under MS-DOS. The final line + feed checks for the inverse of the CR-LF translation problem. (This is a direct descendent of the + PNG + file signature.)
+ This field is present in version 0+ of the superblock.
Version Number of the SuperblockThis value is used to determine the format of the information in the superblock. When the format of + the information in the superblock is changed, the version number is incremented to the next integer + and can be used to determine how the information in the superblock is formatted.
+ Values of 0, 1 and 2 are defined for this field (the format of version 2 is described below, not + here).
+ This field is present in version 0+ of the superblock.
Version Number of the File’s Free Space InformationThis value is used to determine the format of the file’s free space information.
+ The only value currently valid in this field is ‘0’, which indicates that the file’s + free space index is as described in @ref subsec_fmt4_infra_freespaceindex below.
+ This field is present in version 0 and 1 of the superblock.
Version Number of the Root Group Symbol Table EntryThis value is used to determine the format of the information in the Root Group Symbol Table Entry. + When the format of the information in that field is changed, the version number is incremented to the + next integer and can be used to determine how the information in the field is formatted.
+ The only value currently valid in this field is ‘0’, which indicates that the root group + symbol table entry is formatted as described in @ref subsec_fmt4_infra_symboltableentry below.
+ This field is present in version 0 and 1 of the superblock.
Version Number of the Shared Header Message FormatThis value is used to determine the format of the information in a shared object header message. + Since the format of the shared header messages differs from the other private header messages, a + version number is used to identify changes in the format.
+ The only value currently valid in this field is ‘0’, which indicates that shared + header messages are formatted as described in @ref subsubsec_fmt4_dataobject_hdr_msg_shared below.
+ This field is present in version 0 and 1 of the superblock.
\anchor FMT4SizeOfOffsetsV0 Size of OffsetsThis value contains the number of bytes used to store addresses in the file. The values for the + addresses of objects in the file are offsets relative to a base address, usually the address of the + superblock signature. This allows a wrapper to be added after the file is created without invalidating + the internal offset locations.
+ This field is present in version 0+ of the superblock.
\anchor FMT4SizeOfLengthsV0 Size of LengthsThis value contains the number of bytes used to store the size of an object.
+ This field is present in version 0+ of the superblock.
Group Leaf Node KEach leaf node of a group B-tree will have at least this many entries but not more than twice this + many. If a group has a single leaf node then it may have fewer entries.
+ This value must be greater than zero.
+ See the @ref subsec_fmt4_infra_btrees below.
+ This field is present in version 0 and 1 of the superblock.
Group Internal Node KEach internal node of a group B-tree will have at least this many entries but not more than twice this + many. If the group has only one internal node then it might have fewer entries.
+ This value must be greater than zero.
+ See the @ref subsec_fmt4_infra_btrees below.
+ This field is present in version 0 and 1 of the superblock.
File Consistency FlagsThis field is unused and should be ignored.
+ This field is present in version 0+ of the superblock.
Indexed Storage Internal Node KEach internal node of a indexed storage B-tree will have at least this many entries but not more than + twice this many. If the ndex storage B-tree has only one internal node then it might have fewer + entries.
+ This value must be greater than zero.
+ See the @ref subsec_fmt4_infra_btrees below.
+ This field is present in version 1 of the superblock.
Base AddressThis is the absolute file address of the first byte of the HDF5 data within the file. The library + currently constrains this value to be the absolute file address of the superblock itself when creating + new files; future versions of the library may provide greater flexibility. When opening an existing + file and this address does not match the offset of the superblock, the library assumes that the entire + contents of the HDF5 file have been adjusted in the file and adjusts the base address and end of file + address to reflect their new positions in the file. Unless otherwise noted, all other file addresses + are relative to this base address.
+ This field is present in version 0+ of the superblock.
Address of Global Free Space IndexThe file’s free space management is not persistent for version 0 and 1 of the superblock. + Currently this field always contains the @ref FMT4UndefinedAddress "undefined address".
+ This field is present in version 0 and 1 of the superblock.
End of File AddressThis is the absolute file address of the first byte past the end of all HDF5 data. It is used to + determine whether a file has been accidentally truncated and as an address where file data allocation + can occur if space from the free list is not used.
+ This field is present in version 0+ of the superblock.
Driver Information Block AddressThis is the relative file address of the file driver information block which contains driver-specific + information needed to reopen the file. If there is no driver information block then this entry should + be the @ref FMT4UndefinedAddress "undefined address".
+ This field is present in version 0 and 1 of the superblock.
Root Group Symbol Table EntryThis is the @ref subsec_fmt4_infra_symboltableentry of the root group, which serves as the entry-point + into the group graph for the file.
+ This field is present in version 0 and 1 of the superblock.
+ +Versions 2 and 3 of the superblock is described below: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Superblock (Versions 2 and 3)
bytebytebytebyte

Format Signature (8 bytes)

Version \# of SuperblockSize of OffsetsSize of LengthsFile Consistency Flags

Base AddressO


Superblock Extension AddressO


End of File AddressO


Root Group Object Header AddressO

Superblock Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Superblock (Versions 2 and 3)
Field NameDescription
Format SignatureThis field is the same as described for versions 0 and 1 of the superblock.
Version Number of the SuperblockThis field has a value of 2 and has the same meaning as for versions 0 and 1.
Size of OffsetsThis field is the same as described for @ref FMT4SizeOfOffsetsV0 "versions 0 and 1" of the superblock.
Size of LengthsThis field is the same as described for @ref FMT4SizeOfLengthsV0 "versions 0 and 1" of the superblock.
File Consistency FlagsFor superblock version 2: This field is unused and should be ignored.
+ For superblock version 3: This value contains flags to ensure file consistency for file locking. + Currently, the following bit flags are defined: +
    +
  • Bit 0 if set indicates that the file has been opened for write access.
  • +
  • Bit 1 is reserved for future use.
  • +
  • Bit 2 if set indicates that the file has been opened for single-writer/multiple-reader + (SWMR) write access.
  • +
  • Bits 3-7 are reserved for future use.
  • +

+ Bit 0 should be set as the first action when a file has been opened for write access. Bit 2 should + be set when a file has been opened for SWMR write access. These two bits should be cleared only as + the final action when closing a file.
+ This field is present in version 0+ of the superblock.
+ The size of this field has been reduced from 4 bytes in superblock format versions 0 and 1 to + 1 byte.
Base AddressThis field is the same as described for versions 0 and 1 of the superblock.
Superblock Extension AddressThe field is the address of the object header for the @ref subsec_fmt4_boot_supext. If there is no + extension then this entry should be the @ref FMT4UndefinedAddress "undefined address".
End of File AddressThis field is the same as described for versions 0 and 1 of the superblock.
Root Group Object Header AddressThis is the address of the @ref sec_fmt4_dataobject, which serves as the entry point into the group + graph for the file.
Superblock ChecksumThe checksum for the superblock.
+ +\subsection subsec_fmt4_boot_driver II.B. Disk Format: Level 0B - File Driver Info +The driver information block is an optional region of the file which contains information +needed by the file driver in order to reopen a file. The format is described below: + + + + + + + + + + + + + + + + + + + + + +
Layout: Driver Information Block
bytebytebytebyte
VersionReserved
Driver Information Size

Driver Identification (8 bytes)



Driver Information (variable size)


+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Driver Information Block
Field NameDescription
VersionThe version number of the Driver Information Block. This document describes version 0.
Driver Information SizeThe size in bytes of the Driver Information field.
Driver IdentificationThis is an eight-byte ASCII string without null termination which identifies the driver and/or version number + of the Driver Information block. The predefined driver encoded in this field by the HDF5 library is identified + by the letters NCSA followed by the first four characters of the driver name. If the Driver Information + Block is not the original version then the last letter(s) of the identification will be replaced by a version + number in ASCII, starting with 0.
+ Identification for user-defined drivers is also eight-byte long. It can be arbitrary but should be unique to + avoid the four character prefix “NCSA”.
Driver InformationDriver information is stored in a format defined by the file driver (see description below).
+ +The two drivers encoded in the Driver Identification field are as follows: +\li Multi driver:
The identifier for this driver is “NCSAmulti”. This driver provides + a mechanism for segregating raw data and different types of metadata into multiple files. These files + are viewed by the library as a single virtual HDF5 file with a single file address. A maximum of 6 + files will be created for the following data: superblock, B-tree, raw data, global heap, local heap, + and object header. More than one type of data can be written to the same file. +\li Family driver:
The identifier for this driver is “NCSAfami” and is encoded in this + field for library version 1.8 and after. This driver is designed for systems that do not support files + larger than 2 gigabytes by splitting the HDF5 file address space across several smaller files. It does + nothing to segregate metadata and raw data; they are mixed in the address space just as they would be + in a single contiguous file. + +The format of the Driver Information field for the above two drivers are described below: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Multi Driver Information
bytebytebytebyte
Member MappingMember MappingMember MappingMember Mapping
Member MappingMember MappingReservedReserved

Address of Member File 1


End of Address for Member File 1


Address of Member File 2


End of Address for Member File 2


... ...


Address of Member File N


End of Address for Member File N


Name of Member File 1 (variable size)


Name of Member File 2 (variable size)


... ...


Name of Member File N (variable size)

+ + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Multi Driver Information
Field NameDescription
Member MappingThese fields are integer values from 1 to 6 indicating how the data can be mapped to or + merged with another type of data. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Member MappingDescription
1The superblock data.
2The B-tree data.
3The raw data.
4The global heap data.
5The local heap data.
6The object header data.

+ For example, if the third field has the value 3 and all the rest have the + value 1, it means there are two files, one for raw data, and one for superblock, + B-tree, global heap, local heap, and object header.
ReservedThese fields are reserved and should always be zero.
Address of Member File NThis field specifies the virtual address at which the member file starts.
+ N is the number of member files.
End of Address for Member File NThis field is the end of allocated address for the member file.
Name of Member File NThis field is the null-terminated name of member file. And its length should be multiples + of 8 bytes. Additional bytes will be padded with NULLs. The default naming convention is + %%s-X.h5, where X is one of the letters s (for superblock), + b (for B-tree), r (for raw data), g (for global heap), + l (for local heap), and o (for object header). The name for the whole + HDF5 file will substitute the %s in the string.
+
+ + + + + + + + + + + +
Layout: Family Driver Information
bytebytebytebyte

Size of Member File

+
+ + + + + + + + + + +
Fields: Family Driver Information
Field NameDescription
Size of Member FileThis field is the size of the member file in the family of files.
+ +\subsection subsec_fmt4_boot_supext II.C. Disk Format: Level 0C - Superblock Extension +The superblock extension is used to store superblock metadata which is either optional, or added +after the version of the superblock was defined. Superblock extensions may only exist when version 2+ or +later of the superblock is used. A superblock extension is an object header which may hold the following messages: +\li \ref subsec_fmt4_infra_sohm containing information to locate the master table of shared object + header message indices. +\li \ref subsubsec_fmt4_dataobject_hdr_msg_btreek containing non-default B-tree ‘K’ values. +\li \ref subsubsec_fmt4_dataobject_hdr_msg_drvinfo containing information needed by the file driver in + order to reopen a file. See also the \ref subsec_fmt4_boot_driver section above. +\li \ref subsubsec_fmt4_dataobject_hdr_msg_fsinfo containing information about file space handling in the file. + +\section sec_fmt4_infra III. Disk Format: Level 1 - File Infrastructure + +\subsection subsec_fmt4_infra_btrees III.A. Disk Format: Level 1A - B-trees and B-tree Nodes +B-trees allow flexible storage for objects which tend to grow in ways that cause the object to be stored +discontiguously. B-trees are described in various algorithms books including "Introduction to Algorithms" by +Thomas H. Cormen, Charles E. Leiserson, and Ronald L. Rivest. The B-trees are used in several places in +the HDF5 file format, when an index is needed for another data structure. + +The version 1 B-tree structure described below is the original index structure. The version 1 B-trees are +being phased out in favor of the version 2 B-trees described below. Note that both types of structures may +be found in the same file depending on the application settings when creating the file. + +\subsubsection subsubsec_fmt4_infra_btrees_v1 III.A.1. Disk Format: Level 1A1 - Version 1 B-trees +Version 1 B-trees in HDF5 files an implementation of the B-tree. The sibling nodes at a +particular level in the tree are stored in a doubly-linked list. See the “Efficient Locking +for Concurrent Operations on B-trees” paper by Phillip Lehman and S. Bing Yao as published in the + ACM Transactions on Database Systems, Vol. 6, No. 4, December 1981. + +The B-trees implemented by the file format contain one more key than the number of children. In other +words, each child pointer out of a B-tree node has a left key and a right key. The pointers out of internal +nodes point to sub-trees while the pointers out of leaf nodes point to symbol nodes and raw data chunks. +Aside from that difference, internal nodes and leaf nodes are identical. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: B-tree Nodes
bytebytebytebyte
Signature
Node TypeNode LevelEntries Used

Address of Left SiblingO


Address of Right SiblingO

Key 1 (variable size)

Address of Child 1O

Key 2 (variable size)

Address of Child 2O

...
Key 2K (variable size)

Address of Child 2KO

Key 2K+1 (variable size)
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: B-tree Nodes
Field NameDescription
SignatureThe ASCII character string “TREE” is used to indicate the beginning of a + B-tree node. This gives file consistency checking utilities a better chance of reconstructing + a damaged file.
Node TypeEach B-tree points to a particular type of data. This field indicates the type of data as well as + implying the maximum degree K of the tree and the size of each Key field.
+ + + + + + + + + + + + + +
Node TypeDescription
0This tree points to group nodes.
1This tree points to a raw data chunk.
+
Node LevelThe node level indicates the level at which this node appears in the tree (leaf nodes are at level + zero). Not only does the level indicate whether child pointers point to sub-trees or to data, but it + can also be used to help file consistency checking utilities reconstruct damaged trees.
Entries UsedThis determines the number of children to which this node points. All nodes of a particular type of + tree have the same maximum degree, but most nodes will point to less than that number of children. The + valid child pointers and keys appear at the beginning of the node and the unused pointers and keys + appear at the end of the node. The unused pointers and keys have undefined values.
Address of Left SiblingThis is the relative file address of the left sibling of the current node. If the current node is the + left-most node at this level then this field is the @ref FMT4UndefinedAddress "undefined address".
Address of Right SiblingThis is the relative file address of the right sibling of the current node. If the current node is the + right-most node at this level then this field is the @ref FMT4UndefinedAddress "undefined address".
Keys and Child PointersEach tree has 2K+1 keys with 2K child pointers interleaved between the keys. The number + of keys and child pointers actually containing valid values is determined by the node’s + Entries Used field. If that field is N then the B-tree contains N child + pointers and N+1 keys.
KeyThe format and size of the key values is determined by the type of data to which this tree points. The + keys are ordered and are boundaries for the contents of the child pointer; that is, the key values + represented by child N fall between Key N and Key N+1. Whether the interval + is open or closed on each end is determined by the type of data to which the tree points.
+ The format of the key depends on the node type. For nodes of node type 0 (group nodes), the key is + formatted as follows: + + + + + +
A single field of @ref FMT4SizeOfLengthsV0 "Size of Lengths" bytes.Indicates the byte offset into the local heap for the first object name in the subtree which + that key describes.
+
+ For nodes of node type 1 (chunked raw data nodes), the key is formatted as follows: + + + + + + + + + + + + + +
Bytes 1-4Size of chunk in bytes.
Bytes 4-8Filter mask, a 32-bit bit field indicating which filters have been skipped for this chunk. Each + filter has an index number in the pipeline (starting at 0, with the first filter to apply) and + if that filter is skipped, the bit corresponding to its index is set.
(D + 1) 64-bit fieldsThe offset of the chunk within the dataset where D is the number + of dimensions of the dataset, and the last value is the offset within the dataset’s + datatype and should always be zero. For example, if a chunk in a 3-dimensional dataset begins at the + position [5,5,5], there will be three such 64-bit indices, each with the value of + 5, followed by a 0 value.
+
Child PointerThe tree node contains file addresses of subtrees or data depending on the node level. Nodes at Level + 0 point to data addresses, either raw data chunks or group nodes. Nodes at non-zero levels point to other + nodes of the same B-tree.
+ For raw data chunk nodes, the child pointer is the address of a single raw data chunk. For group nodes, + the child pointer points to a @ref subsec_fmt4_infra_symboltableentry, which contains + information for multiple symbol table entries.
+ +Conceptually, each B-tree node looks like this: + + + + + + + + + + + + + + + + + + + + + + +
key[0] child[0] key[1] child[1] key[2]; ... ... key[N-1] child[N-1] key[N]
+where child[i] is a pointer to a sub-tree (at a level above Level 0) or to data (at Level 0). +Each key[i] describes an item stored by the B-tree (a chunk or an object of a group node). +The range of values represented by child[i] is indicated by key[i] and key[i+1]. + +The following question must next be answered: “Is the value described by key[i] contained in +child[i-1] or in child[i]?” The answer depends on the type of tree. In trees for groups (node +type 0) the object described by key[i] is the greatest object contained in child[i-1] while +in chunk trees (node type 1) the chunk described by key[i] is the least chunk in child[i]. + +That means that key[0] for group trees is sometimes unused; it points to offset zero in the heap, which is +always the empty string and compares as "less-than" any valid object name. + +And key[N] for chunk trees is sometimes unused; it contains a chunk offset which compares as +"greater-than" any other chunk offset and has a chunk byte size of zero to indicate that it is not actually +allocated. + +\subsubsection subsubsec_fmt4_infra_btrees_v2 III.A.2. Disk Format: Level 1A2 - Version 2 B-trees +Version 2 (v2) B-trees are “traditional” B-trees, with one major difference. Instead of just using +a simple pointer (or address in the file) to a child of an internal node, the pointer to the child node +contains two additional pieces of information: the number of records in the child node itself, and the +total number of records in the child node and all its descendants. Storing this additional information +allows fast array-like indexing to locate the nth record in the B-tree. + +The entry into a version 2 B-tree is a header which contains global information about the structure of +the B-tree. The root node address field in the header points to the B-tree root node, which is +either an internal or leaf node, depending on the value in the header’s depth field. An +internal node consists of records plus pointers to further leaf or internal nodes in the tree. A leaf +node consists of solely of records. The format of the records depends on the B-tree type (stored in +the header). + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree Header
bytebytebytebyte
Signature
VersionTypeThis space inserted only to align table nicely
Node Size
Record SizeDepth
Split PercentMerge PercentThis space inserted only to align table nicely

Root Node AddressO

Number of Records in Root NodeThis space inserted only to align table nicely

Total Number of Records in B-treeL

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree Header
Field NameDescription
SignatureThe ASCII character string “BTHD” is used to indicate the header of a + version 2 (v2) B-tree node.
VersionThe version number for this B-tree header. This document describes version 0.
TypeThis field indicates the type of B-tree: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0This B-tree is used for testing only. This value should not be used for storing + records in actual HDF5 files.
1This B-tree is used for indexing indirectly accessed, non-filtered ‘huge’ + fractal heap objects.
2This B-tree is used for indexing indirectly accessed, filtered ‘huge’ + fractal heap objects.
3This B-tree is used for indexing directly accessed, non-filtered ‘huge’ + fractal heap objects.
4This B-tree is used for indexing directly accessed, filtered ‘huge’ + fractal heap objects.
5This B-tree is used for indexing the ‘name’ field for links in indexed + groups.
6This B-tree is used for indexing the ‘creation order’ field for links + in indexed groups.
7This B-tree is used for indexing shared object header messages.
8This B-tree is used for indexing the ‘name’ field for indexed + attributes.
9This B-tree is used for indexing the ‘creation order’ field for + indexed attributes.
10This B-tree is used for indexing chunks of datasets with no filters and with more + than one dimension of unlimited extent.
11This B-tree is used for indexing chunks of datasets with filters and more than one + dimension of unlimited extent.
+ The format of records for each type is described below.
Node SizeThis is the size in bytes of all B-tree nodes.
Record SizeThis field is the size in bytes of the B-tree record.
DepthThis is the depth of the B-tree.
Split PercentThe percent full that a node needs to increase above before it is split.
Merge PercentThe percent full that a node needs to be decrease below before it is split.
Root Node AddressThis is the address of the root B-tree node. A B-tree with no records will have the + @ref FMT4UndefinedAddress "undefined address" in this field.
Number of Records in Root NodeThis is the number of records in the root node.
Total Number of Records in B-treeThis is the total number of records in the entire B-tree.
ChecksumThis is the checksum for the B-tree header.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree Internal Node
bytebytebytebyte
Signature
VersionTypeRecords 0, 1, 2...N-1 (variable size)

Child Node Pointer 0O


Number of Records N0 for Child Node 0 (variable size)

Total Number of Records for Child Node 0 (optional, variable size)

Child Node Pointer 1O


Number of Records N1 for Child Node 1 (variable size)

Total Number of Records for Child Node 1 (optional, variable size)
...

Child Node Pointer NO


Number of Records Nn for Child Node N (variable size)

Total Number of Records for Child Node N (optional, variable size)
Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree Internal Node
Field NameDescription
SignatureThe ASCII character string “ BTIN ” is used to indicate the internal node + of a B-tree.
VersionThe version number for this B-tree internal node. This document describes version 0.
TypeThis field is the type of the B-tree node. It should always be the same as the B-tree type in + the header.
RecordsThe size of this field is determined by the number of records for this node and the record size + (from the header). The format of records depends on the type of B-tree.
Child Node PointerThis field is the address of the child node pointed to by the internal node.
Number of Records in Child NodeThis is the number of records in the child node pointed to by the corresponding Node Pointer.
+ The number of bytes used to store this field is determined by the maximum possible number of records able + to be stored in the child node.
+ The maximum number of records in a child node is computed in the following way: +
    +
  • Subtract the fixed size overhead for the child node (for example, its signature, version, + checksum, and so on and one pointer triplet of information for the child node + (because there is one more pointer triplet than records in each internal node)) from the size + of nodes for the B-tree.
  • +
  • Divide that result by the size of a record plus the pointer triplet of information stored to + reach each child node from this node.
  • +

+ Note that leaf nodes do not encode any child pointer triplets, so the maximum number of records in a + leaf node is just the node size minus the leaf node overhead, divided by the record size.
+ Also note that the first level of internal nodes above the leaf nodes do not encode the Total + Number of Records in Child Node value in the child pointer triplets (since it is the same as + the Number of Records in Child Node), so the maximum number of records in these nodes is + computed with the equation above, but using (Child Pointer, Number of Records in Child + Node) pairs instead of triplets.
+ The number of bytes used to encode this field is the least number of bytes required to encode the + maximum number of records in a child node value for the child nodes below this level in the B-tree.
+ For example, if the maximum number of child records is 123, one byte will be used to encode these + values in this node; if the maximum number of child records is 20000, two bytes will be used to + encode these values in this node; and so on. The maximum number of bytes used to encode these values + is 8 (in other words, an unsigned 64-bit integer).
Total Number of Records in Child NodeThis is the total number of records for the node pointed to by the corresponding Node Pointer + and all its children. This field exists only in nodes whose depth in the B-tree node is greater than 1 + (in other words, the “twig” internal nodes, just above leaf nodes, do not store this field + in their child node pointers).
+ The number of bytes used to store this field is determined by the maximum possible number of records + able to be stored in the child node and its descendants.
+ The maximum possible number of records able to be stored in a child node and its descendants is + computed iteratively, in the following way: The maximum number of records in a leaf node is + computed, then that value is used to compute the maximum possible number of records in the first + level of internal nodes above the leaf nodes. Multiplying these two values together determines the + maximum possible number of records in child node pointers for the level of nodes two levels above + leaf nodes. This process is continued up to any level in the B-tree.
+ The number of bytes used to encode this value is computed in the same way as for the Number + of Records in Child Node field.
ChecksumThis is the checksum for this node.
+ + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree Leaf Node
bytebytebytebyte
Signature
VersionTypeRecord 0, 1, 2...N-1 (variable size)
Checksum
+ + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree Leaf Node
Field NameDescription
SignatureThe ASCII character string “ BTLF “ is used to indicate the leaf node + of a version 2 (v2) B-tree.
VersionThe version number for this B-tree leaf node. This document describes version 0.
TypeThis field is the type of the B-tree node. It should always be the same as the B-tree type in + the header.
RecordsThe size of this field is determined by the number of records for this node and the record size + (from the header). The format of records depends on the type of B-tree.
ChecksumThis is the checksum for this node.
+ +The record layout for each stored (in other words, non-testing) B-tree type is as follows: + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 1 Record Layout - Indirectly Accessed, Non-Filtered, + ‘Huge’ Fractal Heap Objects
bytebytebytebyte

Huge Object AddressO


Huge Object LengthL


Huge Object IDL

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 1 Record Layout - Indirectly Accessed, Non-Filtered, + ‘Huge’ Fractal Heap Objects
Field NameDescription
Huge Object AddressThe address of the huge object in the file.
Huge Object LengthThe length of the huge object in the file.
Huge Object IDThe heap ID for the huge object.
+ + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 2 Record Layout - Indirectly Accessed, Filtered, ‘Huge’ + Fractal Heap Objects
bytebytebytebyte

Filtered Huge Object AddressO


Filtered Huge Object LengthL

Filter Mask

Filtered Huge Object Memory SizeL


Huge Object IDL

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 2 Record Layout - Indirectly Accessed, Filtered, ‘Huge’ + Fractal Heap Objects
Field NameDescription
Filtered Huge Object AddressThe address of the filtered huge object in the file.
Filtered Huge Object LengthThe length of the filtered huge object in the file.
Filter MaskA 32-bit bit field indicating which filters have been skipped for this chunk. Each filter has + an index number in the pipeline (starting at 0, with the first filter to apply) and if that filter + is skipped, the bit corresponding to its index is set.
Filtered Huge Object Memory SizeThe size of the de-filtered huge object in memory.
Huge Object IDThe heap ID for the huge object.
+ + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 3 Record Layout - Directly Accessed, Non-Filtered, ‘Huge’ + Fractal Heap Objects
bytebytebytebyte

Huge Object AddressO


Huge Object LengthL

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 3 Record Layout - Directly Accessed, Non-Filtered, ‘Huge’ + Fractal Heap Objects
Field NameDescription
Huge Object AddressThe address of the huge object in the file.
Huge Object LengthThe length of the huge object in the file.
+ + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 4 Record Layout - Directly Accessed, Filtered, ‘Huge’ + Fractal Heap Objects
bytebytebytebyte

Filtered Huge Object AddressO


Filtered Huge Object LengthL

Filter Mask

Filtered Huge Object Memory SizeL

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 4 Record Layout - Directly Accessed, Filtered, ‘Huge’ + Fractal Heap Objects
Field NameDescription
Filtered Huge Object AddressThe address of the filtered huge object in the file.
Filtered Huge Object LengthThe length of the filtered huge object in the file.
Filter MaskA 32-bit bit field indicating which filters have been skipped for this chunk. Each filter has an + index number in the pipeline (starting at 0, with the first filter to apply) and if that filter + is skipped, the bit corresponding to its index is set.
Filtered Huge Object Memory SizeThe size of the de-filtered huge object in memory.
+ + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 5 Record Layout - Link Name for Indexed Group
bytebytebytebyte
Hash of Name
ID (bytes 1-4)
ID (bytes 5-7)
+ + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 5 Record Layout - Link Name for Indexed Group
Field NameDescription
HashThis field is hash value of the name for the link. The hash value is the Jenkins’ lookup3 + checksum algorithm applied to the link’s name.
IDThis is a 7-byte sequence of bytes and is the heap ID for the link record in the group’s + fractal heap.
+ + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 6 Record Layout - Creation Order for Indexed Group
bytebytebytebyte

Creation Order (8 bytes)

ID (bytes 1-4)
ID (bytes 5-7)
+ + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 6 Record Layout - Creation Order for Indexed Group
Field NameDescription
Creation OrderThis field is the creation order value for the link.
IDThis is a 7-byte sequence of bytes and is the heap ID for the link record in the group’s + fractal heap.
+ + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 7 Record Layout - Shared Object Header Messages + (Sub-Type 0 - Message in Heap)
bytebytebytebyte
Message LocationThis space inserted only to align table nicely
Hash
Reference Count

Heap ID (8 bytes)

+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 7 Record Layout - Shared Object Header Messages + (Sub-Type 0 - Message in Heap)
Field NameDescription
Message LocationThis field Indicates the location where the message is stored: + + + + + + + + + + + + + +
ValueDescription
0Shared message is stored in shared message index heap.
1Shared message is stored in object header.
+
HashThis field is hash value of the shared message. The hash value is the Jenkins’ lookup3 + checksum algorithm applied to the shared message.
Reference CountThe number of objects which reference this message.
Heap IDThis is an 8-byte sequence of bytes and is the heap ID for the shared message in the shared + message index’s fractal heap.
+ + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 7 Record Layout - Shared Object Header Messages + (Sub-Type 1 - Message in Object Header)
bytebytebytebyte
Message LocationThis space inserted only to align table nicely
Hash
Reserved (zero)Message TypeObject Header Index

Object Header AddressO

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 7 Record Layout - Shared Object Header Messages + (Sub-Type 1 - Message in Object Header)
Field NameDescription
Message LocationThis field Indicates the location where the message is stored: + + + + + + + + + + + + + +
ValueDescription
0Shared message is stored in shared message index heap.
1Shared message is stored in object header.
+
HashThis field is hash value of the shared message. The hash value is the Jenkins’ lookup3 + checksum algorithm applied to the shared message.
Message TypeThe object header message type of the shared message.
Object Header IndexThis field indicates that the shared message is the nth message of its type in the + specified object header.
Object Header AddressThe address of the object header containing the shared message.
+ + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 8 Record Layout - Attribute Name for Indexed Attributes
bytebytebytebyte

Heap ID (8 bytes)

Message FlagsThis space inserted only to align table nicely
Creation Order
Hash of Name
+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 8 Record Layout - Attribute Name for Indexed Attributes
Field NameDescription
Heap IDThis is an 8-byte sequence of bytes and is the heap ID for the attribute in the object’s + attribute fractal heap.
Message FlagsThe object header message flags for the attribute message.
Creation OrderThis field is the creation order value for the attribute.
HashThis field is hash value of the name for the attribute. The hash value is the Jenkins’ + lookup3 checksum algorithm applied to the attribute’s name.
+ + + + + + + + + + + + + + + + + + + +
Layout: Version 2 B-tree, Type 9 Record Layout- Creation Order for Indexed Attributes
bytebytebytebyte

Heap ID (8 bytes)

Message FlagsThis space inserted only to align table nicely
Creation Order
+ + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 9 Record Layout- Creation Order for Indexed Attributes
Field NameDescription
Heap IDThis is an 8-byte sequence of bytes and is the heap ID for the attribute in the object’s + attribute fractal heap.
Message FlagsThe object header message flags for the attribute message.
Creation OrderThis field is the creation order value for the attribute.
+ + + + + + + + + + + + + + + + + + + + + + + + +
\anchor FMT4V2BtType10 Layout: Version 2 B-tree, Type 10 Record Layout - Non-filtered Dataset Chunks
bytebytebytebyte

AddressO


Dimension 0 Scaled Offset (8 bytes)


Dimension 1 Scaled Offset (8 bytes)


...


Dimension \#n Scaled Offset (8 bytes)

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 11 Record Layout - Filtered Dataset Chunks
Field NameDescription
AddressThis field is the address of the dataset chunk in the file.
Dimension \#n Scaled OffsetThis field is the scaled offset of the chunk within the dataset. n is the number of + dimensions for the dataset. The first scaled offset stored in the list is for the slowest + changing dimension, and the last scaled offset stored is for the fastest changing dimension. + Scaled offset is calculated by dividing the chunk dimension sizes into the chunk offsets.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
\anchor FMT4V2BtType11 Layout: Version 2 B-tree, Type 11 Record Layout - Filtered Dataset Chunks
bytebytebytebyte

AddressO


Chunk Size (variable size; at most 8 bytes)

Filter Mask

Dimension 0 Scaled Offset (8 bytes)


Dimension 1 Scaled Offset (8 bytes)


...


Dimension \#n Scaled Offset (8 bytes)

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 B-tree, Type 5 Record Layout - Non-filtered Dataset Chunks
Field NameDescription
AddressThis field is the address of the dataset chunk in the file.
Chunk SizeThis field is the size of the dataset chunk in bytes.
Filter MaskThis field is the filter mask which indicates the filter + to skip for the dataset chunk. Each filter has an index + number in the pipeline and if that filter is skipped, + the bit corresponding to its index is set.
Dimension \#n Scaled OffsetThis field is the scaled offset of the chunk within the dataset. n is the number of + dimensions for the dataset. The first scaled offset stored in the list is for the slowest + changing dimension, and the last scaled offset stored is for the fastest changing dimension.
+ +\subsection subsec_fmt4_infra_symboltable III.B. Disk Format: Level 1B - Group Symbol Table Nodes +A group is an object internal to the file that allows arbitrary nesting of objects within the file (including +other groups). A group maps a set of link names in the group to a set of relative file addresses of objects +in the file. Certain metadata for an object to which the group points can be cached in +object’s header. + +An HDF5 object name space can be stored hierarchically by partitioning the name into components and storing +each component as a link in a group. The link for a non-ultimate component points to the group containing the +next component. The link for the last component points to the object being named. + +One implementation a group is a collection of symbol table nodes indexed by a B-tree. Each symbol table +node contains entries for one or more links. If an attempt is made to add a link to an already full +symbol table node containing 2K entries, then the node is split and one node contains K +symbols and the other contains K+1 symbols. + + + + + + + + + + + + + + + + + + + + +
Layout: Symbol Table Node (A Leaf of a B-tree)
bytebytebytebyte
Signature
Version NumberReserved (zero)Number of Symbols


Group Entries


+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Symbol Table Node (A Leaf of a B-tree)
Field NameDescription
SignatureThe ASCII character string SNOD is used to indicate the beginning of a symbol table node. This + gives file consistency checking utilities a better chance of reconstructing a damaged file.
Version NumberThe version number for the symbol table node. This document describes version 1. (There is no version + ‘0’ of the symbol table node)
Number of SymbolsAlthough all symbol table nodes have the same length, most contain fewer than the maximum possible number of + link entries. This field indicates how many entries contain valid data. The valid entries are packed + at the beginning of the symbol table node while the remaining entries contain undefined values.
Group EntriesEach link has an entry in the symbol table node. The format of the entry is described below. There are + 2K entries in each group node, where K is the “Group Leaf Node K” value + from the @ref subsec_fmt4_boot_super.
+ +\subsection subsec_fmt4_infra_symboltableentry III.C. Disk Format: Level 1C - Symbol Table Entry +Each symbol table entry in a symbol table node is designed to allow for very fast browsing of stored objects. +Toward that design goal, the symbol table entries include space for caching certain constant metadata from the +object header. + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Symbol Table Entry
bytebytebytebyte
Link Name OffsetO
Object Header AddressO
Cache Type
Reserved (zero)


Scratch-pad Space (16 bytes)


+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Symbol Table Entry
Field NameDescription
Link Name OffsetThis is the byte offset into the group’s local heap for the name of the link. The name is null + terminated.
Object Header AddressEvery object has an object header which serves as a permanent location for the object’s metadata. + In addition to appearing in the object header, some of the object’s metadata can be cached in the + scratch-pad space.
Cache TypeThe cache type is determined from the object header. It also determines the format for the scratch-pad + space.
+ + + + + + + + + + + + + + + + + +
Type:Description:
0No data is cached by the group entry. This is guaranteed to be the case when an object header has + a link count greater than one.
1Group object header metadata is cached in the scratch-pad space. This implies that the symbol table + entry refers to another group.
2The entry is a symbolic link. The first four bytes of the scratch-pad space are the offset into + the local heap for the link value. The object header address will be undefined.
+
ReservedThese four bytes are present so that the scratch-pad space is aligned on an eight-byte boundary. They + are always set to zero.
Scratch-pad SpaceThis space is used for different purposes, depending on the value of the Cache Type field. Any meta-data + about an object represented in the scratch-pad space is duplicated in the object header for + that object.
+ Furthermore, no data is cached in the group entry scratch-pad space if the object header for the object + has a link count greater than one.
+ +\subsubsection subsubsec_fmt4_infra_symboltableentry_scratch Format of the Scratch-pad Space +The symbol table entry scratch-pad space is formatted according to the value in the Cache Type field. + +If the Cache Type field contains the value zero ((0)) then no information is stored in the +scratch-pad space. + +If the Cache Type field contains the value one (1), then the scratch-pad space contains +cached metadata for another object header in the following format: + + + + + + + + + + + + + +
Layout: Object Header Scratch-pad Format
bytebytebytebyte
Address of B-treeO
Address of Name HeapO
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Object Header Scratch-pad Format
Field NameDescription
Address of B-treeThis is the file address for the root of the group’s B-tree.
Address of Name HeapThis is the file address for the group’s local heap, in which are stored the group’s + symbol names.
+ +If the Cache Type field contains the value two ((2)), then the scratch-pad space contains +cached metadata for a symbolic link in the following format: + + + + + + + + + + + +
Layout: Symbolic Link Scratch-pad Format
bytebytebytebyte
Offset to Link Value
+ + + + + + + + + + + +
Fields: Symbolic Link Scratch-pad Format
Field NameDescription
Offset to Link ValueThe value of a symbolic link (that is, the name of the thing to which it points) is stored in the + local heap. This field is the 4-byte offset into the local heap for the start of the link value, which + is null terminated.
+ +\subsection subsec_fmt4_infra_localheap III.D. Disk Format: Level 1D - Local Heaps +A local heap is a collection of small pieces of data that are particular to a single object in the HDF5 file. +Objects can be inserted and removed from the heap at any time. The address of a heap does not change once +the heap is created. For example, a group stores addresses of objects in symbol table nodes with the names +of links stored in the group’s local heap. + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Local Heap
bytebytebytebyte
Signature
VersionReserved (zero)
Data Segment SizeL
Offset to Head of Free-listL
Address of Data SegmentO
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Local Heap
Field NameDescription
SignatureThe ASCII character string “HEAP ” is used to indicate the beginning of a heap. + This gives file consistency checking utilities a better chance of reconstructing a damaged file.
VersionEach local heap has its own version number so that new heaps can be added to old files. This document + describes version zero (0) of the local heap.
Data Segment SizeThe total amount of disk memory allocated for the heap data. This may be larger than the amount of space + required by the objects stored in the heap. The extra unused space in the heap holds a linked list of + free blocks.
Offset to Head of Free-listThis is the offset within the heap data segment of the first free block (or the + @ref FMT4UndefinedAddress "undefined address" if there is no no free block). The free block + contains “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” bytes that are the offset of the next free block (or the value + ‘1’ if this is the last free block) followed by “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” bytes that + store the size of this free block. The size of the free block includes the space used to store the + offset of the next free block and the size of the current block, making the minimum size of a free + block 2 * “@ref FMT4SizeOfLengthsV0 "Size of Lengths"”.
Address of Data SegmentThe data segment originally starts immediately after the heap header, but if the data segment must grow + as a result of adding more objects, then the data segment may be relocated, in its entirety, to another + part of the file.
+ +Objects within a local heap should be aligned on an 8-byte boundary. + +\subsection subsec_fmt4_infra_globalheap III.E. Disk Format: Level 1E - Global Heap +Each HDF5 file has a global heap which stores various types of information which is typically shared between +datasets. The global heap was designed to satisfy these goals: +
    +
  1. Repeated access to a heap object must be efficient without resulting in repeated file I/O requests. + Since global heap objects will typically be shared among several datasets, it is probable that the + object will be accessed repeatedly.
  2. +
  3. Collections of related global heap objects should result in fewer and larger I/O requests. For + instance, a dataset of object references will have a global heap object for each reference. Reading + the entire set of object references should result in a few large I/O requests instead of one small + I/O request for each reference.
  4. +
  5. It should be possible to remove objects from the global heap and the resulting file hole should be + eligible to be reclaimed for other uses.
  6. +
+ +The implementation of the heap makes use of the memory management already available at the file level and +combines that with a new object called a collection to achieve goal B. The global heap is +the set of all collections. Each global heap object belongs to exactly one collection and each collection +contains one or more global heap objects. For the purposes of disk I/O and caching, a collection is treated +as an atomic object, addressing goal A. + +When a global heap object is deleted from a collection (which occurs when its reference count falls to zero), +objects located after the deleted object in the collection are packed down toward the beginning of the +collection and the collection’s global heap object 0 is created (if possible) or its size is increased +to account for the recently freed space. There are no gaps between objects in each collection, with the possible +exception of the final space in the collection, if it is not large enough to hold the header for the +collection’s global heap object 0. These features address goal C. + +The HDF5 library creates global heap collections as needed, so there may be multiple collections throughout +the file. The set of all of them is abstractly called the “global heap”, although they do not +actually link to each other, and there is no global place in the file where you can discover all of the +collections. The collections are found simply by finding a reference to one through another object in the file. +For example, data of variable-length datatype elements is stored in the global heap and is accessed via a +global heap ID. The format for global heap IDs is described at the end of this section. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: A Global Heap Collection
bytebytebytebyte
Signature
VersionReserved (zero)

Collection SizeL


Global Heap Object 1


Global Heap Object 2


...


Global Heap Object N


Global Heap Object 0 (free space)

+\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: A Global Heap Collection
Field NameDescription
SignatureThe ASCII character string “GCOL” is used to indicate the beginning of a collection. + This gives file consistency checking utilities a better chance of reconstructing a damaged file.
VersionEach collection has its own version number so that new collections can be added to old files. This + document describes version one (1) of the collections (there is no version zero (0)).
Collection SizeThis is the size in bytes of the entire collection including this field. The default (and minimum) + collection size is 4096 bytes which is a typical file system block size. This allows for 127 16-byte + heap objects plus their overhead (the collection header of 16 bytes and the 16 bytes of information + about each heap object).
Global Heap Object 1 through NThe objects are stored in any order with no intervening unused space.
Global Heap Object 0Global Heap Object 0 (zero), when present, represents the free space in the collection. Free space always + appears at the end of the collection. If the free space is too small to store the header for Object 0 + (described below) then the header is implied and is not written.
+ The field Object Size for Object 0 indicates the amount of possible free space in the collection + including the 16-byte header size of Object 0.
+ + + + + + + + + + + + + + + + + + + + + + +
Layout: Global Heap Object
bytebytebytebyte
Heap Object IndexReference Count
Reserved (zero)

Object SizeL


Object Data

+\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Global Heap Object
Field NameDescription
Heap Object IndexEach object has a unique identification number within a collection. The identification numbers are + chosen so that new objects have the smallest value possible with the exception that the identifier + 0 always refers to the object which represents all free space within the collection.
Reference CountAll heap objects have a reference count field. An object which is referenced from some other part of the + file will have a positive reference count. The reference count for Object 0 is always zero.
ReservedZero padding to align next field on an 8-byte boundary.
Object Size This is the size of the object data stored for the object. The actual storage space + allocated for the object data is rounded up to a multiple of eight.
Object DataThe object data is treated as a one-dimensional array of bytes to be interpreted by the caller.
+ +
+\anchor FMT4GlobalHeapID

The format for the ID used to locate an object in the global heap is described here:

+ + + + + + + + + + + + + + +
Layout: Global Heap ID
bytebytebytebyte

Collection AddressO

Object Index
+\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Global Heap ID
Field NameDescription
Collection AddressThis field is the address of the global heap collection where the data object is stored.
IDThis field is the index of the data object within the global heap collection.
+ +\subsection subsec_fmt4_infra_globalheapvds III.F. Disk Format: Level 1F - Global Heap Block for Virtual Datasets +The layout for the global heap block used with virtual datasets is described below. For more information +on global heaps, see “ @ref subsec_fmt4_infra_globalheap ” + +There are two versions of the Virtual Dataset Global Heap Block format: + +\li Version 0 is the original format without shared string support. +\li Version 1 introduces a new flags field that supports shared filenames and dataset names to reduce + storage requirements when the same names are used across multiple mappings. Version 1 will only + be used when the lower bound for the HDF5 library version bounds is set to 2.0 or later. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Global Heap Block for Virtual Dataset (Version 0)
bytebytebytebyte
VersionThis space inserted only to align table nicely

Num EntriesL


Source Filename \#1 (variable size)


Source Dataset \#1 (variable size)


Source Selection \#1 (variable size)


Virtual Selection \#1 (variable size)

.
.
.

Source Filename \#n (variable size)


Source Dataset \#n (variable size)


Source Selection \#n (variable size)


Virtual Selection \#n (variable size)

Checksum
+ +
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Global Heap Block for Virtual Dataset (Version 1)
bytebytebytebyte
VersionThis space inserted only to align table nicely

Num EntriesL

Flags \#1This space inserted only to align table nicely

Source Filename \#1 (variable size)


Source Dataset \#1 (variable size)


Source Selection \#1 (variable size)


Virtual Selection \#1 (variable size)

.
.
.
Flags \#nThis space inserted only to align table nicely

Source Filename \#n (variable size)


Source Dataset \#n (variable size)


Source Selection \#n (variable size)


Virtual Selection \#n (variable size)

Checksum
+ +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Global Heap Block for Virtual Dataset
Field NameDescription
VersionThe version number for the block; the value is 0 for version 0 format, 1 for version 1 format.
Num EntriesLThe number of entries in the block.
Flags \#nThis field is a bit field containing additional information about the mapping. + This field is present only in version 1 of the format. + + + + + + + + + + + + + + + + + + + + + +
BitDescription
0If set, the source filename is shared
1If set, the source dataset name is shared
2If set, the source file is the same as the virtual file
3-7Reserved for future use
+
Source Filename \#n (variable size)The source file name where the source dataset is located.

+ If the source filename is shared (bit 0 of flags is set), this is a "Size of Lengths" + length unsigned integer containing the index into the array of mappings where the source + filename is stored. For example, if the value stored here is 1, then the actual string + value for this field is stored in the Source Filename 1 field. If the source filename + is not shared, this is stored as a NULL terminated string. If the source file is the + same as the virtual file (bit 2 of flags is set), this field is not present.

+ For version 0 format, this is always stored as a NULL terminated string.
Source Dataset \#n (variable size)The source dataset name that is mapped to the virtual dataset.

+ If the source dataset name is shared (bit 1 of flags is set), this is a "Size of Lengths" + length unsigned integer containing the index into the array of mappings where the source + dataset name is stored. For example, if the value stored here is 1, then the actual string + value for this field is stored in the Source Dataset 1 field. If the source dataset name + is not shared, this is stored as a NULL terminated string.

+ For version 0 format, this is always stored as a NULL terminated string.
Source Selection \#n (variable size)The @ref FMT4DataspaceSEL "dataspace selection" in the source dataset that is mapped + to the virtual selection.
Virtual Selection \#n (variable size)This is the @ref FMT4DataspaceSEL "dataspace selection" in the virtual dataset that + is mapped to the source selection.
ChecksumThis is the checksum for the block.
+ +\subsection subsec_fmt4_infra_fractalheap III.G. Disk Format: Level 1G - Fractal Heap +Each fractal heap consists of a header and zero or more direct and indirect blocks (described below). +The header contains general information as well as initialization parameters for the doubling +table. The Address of Root Block field in the header points to the first direct or indirect block in +the heap. + +Fractal heaps are based on a data structure called a doubling table. A doubling table provides +a mechanism for quickly extending an array-like data structure that minimizes the number of empty blocks +in the heap, while retaining very fast lookup of any element within the array. More information on +fractal heaps and doubling tables can be found in the RFC +“\ref_rfc20070115 .” + +The fractal heap implements the doubling table structure with indirect and direct blocks. Indirect +blocks in the heap do not actually contain data for objects in the heap, their “size” is +abstract - they represent the indexing structure for locating the direct blocks in the doubling table. +Direct blocks contain the actual data for objects stored in the heap. + +All indirect blocks have a constant number of block entries in each row, called the width +of the doubling table (see Table Width field in the header). The number of rows for each indirect +block in the heap is determined by the size of the block that the indirect block represents in the doubling table +(calculation of this is shown below) and is constant, except for the “root” indirect block, +which expands and shrinks its number of rows as needed. + +Blocks in the first two rows of an indirect block are Starting Block Size number of +bytes in size. For example, if the row width of the doubling table is 4, then the first eight block +entries in the indirect block are Starting Block Size number of bytes in size. The blocks in each +subsequent row are twice the size of the blocks in the previous +row. In other words, blocks in the third row are twice the Starting Block Size, blocks in the +fourth row are four times the Starting Block Size, and so on. Entries for blocks up to the +Maximum Direct Block Size point to direct blocks, and entries for blocks greater than that size +point to further indirect blocks (which have their own entries for direct and indirect blocks). Starting +Block Size and Maximum Direct Block Size are fields stored in the header. + +The number of rows of blocks, nrows, in an indirect block is calculated +by the following expression:

+nrows = (log2(iblock_size) - log2(<Starting Block Size>)) + 1 +where block_size is the size of the block that the indirect block +represents in the doubling table. For example, to represent a block with block_size equals to 1024, +and Starting Block Size equals to 256, three rows are needed. + +The maximum number of rows of direct blocks, max_dblock_rows, in any indirect block of a fractal +heap is given by the following expression:

+max_dblock_rows = (log2(<Maximum Direct Block Size>) - +log2(<Starting Block Size>)) + 2 + +Using the computed values for nrows and max_dblock_rows, along with the Width +of the doubling table, the number of direct and indirect block entries (K and N in the +indirect block description, below) in an indirect block can be computed:

+K = MIN(nrows, max_dblock_rows) * Table Width

+If nrows is less than or equal to max_dblock_rows, N is 0. Otherwise, N +is simply computed:

+N = K - (max_dblock_rows * Table Width) + +The size of indirect blocks on disk is determined by the number of rows in the indirect block +(computed above). The size of direct blocks on disk is exactly the size of the block in the doubling table. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Fractal Heap Header
bytebytebytebyte
Signature
VersionThis space inserted only to align table nicely
Heap ID LengthI/O Filters’ Encoded Length
FlagsThis space inserted only to align table nicely
Maximum Size of Managed Objects

Next Huge Object IDL


v2 B-tree Address of Huge ObjectsO


Amount of Free Space in Managed BlocksL


Address of Managed Block Free Space ManagerO


Amount of Managed Space in HeapL


Amount of Allocated Managed Space in HeapL


Offset of Direct Block Allocation Iterator in Managed SpaceL


Number of Managed Objects in HeapL


Size of Huge Objects in HeapL


Number of Huge Objects in HeapL


Size of Tiny Objects in HeapL


Number of Tiny Objects in HeapL

Table WidthThis space insertedonly to align table nicely

Starting Block SizeL


Maximum Direct Block SizeL

Maximum Heap SizeStarting \# of Rows in Root Indirect Block

Address of Root BlockO

Current \# of Rows in Root Indirect BlockThis space inserted only to align table nicely

Size of Filtered Root Direct Block (optional)L

I/O Filter Mask (optional)
I/O Filter Information (optional, variable size)
Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Fractal Heap Header
Field NameDescription
SignatureThe ASCII character string “FRHP” is used to indicate the beginning of a + fractal heap header. This gives file consistency checking utilities a better chance of reconstructing + a damaged file.
VersionThis document describes version 0.
Heap ID LengthThis is the length in bytes of heap object IDs for this heap.
I/O Filters’ Encoded LengthThis is the size in bytes of the encoded I/O Filter Information.
FlagsThis field is the heap status flag and is a bit field indicating additional information about + the fractal heap. + + + + + + + + + + + + + + + + + +
Bit(s)Description
0If set, the ID value to use for huge object has wrapped around. If the value for the + Next Huge Object ID has wrapped around, each new huge object inserted into the + heap will require a search for an ID value. +
1If set, the direct blocks in the heap are checksummed.
2-7Reserved
Maximum Size of Managed ObjectsThis is the maximum size of managed objects allowed in the heap. Objects greater than this this + are ‘huge’ objects and will be stored in the file directly, rather than in a direct + block for the heap.
Next Huge Object IDThis is the next ID value to use for a huge object in the heap.
v2 B-tree Address of Huge ObjectsThis is the address of the @ref subsubsec_fmt4_infra_btrees_v2 used to track huge objects in the heap. + The type of records stored in the v2 B-tree will be determined by whether the address and + length of a huge object can fit into a heap ID (if yes, it is a “directly” accessed huge + object) and whether there is a filter used on objects in the heap.
Amount of Free Space in Managed BlocksThis is the total amount of free space in managed direct blocks (in bytes).
Address of Managed Block Free Space ManagerThis is the address of the @ref subsec_fmt4_infra_freespaceindex + for managed blocks.
Amount of Managed Space in HeapThis is the total amount of managed space in the heap (in bytes), essentially the + upper bound of the heap’s linear address space.
Amount of Allocated Managed Space in HeapThis is the total amount of managed space (in bytes) actually allocated in the heap. + This can be less than the Amount of Managed Space in Heap field, if some direct + blocks in the heap’s linear address space are not allocated.
Offset of Direct Block Allocation Iterator in Managed SpaceThis is the linear heap offset where the next direct block should be allocated at (in bytes). + This may be less than the Amount of Managed Space in Heap value because the heap’s + address space is increased by a “row” of direct blocks at a time, rather than by single + direct block increments.
Number of Managed Objects in HeapThis is the number of managed objects in the heap.
Size of Huge Objects in HeapThis is the total size of huge objects in the heap (in bytes).
Number of Huge Objects in HeapThis is the number of huge objects in the heap.
Size of Tiny Objects in HeapThis is the total size of tiny objects that are packed in heap IDs (in bytes).
Number of Tiny Objects in HeapThis is the number of tiny objects that are packed in heap IDs.
Table WidthThis is the number of columns in the doubling table for managed blocks. This value + must be a power of two.
Starting Block SizeThis is the starting block size to use in the doubling table for managed blocks (in bytes). + This value must be a power of two.
Maximum Direct Block SizeThis is the maximum size allowed for a managed direct block. Objects inserted into the heap that + are larger than this value (less the number of bytes of direct block prefix/suffix) are stored as + ‘huge’ objects. This value must be a power of two.
Maximum Heap SizeThis is the maximum size of the heap’s linear address space for managed objects (in bytes). + The value stored is the log2 of the actual value, that is: the number of bits of the address space. + ‘Huge’ and ‘tiny’ objects are not counted in this value, since they do not + store objects in the linear address space of the heap.
Starting \# of Rows in Root Indirect BlockThis is the starting number of rows for the root indirect block. A value of 0 indicates that the + root indirect block will have the maximum number of rows needed to address the heap’s + Maximum Heap Size.
Address of Root BlockThis is the address of the root block for the heap. It can be the + @ref FMT4UndefinedAddress "undefined address" if there is no data in the heap. It either + points to a direct block (if the Current \# of Rows in the Root Indirect Block value is 0), + or an indirect block.
Current \# of Rows in Root Indirect BlockThis is the current number of rows in the root indirect block. A value of 0 indicates that + Address of Root Block points to direct block instead of indirect block.
Size of Filtered Root Direct BlockThis is the size of the root direct block, if filters are applied to heap objects (in bytes). + This field is only stored in the header if the I/O Filters’ Encoded Length is + greater than 0.
I/O Filter MaskThis is the filter mask for the root direct block, if filters are applied to heap objects. This + mask has the same format as that used for the filter mask in chunked raw data records in a + @ref subsubsec_fmt4_infra_btrees_v1. This field is only stored in the header if the I/O Filters’ + Encoded Length is greater than 0.
I/O Filter InformationThis is the I/O filter information encoding direct blocks and huge objects, if filters are applied to + heap objects. This field is encoded as a @ref subsubsec_fmt4_dataobject_hdr_msg_filter message. The size + of this field is determined by I/O Filters’ Encoded Length.
ChecksumThis is the checksum for the header.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Fractal Heap Direct Block
bytebytebytebyte
Signature
VersionThis space inserted only to align table nicely

Heap Header AddressO

Block Offset (variable size)
Checksum (optional)

Object Data (variable size)

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Fractal Heap Direct Block
Field NameDescription
SignatureThe ASCII character string “FHDB” is used to indicate the beginning of + a fractal heap direct block. This gives file consistency checking utilities a better chance of + reconstructing a damaged file.
VersionThis document describes version 0.
Heap Header AddressThis is the address for the fractal heap header that this block belongs to. This field is + principally used for file integrity checking.
Block OffsetThis is the offset of the block within the fractal heap’s address space (in bytes). The + number of bytes used to encode this field is the Maximum Heap Size (in the heap’s + header) divided by 8 and rounded up to the next highest integer, for values that are not a multiple + of 8. This value is principally used for file integrity checking.
ChecksumThis is the checksum for the direct block. This field is only present if bit 1 of Flags + in the heap’s header is set.
Object DataThis section of the direct block stores the actual data for objects in the heap. The size of this + section is determined by the direct block’s size minus the size of the other fields stored in the + direct block (for example, the Signature, Version, and others including the + Checksum if it is present).
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Fractal Heap Indirect Block
bytebytebytebyte
Signature
VersionThis space inserted only to align table nicely

Heap Header AddressO

Block Offset (variable size)

Child Direct Block \#0 AddressO


Size of Filtered Direct Block \#0 (optional) L

Filter Mask for Direct Block \#0 (optional)

Child Direct Block \#1 AddressO


Size of Filtered Direct Block \#1 (optional)L

Filter Mask for Direct Block \#1 (optional)
...

Child Direct Block \#K-1 AddressO


Size of Filtered Direct Block \#K-1 (optional)L

Filter Mask for Direct Block \#K-1 (optional)

Child Indirect Block \#0 AddressO


Child Indirect Block \#1 AddressO

...

Child Indirect Block \#N-1 AddressO

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Fractal Heap Indirect Block
Field NameDescription
SignatureThe ASCII character string “FHIB” is used to indicate the beginning of a + fractal heap indirect block. This gives file consistency checking utilities a better chance of + reconstructing a damaged file.
VersionThis document describes version 0.
Heap Header AddressThis is the address for the fractal heap header that this block belongs to. This field is principally + used for file integrity checking.
Block OffsetThis is the offset of the block within the fractal heap’s address space (in bytes). The number + of bytes used to encode this field is the Maximum Heap Size (in the heap’s header) + divided by 8 and rounded up to the next highest integer, for values that are not a multiple of 8. This + value is principally used for file integrity checking.
Child Direct Block \#K AddressThis field is the address of the child direct block. The size of the [uncompressed] direct block can + be computed by its offset in the heap’s linear address space.
Size of Filtered Direct Block \#KThis is the size of the child direct block after passing through the I/O filters defined for this heap + (in bytes). If no I/O filters are present for this heap, this field is not present.
Filter Mask for Direct Block \#KThis is the I/O filter mask for the filtered direct block. This mask has the same format as that + used for the filter mask in chunked raw data records in a @ref subsubsec_fmt4_infra_btrees_v1. If + no I/O filters are present for this heap, this field is not present.
Child Indirect Block \#N AddressThis field is the address of the child indirect block. The size of the indirect block can be computed + by its offset in the heap’s linear address space.
ChecksumThis is the checksum for the indirect block.
+ +An object in the fractal heap is identified by means of a fractal heap ID, which encodes information to +locate the object in the heap. Currently, the fractal heap stores an object in one of three ways, +depending on the object’s size: + + + + + + + + + + + + + + + + + +
TypeDescription
TinyWhen an object is small enough to be encoded in the heap ID, the object’s data is embedded + in the fractal heap ID itself. There are two sub-types for this type of object: normal and extended. + The sub-type for tiny heap IDs depends on whether the heap ID is large enough to store objects + greater than 16 bytes or not. If the heap ID length is 18 bytes or smaller, the ‘normal’ + tiny heap ID form is used. If the heap ID length is greater than 18 bytes in length, the + “extended” form is used. See format description below for both sub-types.
HugeWhen the size of an object is larger than Maximum Size of Managed Objects in the + Fractal Heap Header, the object’s data is stored on its own in the file and the object + is tracked/indexed via a version 2 B-tree. All huge objects for a particular fractal heap use the same + v2 B-tree. All huge objects for a particular fractal heap use the same format for their huge object IDs. +
Depending on whether the IDs for a heap are large enough to hold the object’s retrieval + information and whether I/O pipeline filters are applied to the heap’s objects, 4 sub-types are + derived for huge object IDs for this heap: + + + + + + + + + + + + + + + + + + + + + +
Sub-typeDescription
Directly accessed, non-filteredThe object’s address and length are embedded in the fractal heap ID itself and the + object is directly accessed from them. This allows the object to be accessed without resorting + to the B-tree.
Directly accessed, filteredThe filtered object’s address, length, filter mask and de-filtered size are embedded + in the fractal heap ID itself and the object is accessed directly with them. This allows the + object to be accessed without resorting to the B-tree.
Indirectly accessed, non-filteredThe object is located by using a B-tree key embedded in the fractal heap ID to retrieve the + address and length from the version 2 B-tree for huge objects. Then, the address and length + are used to access the object.
Indirectly accessed, filteredThe object is located by using a B-tree key embedded in the fractal heap ID to retrieve the + filtered object’s address, length, filter mask and de-filtered size from the version + 2 B-tree for huge objects. Then, this information is used to access the object.
ManagedWhen the size of an object does not meet the above two conditions, the object is stored and managed + via the direct and indirect blocks based on the doubling table.
+ +The specific format for each type of heap ID is described below: + + + + + + + + + + + + + + + +
Layout: Fractal Heap ID for Tiny Objects (sub-type 1 - ‘Normal’)
bytebytebytebyte
Version, Type & LengthThis space inserted only to align table nicely

Data (variable size)
+ + + + + + + + + + + + + + + +
Fields: Fractal Heap ID for Tiny Objects (sub-type 1 - ‘Normal’)
Field NameDescription
Version, Type, and LengthThis is a bit field with the following definition: + + + + + + + + + + + + + + + + + +
BitDescription
6-7The current version of ID format. This document describes version 0.
4-5The ID type. Tiny objects have a value of 2. +
0-3The length of the tiny object. The value stored is one less than the actual length (since + zero-length objects are not allowed to be stored in the heap). For example, an object of + actual length 1 has an encoded length of 0, an object of actual length 2 has an encoded + length of 1, and so on.
DataThis is the data for the object.
+ + + + + + + + + + + + + + + + + +
Layout: Fractal Heap ID for Tiny Objects (sub-type 2 - ‘Extended’)
bytebytebytebyte
Version, Type, and LengthExtended LengthThis space inserted only to align table nicely
Data (variable size)
+ + + + + + + + + + + + + + + + + + + +
Fields: Fractal Heap ID for Tiny Objects (sub-type 2 - ‘Extended’)
Field NameDescription
Version, Type, and LengthThis is a bit field with the following definition: + + + + + + + + + + + + + + + + + +
BitDescription
6-7The current version of ID format. This document describes version 0.
4-5The ID type. Tiny objects have a value of 2.
0-3These 4 bits, together with the next byte, form an unsigned 12-bit integer for holding the + length of the object. These 4-bits are bits 8-11 of the 12-bit integer. See description + for the Extended Length field below.
Extended LengthThis byte, together with the 4 bits in the previous byte, forms an unsigned 12-bit integer for + holding the length of the tiny object. These 8 bits are bits 0-7 of the 12-bit integer formed. The + value stored is one less than the actual length (since zero-length objects are not allowed to be + stored in the heap). For example, an object of actual length 1 has an encoded length of 0, an object of + actual length 2 has an encoded length of 1, and so on.
DataThis is the data for the object.
+ + + + + + + + + + + + + + + + +
Layout: Fractal Heap ID for Huge Objects (sub-type 1 & 2): indirectly accessed, + non-filtered/filtered
bytebytebytebyte
Version and TypeThis space inserted only to align table nicely

v2 B-tree KeyL (variable size)

+\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Fractal Heap ID for Huge Objects (sub-type 1 & 2): indirectly accessed, + non-filtered/filtered
Field NameDescription
Version and TypeThis is a bit field with the following definition: + + + + + + + + + + + + + + + + + +
BitDescription
6-7The current version of ID format. This document describes version 0.
4-5The ID type. Huge objects have a value of 1.
0-3Reserved.
v2 B-tree KeyThis field is the B-tree key for retrieving the information from the version 2 B-tree for huge + objects needed to access the object. See the description of @ref subsubsec_fmt4_infra_btrees_v2 + records sub-type 1 & 2 for a description of the fields. New key values are derived from Next + Huge Object ID in the Fractal Heap Header.
+ + + + + + + + + + + + + + + + + + + +
Layout: Fractal Heap ID for Huge Objects (sub-type 3): directly accessed, non-filtered
bytebytebytebyte
Version and TypeThis space inserted only to align table nicely

Address O


Length L

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + +
Fields: Fractal Heap ID for Huge Objects (sub-type 3): directly accessed, non-filtered
Field NameDescription
Version and TypeThis is a bit field with the following definition: + + + + + + + + + + + + + + + + + +
BitDescription
6-7The current version of ID format. This document describes version 0.
4-5The ID type. Huge objects have a value of 1.
0-3Reserved.
AddressThis field is the address of the object in the file.
LengthThis field is the length of the object in the file.
+ + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Fractal Heap ID for Huge Objects (sub-type 4): directly accessed, filtered
bytebytebytebyte
Version and TypeThis space inserted only to align table nicely

Address O


Length L

Filter Mask

De-filtered Size L

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Fractal Heap ID for Huge Objects (sub-type 4): directly accessed, filtered
Field NameDescription
Version and TypeThis is a bit field with the following definition: + + + + + + + + + + + + + + + + + +
BitDescription
6-7The current version of ID format. This document describes version 0.
4-5The ID type. Huge objects have a value of 1.
0-3Reserved.
AddressThis field is the address of the filtered object in the file.
LengthThis field is the length of the filtered object in the file.
Filter MaskThis field is the I/O pipeline filter mask for the filtered object in the file.
Filtered SizeThis field is the size of the de-filtered object in the file.
+ + + + + + + + + + + + + + + + + + + +
Layout: Fractal Heap ID for Managed Objects
bytebytebytebyte
Version and TypeThis space inserted only to align table nicely
Offset (variable size)
Length (variable size)
+ + + + + + + + + + + + + + + + + + + +
Fields: Fractal Heap ID for Managed Objects
Field NameDescription
Version and TypeThis is a bit field with the following definition: + + + + + + + + + + + + + + + + + +
BitDescription
6-7The current version of ID format. This document describes version 0.
4-5The ID type. Managed objects have a value of 0.
0-3Reserved.
OffsetThis field is the offset of the object in the heap. This field’s size is the minimum number of + bytes necessary to encode the Maximum Heap Size value (from the Fractal Heap Header). + For example, if the value of the Maximum Heap Size is less than 256 bytes, this field is 1 + byte in length, a Maximum Heap Size of 256-65535 bytes uses a 2 byte length, and so on.
LengthThis field is the length of the object in the heap. It is determined by taking the minimum value + of Maximum Direct Block Size and Maximum Size of Managed Objects in the Fractal + Heap Header. Again, the minimum number of bytes needed to encode that value is used for the size + of this field.
+ +\subsection subsec_fmt4_infra_freespaceindex III.H. Disk Format: Level 1H - Free-space Index +Free-space managers are used to describe space within a heap or the entire HDF5 file that is not currently +used for that heap or file. + +The free-space manager header contains metadata information about the space being tracked, along +with the address of the list of free space sections which actually describes the free space. The +header records information about free-space sections being tracked, creation parameters for handling +free-space sections of a client, and section information used to locate the collection of free-space sections. + +The free-space section list stores a collection of free-space sections that is specific to each +client of the free-space manager. For example, the fractal heap is a client of the free space +manager and uses it to track unused space within the heap. There are 4 types of section records for the +fractal heap, each of which has its own format, listed below. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Free-space Manager Header
bytebytebytebyte
Signature
VersionClient IDThis space inserted only to align table nicely

Total Space TrackedL


Total Number of SectionsL


Number of Serialized SectionsL


Number of Un-Serialized SectionsL

Number of Section ClassesThis space inserted only to align table nicely
Shrink PercentExpand Percent
Size of Address SpaceThis space inserted only to align table nicely

Maximum Section Size L


Address of Serialized Section ListO


Size of Serialized Section List UsedL


Allocated Size of Serialized Section ListL

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Free-space Manager Header
Field NameDescription
SignatureThe ASCII character string “FSHD” is used to indicate the beginning of the + Free-space Manager Header. This gives file consistency checking utilities a better chance of + reconstructing a damaged file.
VersionThis is the version number for the Free-space Manager Header and this document describes version 0.
Client IDThis is the client ID for identifying the user of this free-space manager: + + + + + + + + + + + + + + + + + +
IDDescription
0Fractal heap
1File
2+Reserved.
Total Space TrackedThis is the total amount of free space being tracked, in bytes.
Total Number of SectionsThis is the total number of free-space sections being tracked.
Number of Serialized SectionsThis is the number of serialized free-space sections being tracked.
Number of Un-Serialized SectionsThis is the number of un-serialized free-space sections being managed. Un-serialized sections are + created by the free-space client when the list of sections is read in.
Number of Section ClassesThis is the number of section classes handled by this free space manager for the free-space client.
Shrink PercentThis is the percent of current size to shrink the allocated serialized free-space section list.
Expand PercentThis is the percent of current size to expand the allocated serialized free-space section list.
Size of Address SpaceThis is the size of the address space that free-space sections are within. This is stored as the + log2 of the actual value (in other words, the number of bits required to store values + within that address space).
Maximum Section SizeThis is the maximum size of a section to be tracked.
Address of Serialized Section ListThis is the address where the serialized free-space section list is stored.
Size of Serialized Section List UsedThis is the size of the serialized free-space section list used (in bytes). This value must be + less than or equal to the allocated size of serialized section list, below.
Allocated Size of Serialized Section ListThis is the size of serialized free-space section list actually allocated (in bytes).
ChecksumThis is the checksum for the free-space manager header.
+ +The free-space sections being managed are stored in a free-space section list, described below. +The sections in the free-space section list are stored in the following way: a count of the number of sections +describing a particular size of free space and the size of the free-space described (in bytes), followed +by a list of section description records; then another section count and size, followed by the list of +section descriptions for that size; and so on. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Free-space Section List
bytebytebytebyte
Signature
VersionThis space inserted only to align table nicely

Free-space Manager Header AddressO

Number of Section Records in Set \#0 (variable size)
Size of Free-space Section Described in Record Set \#0 (variable size)
Record Set \#0 Section Record \#0 Offset (variable size)
Record Set \#0 Section Record #0 TypeThis space inserted only to align table nicely
Record Set \#0 Section Record \#0 Data (variable size)
...
Record Set \#0 Section Record \#K-1 Offset (variable size)
Record Set \#0 Section Record \#K-1 TypeThis space inserted only to align table nicely
Record Set \#0 Section Record \#K-1 Data (variable size)
Number of Section Records in Set \#1 (variable size)
Size of Free-space Section Described in Record Set \#1 (variable size)
Record Set \#1 Section Record \#0 Offset (variable size)
Record Set \#1 Section Record \#0 TypeThis space inserted only to align table nicely
Record Set \#1 Section Record \#0 Data (variable size)
...
Record Set \#1 Section Record \#K-1 Offset (variable size)
Record Set \#1 Section Record \#K-1 TypeThis space inserted only to align table nicely
Record Set \#1 Section Record \#K-1 Data (variable size)
...
...
Number of Section Records in Set \#N-1 (variable size)
Size of Free-space Section Described in Record Set \#N-1 (variable size)
Record Set \#N-1 Section Record \#0 Offset (variable size)
Record Set \#N-1 Section Record \#0 TypeThis space inserted only to align table nicely
Record Set \#N-1 Section Record \#0 Data (variable size)
...
Record Set \#N-1 Section Record \#K-1 Offset (variable size)
Record Set \#N-1 Section Record \#K-1 TypeThis space inserted only to align table nicely
Record Set \#N-1 Section Record \#K-1 Data (variable size)
Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Free-space Section List
Field NameDescription
SignatureThe ASCII character string “FSSE” is used to indicate the beginning of the + Free-space Section Information. This gives file consistency checking utilities a better chance of + reconstructing a damaged file.
VersionThis is the version number for the Free-space Section List and this document describes version 0.
Free-space Manager Header AddressThis is the address of the Free-space Manager Header. This field is principally used for file + integrity checking.
Number of Section Records for Set \#NThis is the number of free-space section records for set \#N. The length of this field is the minimum + number of bytes needed to store the number of serialized sections (from the free-space + manager header).
+ The number of sets of free-space section records is determined by the size of serialized section + list in the free-space manager header.
Section Size for Record Set \#NThis is the size (in bytes) of the free-space section described for all the section records + in set \#N.
+ The length of this field is the minimum number of bytes needed to store the maximum section + size (from the free-space manager header).
Record Set \#N Section \#K OffsetThis is the offset (in bytes) of the free-space section within the client for the free-space manager. +
The length of this field is the minimum number of bytes needed to store the size of address + space (from the free-space manager header).
Record Set \#N Section \#K TypeThis is the type of the section record, used to decode the record set \#N section \#K data + information. The defined record type for file client is: + + + + + + + + + + + + + +
TypeDescription
0File’s section (a range of actual bytes in file)
1+Reserved.
+
The defined record types for a fractal heap client are: + + + + + + + + + + + + + + + + + + + + + + + + + +
TypeDescription
0Fractal heap “single” section
1Fractal heap “first row” section
2Fractal heap “normal row” section
3Fractal heap “indirect” section
4+Reserved.
Record Set \#N Section \#K DataThis is the section-type specific information for each record in the record set, described below.
ChecksumThis is the checksum for the Free-space Section List.
+ +The section-type specific data for each free-space section record is described below: + + + + + +
Layout: File’s Section Data Record
No additional record data stored
+
+ + + + + +
Layout: Fractal Heap “Single” Section Data Record
No additional record data stored
+
+ + + + + +
Layout: Fractal Heap “First Row” Section Data Record
Same format as “indirect” section data
+
+ + + + + +
Layout: Fractal Heap “Normal Row” Section Data Record
No additional record data stored
+
+ + + + + + + + + + + + + + + + + + + +
Layout: Fractal Heap “Indirect” Section Data Record
bytebytebytebyte
Fractal Heap Indirect Block Offset (variable size)
Block Start RowBlock Start Column
Number of BlocksThis space inserted only to align table nicely
+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Fractal Heap “Indirect” Section Data Record
Field NameDescription
Fractal Heap Block OffsetThe offset of the indirect block in the fractal heap’s address space containing the empty + blocks.
+ The number of bytes used to encode this field is the minimum number of bytes needed to encode + values for the Maximum Heap Size (in the fractal heap’s header).
Block Start RowThis is the row that the empty blocks start in.
Block Start ColumnThis is the column that the empty blocks start in.
Number of BlocksThis is the number of empty blocks covered by the section.
+ +\subsection subsec_fmt4_infra_sohm III.I. Disk Format: Level 1I - Shared Object Header Message Table +The shared object header message table is used to locate object header messages that are shared +between two or more object headers in the file. Shared object header messages are stored and indexed in +the file in one of two ways: indexed sequentially in a shared header message list or indexed +with a v2 B-tree. The shared messages themselves are either stored in a fractal heap (when two or more +objects share the message), or remain in an object’s header (when only one object uses the message +currently, but the message can be shared in the future). + +The shared object header message table contains a list of shared message index headers. Each +index header records information about the version of the index format, the index storage type, flags +for the message types indexed, the number of messages in the index, the address where the index resides, +and the fractal heap address if shared messages are stored there. + +Each index can be either a list or a v2 B-tree and may transition between those two forms as the number +of messages in the index varies. Each shared message record contains information used to locate the +shared message from either a fractal heap or an object header. The types of messages that can be shared +are: Dataspace, Datatype, Fill Value, Filter Pipeline and Attribute. + +The shared object header message table is pointed to from a +@ref subsubsec_fmt4_dataobject_hdr_msg_shared message in the superblock extension for a file. This +message stores the version of the table format, along with the number of index headers in the table. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Shared Object Header Message Table
bytebytebytebyte
Signature
Version for index \#0Index Type for index #0Message Type Flags for index \#0
Minimum Message Size for index \#0
List Cutoff for index \#0v2 B-tree Cutoff for index \#0
Number of Messages for index \#0This space inserted only to align table nicely

Index AddressO for index \#0


Fractal Heap AddressO for index \#0

...
...
Version for index \#N-1Index Type for index \#N-1Message Type Flags for index \#N-1
Minimum Message Size for index \#N-1
List Cutoff for index \#N-1v2 B-tree Cutoff for index \#N-1
Number of Messages for index \#N-1This space inserted only to align table nicely

Index AddressO for index \#N-1


Fractal Heap AddressO for index \#N-1

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Shared Object Header Message Table
Field NameDescription
SignatureThe ASCII character string “SMTB” is used to indicate the beginning of the + Shared Object Header Message table. This gives file consistency checking utilities a better chance + of reconstructing a damaged file.
Version for index \#NThis is the version number for the list of shared object header message indexes and this document + describes version 0.
Index Type for index \#NThe type of index can be an unsorted list or a v2 B-tree.
Message Type Flags for index \#NThis field indicates the type of messages tracked in the index, as follows: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
BitsDescription
0If set, the index tracks Dataspace Messages.
1If set, the message tracks Datatype Messages.
2If set, the message tracks Fill Value Messages.
3If set, the message tracks Filter Pipeline Messages.
4If set, the message tracks Attribute Messages.
5-15Reserved (zero).
+ An index can track more than one type of message, but each type of message can only by in one index.
Minimum Message Size for index \#NThis is the message size sharing threshold for the index. If the encoded size of the message is + less than this value, the message is not shared.
List Cutoff for index \#NThis is the cutoff value for the indexing of messages to switch from a list to a v2 B-tree. If the + number of messages is greater than this value, the index should be a v2 B-tree.
v2 B-tree Cutoff for index \#NThis is the cutoff value for the indexing of messages to switch from a v2 B-tree back to a list. + If the number of messages is less than this value, the index should be a list.
Number of Messages for index \#NThe number of shared messages being tracked for the index.
Index Address for index \#NThis field is the address of the list or v2 B-tree where the index nodes reside.
Fractal Heap Address for index \#NThis field is the address of the fractal heap if shared messages are stored there.
ChecksumThis is the checksum for the table.
+ +Shared messages are indexed either with a shared message record list, described below, +or using a v2 B-tree (using record type 7). The number of records in the shared message record +list is determined in the index’s entry in the shared object header message table. + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Shared Message Record List
bytebytebytebyte
Signature
Shared Message Record \#0
Shared Message Record \#1
...
Shared Message Record \#N-1
Checksum
+ + + + + + + + + + + + + + + + + + + +
Fields: Shared Message Record List
Field NameDescription
SignatureThe ASCII character string “SMLI” is used to indicate the beginning of a + list of index nodes. This gives file consistency checking utilities a better chance of + reconstructing a damaged file.
Shared Message Record \#NThe record for locating the shared message, either in the fractal heap for the index, or an object + header (see format for index nodes below).
ChecksumThis is the checksum for the list.
+ +The record for each shared message in an index is stored in one of the following forms: + + + + + + + + + + + + + + + + + + + + + +
Layout: Shared Message Record, for Messages Stored in a Fractal Heap
bytebytebytebyte
Message LocationThis space inserted only to align table nicely
Hash Value
Reference Count

Fractal Heap ID

+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Shared Message Record, for Messages Stored in a Fractal Heap
Field NameDescription
Message LocationThis has a value of 0 indicating that the message is stored in the heap.
Hash ValueThis is the hash value for the message.
Reference CountThis is the number of times the message is used in the file.
Fractal Heap IDThis is an 8-byte fractal heap ID for the message as stored in the fractal heap for the index.
+ + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Shared Message Record, for Messages Stored in an Object Header
bytebytebytebyte
Message LocationThis space inserted only to align table nicely
Hash Value
ReservedMessage TypeCreation Index

Object Header AddressO

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Shared Message Record, for Messages Stored in an Object Header
Field NameDescription
Message LocationThis has a value of 1 indicating that the message is stored in an object header.
Hash ValueThis is the hash value for the message.
Message TypeThis is the message type in the object header.
Creation IndexThis is the creation index of the message within the object header.
Object Header AddressThis is the address of the object header where the message is located.
+ +\section sec_fmt4_dataobject IV. Disk Format: Level 2 - Data Objects +Data objects contain the “real” user-visible information in the file. These objects compose +the scientific data and other information which are generally thought of as “data” by the +end-user. All the other information in the file is provided as a framework for storing and accessing +these data objects. + +A data object is composed of header and data information. The header information contains the +information needed to interpret the data information for the object as well as additional “metadata” +or pointers to additional “metadata” used to describe or annotate each object. + +\subsection subsec_fmt4_dataobject_hdr IV.A. Disk Format: Level 2A - Data Object Headers +The header information of an object is designed to encompass all the information about an object, except for +the data itself. This information includes the dataspace, datatype, information about how the data is stored +on disk (in external files, compressed, broken up in blocks, and so on), as well as other information used by the +library to speed up access to the data objects or maintain a file’s integrity. Information stored by user +applications as attributes is also stored in the object’s header. The header of each object is not necessarily +located immediately prior to the object’s data in the file and in fact may be located in any position in the +file. The order of the messages in an object header is not significant. + +Object headers are composed of a prefix and a set of messages. The prefix contains the information needed to +interpret the messages and a small amount of metadata about the object, and the messages contain the majority +of the metadata about the object. + +\subsection subsec_fmt4_dataobject_hdr_prefix IV.A.1 Disk Format: Level 2A1 - Data Object Header Prefix + +\subsubsection subsubsec_fmt4_dataobject_hdr_prefix_one IV.A.1.a Version 1 Data Object Header Prefix +Header messages are aligned on 8-byte boundaries for version 1 object headers. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 1 Object Header
bytebytebytebyte
VersionReserved (zero)Total Number of Header Messages
Object Reference Count
Object Header Size
Reserved (zero)
Header Message Type \#1Size of Header Message Data \#1
Header Message \#1 FlagsReserved (zero)

Header Message Data \#1

.
.
.
Header Message Type \#nSize of Header Message Data \#n
Header Message \#n FlagsReserved (zero)

Header Message Data \#n

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 1 Object Header
Field NameDescription
VersionThis value is used to determine the format of the information in the object header. When the format of + the object header is changed, the version number is incremented and can be used to determine how the + information in the object header is formatted. This is version one (1) (there was no version zero (0)) + of the object header.
Total Number of Header MessagesThis value determines the total number of messages listed in object headers for this object. This value + includes the messages in continuation messages for this object.
Object Reference CountThis value specifies the number of “hard links” to this object within the current file. + References to the object from external files, “soft links” in this file and object + references in this file are not tracked.
Object Header SizeThis value specifies the number of bytes of header message data following this length field that + contain object header messages for this object header. This value does not include the size of object header + continuation blocks for this object elsewhere in the file.
Header Message \#n TypeThis value specifies the type of information included in the following header message data. The + message types for header messages are defined in sections below.
Size of Header Message \#n DataThis value specifies the number of bytes of header message data following the header message type and + length information for the current message. The size includes padding bytes to make the message a multiple + of eight bytes.
Header Message \#n FlagsThis is a bit field with the following definition: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
BitDescription
0If set, the message data is constant. This is used for messages like the datatype message of + a dataset.
1If set, the message is shared and stored in another location than the object header. + The Header Message Data field contains a Shared Message (described in the @ref + subsec_fmt4_dataobject_hdr_msg section below) and the Size of Header Message Data field contains + the size of that Shared Message.
2If set, the message should not be shared.
3If set, the HDF5 decoder should fail to open this object if it does not understand the + message’s type and the file is open with permissions allowing write access to the file. + (Normally, unknown messages can just be ignored by HDF5 decoders)
4If set, the HDF5 decoder should set bit 5 of this message’s flags (in other words, this + bit field) if it does not understand the message’s type and the object is modified in any + way. (Normally, unknown messages can just be ignored by HDF5 decoders)
5If set, this object was modified by software that did not understand this message. (Normally, + unknown messages should just be ignored by HDF5 decoders) (Can be used to invalidate an index + or a similar feature)
6If set, this message is shareable.
7If set, the HDF5 decoder should always fail to open this object if it does not understand the + message’s type (whether it is open for read-only or read-write access). (Normally, unknown + messages can just be ignored by HDF5 decoders)
+
Header Message \#n DataThe format and length of this field is determined by the header message type and size respectively. + Some header message types do not require any data and this information can be eliminated by setting the + length of the message to zero. The data is padded with enough zeros to make the size a multiple of + eight.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_prefix_two IV.A.1.b Version 2 Data Object Header Prefix +Note that the “total number of messages” field has been dropped from the data object header +prefix in this version. The number of messages in the data object header is just determined by the +messages encountered in all the object header blocks. + +Note also that the fields and messages in this version of data object headers have no alignment +or padding bytes inserted - they are stored packed together. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 Object Header
bytebytebytebyte
Signature
VersionFlagsThis space inserted only to align table nicely
Access time (optional)
Modification Time (optional)
Change Time (optional)
Birth Time (optional)
Maximum \# of compact attributes (optional)Minimum \# of dense attributes (optional)
Size of Chunk \#0 (variable size)This space inserted only to align table nicely
Header Message Type \#1Size of Header Message Data \#1Header Message \#1 Flags
Header Message \#1 Creation Order (optional)This space inserted only to align table nicely

Header Message Data \#1

.
.
.
Header Message Type \#nSize of Header Message Data \#nHeader Message \#n Flags
Header Message \#n Creation Order (optional)This space inserted only to align table nicely

Header Message Data \#n

Gap (optional, variable size)
Checksum
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 Object Header
Field NameDescription
SignatureThe ASCII character string “OHDR” is used to indicate the beginning of + an object header. This gives file consistency checking utilities a better chance of reconstructing + a damaged file.
VersionThis field has a value of 2 indicating version 2 of the object header.
FlagsThis field is a bit field indicating additional information about the object header. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Bit(s)Description
0-1This two bit field determines the size of the Size of Chunk \#0 field. The values are: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0The Size of Chunk \#0 field is 1 byte.
1The Size of Chunk \#0 field is 2 bytes.
2The Size of Chunk \#0 field is 4 bytes.
3The Size of Chunk \#0 field is 8 bytes.
2If set, attribute creation order is tracked.
3If set, attribute creation order is indexed.
4If set, non-default attribute storage phase change values are stored.
5If set, access, modification, change and birth times are stored.
6-7Reserved
Access TimeThis 32-bit value represents the number of seconds after the UNIX epoch when the object’s + raw data was last accessed (in other words, read or written). This field is present if bit 5 of + flags is set.
Modification TimeThis 32-bit value represents the number of seconds after the UNIX epoch when the object’s + raw data was last modified (in other words, written). This field is present if bit 5 of + flags is set.
Change TimeThis 32-bit value represents the number of seconds after the UNIX epoch when the object’s + metadata was last changed. This field is present if bit 5 of flags is set.
Birth TimeThis 32-bit value represents the number of seconds after the UNIX epoch when the object was + created. This field is present if bit 5 of flags is set.
Maximum \# of compact attributesThis is the maximum number of attributes to store in the compact format before switching to the + indexed format. This field is present if bit 4 of flags is set.
Minimum \# of dense attributesThis is the minimum number of attributes to store in the indexed format before switching to the + compact format. This field is present if bit 4 of flags is set.
Size of Chunk \#0This unsigned value specifies the number of bytes of header message data following this field + that contain object header information. This value does not include the size of object header + continuation blocks for this object elsewhere in the file. The length of this field varies + depending on bits 0 and 1 of the flags field.
Header Message \#n TypeSame format as version 1 of the object header, described above.
Size of Header Message \#n DataThis value specifies the number of bytes of header message data following the header message + type and length information for the current message. The size of messages in this version does + not include any padding bytes.
Header Message \#n FlagsSame format as version 1 of the object header, described above.
Header Message \#n Creation OrderThis field stores the order that a message of a given type was created in.
+ This field is present if bit 2 of flags is set.
Header Message \#n DataSame format as version 1 of the object header, described above.
GapA gap in an object header chunk is inferred by the end of the messages for the chunk before the + beginning of the chunk’s checksum. Gaps are always smaller than the size of an object header + message prefix (message type + message size + message flags).
+ Gaps are formed when a message (typically an attribute message) in an earlier chunk is deleted + and a message from a later chunk that does not quite fit into the free space is moved into the + earlier chunk.
ChecksumThis is the checksum for the object header chunk.
+ +The header message types and the message data associated with them compose the critical "meta-data" about +each object. Some header messages are required for each object while others are optional. Some optional +header messages may also be repeated several times in the header itself, the requirements and number of +times allowed in the header will be noted in each header message description below. + +\subsection subsec_fmt4_dataobject_hdr_msg IV.A.2 Disk Format: Level 2A2 - Data Object Header Messages +Data object header messages are small pieces of metadata that are stored in the data object header for +each object in an HDF5 file. Data object header messages provide the metadata required to describe an +object and its contents, as well as optional pieces of metadata that annotate the meaning or purpose of +the object. + +Data object header messages are either stored directly in the data object header for the object or are +shared between multiple objects in the file. When a message is shared, a flag in the Message Flags +indicates that the actual Message Data portion of that message is stored in another location +(such as another data object header, or a heap in the file) and the Message Data field +contains the information needed to locate the actual information for the message. + +The format of shared message data is described here: + + + + + + + + + + + + + + + + + + + +
Layout: Shared Message (Version 1)
bytebytebytebyte
VersionTypeReserved (zero)
Reserved (zero)

AddressO

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + +
Fields: Shared Message (Version 1)
Field NameDescription
VersionThe version number is used when there are changes in the format of a shared object message and is + described here: + + + + + + + + + + + + + +
VersionDescription
0Never used.
1Used by the library before version 1.6.1.
TypeThe type of shared message location: + + + + + + + + + +
ValueDescription
0Message stored in another object’s header (a committed message).
AddressThe address of the object header containing the message to be shared.
+ + + + + + + + + + + + + + + + + +
Layout: Shared Message (Version 2)
bytebytebytebyte
VersionTypeThis space inserted only to align table nicely

AddressO

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + +
Fields: Shared Message (Version 2)
Field NameDescription
VersionThe version number is used when there are changes in the format of a shared object message and is + described here: + + + + + + + + + +
VersionDescription
2Used by the library of version 1.6.1 and after.
TypeThe type of shared message location: + + + + + + + + + +
ValueDescription
0Message stored in another object’s header (a committed message).
AddressThe address of the object header containing the message to be shared.
+ + + + + + + + + + + + + + + + + +
Layout: Shared Message (Version 3)
bytebytebytebyte
VersionTypeThis space inserted only to align table nicely
Location (variable size)
+ + + + + + + + + + + + + + + + + + + +
Fields: Shared Message (Version 3)
Field NameDescription
VersionThe version number indicates changes in the format of shared object message and is described here: + + + + + + + + + +
VersionDescription
3Used by the library of version 1.8 and after. In this version, the Type field can + indicate that the message is stored in the fractal heap.
TypeThe type of shared message location: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0Message is not shared and is not shareable.
1Message stored in file’s shared object header message heap + (a shared message).
2Message stored in another object’s header (a committed message).
3Message stored is not shared, but is shareable.
LocationThis field contains either a @ref FMT4SizeOfOffsetsV0 "Size of Offsets"-bytes address of the object header + containing the message to be shared, or an 8-byte fractal heap ID for the message in the + file’s shared object header message heap.
+ +The following is a list of currently defined header messages: + +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_nil IV.A.2.a. The NIL Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: NIL
Header Message Type: 0x0000
Length: Varies
Status: Optional; may be repeated.
Description:The NIL message is used to indicate a message which is to be ignored when reading the header messages + for a data object. [Possibly one which has been deleted for some reason.]
Format of Data: Unspecified
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_simple IV.A.2.b. The Dataspace Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Dataspace
Header Message Type: 0x0001
Length: Varies according to the number of dimensions, as described in the following + table.
Status: Required for dataset objects; may not be repeated.
Description:The dataspace message describes the number of dimensions (in other words, “rank”) and size + of each dimension that the data object has. This message is only used for datasets which have a + simple, rectilinear, array-like layout; datasets requiring a more complex layout are not yet supported.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Dataspace Message - Version 1
bytebytebytebyte
VersionDimensionalityFlagsReserved
Reserved
Dimension \#1 SizeL

.
.
.

Dimension \#n SizeL


Dimension \#1 Maximum SizeL

.
.
.

Dimension \#n Maximum SizeL


Permutation Index \#1L

.
.
.

Permutation Index \#nL

+\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Dataspace Message - Version 1
Field NameDescription
Version This value is used to determine the format of the Dataspace Message. When the format of the + information in the message is changed, the version number is incremented and can be used to determine + how the information in the object header is formatted. This document describes version one (1) (there + was no version zero (0)).
DimensionalityThis value is the number of dimensions that the data object has.
FlagsThis field is used to store flags to indicate the presence of parts of this message. Bit 0 (the least + significant bit) is used to indicate that maximum dimensions are present. Bit 1 is used to indicate + that permutation indices are present.
Dimension \#n SizeThis value is the current size of the dimension of the data as stored in the file. The first dimension + stored in the list of dimensions is the slowest changing dimension and the last dimension stored is the + fastest changing dimension.
Dimension \#n Maximum SizeThis value is the maximum size of the dimension of the data as stored in the file. This value may be + the special “@ref FMT4UnlimitedDim "unlimited size"” which indicates that the data + may expand along this dimension indefinitely. If these values are not stored, the maximum size of each + dimension is assumed to be the dimension’s current size.
Permutation Index \#nThis value is the index permutation used to map each dimension from the canonical representation to an + alternate axis for each dimension. If these values are not stored, the first dimension stored in the list + of dimensions is the slowest changing dimension and the last dimension stored is the fastest changing + dimension.
+ +Version 2 of the dataspace message dropped the optional permutation index value support, as it was never +implemented in the HDF5 Library: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Dataspace Message - Version 2
bytebytebytebyte
VersionDimensionalityFlagsType
Dimension \#1 SizeL

.
.
.

Dimension \#n SizeL


Dimension \#1 Maximum SizeL

.
.
.

Dimension \#n Maximum SizeL

+\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Dataspace Message - Version 2
Field NameDescription
Version This value is used to determine the format of the Dataspace Message. This field should be ‘2’ + for version 2 format messages.
DimensionalityThis value is the number of dimensions that the data object has.
FlagsThis field is used to store flags to indicate the presence of parts of this message. Bit 0 (the least + significant bit) is used to indicate that maximum dimensions are present.
Typeindicates the type of the dataspace: + + + + + + + + + + + + + + + + + +
ValueDescription
0A scalar dataspace; in other words, a dataspace with a single, dimensionless element.
1A simple dataspace; in other words, a dataspace with a rank > 0 and an appropriate number + of dimensions.
2A null dataspace; in other words, a dataspace with no elements.
Dimension \#n SizeThis value is the current size of the dimension of the data as stored in the file. The first dimension + stored in the list of dimensions is the slowest changing dimension and the last dimension stored is the + fastest changing dimension.
Dimension \#n Maximum SizeThis value is the maximum size of the dimension of the data as stored in the file. This value may be + the special “@ref FMT4UnlimitedDim "unlimited size"” which indicates that the data + may expand along this dimension indefinitely. If these values are not stored, the maximum size of each + dimension is assumed to be the dimension’s current size.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_linkinfo IV.A.2.c. The Link Info Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Link Info
Header Message Type: 0x002
Length: Varies
Status: Optional; may not be repeated.
Description:The link info message tracks variable information about the current state of the links for a + “new style” group’s behavior. Variable information will be stored in this + message and constant information will be stored in the @ref + subsubsec_fmt4_dataobject_hdr_msg_groupinfo message.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Link Info
bytebytebytebyte
VersionFlagsThis space inserted only to align table nicely

Maximum Creation Index (8 bytes, optional)


Fractal Heap AddressO


Address of v2 B-tree for Name IndexO


Address of v2 B-tree for Creation Order IndexO (optional)

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Link Info
Field NameDescription
VersionThe version number for this message. This document describes version 0.
FlagsThis field determines various optional aspects of the link info message: + + + + + + + + + + + + + + + + + +
BitDescription
0If set, creation order for the links is tracked.
1If set, creation order for the links is indexed.
2-7Reserved
Maximum Creation IndexThis 64-bit value is the maximum creation order index value stored for a link in this group.
+ This field is present if bit 0 of flags is set.
Fractal Heap AddressThis is the address of the fractal heap to store dense links. Each link stored in the fractal heap + is stored as a @ref subsubsec_fmt4_dataobject_hdr_msg_link.
+ If there are no links in the group, or the group’s links are stored “compactly” + (as object header messages), this value will be the @ref FMT4UndefinedAddress "undefined address".
Address of v2 B-tree for Name IndexThis is the address of the version 2 B-tree to index names of links.
+ If there are no links in the group, or the group’s links are stored “compactly” + (as object header messages), this value will be the @ref FMT4UndefinedAddress "undefined address".
Address of v2 B-tree for Creation Order IndexThis is the address of the version 2 B-tree to index creation order of links.
+ If there are no links in the group, or the group’s links are stored “compactly” + (as object header messages), this value will be the @ref FMT4UndefinedAddress "undefined address".
+ This field exists if bit 1 of flags is set.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_dtmessage IV.A.2.d. The Datatype Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Datatype
Header Message Type: 0x0003
Length: Variable
Status: Required for dataset or committed datatype (formerly named datatype) + objects; may not be repeated.
Description:The datatype message defines the datatype for each element of a dataset or a common datatype for + sharing between multiple datasets. A datatype can describe an atomic type like a fixed- or + floating-point type or more complex types like a C struct (compound datatype), array (array datatype) + or C++ vector (variable-length datatype).
+ Datatype messages that are part of a dataset object do not describe how elements are related to one + another; the dataspace message is used for that purpose. Datatype messages that are part of a committed + datatype (formerly named datatype) message describe a common datatype that can be shared by multiple + datasets in the file.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + +
Layout: Datatype Message
bytebytebytebyte
Class and VersionClass Bit Field, Bits 0-7Class Bit Field, Bits 8-15Class Bit Field, Bits 16-23
Size


Properties

/
+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Datatype Message
Field NameDescription
Class and VersionThe version of the datatype message and the datatype’s class information are packed together in + this field. The version number is packed in the top 4 bits of the field and the class is contained + in the bottom 4 bits.
+ The version number information is used for changes in the format of the datatype message and is + described here: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
VersionDescription
0Never used
1Used by early versions of the library to encode compound datatypes with explicit array fields. + See the compound datatype description below for further details.
2Used when an array datatype needs to be encoded.
3Used when a VAX byte-ordered type needs to be encoded. Packs various other datatype classes more + efficiently also.
4Used to encode the revised reference datatype.
5Used when a complex number datatype needs to be encoded.

+ The class of the datatype determines the format for the class bit field and properties portion of the + datatype message, which are described below. The following classes are currently defined: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0\ref FMT4ClassFixedPoint "Fixed-Point"
1\ref FMT4ClassFloatingPoint "Floating-Point"
2\ref FMT4ClassTime "Time"
3\ref FMT4ClassString "String"
4\ref FMT4ClassBitField "Bit field"
5\ref FMT4ClassOpaque "Opaque"
6\ref FMT4ClassCompound "Compound"
7\ref FMT4ClassReference "Reference"
8\ref FMT4ClassEnum "Enumerated"
9\ref FMT4ClassVarLen "Variable-Length"
10\ref FMT4ClassArray "Array"
11\ref FMT4ClassComplex "Complex"
Class Bit FieldsThe information in these bit fields is specific to each datatype class and is described below. + All bits not defined for a datatype class are set to zero.
SizeThe size of a datatype element in bytes.
PropertiesThis variable-sized sequence of bytes encodes information specific to each datatype class and is + described for each class below. If there is no property information specified for a datatype class, + the size of this field is zero bytes.
+ +\anchor FMT4ClassFixedPoint

Class specific information for Fixed-Point Numbers (Class 0):

+ + + + + + + + + + + + + + + + + + + + + + +
Bits: Fixed-point Bit Field Description
BitsMeaning
0Byte Order. If zero, byte order is little-endian; otherwise, byte order is big endian.
1, 2Padding type. Bit 1 is the lo_pad type and bit 2 is the hi_pad type. If a datum has + unused bits at either end, then the lo_pad or hi_pad bit is copied to those locations.
3Signed. If this bit is set then the fixed-point number is in 2’s complement form.
4-23Reserved (zero).
+ + + + + + + + + + + + + +
Layout: Fixed-Point Property Descriptions
ByteByteByteByte
Bit OffsetBit Precision
+ + + + + + + + + + + + + + + +
Fields: Fixed-Point Property Descriptions
Field NameDescription
Bit OffsetThe bit offset of the first significant bit of the fixed-point value within the datatype. The + bit offset specifies the number of bits “to the right of” the value (which are + set to the lo_pad bit value).
Bit PrecisionThe number of bits of precision of the fixed-point value within the datatype. This value, + combined with the datatype element’s size and the Bit Offset field specifies the number + of bits “to the left of” the value (which are set to the hi_pad bit value).
+ +\anchor FMT4ClassFloatingPoint

Class specific information for Floating-Point Numbers (Class 1):

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Bits: Floating-Point Bit Field Description
BitsMeaning
0Byte Order. These two non-contiguous bits specify the “endianness” of + the bytes in the datatype element. + + + + + + + + + + + + + + + + + + + + + + + + + + +
Bit 6Bit 0Description
00Byte order is little-endian
01Byte order is big-endian
10Reserved
11Byte order is VAX-endian
1, 2, 3Padding type. Bit 1 is the low bits pad type, bit 2 is the high bits pad type, and bit + 3 is the internal bits pad type. If a datum has unused bits at either end or between the sign bit, exponent, + or mantissa, then the value of bit 1, 2, or 3 is copied to those locations.
4-5Mantissa Normalization. This 2-bit bit field specifies how the most significant bit of + the mantissa is managed. + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0No normalization
1The most significant bit of the mantissa is always set (except for 0.0).
2The most significant bit of the mantissa is not stored, but is implied to be set.
3Reserved.
6-7Reserved (zero).
8-15Sign Location. This is the bit position of the sign bit. Bits are numbered with the least + significant bit zero.
16-23Reserved (zero).
+ + + + + + + + + + + + + + + + + + + + + + +
Layout: Floating-Point Property Description
ByteByteByteByte
Bit OffsetBit Precision
Exponent LocationExponent SizeMantissa LocationMantissa Size
Exponent Bias
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Floating-Point Property Description
Field NameProperty Description
Bit OffsetThe bit offset of the first significant bit of the floating-point value within the datatype. The + bit offset specifies the number of bits “to the right of” the value.
Bit PrecisionThe number of bits of precision of the floating-point value within the datatype.
Exponent LocationThe bit position of the exponent field. Bits are numbered with the least significant + bit number zero.
Exponent SizeThe size of the exponent field in bits.
Mantissa LocationThe bit position of the mantissa field. Bits are numbered with the least significant bit number + zero.
Mantissa SizeThe size of the mantissa field in bits.
Exponent BiasThe bias of the exponent field.
+ +\anchor FMT4ClassTime

Class specific information for Time (Class 2):

+ + + + + + + + + + + + + + +
Bits: Time Bit Field Description
BitsMeaning
0Byte Order. If zero, byte order is little-endian; otherwise, byte order is big endian.
1-23Reserved (zero).
+
+ + + + + + + + + + +
Layout: Time Property Description
ByteByte
Bit Precision
+
+ + + + + + + + + + + +
Fields: Time Property Description
Field NameDescription
Bit PrecisionThe number of bits of precision of the time value.
+
+ +\anchor FMT4ClassString

Class specific information for Strings (Class 3):

+ + + + + + + + + + + + + + + + + +
Bits: String Bit Field Description
BitsMeaning
0-3Padding type. This four-bit value determines the type of padding to use for the + string. The values are: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0Null Terminate: A zero byte marks the end of the string and is guaranteed to be present after + converting a long string to a short string. When converting a short string to a long string the + value is padded with additional null characters as necessary.
1Null Pad: Null characters are added to the end of the value during conversions from short values + to long values but conversion in the opposite direction simply truncates the value.
2Space Pad: Space characters are added to the end of the value during conversions from short values + to long values but conversion in the opposite direction simply truncates the value. This is the + Fortran representation of the string.
3-15Reserved.
+
4-7Character Set. The character set used to encode the string. + + + + + + + + + + + + + + + + + +
ValueDescription
0ASCII character set encoding
1UTF-8 character set encoding
2-15Reserved
8-23Reserved (zero).
+ +There are no properties defined for the string class. + +\anchor FMT4ClassBitField

Class specific information for Bitfields (Class 4):

+ + + + + + + + + + + + + + + + + + +
Bits: Bitfield Bit Field Description
BitsMeaning
0Byte Order. If zero, byte order is little-endian; otherwise, byte order is big endian.
1, 2Padding type. Bit 1 is the lo_pad type and bit 2 is the hi_pad type. If a datum has + unused bits at either end, then the lo_pad or hi_pad bit is copied to those locations.
3-23Reserved (zero).
+ + + + + + + + + + + + + +
Layout: Bit Field Property Description
ByteByteByteByte
Bit OffsetBit Precision
+ + + + + + + + + + + + + + + +
Fields: Bit Field Property Description
Field NameDescription
Bit OffsetThe bit offset of the first significant bit of the bitfield within the datatype. The bit + offset specifies the number of bits “to the right of” the value.
Bit PrecisionThe number of bits of precision of the bit field within the datatype.
+ +\anchor FMT4ClassOpaque

Class specific information for Opaque (Class 5):

+ + + + + + + + + + + + + + +
Bits: Opaque Bit Field Description
BitsMeaning
0-7Length of ASCII tag in bytes.
8-23Reserved (zero).
+ + + + + + + + + + + + +
Layout: Opaque Property Description
ByteByteByteByte

ASCII Tag

+ + + + + + + + + + + +
Fields: Opaque Property Description
Field NameDescription
ASCII TagThis NUL-terminated string provides a description for the opaque type. It is NUL-padded to a + multiple of 8 bytes.
+ +\anchor FMT4ClassCompound

Class specific information for Compound Types (Class 6):

+ + + + + + + + + + + + + + +
Bits: Compound Bit Field Description
BitsMeaning
0-15Number of Members. This field contains the number of members defined for the + compound datatype. The member definitions are listed in the Properties field of the data type + message.
16-23Reserved (zero).
+ +The Properties field of a compound datatype is a list of the member definitions of the compound datatype. +The member definitions appear one after another with no intervening bytes. The member types are described +with a (recursively) encoded datatype message. + +Note that the property descriptions are different for different versions of the datatype version. Additionally +note that the version 0 properties are deprecated and has been replaced with later encodings in +versions of the HDF5 library from the 1.4 release onward. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Compound Properties Description for Datatype Version 1
ByteByteByteByte


Name

/
Byte Offset of Member
DimensionalityReserved (zero)
Dimension Permutation
Reserved (zero)
Dimension \#1 Size (required)
Dimension \#2 Size (required)
Dimension \#3 Size (required)
Dimension \#4 Size (required)

Member Type Message

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Compound Properties Description for Datatype Version 1
Field NameDescription
NameThis NUL-terminated string provides a description for the opaque type. It is NUL-padded to a + multiple of 8 bytes.
Byte Offset of MemberThis is the byte offset of the member within the datatype.
DimensionalityIf set to zero, this field indicates a scalar member. If set to a value greater than zero, + this field indicates that the member is an array of values. For array members, the size of + the array is indicated by the ‘Size of Dimension n’ field in this message.
Dimension PermutationThis field was intended to allow an array field to have its dimensions permuted, but this was + never implemented. This field should always be set to zero.
Dimension \#n SizeThis field is the size of a dimension of the array field as stored in the file. The first + dimension stored in the list of dimensions is the slowest changing dimension and the last + dimension stored is the fastest changing dimension.
Member Type MessageThis field is a datatype message describing the datatype of the member.
+ + + + + + + + + + + + + + + + + + +
Layout: Compound Properties Description for Datatype Version 2
ByteByteByteByte

Name

Byte Offset of Member

Member Type Message

+ + + + + + + + + + + + + + + + + + + +
Fields: Compound Properties Description for Datatype Version 2
Field NameDescription
NameThis NUL-terminated string provides a description for the opaque type. It is NUL-padded to a multiple + of 8 bytes.
Byte Offset of MemberThis is the byte offset of the member within the datatype.
Member Type MessageThis field is a datatype message describing the datatype of the member.
+ + + + + + + + + + + + + + + + + + +
Layout: Compound Properties Description for Datatype Version 3
ByteByteByteByte

Name

Byte Offset of Member

Member Type Message

+ + + + + + + + + + + + + + + + + + + +
Fields: Compound Properties Description for Datatype Version 3
Field NameDescription
NameThis NUL-terminated string provides a description for the opaque type. It is not NUL-padded to a + multiple of 8 bytes.
Byte Offset of MemberThis is the byte offset of the member within the datatype. The field size is the minimum number of bytes + necessary, based on the size of the datatype element. For example, a datatype element size of less than + 256 bytes uses a 1 byte length, a datatype element size of 256-65535 bytes uses a 2 byte length, and + so on.
Member Type MessageThis field is a datatype message describing the datatype of the member.
+ +\anchor FMT4ClassReference

Class specific information for Reference (Class 7):


+Note that for region references, the stored data is a \ref FMT4GlobalHeapID "Global Heap ID" pointing to information +about the region stored in the global heap. + + + + + + + + + + + + + + + +
Bits: Reference Bit Field Description for Datatype Version < 4
BitsMeaning
0-3Type. This four-bit value contains the reference types which are supported + for backward compatibility. The values defined are: + + + + + + + + + + + + + + + + + +
ValueDescription
0Object Reference (#H5R_OBJECT1): A reference to another object in this HDF5 file.
1Dataset Region Reference (#H5R_DATASET_REGION1): A reference to a region within a dataset + in this HDF5 file.
2-15Reserved
4-23Reserved (zero).
+ + + + + + + + + + + + + + + + + + + +
Bits: Reference Bit Field Description for Datatype Version 4
BitsMeaning
0-3Type. This four-bit value contains the revised reference types. The values + defined are: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
2Object Reference (#H5R_OBJECT2): A reference to another object in this file or an + external file.
3Dataset Region Reference (#H5R_DATASET_REGION2): A reference to a region within a + dataset in this file or an external file.
4Attribute Reference (#H5R_ATTR): A reference to an attribute attached to an object + in this file or an external file.
5-15Reserved
+
4-7Version. This four-bit value contains the version for encoding the revised + reference types. The values defined are: + + + + + + + + + + + + + + + + + +
ValueDescription
0Unused
1The version for encoding the revised reference types: Object Reference (2), + Dataset Region Reference (3) and Attribute Reference (4).
2-15Reserved
8-23Reserved (zero).
+ +There are no properties defined for the reference class. + +\anchor FMT4ClassEnum

Class specific information for Enumeration (Class 8):

+ + + + + + + + + + + + + + +
Bits: Enumeration Bit Field Description
BitsMeaning
0-15Number of Members. The number of name/value pairs defined for the enumeration type.
16-23Reserved (zero).
+ + + + + + + + + + + + + + + + + + +
Layout: Enumeration Property Description for Datatype Versions 1 and 2
ByteByteByteByte

Base Type
/

Names


Values

+ + + + + + + + + + + + + + + + + + + +
Fields: Enumeration Property Description for Datatype Versions 1 and 2
Field NameDescription
Base TypeEach enumeration type is based on some parent type, usually an integer. The information + for that parent type is described recursively by this field.
NamesThe name for each name/value pair. Each name is stored as a null terminated ASCII string + in a multiple of eight bytes. The names are in no particular order.
ValuesThe list of values in the same order as the names. The values are packed (no inter-value + padding) and the size of each value is determined by the parent type.
+ + + + + + + + + + + + + + + + + + +
Layout: Enumeration Property Description for Datatype Versions 3
ByteByteByteByte

Base Type
/

Names


Values

+ + + + + + + + + + + + + + + + + + + +
Fields: Enumeration Property Description for Datatype Versions 3
Field NameDescription
Base TypeEach enumeration type is based on some parent type, usually an integer. The information + for that parent type is described recursively by this field.
NamesThe name for each name/value pair. Each name is stored as a null terminated ASCII string, + not padded to a multiple of eight bytes. The names are in no particular order.
ValuesThe list of values in the same order as the names. The values are packed (no inter-value + padding) and the size of each value is determined by the parent type.
+ +\anchor FMT4ClassVarLen

Class specific information for Variable-Length (Class 9):

+ + + + + + + + + + + + + + + + + + + + + + +
Bits: Variable-Length Bit Field Description
BitsMeaning
0-3Type. This four-bit value contains the type of variable-length datatype described. + The values defined are: + + + + + + + + + + + + + + + + + +
ValueDescription
0Sequence: A variable-length sequence of any datatype. Variable-length sequences do not + have padding or character set information.
1String: A variable-length sequence of characters. Variable-length strings have padding and + character set information.
2-15Reserved
4-7Padding type. (variable-length string only). This four-bit value determines the + type of padding used for variable-length strings. The values are the same as for the string padding + type, as follows: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
00 Null terminate: A zero byte marks the end of a string and is guaranteed to be present after + converting a long string to a short string. When converting a short string to a long string, + the value is padded with additional null characters as necessary.
1Null pad: Null characters are added to the end of the value during conversion from a short string to + a longer string. Conversion from a long string to a shorter string simply truncates the value.
2Space pad: Space characters are added to the end of the value during conversion from a short string + to a longer string. Conversion from a long string to a shorter string simply truncates the value. + This is the Fortran representation of the string.
3-15Reserved
+ This value is set to zero for variable-length sequences.
8-11Character Set. (variable-length string only) This four-bit value specifies the + character set to be used for encoding the string: + + + + + + + + + + + + + + + + + +
ValueDescription
0ASCII character set encoding.
1UTF-8 character set encoding.
2-15Reserved
+ This value is set to zero for variable-length sequences.
12-23Reserved (zero).
+ + + + + + + + + + + + +
Layout: Variable-Length Property Description
ByteByteByteByte

Parent Type Message

+ + + + + + + + + + + +
Fields: Variable-Length Property Description
Field NameDescription
Parent TypeEach variable-length type is based on some parent type. This field contains the datatype + message describing that parent type. In the case of nested variable-length types, this parent + datatype message will recursively contain all parent datatype messages. Variable-length strings + are considered to have the parent type #H5T_NATIVE_UCHAR.
+ +\anchor FMT4ClassArray

Class specific information for Array (Class 10):

+ +There are no bit fields defined for the array class. + +Note that the dimension information defined in the property for this datatype class is independent of +dataspace information for a dataset. The dimension information here describes the dimensionality of the +information within a data element (or a component of an element, if the array datatype is nested within +another datatype) and the dataspace for a dataset describes the location of the elements in a dataset. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Array Property Description for Datatype Version 2
ByteByteByteByte
DimensionalityReserved (zero)
Dimension \#1 Size
.
.
.
Dimension \#n Size
Permutation Index \#1
.
.
.
Permutation Index \#n

Base Type

+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Array Property Description for Datatype Version 2
Field NameDescription
DimensionalityThis value is the number of dimensions that the array has.
Dimension \#n SizeThis value is the size of the dimension of the array as stored in the file. The first dimension + stored in the list of dimensions is the slowest changing dimension and the last dimension stored + is the fastest changing dimension.
Permutation Index \#nThis value is the index permutation used to map each dimension from the canonical representation + to an alternate axis for each dimension. Currently, dimension permutations are not supported and + these indices should be set to the index position minus one (i.e. the first dimension should be + set to 0, the second dimension should be set to 1, and so on.)
Base TypeEach array type is based on some parent type. The information for that parent type is described + recursively by this field.
+ + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Array Property Description for Datatype Version 3
ByteByteByteByte
DimensionalityThis space inserted only to align table nicely
Dimension \#1 Size
.
.
.
Dimension \#n Size

Base Type

+ + + + + + + + + + + + + + + + + + + +
Fields: Array Property Description for Datatype Version 3
Field NameDescription
DimensionalityThis value is the number of dimensions that the array has.
Dimension \#n SizeThis value is the size of the dimension of the array as stored in the file. The first dimension + stored in the list of dimensions is the slowest changing dimension and the last dimension stored + is the fastest changing dimension.
Base TypeEach array type is based on some parent type. The information for that parent type is described + recursively by this field.
+ +\anchor FMT4ClassComplex

Class specific information for the Complex class (Class 11):

+ + + + + + + + + + + + + + + + + + +
Bits: Complex Bit Field Description
BitsMeaning
0Homogeneous. If zero, each part of the complex number + datatype is a different floating point datatype (heterogeneous). + Otherwise, each part of the complex number datatype is the same + floating point datatype (homogeneous). Currently, only homogeneous + complex number datatypes are supported.
1,2Complex number form. This two-bit value contains the type of complex number datatype + described. The values defined are: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0Rectangular
1Polar
2Exponential
3Reserved
+ Currently, only rectangular complex number datatypes are supported.
3-23Reserved (zero).
+ + + + + + + + + + + + +
Layout: Complex Property Description
ByteByteByteByte

Parent Type Message

+ + + + + + + + + + + +
Fields: Complex Property Description
Field NameDescription
Parent Type MessageEach complex number type is based on a parent floating point type. This field contains the + datatype message describing that parent type.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_ofvmessage IV.A.2.e. Data Storage - Fill Value (Old) Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Fill Value (old)
Header Message Type: 0x0004
Length: Varies
Status: Optional; may not be repeated.
Description:The fill value message stores a single data value which is returned to the application when an + uninitialized data element is read from a dataset. The fill value is interpreted with the same + datatype as the dataset. If no fill value message is present then a fill value of all zero bytes + is assumed.
+ This fill value message is deprecated in favor of the “new” fill value message (Message + Type 0x0005) and is only written to the file for forward compatibility with versions of the HDF5 + Library before the 1.6.0 version. Additionally, it only appears for datasets with a user-defined + fill value (as opposed to the library default fill value or an explicitly set “undefined” + fill value).
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + +
Layout: Fill Value Message (Old)
bytebytebytebyte
Size (4 bytes)

Fill Value (optional, variable size)

+
+ + + + + + + + + + + + + + +
Fields: Fill Value Message (Old)
Field NameDescription
SizeThis is the size of the Fill Value field in bytes.
Fill ValueThe fill value. The bytes of the fill value are interpreted using the same datatype as for the dataset.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_fvmessage IV.A.2.f. The Data Storage - Fill Value Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Fill Value
Header Message Type: 0x0005
Length: Varies
Status: Required for dataset objects; may not be repeated.
Description:The fill value message stores a single data value which is returned to the application when an + uninitialized data element is read from a dataset. The fill value is interpreted with the same + datatype as the dataset.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + +
Layout: Fill Value Message - Versions 1 & 2
bytebytebytebyte
VersionSpace Allocation TimeFill Value Write TimeFill Value Defined
Size (optional)

Fill Value (optional, variable size)

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Fill Value Message - Versions 1 & 2
Field NameDescription
VersionThe version number information is used for changes in the format of the fill value message and + is described here: + + + + + + + + + + + + + + + + + + + + + +
VersionDescription
0Never used
1Initial version of this message.
2In this version, the Size and Fill Value fields are only present if the Fill Value Defined + field is set to 1.
3This version packs the other fields in the message more efficiently than version 2.
+
Space Allocation TimeWhen the storage space for the dataset’s raw data will be allocated. The allowed values are: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0Not used
1Early allocation. Storage space for the entire dataset should be allocated in the file + when the dataset is created.
2Late allocation. Storage space for the entire dataset should not be allocated until the + dataset is written to.
3Incremental allocation. Storage space for the dataset should not be allocated until the + portion of the dataset is written to. This is currently used in conjunction with chunked + data storage for datasets.
Fill Value Write TimeAt the time that storage space for the dataset’s raw data is allocated, this value indicates + whether the fill value should be written to the raw data storage elements. The allowed values are: + + + + + + + + + + + + + + + + + +
ValueDescription
0On allocation. The fill value is always written to the raw data storage when the storage + space is allocated.
1Never. The fill value should never be written to the raw data storage.
2Fill value written if set by user. The fill value will be written to the raw data storage + when the storage space is allocated only if the user explicitly set the fill value. If the + fill value is the library default or is undefined, it will not be written to the raw data storage.
Fill Value DefinedThis value indicates if a fill value is defined for this dataset. If this value is 0, the fill + value is undefined. If this value is 1, a fill value is defined for this dataset. For version 2 + or later of the fill value message, this value controls the presence of the Size and Fill field.
SizeThis is the size of the Fill Value field in bytes. This field is not present if the Version + field is greater than 1 and the Fill Value Defined field is set to 0.
Fill ValueThe fill value. The bytes of the fill value are interpreted using the same datatype as for the + dataset. This field is not present if the Version field is greater than 1 and the Fill Value + Defined field is set to 0.
+ + + + + + + + + + + + + + + + + + + + + +
Layout: Fill Value Message - Versions 3
bytebytebytebyte
VersionFlagsFill Value Write TimeThis space inserted only to align table nicely
Size (optional)

Fill Value (optional, variable size)

+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Fill Value Message - Versions 3
Field NameDescription
VersionThe version number information is used for changes in the format of the fill value message and + is described here: + + + + + + + + + + + + + + + + + + + + + +
VersionDescription
0Never used
1Initial version of this message.
2In this version, the Size and Fill Value fields are only present if the Fill Value Defined + field is set to 1.
3This version packs the other fields in the message more efficiently than version 2.
+
FlagsWhen the storage space for the dataset’s raw data will be allocated. The allowed values are: + + + + + + + + + + + + + + + + + + + + + + + + + +
BitsDescription
0-1Space Allocation Time, with the same values as versions 1 and 2 of the message.
2-3Fill Value Write Time, with the same values as versions 1 and 2 of the message.
4Fill Value Undefined, indicating that the fill value has been marked as “undefined” + for this dataset. Bits 4 and 5 cannot both be set.
5Fill Value Defined, with the same values as versions 1 and 2 of the message. Bits 4 and 5 + cannot both be set.
6-7Reserved (zero).
SizeThis is the size of the Fill Value field in bytes. This field is not present if the Version + field is greater than 1 and the Fill Value Defined flag is set to 0.
Fill ValueThe fill value. The bytes of the fill value are interpreted using the same datatype as for the + dataset. This field is not present if the Version field is greater than 1 and the Fill Value + Defined flag is set to 0.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_link IV.A.2.g. The Link Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Link
Header Message Type: 0x0006
Length: Varies
Status: Optional; may be repeated.
Description:This message encodes the information for a link in a group’s object header, when the group is + storing its links “compactly”, or in the group’s fractal heap, when the group is + storing its links “densely”.
+ A group is storing its links compactly when the fractal heap address in the + @ref subsubsec_fmt4_dataobject_hdr_msg_linkinfo is set to the + @ref FMT4UndefinedAddress "undefined address" value.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Link Message
bytebytebytebyte
VersionFlagsLink type (optional)This space inserted only to align table nicely

Creation Order (8 bytes, optional)

Link Name Character Set (optional)Length of Link Name (variable size)This space inserted only to align table nicely
Link Name (variable size)

Link Information (variable size)

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Link Message
Field NameDescription
VersionThe version number for this message. This document describes version 1.
FlagsThis field contains information about the link and controls the presence of other fields below. + + + + + + + + + + + + + + + + + + + + + + + + + +
BitsDescription
0-1Determines the size of the Length of Link Name field. + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0The size of the Length of Link Name field is 1 byte.
1The size of the Length of Link Name field is 2 bytes.
2The size of the Length of Link Name field is 4 bytes.
3The size of the Length of Link Name field is 8 bytes.
2Creation Order Field Present: if set, the Creation Order field is present. If + not set, creation order information is not stored for links in this group.
3Link Type Field Present: if set, the link is not a hard link and the Link Type + field is present. If not set, the link is a hard link.
4Link Name Character Set Field Present: if set, the link name is not represented with the + ASCII character set and the Link Name Character Set field is present. If not set, + the link name is represented with the ASCII character set.
5-7Reserved (zero).
Link typeThis is the link class type and can be one of the following values: + + + + + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0A hard link (should never be stored in the file)
1A soft link.
2-63Reserved for future HDF5 internal use.
64An external link.
65-255Reserved, but available for user-defined link types.
+ This field is present if bit 3 of Flags is set.
Creation OrderThis 64-bit value is an index of the link’s creation time within the group. Values start at + 0 when the group is created an increment by one for each link added to the group. Removing a link + from a group does not change existing links’ creation order field.
+ This field is present if bit 2 of Flags is set.
Link Name Character SetThis is the character set for encoding the link’s name: + + + + + + + + + + + + + +
ValueDescription
0ASCII character set encoding (this should never be stored in the file)
1UTF-8 character set encoding
+ This field is present if bit 4 of Flags is set.
Length of link nameThis is the length of the link’s name. The size of this field depends on bits 0 and 1 of Flags.
Link nameThis is the name of the link, non-NULL terminated.
Link informationThe format of this field depends on the link type.
+ For hard links, the field is formatted as follows: + + + + + +
@ref FMT4SizeOfOffsetsV0 "Size of Offsets" bytes:The address of the object header for the object that the link points to.
+
+ For soft links, the field is formatted as follows: + + + + + + + + + +
Bytes 1-2:Length of soft link value.
Length of soft link value bytes:A non-NULL-terminated string storing the value of the soft link.
+
+ For external links, the field is formatted as follows: + + + + + + + + + +
Bytes 1-2:Length of external link value.
Length of external link value bytes:The first byte contains the version number in the upper 4 bits and flags in the lower 4 bits + for the external link. Both version and flags are defined to be zero in this document. The + remaining bytes consist of two NULL-terminated strings, with no padding between them. The first + string is the name of the HDF5 file containing the object linked to and the second string is the + full path to the object linked to, within the HDF5 file’s group hierarchy.
+
+ For user-defined links, the field is formatted as follows: + + + + + + + + + +
Bytes 1-2:Length of user-defined data.
Length of user-defined link value bytes:The data supplied for the user-defined link type.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_external IV.A.2.h. The Data Storage - External Data Files Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: External Data Files
Header Message Type: 0x0007
Length: Varies
Status: Optional; may not be repeated.
Description:The external data storage message indicates that the data for an object is stored outside the HDF5 + file. The filename of the object is stored as a Universal Resource Location (URL) of the actual + filename containing the data. An external file list record also contains the byte offset of the + start of the data within the file and the amount of space reserved in the file for that data.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + +
Layout: External File List Message
bytebytebytebyte
VersionReserved (zero)
Allocated SlotsUsed Slots

Heap AddressO


Slot Definitions...

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: External File List Message
Field NameDescription
VersionThe version number information is used for changes in the format of External Data Storage Message + and is described here: + + + + + + + + + + + + + +
VersionDescription
0Never used.
1The current version used by the library.
Allocated SlotsThe total number of slots allocated in the message. Its value must be at least as large as the value + contained in the Used Slots field. (The current library simply uses the number of Used Slots for this + message)
Used SlotsThe number of initial slots which contain valid information.
Heap AddressThis is the address of a local name heap which contains the names for the external files. (The local + heap information can be found in @ref subsec_fmt4_infra_localheap in this document). The name at offset + zero in the heap is always the empty string.
Slot DefinitionsThe slot definitions are stored in order according to the array addresses they represent.
+ + + + + + + + + + + + + + + + + + +
Layout: External File List Slot
bytebytebytebyte

Name Offset in Local HeapL


File Offset in External Data FileL


Data Size in External FileL

+\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + +
Fields: External File List Slot
Field NameDescription
Name Offset in Local HeapThe byte offset within the local name heap for the name of the file. File names are stored as a URL + which has a protocol name, a host name, a port number, and a file name: + protocol:port//host/file. If the protocol is omitted + then “file:” is assumed. If the port number is omitted then a default port for that protocol + is used. If both the protocol and the port number are omitted then the colon can also be omitted. If the double + slash and host name are omitted then “localhost” is assumed. The file name is the only mandatory part, + and if the leading slash is missing then it is relative to the application’s current working directory + (the use of relative names is not recommended).
Offset in External Data FileThis is the byte offset to the start of the data in the specified file. For files that contain data for + a single dataset this will usually be zero.
Data Size in External FileThis is the total number of bytes reserved in the specified file for raw data storage. For a file that + contains exactly one complete dataset which is not extendable, the size will usually be the exact size of + the dataset. However, by making the size larger one allows HDF5 to extend the dataset. The size can be set + to a value larger than the entire file since HDF5 will read zeros past the end of the file without failing.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_layout IV.A.2.i. The Data Layout Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Data Storage - Layout
Header Message Type: 0x0008
Length: Varies
Status: Required for datasets; may not be repeated.
Description:The Data Layout message describes how the elements of a multi-dimensional array + are stored in the HDF5 file. Four types of data layout are supported: +
    +
  1. Contiguous: The array is stored in one contiguous area of the file. This layout requires that the + size of the array be constant: data manipulations such as chunking, compression, checksums or encryption + are not permitted. The message stores the total storage size of the array. The offset of an element from + the beginning of the storage area is computed as in a C array.
  2. +
  3. Chunked: The array domain is regularly decomposed into chunks, and each chunk is allocated and stored + separately. This layout supports arbitrary element traversals, compression, encryption, and checksums + (these features are described in other messages). The message stores the size of a chunk instead of the + size of the entire array; the size of the entire array can be calculated by traversing the B-tree that + stores the chunk addresses.
  4. +
  5. Compact: The array is stored in one contiguous block, as part of this object header messagei.
  6. +
  7. Virtual: This is only supported for version 4 of the Data Layout message. The message stores information + that is used to locate the global heap collection containing the Virtual Dataset (VDS) mapping information. + The mapping associates the VDS to the source dataset elements that are stored across a collection + of HDF5 files.
  8. +
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Data Layout Message (Versions 1 and 2)
bytebytebytebyte
VersionDimensionalityLayout ClassReserved (zero)
Reserved (zero)

Data AddressO (optional)

Dimension 0 Size
Dimension 1 Size
...
Dataset Element Size (optional)
Compact Data Size (optional)

Compact Data...(variable size, optional)

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Data Layout Message (Versions 1 and 2)
Field NameDescription
VersionThe version number information is used for changes in the format of the data layout message and + is described here: + + + + + + + + + + + + + + + + + +
VersionDescription
0Never used.
1Used by version 1.4 and before of the library to encode layout information. Data space is + always allocated when the data set is created.
2Used by version 1.6.x of the library to encode layout information. Data space is allocated + only when it is necessary.
DimensionalityAn array has a fixed dimensionality. This field specifies the number of dimension size fields later + in the message. The value stored for chunked storage is 1 greater than the number of dimensions in + the dataset’s dataspace. For example, 2 is stored for a 1 dimensional dataset.
Layout ClassThe layout class specifies the type of storage for the data and how the other fields of the layout + message are to be interpreted. + + + + + + + + + + + + + + + + + +
ValueDescription
0Compact Storage
1Contiguous Storage
2Chunked Storage
Data AddressFor contiguous storage, this is the address of the raw data in the file. For chunked storage this is + the address of the @ref subsubsec_fmt4_infra_btrees_v1 that is used to look up the addresses of the + chunks. This field is not present for compact storage. If the version for this message is greater than 1, + the address may have the @ref FMT4UndefinedAddress "undefined address" value, to indicate that storage + has not yet been allocated for this array.
Dimension \#n SizeFor contiguous and compact storage the dimensions define the entire size of the array while for chunked storage + they define the size of a single chunk. In all cases, they are in units of array elements (not bytes). The + first dimension stored in the list of dimensions is the slowest changing dimension and the last dimension + stored is the fastest changing dimension.
Dataset Element SizeThe size of a dataset element, in bytes. This field is only present for chunked storage.
Compact Data SizeThis field is only present for compact data storage. It contains the size of the raw data for the + dataset array, in bytes.
Compact DataThis field is only present for compact data storage. It contains the raw data for the dataset + array.
+ +Version 3 of this message re-structured the format into specific properties that are required for each layout class. + + + + + + + + + + + + + + + + +
Layout: Data Layout Message (Version 3)
bytebytebytebyte
VersionLayout ClassThis space inserted only to align table nicely

Properties (variable size)

+ + + + + + + + + + + + + + + + + + + +
Fields: Data Layout Message (Version 3)
Field NameDescription
VersionThe version number information is used for changes in the format of layout message and is + described here: + + + + + + + + + +
VersionDescription
3Used by the version 1.6.3 and later of the library to store properties for each layout class.
Layout ClassThe layout class specifies how the other fields of the layout message are to be interpreted. + + + + + + + + + + + + + + + + + +
ValueDescription
0Compact Storage
1Contiguous Storage
2Chunked Storage
PropertiesThis variable-sized field encodes information specific to each layout class and is described below. If + there is no property information specified for a layout class, the size of this field is zero bytes.
+ +\anchor FMT4CompactStorage

Class-specific information for compact layout (Class 0):


+(Note: The dimensionality information is in the Dataspace message) + + + + + + + + + + + + + + + +
Layout: Compact Storage Property Description
bytebytebytebyte
SizeThis space inserted only to align table nicely

Raw Data...(variable size)

+ + + + + + + + + + + + + + + +
Fields: Compact Storage Property Description
Field NameDescription
SizeThis field contains the size of the raw data for the dataset array, in bytes.
Raw DataThis field contains the raw data for the dataset array.
+ +\anchor FMT4ContiguousStorage

Class-specific information for contiguous storage (layout class 1):


+(Note: The dimensionality information is in the Dataspace message) + + + + + + + + + + + + + + + +
Layout: Contiguous Storage Property Description
bytebytebytebyte

AddressO


SizeL

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Contiguous Storage Property Description
Field NameDescription
AddressThis is the address of the raw data in the file. The address may have the + @ref FMT4UndefinedAddress "undefined address" value, to indicate that storage has not + yet been allocated for this array.
SizeThis field contains the size allocated to store the raw data, in bytes.
+ +Class-specific information for chunked storage (layout class 2): + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Chunked Storage Property Description
bytebytebytebyte
DimensionalityThis space inserted only to align table nicely

AddressO

Dimension 0 Size
Dimension 1 Size
...
Dimension \#n Size
Dataset Element Size
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Chunked Storage Property Description
Field NameDescription
Dimensionality>A chunk has a fixed dimensionality. This field specifies the number of dimension size fields + later in the message.
AddressThis is the address of the @ref subsubsec_fmt4_infra_btrees_v1 that is used to look up the + addresses of the chunks that actually store portions of the array data. The address may have the + @ref FMT4UndefinedAddress "undefined address" value, to indicate + that storage has not yet been allocated for this array.
Dimension \#n SizeThese values define the dimension size of a single chunk, in units of array elements (not bytes). + The first dimension stored in the list of dimensions is the slowest changing dimension and the + last dimension stored is the fastest changing dimension.
Dataset Element SizeThe size of a dataset element, in bytes.
+ +\anchor FMT4DataLayoutV4

Version 4 of this message is similar to version 3 but has additional +information for the virtual layout class as well as indexing information for the chunked layout class.

+ + + + + + + + + + + + + + + + +
Layout: Data Layout Message (Version 4)
bytebytebytebyte
VersionLayout ClassThis space inserted only to align table nicely

Properties (variable size)

+ + + + + + + + + + + + + + + + + + + +
Fields: Data Layout Message (Version 4)
Field NameDescription
VersionThe value for this field is 4 and is used by version 1.10.0 and later of the library to + store properties for each layout class and indexing information for the chunked layout.
Layout ClassThe layout class specifies specifies the type of storage for the data and how the other fields + of the layout message are to be interpreted. + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0Compact Storage
1Contiguous Storage
2Chunked Storage
3Virtual Storage
PropertiesThis variable-sized field encodes information specific to each layout class as follows: + + + + + + + + + + + + + + + + + + + + + +
Layout ClassDescription
Compact StorageSee @ref FMT4CompactStorage "Compact Storage Property Description" for the version 3 + Data Layout message.
Contiguous StorageSee @ref FMT4ContiguousStorage "Contiguous Storage Property Description" for the version 3 + Data Layout message.
Chunked StorageSee @ref FMT4ChunkedStorage "Chunked Storage Property Description" below.
Virtual StorageSee @ref FMT4VirtualStorage "Virtual Storage Property Description" below.
+ +\anchor FMT4ChunkedStorage

Class-specific information for chunked storage (layout class 2):

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Chunked Storage Property Description
bytebytebytebyte
FlagsDimensionalityDimension Size Encoded LengthThis space inserted only to align table nicely

Dimension 0 Size (variable size)


Dimension 1 Size (variable size)


...


Dimension \#n Size (variable size)

Chunk Indexing TypeThis space inserted only to align table nicely

Indexing Type Information (variable size)


AddressO

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Chunked Storage Property Description
Field NameDescription
FlagsThis is the chunked layout feature flag: + + + + + + + + + + + + + +
ValueDescription
DONT_FILTER_PARTIAL_BOUND_CHUNKS (bit 0)Do not apply filter to a partial edge chunk.
SINGLE_INDEX_WITH_FILTER (bit 1)A filtered chunk for Single Chunk indexing.
DimensionalityA chunk has fixed dimension. This field specifies the number of Dimension Size fields + later in the message.
Dimension Size Encoded LengthThis is the size in bytes used to encode Dimension Size.
Dimension \#n SizeThese values define the dimension size of a single chunk, in units of array elements (not bytes). + The first dimension stored in the list of dimensions is the slowest changing dimension and the + last dimension stored is the fastest changing dimension.
Chunk Indexing TypeThere are five indexing types used to look up addresses of the chunks. For more information on each + type, see @ref sec_fmt4_appendixc
+ + + + + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
1@ref subsec_fmt4_appendixc_chunk indexing type.
2@ref subsec_fmt4_appendixc_implicit indexing type.
3@ref subsec_fmt4_appendixc_fixedarr indexing type.
4@ref subsec_fmt4_appendixc_extarr indexing type.
5@ref subsec_fmt4_appendixc_appv2btree indexing type.
Indexing Type InformationThis variable-sized field encodes information specific to an indexing type. More information on + what is encoded with each type can be found below this table. +
    +
  • See @ref FMT4IndexInfoSingle "Single Chunk" below.
  • +
  • See @ref FMT4IndexInfoImplicit "Implicit" below.
  • +
  • See @ref FMT4IndexInfoFixed "Fixed Array" below.
  • +
  • See @ref FMT4IndexInfoExtensible "Extensible Array" below.
  • +
  • See @ref FMT4IndexInfoV2Btrees "Version 2 B-tree" below.
  • +
AddressThis is the address specific to an indexing type. The address may be undefined if the chunk or + index storage is not allocated yet.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
Single Chunk indexAddress of the single chunk.
Implicit indexAddress of the array of dataset chunks.
Fixed Array indexAddress of the index.
Extensible Array indexAddress of the index.
Version 2 B-tree indexAddress of the index.
+ +
    +
  1. \anchor FMT4IndexInfoSingle

    Index-specific information for Single Chunk:

    +The following information exists only when the chunk is filtered. In other words, when +DONT_FILTER_PARTIAL_BOUND_CHUNKS (bit 0) is enabled in the field flags. + + + + + + + + + + + + + + +
    Layout: Single Chunk Indexing Information
    bytebytebytebyte

    Size of filtered chunkL

    Filters for chunk
    +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + +
    Fields: Single Chunk Indexing Information
    Field NameDescription
    Size of filtered chunkThis field is the size of a filtered chunk.
    Filters for chunkThis field contains filters for the chunk.
    +
  2. + +
  3. \anchor FMT4IndexInfoImplicit

    Index-specific information for Implicit:

    + + + + + + + + + + + +
    Layout: Implicit Indexing Information
    bytebytebytebyte
    No specific indexing information
    +
  4. + +
  5. \anchor FMT4IndexInfoFixed

    Index-specific information for Fixed Array:

    + + + + + + + + + + + + +
    Layout: Fixed Array Indexing Information
    bytebytebytebyte
    Page BitsThis space inserted only to align table nicely
    + + + + + + + + + + +
    Fields: Fixed Array Indexing Information
    Field NameDescription
    Page BitsThis field contains the number of bits needed to store the maximum number of elements in a + data block page.
    +
  6. + +
  7. \anchor FMT4IndexInfoExtensible

    Index-specific information for Extensible Array:

    + + + + + + + + + + + + + + + + + +
    Layout: Extensible Array Indexing Information
    bytebytebytebyte
    Max BitsIndex ElementsMin PointersMin Elements
    Page BitsThis space inserted only to align table nicely
    + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    Fields: Extensible Array Indexing Information
    Field NameDescription
    Max BitsThis field contains the number of bits needed to store the maximum number of elements + in the array.
    Index ElementsThis field contains the number of elements to store in the index block.
    Min PointersThis field contains the minimum number of data block pointers for a superblock.
    Min ElementsThis field contains the minimum number of elements per data block.
    Page BitsThis field contains the number of bits needed to store the maximum number of elements in + a data block page.
    +
  8. + +
  9. \anchor FMT4IndexInfoV2Btrees

    Index-specific information for Version 2 B-tree:

    + + + + + + + + + + + + + + + + +
    Layout: Version 2 B-tree Indexing Information
    bytebytebytebyte
    Node Size
    Split PercentMerge PercentThis space inserted only to align table nicely
    + + + + + + + + + + + + + + + + + + +
    Fields: Version 2 B-tree Indexing Information
    Field NameDescription
    Node SizeThis field is the size in bytes of a B-tree node.
    Split PercentThis field is the percentage full of a B-tree node at which to split the node.
    Merge PercentThis field is the percentage full of a B-tree node at which to merge the node.
    +
  10. +
+ +\anchor FMT4VirtualStorage

Class-specific information for virtual storage (layout class 3):

+ + + + + + + <>byte + + + + + + + +
Layout: Virtual Storage Property Description
bytebytebyte

AddressO

Index
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Virtual Storage Property Description
Field NameDescription
AddressThis is the address of the global heap collection where the VDS mapping entries are stored. + See @ref subsec_fmt4_infra_globalheapvds
IndexThis is the index of the data object within the global heap collection.
+ + +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_bogus IV.A.2.j. The Bogus Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Bogus
Header Message Type:0x0009
Length: 4 bytes
Status: For testing only; should never be stored in a valid file.
Description:This message is used for testing the HDF5 Library’s response to an “unknown” + message type and should never be encountered in a valid HDF5 file.
Format of Data: See the tables below.
+ + + + + + + + + + + + +
Layout: Bogus Message
bytebytebytebyte
Bogus Value
+ + + + + + + + + + + +
Fields: Bogus Message
Field NameDescription
Bogus ValueThis value should always be: 0xdeadbeef.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_groupinfo IV.A.2.k. The Group Info Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Group Info
Header Message Type: 0x000A
Length: Varies
Status: Optional; may not be repeated.
Description:This message stores information for the constants defining a “new style” group’s + behavior. Constant information will be stored in this message and variable information will be stored + in the @ref subsubsec_fmt4_dataobject_hdr_msg_linkinfo message.
+ Note: the “estimated entry” information below is used when determining the size of the + object header for the group when it is created.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + +
Layout: Group Info Message
bytebytebytebyte
VersionFlagsLink Phase Change: Maximum Compact Value (optional)
Link Phase Change: Minimum Dense Value (optional)Estimated Number of Entries (optional)
Estimated Link Name Length of Entries (optional)This space inserted only to align table nicely
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Group Info Message
Field NameDescription
VersionThe version number for this message. This document describes version 0.
FlagsThis is the group information flag with the following definition: + + + + + + + + + + + + + + + + + +
BitDescription
0If set, link phase change values are stored.
1If set, the estimated entry information is non-default + and is stored.
2-7Reserved
Link Phase Change: Maximum Compact ValueThe is the maximum number of links to store “compactly” (in the group’s object header).
+ This field is present if bit 0 of Flags is set.
Link Phase Change: Minimum Dense ValueThis is the minimum number of links to store “densely” (in the group’s fractal + heap). The fractal heap’s address is located in the @ref subsubsec_fmt4_dataobject_hdr_msg_linkinfo + message.
+ This field is present if bit 0 of Flags is set.
Estimated Number of EntriesThis is the estimated number of entries in groups. If this field is not present, the default value of + 4 will be used for the estimated number of group entries.
+ This field is present if bit 1 of Flags is set.
Estimated Link Name Length of EntriesThis is the estimated length of entry name. If this field is not present, the default value of + 8 will be used for the estimated link name length of group entries.
+ This field is present if bit 1 of Flags is set.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_filter IV.A.2.l. The Data Storage - Filter Pipeline Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Data Storage - Filter Pipeline
Header Message Type: 0x000B
Length: Varies
Status: Optional; may not be repeated.
Description:This message describes the filter pipeline which should be applied to the data stream by providing filter + identification numbers, flags, a name, and client data.
+ This message may be present in the object headers of both dataset and group objects. For datasets, it + specifies the filters to apply to raw data. For groups, it specifies the filters to apply to the + group’s fractal heap. Currently, only datasets using chunked data storage use the filter pipeline + on their raw data.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + +
Layout: Filter Pipeline Message - Version 1
bytebytebytebyte
VersionNumber of FiltersReserved (zero)
Reserved (zero)

Filter Description List (variable size)

+ + + + + + + + + + + + + + + + + + + +
Fields: Filter Pipeline Message - Version 1
Field NameDescription
VersionThe version number for this message. This table describes version 1.
Number of FiltersThe total number of filters described in this message. The maximum possible number of filters in a + message is 32.
Filter Description ListA description of each filter. A filter description appears in the next table.
+ + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Filter Description
bytebytebytebyte
Filter IdentificationName Length
FlagsNumber of Values for Client Data

Name (variable size, optional)


Client Data (variable size, optional)

Padding (variable size, optional)
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Filter Description
Field NameDescription
Filter Identification ValueThis value, often referred to as a filter identifier, is designed to be a unique identifier for + the filter. Values from zero through 32,767 are reserved for filters supported by The HDF Group + in the HDF5 library and for filters requested and supported by third parties. Filters supported + by The HDF Group are documented immediately below. Information on 3rd-party filters can be found + at The HDF Group’s + Registered Filters page.
+ 1
To request a filter identifier, + please contact The HDF Group’s Help Desk at HDF Help Desk. + You will be asked to provide the following information: +
    +
  1. Contact information for the developer requesting the new identifier
  2. +
  3. A short description of the new filter
  4. +
  5. Links to any relevant information, including licensing information
  6. +

+ Values from 32768 to 65535 are reserved for non-distributed uses (for example, internal company usage) + or for application usage when testing a feature. The HDF Group does not track or document the use of + the filters with identifiers from this range.
+ The filters currently in library version 1.8.0 are listed below: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
IdentificationNameDescription
0N/AReserved
1deflateGZIP deflate compression
2shuffleData element shuffling
3fletcher32Fletcher32 checksum
4szipSZIP compression
5nbitN-bit packing
6scaleoffsetScale and offset encoded values
Name LengthEach filter has an optional null-terminated ASCII name and this field holds the length of the name + including the null termination padded with nulls to be a multiple of eight. If the filter has no name + then a value of zero is stored in this field.
FlagsThe flags indicate certain properties for a filter. The bit values defined so far are: + + + + + + + + + + + + + +
ValueDescription
0If set then the filter is an optional filter. During output, if an optional filter fails it will be + silently skipped in the pipeline.
1-15Reserved (zero)
Number of Client Data ValuesEach filter can store integer values to control how the filter operates. The number of entries + in the Client Data array is stored in this field.
NameIf the Name Length field is non-zero then it will contain the size of this field, padded + to a multiple of eight. This field contains a null-terminated, ASCII character string to serve as + a comment/name for the filter.
Client DataThis is an array of four-byte integers which will be passed to the filter function. The Client Data + Number of Values determines the number of elements in the array.
PaddingFour bytes of zeroes are added to the message at this point if the Client Data Number of Values field + contains an odd number.
+\anchor FMT4Footnote1Change 1 If you are reading an earlier version of this document, this +link may have changed. If the link does not work, use the latest version of this document on +The HDF Group’s github website, +\ref SPEC; the link there will always be correct. + + + + + + + + + + + + + + + + + +
Layout: Filter Pipeline Message - Version 2
bytebytebytebyte
VersionNumber of FiltersThis space inserted only to align table nicely

Filter Description List (variable size)

+ + + + + + + + + + + + + + + + + + + +
Fields: Filter Pipeline Message - Version 2
Field NameDescription
VersionThe version number for this message. This table describes version 2.
Number of FiltersThe total number of filters described in this message. The maximum possible number of filters in a + message is 32.
Filter Description ListA description of each filter. A filter description appears in the next table.
+ + + + + + + + + + + + + + + + + + + + + + + +
Layout: Filter Description
bytebytebytebyte
Filter IdentificationName Length (optional)
FlagsNumber Client Data Values

Name (variable size, optional)


Client Data (variable size, optional)

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Filter Description
Field NameDescription
Filter Identification ValueThis value, often referred to as a filter identifier, is designed to be a unique identifier for + the filter. Values from zero through 32,767 are reserved for filters supported by The HDF Group + in the HDF5 library and for filters requested and supported by third parties. Filters supported + by The HDF Group are documented immediately below. Information on 3rd-party filters can be found + at The HDF Group’s + Registered Filters page.
+ 1
To request a filter identifier, + please contact The HDF Group’s Help Desk at HDF Help Desk. + You will be asked to provide the following information: +
    +
  1. Contact information for the developer requesting the new identifier
  2. +
  3. A short description of the new filter
  4. +
  5. Links to any relevant information, including licensing information
  6. +

+ Values from 32768 to 65535 are reserved for non-distributed uses (for example, internal company usage) + or for application usage when testing a feature. The HDF Group does not track or document the use of + the filters with identifiers from this range.
+ The filters currently in library version 1.8.0 are listed below: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
IdentificationNameDescription
0N/AReserved
1deflateGZIP deflate compression
2shuffleData element shuffling
3fletcher32Fletcher32 checksum
4szipSZIP compression
5nbitN-bit packing
6scaleoffsetScale and offset encoded values
Name LengthEach filter has an optional null-terminated ASCII name and this field holds the length of the name + including the null termination padded with nulls to be a multiple of eight. If the filter has no name + then a value of zero is stored in this field.
+ Filters with IDs less than 256 (in other words, filters that are defined in this format documentation) + do not store the Name Length or Name fields.
FlagsThe flags indicate certain properties for a filter. The bit values defined so far are: + + + + + + + + + + + + + +
ValueDescription
0If set then the filter is an optional filter. During output, if an optional filter fails it will be + silently skipped in the pipeline.
1-15Reserved (zero)
Number of Client Data ValuesEach filter can store integer values to control how the filter operates. The number of entries + in the Client Data array is stored in this field.
NameIf the Name Length field is non-zero then it will contain the size of this field, padded + to a multiple of eight. This field contains a null-terminated, ASCII character string to serve as + a comment/name for the filter.
Client DataThis is an array of four-byte integers which will be passed to the filter function. The Client Data + Number of Values determines the number of elements in the array.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_attribute IV.A.2.m. The Attribute Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Attribute
Header Message Type: 0x000C
Length: Varies
Status: Optional; may be repeated.
Description:The Attribute message is used to store objects in the HDF5 file which are used as attributes, + or “metadata” about the current object. An attribute is a small dataset; it has a name, + a datatype, a dataspace, and raw data. Since attributes are stored in the object header, they should + be relatively small (in other words, less than 64KB). They can be associated with any type of object + which has an object header (groups, datasets, or committed (named) datatypes).
+ In 1.8.x versions of the library, attributes can be larger than 64KB. See the + “ @ref subsec_attribute_special ” section of the Attributes chapter in + the @ref UG for more information.
+ Note: Attributes on an object must have unique names: the HDF5 Library currently enforces this by + causing the creation of an attribute with a duplicate name to fail. Attributes on different + objects may have the same name, however.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Attribute Message (Version 1)
bytebytebytebyte
VersionReserved (zero)Name Size
Datatype SizeDataspace Size

Name (variable size)


Datatype (variable size)


Dataspace (variable size)


Data (variable size)

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Attribute Message (Version 1)
Field NameDescription
VersionThe version number information is used for changes in the format of the attribute message and is + described here: + + + + + + + + + + + + + +
VersionDescription
0Never used.
1Used by the library before version 1.6 to encode attribute message. This version does not + support shared datatypes.
Name SizeThe length of the attribute name in bytes including the null terminator. Note that the + Name field below may contain additional padding not represented by this field.
Datatype SizeThe length of the datatype description in the Datatype field below. Note that the + Datatype field may contain additional padding not represented by this field.
Dataspace SizeThe length of the dataspace description in the Dataspace field below. Note that the + Dataspace field may contain additional padding not represented by this field.
NameThe null-terminated attribute name. This field is padded with additional null characters to make it a + multiple of eight bytes.
TypeThe datatype description follows the same format as described for the datatype object header message. + This field is padded with additional zero bytes to make it a multiple of eight bytes.
SpaceThe dataspace description follows the same format as described for the dataspace object header message. + This field is padded with additional zero bytes to make it a multiple of eight bytes.
DataThe raw data for the attribute. The size is determined from the datatype and dataspace descriptions. + This field is not padded with additional bytes.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Attribute Message (Version 2)
bytebytebytebyte
VersionFlagsName Size
Datatype SizeDataspace Size

Name (variable size)


Datatype (variable size)


Dataspace (variable size)


Data (variable size)

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Attribute Message (Version 2)
Field NameDescription
VersionThe version number information is used for changes in the format of the attribute message and is + described here: + + + + + + + + + +
VersionDescription
2Used by the library of version 1.6.x and after to encode attribute messages. This version + supports shared datatypes. The fields of name, datatype, and dataspace are not padded with + additional bytes of zero.
Flags>This bit field contains extra information about interpreting the attribute message: + + + + + + + + + + + + + +
BitDescription
0If set, datatype is shared.
1If set, dataspace is shared.
Name SizeThe length of the attribute name in bytes including the null terminator.
Datatype SizeThe length of the datatype description in the Datatype field below.
Dataspace SizeThe length of the dataspace description in the Dataspace field below.
NameThe null-terminated attribute name. This field is not padded with additional bytes.
DatatypeThe datatype description follows the same format as described for the datatype object + header message.
+ If the Flag field indicates this attribute’s datatype is shared, this field will + contain a “shared message” encoding instead of the datatype encoding.
+ This field is not padded with additional bytes.
DataspaceThe dataspace description follows the same format as described for the dataspace object + header message.
+ If the Flag field indicates this attribute’s dataspace is shared, this field will + contain a “shared message” encoding instead of the dataspace encoding.
+ This field is not padded with additional bytes.
DataThe raw data for the attribute. The size is determined from the datatype and dataspace + descriptions.
+ This field is not padded with additional zero bytes.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Attribute Message (Version 3)
bytebytebytebyte
VersionFlagsName Size
Datatype SizeDataspace Size
Name Character Set EncodingThis space inserted only to align table nicely

Name (variable size)


Datatype (variable size)


Dataspace (variable size)


Data (variable size)

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Attribute Message (Version 3)
Field NameDescription
VersionThe version number information is used for changes in the format of the attribute message and is + described here: + + + + + + + + + +
VersionDescription
2Used by the library of version 1.8.x and after to encode attribute messages. This version + supports attributes with non-ASCII names.
Flags>This bit field contains extra information about interpreting the attribute message: + + + + + + + + + + + + + +
BitDescription
0If set, datatype is shared.
1If set, dataspace is shared.
Name SizeThe length of the attribute name in bytes including the null terminator.
Datatype SizeThe length of the datatype description in the Datatype field below.
Dataspace SizeThe length of the dataspace description in the Dataspace field below.
Name Character Set EncodingThe character set encoding for the attribute’s name: + + + + + + + + + + + + + +
ValueDescription
0ASCII character set encoding
1UTF-8 character set encoding
NameThe null-terminated attribute name. This field is not padded with additional bytes.
DatatypeThe datatype description follows the same format as described for the datatype object + header message.
+ If the Flag field indicates this attribute’s datatype is shared, this field will + contain a “shared message” encoding instead of the datatype encoding.
+ This field is not padded with additional bytes.
DataspaceThe dataspace description follows the same format as described for the dataspace object + header message.
+ If the Flag field indicates this attribute’s dataspace is shared, this field will + contain a “shared message” encoding instead of the dataspace encoding.
+ This field is not padded with additional bytes.
DataThe raw data for the attribute. The size is determined from the datatype and dataspace + descriptions.
+ This field is not padded with additional zero bytes.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_comment IV.A.2.n. The Object Comment Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Object Comment
Header Message Type: 0x000D
Length: Varies
Status: Optional; may not be repeated.
Description:The object comment is designed to be a short description of an object. An object comment is a sequence + of non-zero (\0) ASCII characters with no other formatting included by the library.
Format of Data: See the tables below.
+ + + + + + + + + + + + +
Layout: Object Comment Message
bytebytebytebyte

Comment (variable size)

+
+ + + + + + + + + + + +
Fields: Object Comment Message
Field NameDescription
NameA null terminated ASCII character string.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_omodified IV.A.2.o. The Object Modification Time (Old) Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Object Modification Time (Old)
Header Message Type: 0x000E
Length: Fixed
Status: Optional; may not be repeated.
Description:The object modification date and time is a timestamp which indicates (using ISO-8601 date and + time format) the last modification of an object. The time is updated when any object header + message changes according to the system clock where the change was posted. All fields of this + message should be interpreted as coordinated universal time (UTC).
+ This modification time message is deprecated in favor of the “new” + @ref subsubsec_fmt4_dataobject_hdr_msg_mod message and is no longer written to the file in + versions of the HDF5 Library after the 1.6.0 version.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Modification Time Message (Old)
bytebytebytebyte
Year
MonthDay of Month
HourMinute
SecondReserved
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Modification Time Message (Old)
Field NameDescription
YearThe four-digit year as an ASCII string. For example, 1998.
MonthThe month number as a two digit ASCII string where January is 01 and December is + 12.
Day of MonthThe day number within the month as a two digit ASCII string. The first day of the month is + 01.
HourThe hour of the day as a two digit ASCII string where midnight is 00 and 11:00pm + is 23.
MinuteThe minute of the hour as a two digit ASCII string where the first minute of the hour is + 00 and the last is 59.
SecondThe second of the minute as a two digit ASCII string where the first second of the minute is + 00 and the last is 59.
ReservedThis field is reserved and should always be zero.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_shared IV.A.2.p. The Shared Message Table Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Shared Message Table
Header Message Type: 0x000F
Length: Fixed
Status: Optional; may not be repeated.
Description:This message is used to locate the table of shared object header message (SOHM) indexes. Each + index consists of information to find the shared messages from either the heap or object header. + This message is only found in the superblock extension.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + +
Layout: Shared Message Table Message
bytebytebytebyte
VersionThis space inserted only to align table nicely

Shared Object Header Message Table AddressO

Number of IndicesThis space inserted only to align table nicely
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + +
Fields: Shared Message Table Message
Field NameDescription
VersionThe version number for this message. This document describes version 0.
Shared Object Header Message Table AddressThis field is the address of the master table for shared object header message indexes.
Number of IndicesThis field is the number of indices in the master table.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_continuation IV.A.2.q. The Object Header Continuation Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Object Header Continuation
Header Message Type: 0x0010
Length: Fixed
Status: Optional; may be repeated.
Description:The object header continuation is the location in the file of a block containing more header messages + for the current data object. This can be used when header blocks become too large or are likely to + change over time.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + +
Layout: Object Header Continuation Message
bytebytebytebyte

OffsetO


LengthL

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Object Header Continuation Message
Field NameDescription
OffsetThis value is the address in the file where the header continuation block is located.
LengthThis value is the length in bytes of the header continuation block in the file.
+ +The format of the header continuation block that this message points to depends on the version of the +object header that the message is contained within. + +Continuation blocks for version 1 object headers have no special formatting information; they are +merely a list of object header message info sequences (type, size, flags, reserved bytes and data for +each message sequence). See the description of @ref subsubsec_fmt4_dataobject_hdr_prefix_one. + +Continuation blocks for version 2 object headers do have special formatting information as +described here (see also the description of @ref subsubsec_fmt4_dataobject_hdr_prefix_two): + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 Object Header Continuation Block
bytebytebytebyte
Signature
Header Message Type \#1Size of Header Message Data \#1Header Message \#1 Flags
Header Message \#1 Creation Order (optional)This space inserted only to align table nicely

Header Message Data \#1

.
.
.
Header Message Type \#nSize of Header Message Data \#nHeader Message \#n Flags
Header Message \#n Creation Order (optional)This space inserted only to align table nicely

Header Message Data \#n

Gap (optional, variable size)
Checksum
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 Object Header Continuation Block
Field NameDescription
SignatureThe ASCII character string “OCHK” is used to indicate the + beginning of an object header continuation block. This gives file consistency checking + utilities a better chance of reconstructing a damaged file.
Header Message \#n TypeSame format as version 1 of the object header, described above.
Size of Header Message \#n DataSame format as version 1 of the object header, described above.
Header Message \#n FlagsSame format as version 1 of the object header, described above.
Header Message \#n Creation OrderThis field stores the order that a message of a given type was created in.
+ This field is present if bit 2 of flags is set.
Header Message \#n DataSame format as version 1 of the object header, described above.
GapA gap in an object header chunk is inferred by the end of the messages for the chunk before the + beginning of the chunk’s checksum. Gaps are always smaller than the size of an object header + message prefix (message type + message size + message flags).
+ Gaps are formed when a message (typically an attribute message) in an earlier chunk is deleted + and a message from a later chunk that does not quite fit into the free space is moved into the + earlier chunk.
ChecksumThis is the checksum for the object header chunk.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_stmgroup IV.A.2.r. The Symbol Table Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Symbol Table Message
Header Message Type: 0x0011
Length: Fixed
Status: Required for “old style” groups; may not be repeated.
Description:Each “old style” group has a v1 B-tree and a local heap for storing symbol table entries, + which are located with this message.
Format of data: See the tables below.
+ + + + + + + + + + + + + + + +
Layout: Symbol Table Message
bytebytebytebyte

v1 B-tree AddressO


Local Heap AddressO

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Symbol Table Message
Field NameDescription
v1 B-tree AddressThis value is the address of the v1 B-tree containing the symbol table entries for the group.
Local Heap AddressThis value is the address of the local heap containing the link names for the symbol table + entries for the group.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_mod IV.A.2.s. The Object Modification Time Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Object Modification Time
Header Message Type: 0x0012
Length: Fixed
Status: Optional; may not be repeated.
Description:The object modification time is a timestamp which indicates the time of the last modification of + an object. The time is updated when any object header message changes according to the system clock + where the change was posted.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + +
Layout: Modification Time Message
bytebytebytebyte
VersionReserved (zero)
Seconds After UNIX Epoch
+ + + + + + + + + + + + + + + +
Fields: Modification Time Message
Field NameDescription
VersionThe version number is used for changes in the format of Object Modification Time and is described + here: + + + + + + + + + + + + + +
VersionDescription
0Never used.
1Used by Version 1.6.1 and after of the library to encode time. In this version, the time is + the seconds after Epoch.
Seconds After UNIX EpochA 32-bit unsigned integer value that stores the number of seconds since 0 hours, 0 minutes, + 0 seconds, January 1, 1970, Coordinated Universal Time.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_btreek IV.A.2.t. The B-tree ‘K’ Values Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: B-tree ‘K’ Values
Header Message Type: 0x0013
Length: Fixed
Status: Optional; may not be repeated.
Description:This message retrieves non-default ‘K’ values for internal and leaf nodes of a group + or indexed storage v1 B-trees. This message is only found in the superblock extension.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + +
Layout: B-tree ‘K’ Values Message
bytebytebytebyte
VersionIndexed Storage Internal Node KThis space inserted only to align table nicely
Group Internal Node KGroup Leaf Node K
+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: B-tree ‘K’ Values Message
Field NameDescription
VersionThe version number for this message. This document describes version 0.
Indexed Storage Internal Node KThis is the node ‘K’ value for each internal node of an indexed storage v1 B-tree. + See the description of this field in version 0 and 1 of the superblock as well the section on + v1 B-trees.
Group Internal Node KThis is the node ‘K’ value for each internal node of a group v1 B-tree. See the + description of this field in version 0 and 1 of the superblock as well as the section + on v1 B-trees.
Group Leaf Node KThis is the node ‘K’ value for each leaf node of a group v1 B-tree. See the + description of this field in version 0 and 1 of the superblock as well as the section on v1 + B-trees.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_drvinfo IV.A.2.u. The Driver Info Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Driver Info
Header Message Type: 0x0014
Length: Varies
Status: Optional; may not be repeated.
Description:This message contains information needed by the file driver to reopen a file. This message is + only found in the superblock extension: see the @ref subsec_fmt4_boot_supext section + for more information. For more information on the fields in the driver info message, see the + @ref subsec_fmt4_boot_driver section; those who use the multi and family file drivers will find + this section particularly helpful.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + +
Layout: Driver Info Message
bytebytebytebyte
VersionThis space inserted only to align table nicely

Driver Identification
Driver Information SizeThis space inserted only to align table nicely


Driver Information (variable size)


+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Driver Info Message
Field NameDescription
VersionThe version number for this message. This document describes version 0.
Driver IdentificationThis is an eight-byte ASCII string without null termination which identifies the driver.
Driver Information SizeThe size in bytes of the Driver Information field of this message.
Driver InformationDriver information is stored in a format defined by the file driver.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_attrinfo IV.A.2.v. The Attribute Info Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Attribute Info
Header Message Type: 0x0015
Length: Varies
Status: Optional; may not be repeated.
Description:This message stores information about the attributes on an object, such as the maximum creation + index for the attributes created and the location of the attribute storage when the attributes + are stored “densely”.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + +
Layout: Attribute Info Message
bytebytebytebyte
VersionFlagsMaximum Creation Index (optional)

Fractal Heap AddressO


Attribute Name v2 B-tree AddressO


Attribute Creation Order v2 B-tree AddressO (optional)

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Attribute Info Message
Field NameDescription
VersionThe version number for this message. This document + describes version 0.
FlagsThis is the attribute index information flag with the following definition: + + + + + + + + + + + + + + + + + +
BitDescription
0If set, creation order for attributes is tracked.
1If set, creation order for attributes is indexed.
2-7Reserved
Maximum Creation IndexThe is the maximum creation order index value for the attributes on the object.
+ This field is present if bit 0 of Flags is set.
Fractal Heap AddressThis is the address of the fractal heap to store dense attributes. Each attribute stored in the + fractal heap is described by the @ref subsubsec_fmt4_dataobject_hdr_msg_attribute.
Attribute Name v2 B-tree AddressThis is the address of the version 2 B-tree to index the names of densely stored attributes.
Attribute Creation Order v2 B-tree AddressThis is the address of the version 2 B-tree to index the creation order of densely stored + attributes.
+ This field is present if bit 1 of Flags is set.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_refcount IV.A.2.w. The Object Reference Count Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: Object Reference Count
Header Message Type: 0x0016
Length: Fixed
Status: Optional; may not be repeated.
Description:This message stores the number of hard links (in groups or objects) pointing to an object: in + other words, its reference count.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + +
Layout: Object Reference Count
bytebytebytebyte
VersionThis space inserted only to align table nicely
Reference count
+ + + + + + + + + + + + + + + +
Fields: Object Reference Count
Field NameDescription
VersionThe version number for this message. This document describes version 0.
Reference CountThe unsigned 32-bit integer is the reference count for the object. This message is only present + in “version 2” (or later) object headers, and if not present in those object header versions, + the reference count for the object is assumed to be 1.
+ +\subsubsection subsubsec_fmt4_dataobject_hdr_msg_fsinfo IV.A.2.x. The File Space Info Message + + + + + + + + + + + + + + + + + + + + +
Header Message Name: File Space Info
Header Message Type: 0x0017
Length: Fixed
Status: Optional; may not be repeated.
Description:This message stores the file space management information that the library uses in handling file + space requests for the file. Version 0 of the message is used for release 1.10.0 only. Version 1 + of the message is used for release 1.10.1+. There is no File Space Info message before release + 1.10 as the library does not track file space across multiple file opens.
+ Note that version 0 is deprecated starting in release 1.10.1. That means when the 1.10.1+ library + opens an HDF5 file with a version 0 message, the library will decode and map the message to + version 1. On file close, it will encode the message as a version 1 message.
+ The library uses the following three mechanisms to manage file space in an HDF5 file: +
    +
  • Free-space managers
    + They track free-space sections of various sizes in the file that are not currently + allocated. Each free-space manager corresponds to a file space type. There are two main + groups of file space types: metadata and raw data. Metadata is further divided into five + types: superblock, B-tree, global heap, local heap, and object header. See the description + of @ref subsec_fmt4_infra_freespaceindex as well the description of file space allocation + types in @ref sec_fmt4_appendixb.
  • +
  • Aggregators
    + The library manages two aggregators, one for metadata and one for raw data. Aggregator is + a contiguous block of free-space in the file. The size of each aggregator is tunable via + public routines #H5Pset_meta_block_size and #H5Pset_small_data_block_size respectively.
  • +
  • Virtual file drivers
    + The library's virtual file driver interface dispatches requests for additional space to the + allocation routine of the file driver associated with the file. For example, if the sec2 + file driver is being used, its allocation routine will increase the size of the file to + service the requests.
  • +
+ For release 1.10.0, the library derives the following four file space strategies based on the mechanisms: +
    +
  • #H5F_FILE_SPACE_ALL +
      +
    • Mechanisms used: free-space managers, aggregators, and virtual file drivers
    • +
    • Does not persist free-space across file opens
    • +
    • This strategy is the library default
    • +
    +
  • +
  • #H5F_FILE_SPACE_ALL_PERSIST
  • +
      +
    • Mechanisms used: free-space managers, aggregators, and virtual file drivers
    • +
    • Persist free-space across file opens
    • +
    +
  • #H5F_FILE_SPACE_AGGR_VFD
  • +
      +
    • Mechanisms used: aggregators and virtual file drivers
    • +
    • Does not persist free-space across file opens
    • +
    +
  • #H5F_FILE_SPACE_VFD
  • +
      +
    • Mechanisms used: virtual file drivers
    • +
    • Does not persist free-space across file opens
    • +
    +
+ For release 1.10.1+, the free-space manager mechanism is modified to handle paged aggregation + which aggregates small metadata and raw data allocations into constant-sized well-aligned pages + to allow efficient I/O accesses. With the support of this feature, the library derives the + following four file space strategies: +
    +
  • #H5F_FSPACE_STRATEGY_FSM_AGGR
  • +
      +
    • Mechanisms used: free-space managers, aggregators, and virtual file drivers
    • +
    • This strategy is the library default
    • +
    +
  • #H5F_FSPACE_STRATEGY_PAGE
  • +
      +
    • Mechanisms used: free-space managers with embedded paged aggregation and virtual file drivers
    • +
    +
  • #H5F_FSPACE_STRATEGY_AGGR
  • +
      +
    • Mechanisms used: aggregators and virtual file drivers
    • +
    +
  • #H5F_FSPACE_STRATEGY_NONE
  • +
      +
    • Mechanisms used: virtual file drivers
    • +
    +
+ The default is not persisting free-space across file opens for the above four strategies. User can use + the public routine #H5Pset_file_space_strategy to request persisting free-space.
Format of Data: See the tables below.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: File Space Info
bytebytebytebyte
VersionStrategyThresholdL

Free-space manager addressO for #H5FD_MEM_SUPER


Free-space manager addressO for #H5FD_MEM_BTREE


Free-space manager addressO for #H5FD_MEM_DRAW


Free-space manager addressO for #H5FD_MEM_GHEAP


Free-space manager addressO for #H5FD_MEM_LHEAP


Free-space manager addressO for #H5FD_MEM_OHDR

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + +
Fields: File Space Info
Field NameDescription
VersionThis is the version 0 of this message.
StrategyThis is the file space strategy used to manage file space. There are four types: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
1#H5F_FILE_SPACE_ALL_PERSIST
2#H5F_FILE_SPACE_ALL
3#H5F_FILE_SPACE_AGGR_VFD
4#H5F_FILE_SPACE_VFD
ThresholdThis is the smallest free-space section size that the free-space manager will track.
Free-space manager addressesThese are the six free-space manager addresses for the six file space allocation types: +
    +
  • #H5FD_MEM_SUPER
  • +
  • #H5FD_MEM_BTREE
  • +
  • #H5FD_MEM_DRAW
  • +
  • #H5FD_MEM_GHEAP
  • +
  • #H5FD_MEM_LHEAP
  • +
  • #H5FD_MEM_OHDR
  • +
+ Note that these six fields exist only if the value for the field “Strategy” + is #H5F_FILE_SPACE_ALL_PERSIST.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: File Space Info - Version 1
bytebytebytebyte
VersionStrategyPersisting free-spaceThis space inserted only to align table nicely
Free-space Section ThresholdL
File Space Page Size
Page-end Metadata thresholdThis space inserted only to align table nicely
EOAO

AddressO of small-sized free-space manager for #H5FD_MEM_SUPER


AddressO of small-sized free-space manager for #H5FD_MEM_BTREE


AddressO of small-sized free-space manager for #H5FD_MEM_DRAW


AddressO of small-sized free-space manager for #H5FD_MEM_GHEAP


AddressO of small-sized free-space manager for #H5FD_MEM_LHEAP


AddressO of small-sized free-space manager for #H5FD_MEM_OHDR


AddressO of large-sized free-space manager for #H5FD_MEM_SUPER


AddressO of large-sized free-space manager for #H5FD_MEM_BTREE


AddressO of large-sized free-space manager for #H5FD_MEM_DRAW


AddressO of large-sized free-space manager for #H5FD_MEM_GHEAP


AddressO of large-sized free-space manager for #H5FD_MEM_LHEAP


AddressO of large-sized free-space manager for #H5FD_MEM_OHDR

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: File Space Info - Version 1
Field NameDescription
VersionThis is the version 1 of this message.
StrategyThis is the file space strategy used to manage file space. There are four types: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0#H5F_FSPACE_STRATEGY_FSM_AGGR
1#H5F_FSPACE_STRATEGY_PAGE
2#H5F_FSPACE_STRATEGY_AGGR
3#H5F_FSPACE_STRATEGY_NONE
Persisting free-spaceTrue or false in persisting free-space.
Free-space Section ThresholdThis is the smallest free-space section size that the free-space manager will track.
File space page sizeThis is the file space page size, which is used when the paged aggregation feature is enabled.
Page-end metadata thresholdThis is the smallest free-space section size at the end of a page that the free-space manager will + track. This is used when the paged aggregation feature is enabled.
EOAThe EOA before the allocation of free-space manager header and section info for the self-referential + free-space managers when persisting free-space.
+ Note that self-referential free-space managers are managers that involve file space allocation for + the managers' free-space header and section info.
Addresses of small-sized free-space managersThese are the addresses of the six small-sized free-space manager addresses for the six file space + allocation types: +
    +
  • #H5FD_MEM_SUPER
  • +
  • #H5FD_MEM_BTREE
  • +
  • #H5FD_MEM_DRAW
  • +
  • #H5FD_MEM_GHEAP
  • +
  • #H5FD_MEM_LHEAP
  • +
  • #H5FD_MEM_OHDR
  • +
+ Note that these six fields exist only if the value for the field + “Persisting free-space” is true.
Addresses of large-sized free-space managersThese are the addresses of the six large-sized free-space manager addresses for the six file space + allocation types: +
    +
  • #H5FD_MEM_SUPER
  • +
  • #H5FD_MEM_BTREE
  • +
  • #H5FD_MEM_DRAW
  • +
  • #H5FD_MEM_GHEAP
  • +
  • #H5FD_MEM_LHEAP
  • +
  • #H5FD_MEM_OHDR
  • +
+ Note that these six fields exist only if the value for the field + “Persisting free-space” is true.
+ +\subsection subsec_fmt4_dataobject_storage IV.B. Disk Format: Level 2B - Data Object Data Storage +The data for an object is stored separately from the header information in the file and may not actually +be located in the HDF5 file itself if the header indicates that the data is stored externally. The +information for each record in the object is stored according to the dimensionality of the object +(indicated in the dataspace header message). Multi-dimensional array data is stored in C order; in other +words, the “last” dimension changes fastest. + +Data whose elements are composed of atomic datatypes are stored in IEEE format, unless +they are specifically defined as being stored in a different machine format with the architecture-type +information from the datatype header message. This means that each architecture will need to +[potentially] byte-swap data values into the internal representation for that particular machine. + +Data with a variable-length datatype is stored in the global heap of the HDF5 file. Global heap +identifiers are stored in the data object storage. + +Data whose elements are composed of reference datatypes are stored in several different ways depending +on the particular reference type involved. Object pointers are just stored as the offset of the +object header being pointed to with the size of the pointer being the same number of bytes as offsets +in the file. + +Dataset region references are stored as a heap-ID which points to the following information within the +file-heap: an offset of the object pointed to, number-type information (same format as header message), +dimensionality information (same format as header message), sub-set start and end information (in other +words, a coordinate location for each), and field start and end names (in other words, a [pointer to the] +string indicating the first field included and a [pointer to the] string name for the last field). + +Data of a compound datatype is stored as a contiguous stream of the items in the structure, with each +item formatted according to its datatype.
+Description of datatypes for variable-length, references and compound classes can be found in +@ref subsubsec_fmt4_dataobject_hdr_msg_dtmessage.
+Information about global heap and heap ID can be found in @ref subsec_fmt4_infra_globalheap..
+For reference datatype, see also the encoding description for @ref subsec_fmt4_appendixd_encoderv and +@ref subsec_fmt4_appendixd_encodedp in Appendix D. + + +\section sec_fmt4_appendixa V. Appendix A: Definitions +Definitions of various terms used in this document are included in this section. + + + + + + + + + + + + + +
TermDefinition
Undefined Address\anchor FMT4UndefinedAddress The "undefined address" for a file is a file address with all bits + set: in other words, 0xffff...ff.
Unlimited Size\anchor FMT4UnlimitedDim The "unlimited size" for a size is a value with all bits set: in other words, + 0xffff...ff.
+ +\section sec_fmt4_appendixb VI. Appendix B: File Space Allocation Types +There are six basic types of file memory allocation as follows: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Basic Allocation TypeDescription
#H5FD_MEM_SUPERFile space allocated for Superblock.
#H5FD_MEM_BTREEFile space allocated for B-tree.
#H5FD_MEM_DRAWFile space allocated for raw data.
#H5FD_MEM_GHEAPFile space allocated for Global Heap.
#H5FD_MEM_LHEAPFile space allocated for Local Heap.
#H5FD_MEM_OHDRFile space allocated for Object Header.
+ +There are other file space allocation types that are mapped to the above six basic types +because they are similar in nature. The mapping and the corresponding description are +listed in the following two tables: + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Basic Allocation TypeMapping of Allocation Types to Basic Allocation Types
#H5FD_MEM_SUPERnone
#H5FD_MEM_BTREE#H5FD_MEM_SOHM_INDEX
#H5FD_MEM_DRAW#H5FD_MEM_FHEAP_HUGE_OBJ
#H5FD_MEM_GHEAPnone
#H5FD_MEM_LHEAP#H5FD_MEM_FHEAP_DBLOCK, #H5FD_MEM_FSPACE_SINFO
#H5FD_MEM_OHDR#H5FD_MEM_FHEAP_HDR, #H5FD_MEM_FHEAP_IBLOCK, #H5FD_MEM_FSPACE_HDR, #H5FD_MEM_SOHM_TABLE
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Allocation TypeDescription
#H5FD_MEM_FHEAP_HDRFile space allocated for Fractal Heap Header.
#H5FD_MEM_FHEAP_DBLOCKFile space allocated for Fractal Heap Direct Blocks.
#H5FD_MEM_FHEAP_IBLOCKFile space allocated for Fractal Heap Indirect Blocks.
#H5FD_MEM_FHEAP_HUGE_OBJFile space allocated for huge objects in the fractal heap.
#H5FD_MEM_FSPACE_HDRFile space allocated for Free-space Manager Header.
#H5FD_MEM_FSPACE_SINFOFile space allocated for Free-space Section List of the free-space manager.
#H5FD_MEM_SOHM_TABLEFile space allocated for Shared Object Header Message Table.
#H5FD_MEM_SOHM_INDEXFile space allocated for Shared Message Record List.
+ +\section sec_fmt4_appendixc VII. Appendix C: Types of Indexes for Dataset Chunks +For an HDF5 file without the latest format enabled, the library uses the +@ref subsubsec_fmt4_infra_btrees_v1 to index dataset chunks.
+For an HDF5 file with the latest format enabled, the library uses one of the following five +indexing types depending on a chunked dataset’s dimension specification and the way it +is extended. + +\subsection subsec_fmt4_appendixc_chunk VII.A. The Single Chunk Index +The Single Chunk index can be used when the dataset fulfills the following condition: + + +The dataset has only one chunk, and the address of the single chunk is stored in the version 4 +Data Layout message. See the @ref subsec_fmt4_appendixc_chunk layout and field description tables. + +\subsection subsec_fmt4_appendixc_implicit VII.B. The Implicit Index +The Implicit index can be used when the dataset fulfills the following conditions: + + +Since the dataset’s dimension sizes are known and storage space is to be allocated early, an +array of dataset chunks are allocated based on the maximum dimension sizes when the dataset is created. +The base address of the array is stored in the version 4 Data Layout message. See the +@ref subsec_fmt4_appendixc_chunk layout layout and field description tables.
+When accessing a dataset chunk with a specified offset, the address of the chunk in the array is computed +as below: +\code +base address + (size of a chunk in bytes * chunk index associated with the offset) +\endcode + +\anchor FMT4ChunkIndex A chunk index starts at 0 and increases according to the fastest changing +dimension, then the next fastest, and so on. The chunk index for a dataset chunk offset is computed as below: +
    +
  1. Calculate the scaled offset for each dimension in scaled_offset:
    + scaled_offset = chunk_offset/chunk_dims
  2. +
  3. Calculate the # of chunks for each dimension in nchunks:
    + nchunks = (curr_dims + chunk_dims - 1)/chunk_dims
  4. +
  5. Calculate the down chunks for each dimension in down_chunks:
    + + // n is the # of dimensions + for(i = (int)(n-1), acc = 1; i >= 0; i--) { + down_chunks[i] = acc; + acc *= nchunks[i]; + } +
  6. +
  7. Calculate the chunk index in chunk_index:
    + + // n is the # of dimensions + for(u = 0, chunk_index = 0; u < n; u++) + chunk_index += down_chunks[u] * scaled_offset[u]; +
  8. +
+ +For example, for a 2-dimensional dataset with curr_dims[4,5] and +chunk_dims[3,2], there will be a total of 6 chunks, with 3 chunks in the fastest +changing dimension and 2 chunks in the slowest changing dimension. See the figure below. +The chunk index for the chunk offset [3,4] is computed as below: +
    +
  1. scaled_offset[0] = 1, scaled_offset[1] = 2
  2. +
  3. nchunks[0] = 2, nchunks[1] = 3
  4. +
  5. down_chunks[0] = 3, down_chunks[1] = 1
  6. +
  7. chunk_index = 5
  8. +
+ + + + + + + + +
Figure 3: Implicit index chunk diagram
\image html FileFormatSpecChunkDiagram.jpg
+ +\subsection subsec_fmt4_appendixc_fixedarr VII.C. The Fixed Array Index +The Fixed Array index can be used when the dataset fulfills the following condition: + + +Since the maximum number of chunks is known, an array of in-file-on-disk addresses based on the +maximum number of chunks is allocated when data is written to the dataset. To access a dataset +chunk with a specified offset, the @ref FMT4ChunkIndex "chunk index" associated with the offset +is calculated. The index is mapped into the array to locate the disk address for the chunk.
+The Fixed Array (FA) index structure provides space and speed improvements in locating chunks over +index structures that handle more dynamic data accesses like a +@ref subsec_fmt4_appendixc_appv2btree index.The entry into the Fixed Array is the Fixed Array +header which contains metadata about the entries stored in the array. The header contains a +pointer to a data block which stores the array of entries that describe the dataset chunks. +For greater efficiency, the array will be divided into multiple pages if the number of entries +exceeds a threshold value. The space for the data block and possibly data block pages are allocated +as a single contiguous block of space.
+The content of the data block depends on whether paging is activated or not. When paging is not +used, elements that describe the chunks are stored in the data block. If paging is turned on, +the data block contains a bitmap indicating which pages are initialized. Then subsequent data +block pages will contain the entries that describe the chunks.
+An entry describes either a filtered or non-filtered dataset chunk. The formats for both element +types are described below.
+ + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Fixed Array Header
bytebytebytebyte
Signature
VersionClient IDEntry SizePage Bits

Max Num EntriesL


Data Block AddressO

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Fixed Array Header
Field NameDescription
SignatureThe ASCII character string “FAHD” + is used to indicate the beginning of a Fixed Array header. + This gives file consistency checking utilities a better + chance of reconstructing a damaged file.
VersionThis document describes version 0.
Client IDThe ID for identifying the client of the Fixed Array: + + + + + + + + + + + + + + + + + +
IDDescription
0Non-filtered dataset chunks
1Filtered dataset chunks
2+Reserved
Entry SizeThe size in bytes of an entry in the Fixed Array.
Page BitsThe number of bits needed to store the maximum number of entries in a + @ref FMT4FADataBlockPage "data block page"
Max Num EntriesThe maximum number of entries in the Fixed Array.
Data Block AddressThe address of the data block in the Fixed Array.
ChecksumThe checksum for the header.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Fixed Array Data Block
bytebytebytebyte
Signature
VersionClient IDThis space inserted only to align table nicely

Header AddressO


Page Bitmap (variable size and optional)


Elements (variable size and optional)

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Fixed Array Data Block
Field NameDescription
SignatureThe ASCII character string “FADB” is used to indicate the + beginning of a Fixed Array data block. This gives file consistency checking utilities a + better chance of reconstructing a damaged file.
VersionThis document describes version 0.
Client IDThe ID for identifying the client of the Fixed Array: + + + + + + + + + + + + + + + + + +
IDDescription
0Non-filtered dataset chunks
1Filtered dataset chunks
2+Reserved.
Header AddressThe address of the Fixed Array header. Principally used for file integrity checking.
Page BitmapA bitmap indicating which data block pages are initialized.
+ Exists only if the data block is paged.
ElementsContains the elements stored in the data block and exists only if the data block is not paged. + There are two element types: + + + + + + + + + + + + + +
IDDescription
0@ref FMT4FaNonFilterChunk "Non-filtered dataset chunks"
1@ref FMT4FaFilterChunk "Filtered dataset chunks"
ChecksumThe checksum for the Fixed Array data block.
+ + + + + + + + + + + + + + + +
\anchor FMT4FADataBlockPage Layout: Fixed Array Data Block Page
bytebytebytebyte

Elements (variable size)

Checksum
+ + + + + + + + + + + + + + + +
Fields: Fixed Array Data Block Page
Field NameDescription
ElementsContains the elements stored in the data block page. + There are two element types: + + + + + + + + + + + + + +
IDDescription
0@ref FMT4FaNonFilterChunk "Non-filtered dataset chunks"
1@ref FMT4FaFilterChunk "Filtered dataset chunks"
ChecksumThe checksum for a Fixed Array data block page.
+ +\anchor FMT4FaNonFilterChunk + + + + + + + + + + + +
Layout: Data Block Element for Non-filtered Dataset Chunk
bytebytebytebyte

AddressO

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + +
Fields: Data Block Element for Non-filtered Dataset Chunk
Field NameDescription
AddressThe address of the dataset chunk in the file.
+ +\anchor FMT4FaFilterChunk + + + + + + + + + + + + + + + + + +
Layout: Data Block Element for Filtered Dataset Chunk
bytebytebytebyte

AddressO


Chunk Size (variable size; at most 8 bytes)

Filter Mask
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + +
Fields: Data Block Element for Filtered Dataset Chunk
Field NameDescription
AddressThe address of the dataset chunk in the file.
Chunk SizeThe size of the dataset chunk in bytes.
Filter MaskIndicates the filter to skip for the dataset chunk. Each + filter has an index number in the pipeline; if that filter is + skipped, the bit corresponding to its index is set.
+ +\section subsec_fmt4_appendixc_extarr VII.D. The Extensible Array Index +The Extensible Array index can be used when the dataset fulfills the following condition: + + +The Extensible Array (EA) is a data structure that is used as a chunk index in datasets where the +dataspace has a single unlimited dimension. In other words, one dimension is set to +H5S_UNLIMITED, and the other dimensions are any number of fixed-size dimensions. The +idea behind the extensible array is that a particular data object can be located via a lightweight +indexing structure of fixed depth for a given address space. This indexing structure requires only +a few (2-3) file operations per element lookup and gives good cache performance. Unlike the B-tree +structure, the extensible array is optimized for appends. Where a B-tree would always add at the +rightmost node under these circumstances, either creating a deep tree (version 1) or requiring +expensive rebalances to correct (version 2), the extensible array has already mapped out a pre-balanced +internal structure. This optimized internal structure is instantiated as needed when chunk +records are inserted into the structure.
+An Extensible Array consists of a header, an index block, secondary blocks, data blocks, and +(optional) data block pages. The general scheme is that the index block is used to reference a +secondary block, which is, in turn, used to reference the data block page where the chunk information +is stored. The data blocks will be paged for efficiency when their size passes a threshold value. +These pages are laid out contiguously on the disk after the data block, are initialized as needed, +and are tracked via bitmaps stored in the secondary block. The number of secondary and data +blocks/pages in a chunk index varies as they are allocated as needed and the first few are +(conceptually) stored in parent elements as an optimization. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+ Layout: Extensible Array Header +
bytebytebytebyte
Signature
VersionClient IDElement SizeMax Nelmts Bits
Index Blk ElmtsData Blk Min ElmtsSecondary Blk Min Data PtrsMax Data Blk Page Nelmts Bits

Num Secondary BlksL


Secondary Blk SizeL


Num Data BlksL


Data Blk SizeL


Max Index SetL


Num ElementsL


Index Block AddressO

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. +\li Items marked with an ‘L’ in the above table are of the size specified in + “@ref FMT4SizeOfLengthsV0 "Size of Lengths"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Extensible Array Header
Field NameDescription
SignatureThe ASCII character string “EAHD” is used to indicate the beginning + of an Extensible Array header. This gives file consistency checking utilities a better chance + of reconstructing a damaged file.
VersionThis document describes version 0.
Client IDThe ID for identifying the client of the Fixed Array: + + + + + + + + + + + + + + + + + +
IDDescription
0Non-filtered dataset chunks
1Filtered dataset chunks
2+Reserved.
Element SizeThe size in bytes of an element in the Extensible Array.
Max Nelmts BitsThe number of bits needed to store the maximum number of elements in the Extensible Array.
Index Blk ElmtsThe number of elements to store in the index block.
Data Blk Min ElmtsThe minimum number of elements per data block.
Secondary Blk Min Data PtrsThe minimum number of data block pointers for a secondary block.
Max Dblk Page Nelmts BitsThe number of bits needed to store the maximum number of elements in a data block page.
Num Secondary BlksThe number of secondary blocks created.
Secondary Blk SizeThe size of the secondary blocks created.
Num Data BlksThe number of data blocks created.
Data Blk SizeThe size of the data blocks created.
Max Index SetThe maximum index set.
Num ElmtsThe number of elements realized.
Index Block AddressThe address of the index block.
ChecksumThe checksum for the header.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Extensible Array Index Block
bytebytebytebyte
Signature
VersionClient IDThis space inserted only to align table nicely

Header AddressO


Elements (variable size and optional)


Data Block Addresses (variable size and optional)


Secondary Block Addresses (variable size and optional)

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Extensible Array Index Block
Field NameDescription
SignatureThe ASCII character string “EAIB” is used to indicate the beginning + of an Extensible Array Index Block. This gives file consistency checking utilities a better + chance of reconstructing a damaged file.
VersionThis document describes version 0.
Client IDThe client ID for identifying the user of the Extensible Array: + + + + + + + + + + + + + + + + + +
IDDescription
0Non-filtered dataset chunks
1Filtered dataset chunks
2+Reserved.
Header AddressThe address of the Extensible Array header. Principally used for file integrity checking.
ElementsContains the elements that are stored directly in the index block. An optimization to avoid unnecessary + secondary blocks.
There are two element types: + + + + + + + + + + + + + +
IDDescription
0@ref FMT4EaNonFilterChunk "Non-filtered dataset chunks"
1@ref FMT4EaFilterChunk "Filtered dataset chunks"
Data Block AddressesContains the addresses of the data blocks that are stored directly in the Index Block. An + optimization to avoid unnecessary secondary blocks.
Secondary Block AddressesContains the addresses of the secondary blocks.
ChecksumThe checksum for the Extensible Array Index Block.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Extensible Array Secondary Block
bytebytebytebyte
Signature
VersionClient IDThis space inserted only to align table nicely

Header AddressO


Block Offset (variable size)


Page Bitmap (variable size and optional)


Data Block Addresses (variable size and optional)

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Extensible Array Secondary Block
Field NameDescription
SignatureThe ASCII character string “EASB” is used to indicate the beginning + of an Extensible Array Secondary Block. This gives file consistency checking utilities + a better chance of reconstructing a damaged file.
VersionThis document describes version 0.
Client IDThe ID for identifying the client of the Extensible Array: + + + + + + + + + + + + + + + + + +
IDDescription
0Non-filtered dataset chunks
1Filtered dataset chunks
2+Reserved.
Header AddressThe address of the Extensible Array header. Principally used for file integrity checking.
Block OffsetStores the offset of the block in the array.
Page BitmapA bitmap indicating which data block pages are initialized.
+ Exists only if the data block is paged.
Data Block AddressesContains the addresses of the data blocks referenced by this secondary block.
ChecksumThe checksum for the Extensible Array Secondary Block.
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Extensible Array Data Block
bytebytebytebyte
Signature
VersionClient IDThis space inserted only to align table nicely

Header AddressO


Block Offset (variable size)


Elements (variable size and optional)

Checksum
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Extensible Array Data Block
Field NameDescription
SignatureThe ASCII character string “EADB” is used to indicate the beginning + of an Extensible Array data block. This gives file consistency checking utilities a better + chance of reconstructing a damaged file.
VersionThis document describes version 0.
Client IDThe ID for identifying the client of the Extensible Array: + + + + + + + + + + + + + + + + + +
IDDescription
0Non-filtered dataset chunks
1Filtered dataset chunks
2+Reserved.
Header AddressThe address of the Extensible Array header. Principally used for file integrity checking.
Block OffsetThe offset of the block in the array.
ElementsContains the elements stored in the data block and exists only if the data block is not paged. +
There are two element types: + + + + + + + + + + + + + +
IDDescription
0@ref FMT4EaNonFilterChunk "Non-filtered dataset chunks"
1@ref FMT4EaFilterChunk "Filtered dataset chunks"
ChecksumThe checksum for the Extensible Array data block.
+ + + + + + + + + + + + + + + +
Layout: Extensible Array Data Block Page
bytebytebytebyte

Elements (variable size)

Checksum
+
+ + + + + + + + + + + + + + +
Fields: Extensible Array Data Block Page
Field NameDescription
ElementsContains the elements stored in the data block page.
+ There are two element types: + + + + + + + + + + + + + +
IDDescription
0@ref FMT4EaNonFilterChunk "Non-filtered dataset chunks"
1@ref FMT4EaFilterChunk "Filtered dataset chunks"
ChecksumThe checksum for an Extensible Array data block page.
+ +\anchor FMT4EaNonFilterChunk + + + + + + + + + + + +
Layout: Data Block Element for Non-filtered Dataset Chunk
bytebytebytebyte

AddressO

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + +
Fields: Data Block Element for Non-filtered Dataset Chunk
Field NameDescription
AddressThe address of the dataset chunk in the file.
+ +\anchor FMT4EaFilterChunk + + + + + + + + + + + + + + + + + +
Layout: Data Block Element for Filtered Dataset Chunk
bytebytebytebyte

AddressO


Chunk Size (variable size; at most 8 bytes)

Filter Mask
+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + + + + + +
Fields: Data Block Element for Filtered Dataset Chunk
Field NameDescription
AddressThe address of the dataset chunk in the file.
Chunk SizeThe size of the dataset chunk in bytes.
Filter MaskIndicates the filter to skip for the dataset chunk. + Each filter has an index number in the pipeline; if that + filter is skipped, the bit corresponding to its index is set.
+ +\section subsec_fmt4_appendixc_appv2btree VII.E. The Version 2 B-trees Index +The Version 2 B-trees index can be used when the dataset fulfills the following condition: + + +Version 2 B-trees can be used to index various objects in the library. See +@ref subsubsec_fmt4_infra_btrees_v2 for more information. The B-tree types +@ref FMT4V2BtType10 "10" and @ref FMT4V2BtType11 "11" record layouts are for +indexing dataset chunks. + +\section sec_fmt4_appendixd VIII. Appendix D: Encoding for Dataspace and Reference + +\section subsec_fmt4_appendixd_encode VIII.A. Dataspace Encoding +#H5Sencode is a public routine that encodes a dataspace description into a buffer while +#H5Sdecode is the corresponding routine that decodes the description encoded in the buffer. +See the reference manual description for these two public routines. + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Dataspace Description for #H5Sencode/#H5Sdecode
bytebytebytebyte
Dataspace IDEncode VersionSize of SizeThis space inserted only to align table nicely

Size of Extent



Dataspace Message (variable size)



Dataspace Selection (variable size)

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Dataspace Description for #H5Sencode/#H5Sdecode
Field NameDescription
Dataspace IDThe datspace message ID which is 1.
Encode VersionH5S_ENCODE_VERSION which is 0.
Size of SizeThe number of bytes used to store the size of an object.
Size of ExtentSize of the dataspace message.
Dataspace MessageThe dataspace message information. See @ref subsubsec_fmt4_dataobject_hdr_msg_simple
Dataspace SelectionThe dataspace selection information. See @ref FMT4DataspaceSEL"Dataspace Selection".
+ +\anchor FMT4DataspaceSEL + + + + + + + + + + + + + + +
Layout: Dataspace Selection
bytebytebytebyte
Selection Type

Selection Info (variable size)

+
+ + + + + + + + + + + + + + +
Fields: Dataspace Selection
Field NameDescription
Selection TypeThere are 4 types of selection: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0#H5S_SEL_NONE: Nothing selected
1#H5S_SEL_POINTS: Sequence of points selected
2#H5S_SEL_HYPERSLABS: Hyperslab selected
3#H5S_SEL_ALL: Entire extent selected
Selection InfoThere are 4 types of selection info: + + + + + + + + + + + + + + + + + + + + + +
ValueDescription
0Selection info for #H5S_SEL_NONE Layout and Fields @ref FMT4SelNONE "tables"
1Selection info for #H5S_SEL_POINTS Layout and Fields @ref FMT4SelPOINTS "tables"
2Selection info for #H5S_SEL_HYPERSLABS Layout and Fields @ref FMT4SelHYPER "tables"
3Selection for #H5S_SEL_ALL Layout and Fields @ref FMT4SelALL "tables"
+ +\anchor FMT4SelNONE + + + + + + + + + + + + + + +
Layout: Selection Info for #H5S_SEL_NONE
bytebytebytebyte
Version

Reserved (zero, 8 bytes)

+
+ + + + + + + + + + +
Fields: Selection Info for #H5S_SEL_NONE
Field NameDescription
VersionThe version number for the #H5S_SEL_NONE Selection Info. The value is 1.
+ +\anchor FMT4SelPOINTS + + + + + + + + + + + + + + +
Layout: Selection Info for #H5S_SEL_POINTS
bytebytebytebyte
Version


Points Selection Info (variable size)

+ + + + + + + + + + + + + + + +
Fields: Selection Info for #H5S_SEL_POINTS
Field NameDescription
VersionThe version number for the #H5S_SEL_POINTS Selection Info. The value is either 1 or 2.
Points Selection InfoDepending on version: + + + + + + + + + + + + + +
VersionDescription
1See @ref FMT4SelPOINTSV1 "Version 1 Points Selection Info"
2See @ref FMT4SelPOINTSV2 "Version 2 Points Selection Info"
+ +\anchor FMT4SelPOINTSV1 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 1 Points Selection Info
bytebytebytebyte
Reserved (zero)
Length
Rank
Num Points
Point \#1: coordinate \#1
.
.
.
Point \#1: coordinate \#u
.
.
.
Point \#n: coordinate \#1
.
.
.
Point \#n: coordinate \#u
+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 1 Points Selection Info
Field NameDescription
LengthThe size in bytes from Length to the end of the selection info.
RankThe number of dimensions.
Num PointsThe number of points in the selection.
Point \#n: coordinate \#uThe array of points in the selection. The points selected are \#1 to \#n where n is + Num Points. The list of coordinates for each point are \#1 to \#u where u is + Rank.
+ +\anchor FMT4SelPOINTSV2 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 Points Selection Info
bytebytebytebyte
Encode SizeThis space inserted only to align table nicely
Rank
Num Points(2, 4 or 8 bytes)
Point \#1: coordinate \#1(2, 4 or 8 bytes)
.
.
.
Point \#1: coordinate \#u(2, 4 or 8 bytes)
.
.
.
Point \#n: coordinate \#1 (2, 4 or 8 bytes)
.
.
.
Point \#n: coordinate \#u(2, 4 or 8 bytes)
+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 Points Selection Info
Field NameDescription
Encode SizeThe size for encoding the points selection info which can be 2, 4 or 8 bytes.
RankThe number of dimensions.
Num PointsThe number of points in the selection. The field Encode Size indicates the size + of this field
Point \#n: coordinate \#uThe array of points in the selection. The points selected are \#1 to \#n where n is + Num Points. The list of coordinates for each point are \#1 to \#u where u is + Rank. The field Encode Size indicates the size of this field
+ +\anchor FMT4SelHYPER + + + + + + + + + + + + + + +
Layout: Selection Info for #H5S_SEL_HYPERSLABS
bytebytebytebyte
Version

Hyperslab Selection Info (variable size)

+ + + + + + + + + + + + + + + +
Fields: Selection Info for #H5S_SEL_HYPERSLABS
Field NameDescription
VersionThe version number for the #H5S_SEL_HYPERSLABS selection info. The value is 1, 2 or 3.
Hyperslab Selection InfoDepending on version: + + + + + + + + + + + + + + + + + +
VersionDescription
1See @ref FMT4SelHYPERV1 "Version 1 Hyperslab Selection Info".
2See @ref FMT4SelHYPERV2 "Version 2 Hyperslab Selection Info"
3See @ref FMT4SelHYPERV3 "Version 3 Hyperslab Selection Info"
+ +\anchor FMT4SelHYPERV1 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 1 Hyperslab Selection Info
bytebytebytebyte
Reserved
Length
Rank
Num Blocks
Starting Offset \#1 for Block \#1
.
.
.
Starting Offset \#n for Block \#1
Ending Offset \#1 for Block \#1
.
.
.
Ending Offset \#n for Block \#1
.
.
.
.
.
.
.
.
.
Starting Offset \#1 for Block \#u
.
.
.
Starting Offset \#n for Block \#u
Ending Offset \#1 for Block \#u
.
.
.
Ending Offset \#n for Block \#u
+ + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 1 Hyperslab Selection Info
Field NameDescription
LengthThe size in bytes from the field Rank to the end of the Selection Info.
RankThe number of dimensions in the dataspace.
Num BlocksThe number of blocks in the selection.
Starting Offset \#n for Block \#uThe offset \#n of the starting element in block \#u. \#n is from 1 to Rank. + \#u is from 1 to Num Blocks moving from the fastest changing dimension to + the slowest changing dimension.
Ending Offset \#n for Block \#uThe offset \#n of the ending element in block \#u. \#n is from 1 to Rank. + \#u is from 1 to Num Blocks moving from the fastest changing dimension to + the slowest changing dimension.
+ +\anchor FMT4SelHYPERV2 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 2 Hyperslab Selection Info
bytebytebytebyte
FlagsThis space inserted only to align table nicely
Length
Rank
Start \#1 (8 bytes)
Stride \#1 (8 bytes)
Count \#1 (8 bytes)
Block \#1 (8 bytes)
.
.
.
Start \#n (8 bytes)
Stride \#n (8 bytes)
Count \#n (8 bytes)
Block \#n (8 bytes)
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 2 Hyperslab Selection Info
Field NameDescription
FlagsThis is a bit field with the following definition. Currently, this is always set to 0x1. + + + + + + + + + +
BitDescription
0If set, it is a regular hyperslab, otherwise, irregular.
LengthThe size in bytes from the field Rank to the end of the Selection Info.
RankThe number of dimensions in the dataspace.
Start \#nThe offset of the starting element in the block. \#n is from 1 to Rank.
Stride \#nThe number of elements to move in each dimension. \#n is from 1 to Rank.
Count \#nThe number of blocks to select in each dimension. \#n is from 1 to Rank.
Block \#nThe size (in elements) of each block in each dimension. \#n is from 1 to Rank.
+ +\anchor FMT4SelHYPERV3 + + + + + + + + + + + + + + + + + + + +
Layout: Version 3 Hyperslab Selection Info
bytebytebytebyte
FlagsEncode SizeThis space inserted only to align table nicely
Rank

Regular/Irregular Hyperslab Selection Info (variable size)

+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 3 Hyperslab Selection Info
Field NameDescription
FlagsThis is a bit field with the following definition: + + + + + + + + + +
BitDescription
0If set, it is a regular hyperslab, otherwise, irregular.
Encode SizeThe size for encoding hyperslab selection info, which can 2, 4 or 8 bytes.
RankThe number of dimensions in the dataspace.
Regular/Irregular Hyperslab Selection InfoThis is the selection info for version 3 hyperslab which can be regular or irregular. + If bit 0 of the field Flags is set, see + @ref FMT4SelHYPERV3REG "Version 3 Regular Hyperslab Selection Info" + Otherwise, see @ref FMT4SelHYPERV3IRREG "Version 3 Irregular Hyperslab Selection Info"
+ +\anchor FMT4SelHYPERV3REG + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 3 Regular Hyperslab Selection Info
bytebytebytebyte
Start \#1 (2, 4 or 8 bytes)
Stride \#1 (2, 4 or 8 bytes)
Count \#1 (2, 4 or 8 bytes)
Block \#1 (2, 4 or 8 bytes)
.
.
.
Start \#n (2, 4 or 8 bytes)
Stride \#n (2, 4 or 8 bytes)
Count \#n (2, 4 or 8 bytes)
Block \#n (2, 4 or 8 bytes)
+ + + + + + + + + + + + + + + + + + + + + + + +
Fields: Version 3 Regular Hyperslab Selection Info
Field NameDescription
Start \#nThe offset of the starting element in the block. \#n is from 1 to Rank. + The field Encode Size indicates the size of this field.
Stride \#nThe number of elements to move in each dimension. \#n is from 1 to Rank. + The field Encode Size indicates the size of this field.
Count \#nThe number of blocks to select in each dimension. \#n is from 1 to Rank. + The field Encode Size indicates the size of this field.
Block \#nThe size (in elements) of each block in each dimension. \#n is from 1 to Rank. + The field Encode Size indicates the size of this field.
+ +\anchor FMT4SelHYPERV3IRREG + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Version 3 Irregular Hyperslab Selection Info
bytebytebytebyte
Num Blocks (2, 4 or 8 bytes)
Starting Offset \#1 for Block \#1 (2, 4 or 8 bytes)
.
.
.
Starting Offset \#n for Block \#1 (2, 4 or 8 bytes)
Ending Offset \#1 for Block \#1 (2, 4 or 8 bytes)
.
.
.
Ending Offset \#n for Block \#1 (2, 4 or 8 bytes)
.
.
.
.
.
.
.
.
.
Starting Offset \#1 for Block \#u (2, 4 or 8 bytes)
.
.
.
Starting Offset \#n for Block \#u (2, 4 or 8 bytes)
Ending Offset \#1 for Block \#u (2, 4 or 8 bytes)
.
.
.
Ending Offset \#n for Block \#u (2, 4 or 8 bytes)
+ + + + + + + + + + + + + + + +
Fields: Version 3 Irregular Hyperslab Selection Info
Num BlocksThe number of blocks in the selection. The field Encode Size indicates the size of + this field
Starting Offset \#n for Block \#uThe offset \#n of the starting element in block \#u. \#n is from 1 to Rank. + \#u is from 1 to Num Blocks moving from the fastest changing dimension to the slowest + changing dimension. The field Encode Size indicates the size of this field
Ending Offset \#n for Block \#uThe offset \#n of the ending element in block \#u. \#n is from 1 to Rank. \#u is from + 1 to Num Blocks moving from the fastest changing dimension to the slowest changing + dimension. The field Encode Size indicates the size of this field
+ +\anchor FMT4SelALL + + + + + + + + + + + + + + +
Layout: Selection Info for #H5S_SEL_ALL
bytebytebytebyte
Version

Reserved (zero, 8 bytes)

+
+ + + + + + + + + + +
Fields: Selection Info for #H5S_SEL_ALL
Field NameDescription
VersionThe version number for the #H5S_SEL_ALL Selection Info; the value is 1.
+ +\section subsec_fmt4_appendixd_encoderv VIII.B. Reference Encoding (Revised) +For the following reference type, the Reference Header and Reference Block are stored together +as the dataset's raw data: + + +For the following reference types, the Reference Header plus the @ref FMT4GlobalHeapID "Global Heap ID" +are stored as the dataset's raw data in the file. The global heap ID is used to locate the +Reference Block stored in the global heap: + + + + + + + + + + + + + + + +
Layout: Reference Header
bytebytebytebyte
Reference TypeFlagsThis space inserted only to align table nicely
+ + + + + + + + + + + + + + + +
Fields: Reference Header
Field NameDescription
Reference TypeThere are 3 types of references: + + + + + + + + + + + + + + + + + +
ValueDescription
2#H5R_OBJECT2: Object Reference
3#H5R_DATASET_REGION2: Dataset Region Reference
4#H5R_ATTR: Attribute Reference
FlagsThis field describes the reference: + + + + + + + + + + + + + +
BitDescription
0If set, the reference is to an external file.
1-7Reserved
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Layout: Reference Block
bytebytebytebyte
Token SizeThis space inserted only to align table nicely

Token (variable size)

Length of External File NameThis space inserted only to align table nicely

External File Name (variable size)

Size of Dataspace Selection
Rank of Dataspace Selection

Dataspace Selection Information (variable size)

Length of Attribute Name This space inserted only to align table nicely

Attribute Name (variable size)

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Fields: Reference Block
Field NameDescription
Token sizeThis is the size of the token for the object.
TokenThis is the token for the object.
Length of External File NameThis is the length for the external file name.
+ This field exists if bit 0 of flags is set.
External File NameThis is the name of the external file being referenced.
+ This field exists if bit 0 of flags is set. +
Dataspace Selection InformationSee @ref FMT4DataspaceSEL "Dataspace Selection".
+ This field exists if the Reference Type is #H5R_DATASET_REGION2.
Length of Attribute NameThis is the length of the attribute name.
+ This field exists if the Reference Type is #H5R_ATTR.
Attribute NameThis is the name of the attribute being referenced.
+ This field exists if the Reference Type is #H5R_ATTR.
+ +\section subsec_fmt4_appendixd_encodedp VIII.C. Reference Encoding (Backward Compatibility) +The two references described below are maintained to preserve compatibility with previous versions +of the library.
+For the following reference type, the reference encoding is stored as the dataset's raw data in the file: + + +For the following reference type, the @ref FMT4GlobalHeapID "Global Heap ID" is stored as the dataset's +raw data in the file. The global heap ID is used to locate the reference encoding stored in the global heap: + + + + + + + + + + + + + +
Layout: Reference for #H5R_OBJECT1
bytebytebytebyte

Object AddressO

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + +
Fields: Reference for #H5R_OBJECT1
Field NameDescription
Object AddressAddress of the object being referenced
+
+ + + + + + + + + + + + + + +
Layout: Reference for #H5R_DATASET_REGION1
bytebytebytebyte

Object AddressO


Dataspace Selection Information (variable size)

+\li Items marked with an ‘O’ in the above table are of the size specified in + “@ref FMT4SizeOfOffsetsV0 "Size of Offsets"” field in the superblock. + + + + + + + + + + + + + + + +
Fields: Reference for #H5R_DATASET_REGION1
Field NameDescription
Object AddressThis is the address of the object being referenced.
Dataspace Selection InformationThis is the dataspace selection for the object being referenced. See + @ref FMT4DataspaceSEL "Dataspace Selection".
+ +
+Navigate back: \ref index "Main" / \ref SPEC + +*/ diff --git a/doxygen/dox/Specifications.dox b/doxygen/dox/Specifications.dox index 6ee13e9ca23..7c08eda079a 100644 --- a/doxygen/dox/Specifications.dox +++ b/doxygen/dox/Specifications.dox @@ -16,6 +16,7 @@ Navigate back: \ref index "Main" \li \ref FMT11 \li \ref FMT2 \li \ref FMT3 +\li \ref FMT4 \section sec_spec_other Other diff --git a/release_docs/RELEASE.txt b/release_docs/RELEASE.txt index 219c09e76f1..1a8a439cf13 100644 --- a/release_docs/RELEASE.txt +++ b/release_docs/RELEASE.txt @@ -200,6 +200,14 @@ New Features Library: -------- + - The file format has been updated to 4.0 + + The Virtual Dataset Global Heap Block format has been updated to version 1 + to support shared string storage for source filenames and dataset names, + reducing file size when multiple mappings reference the same sources. This + new format is only used when the HDF5 library version bounds lower bound + is set to 2.0 or later. + - The H5Dread_chunk() signature has changed A new parameter, nalloc, has been added to H5Dread_chunk(). This parameter diff --git a/src/H5Dvirtual.c b/src/H5Dvirtual.c index bdb093ed07b..ba38c59da81 100644 --- a/src/H5Dvirtual.c +++ b/src/H5Dvirtual.c @@ -399,14 +399,16 @@ herr_t H5D__virtual_store_layout(H5F_t *f, H5O_layout_t *layout) { H5O_storage_virtual_t *virt = &layout->storage.u.virt; - uint8_t *heap_block = NULL; /* Block to add to heap */ - size_t *str_size = NULL; /* Array for VDS entry string lengths */ - uint8_t *heap_block_p; /* Pointer into the heap block, while encoding */ - size_t block_size; /* Total size of block needed */ - hsize_t tmp_nentries; /* Temp. variable for # of VDS entries */ - uint32_t chksum; /* Checksum for heap data */ - size_t i; /* Local index variable */ - herr_t ret_value = SUCCEED; /* Return value */ + uint8_t *heap_block = NULL; /* Block to add to heap */ + size_t *str_size = NULL; /* Array for VDS entry string lengths */ + uint8_t *heap_block_p; /* Pointer into the heap block, while encoding */ + size_t block_size; /* Total size of block needed */ + hsize_t tmp_hsize; /* Temp. variable for encoding hsize_t */ + uint32_t chksum; /* Checksum for heap data */ + uint8_t max_version; /* Maximum encoding version allowed by version bounds */ + uint8_t version = H5O_LAYOUT_VDS_GH_ENC_VERS_0; /* Encoding version */ + size_t i; /* Local index variable */ + herr_t ret_value = SUCCEED; /* Return value */ FUNC_ENTER_PACKAGE @@ -421,6 +423,12 @@ H5D__virtual_store_layout(H5F_t *f, H5O_layout_t *layout) /* Set the low/high bounds according to 'f' for the API context */ H5CX_set_libver_bounds(f); + /* Calculate maximum encoding version. Currently there are no features that require a later version, + * so we only upgrade if the lower bound is high enough that we don't worry about backward + * compatibility, and if there is a benefit (will calculate the benefit later). */ + max_version = + H5F_LOW_BOUND(f) >= H5F_LIBVER_V200 ? H5O_LAYOUT_VDS_GH_ENC_VERS_1 : H5O_LAYOUT_VDS_GH_ENC_VERS_0; + /* Allocate array for caching results of strlen */ if (NULL == (str_size = (size_t *)H5MM_malloc(2 * virt->list_nused * sizeof(size_t)))) HGOTO_ERROR(H5E_OHDR, H5E_RESOURCE, FAIL, "unable to allocate string length array"); @@ -459,11 +467,64 @@ H5D__virtual_store_layout(H5F_t *f, H5O_layout_t *layout) if ((select_serial_size = H5S_SELECT_SERIAL_SIZE(ent->source_dset.virtual_select)) < 0) HGOTO_ERROR(H5E_OHDR, H5E_CANTENCODE, FAIL, "unable to check dataspace selection size"); block_size += (size_t)select_serial_size; - } /* end for */ + } /* Checksum */ block_size += 4; + /* + * Calculate_heap_block_size for version 1, if available + */ + if (max_version >= H5O_LAYOUT_VDS_GH_ENC_VERS_1) { + size_t block_size_1; /* Block size if we use version 1 */ + /* Version and number of entries */ + block_size_1 = (size_t)1 + H5F_SIZEOF_SIZE(f); + + /* Calculate size of each entry */ + for (i = 0; i < virt->list_nused; i++) { + H5O_storage_virtual_ent_t *ent = &virt->list[i]; + hssize_t select_serial_size; /* Size of serialized selection */ + + /* Flags */ + block_size_1 += (size_t)1; + + /* Source file name (no encoding necessary for ".") */ + if (strcmp(ent->source_file_name, ".")) { + if (ent->source_file_orig == SIZE_MAX) + block_size_1 += str_size[2 * i]; + else + block_size_1 += MIN(str_size[2 * i], H5F_SIZEOF_SIZE(f)); + } + + /* Source dset name */ + if (ent->source_dset_orig == SIZE_MAX) + block_size_1 += str_size[(2 * i) + 1]; + else + block_size_1 += MIN(str_size[(2 * i) + 1], H5F_SIZEOF_SIZE(f)); + + /* Source selection */ + if ((select_serial_size = H5S_SELECT_SERIAL_SIZE(ent->source_select)) < 0) + HGOTO_ERROR(H5E_OHDR, H5E_CANTENCODE, FAIL, "unable to check dataspace selection size"); + block_size_1 += (size_t)select_serial_size; + + /* Virtual dataset selection */ + if ((select_serial_size = H5S_SELECT_SERIAL_SIZE(ent->source_dset.virtual_select)) < 0) + HGOTO_ERROR(H5E_OHDR, H5E_CANTENCODE, FAIL, "unable to check dataspace selection size"); + block_size_1 += (size_t)select_serial_size; + } + + /* Checksum */ + block_size_1 += 4; + + /* Determine which version to use. Only use version 1 if we save space. In the case of a tie, use + * version 1 since it will allow faster decoding since we know (some of) which strings are shared + * and won't need to do hash table lookups for those. */ + if (block_size_1 <= block_size) { + version = H5O_LAYOUT_VDS_GH_ENC_VERS_1; + block_size = block_size_1; + } + } + /* Allocate heap block */ if (NULL == (heap_block = (uint8_t *)H5MM_malloc(block_size))) HGOTO_ERROR(H5E_OHDR, H5E_RESOURCE, FAIL, "unable to allocate heap block"); @@ -474,22 +535,57 @@ H5D__virtual_store_layout(H5F_t *f, H5O_layout_t *layout) heap_block_p = heap_block; /* Encode heap block encoding version */ - *heap_block_p++ = (uint8_t)H5O_LAYOUT_VDS_GH_ENC_VERS; + *heap_block_p++ = version; /* Number of entries */ - tmp_nentries = (hsize_t)virt->list_nused; - H5F_ENCODE_LENGTH(f, heap_block_p, tmp_nentries); + H5_CHECK_OVERFLOW(virt->list_nused, size_t, hsize_t); + tmp_hsize = (hsize_t)virt->list_nused; + H5F_ENCODE_LENGTH(f, heap_block_p, tmp_hsize); /* Encode each entry */ for (i = 0; i < virt->list_nused; i++) { - H5O_storage_virtual_ent_t *ent = &virt->list[i]; + H5O_storage_virtual_ent_t *ent = &virt->list[i]; + uint8_t flags = 0; + + /* Flags */ + if (version >= H5O_LAYOUT_VDS_GH_ENC_VERS_1) { + if (!strcmp(ent->source_file_name, ".")) + /* Source file in same file as VDS */ + flags |= H5O_LAYOUT_VDS_SOURCE_SAME_FILE; + else if ((ent->source_file_orig != SIZE_MAX) && (str_size[2 * i] >= H5F_SIZEOF_SIZE(f))) + /* Source file name is shared (stored in another entry) */ + flags |= H5O_LAYOUT_VDS_SOURCE_FILE_SHARED; + + if ((ent->source_dset_orig != SIZE_MAX) && (str_size[(2 * i) + 1] >= H5F_SIZEOF_SIZE(f))) + /* Source dataset name is shared (stored in another entry) */ + flags |= H5O_LAYOUT_VDS_SOURCE_DSET_SHARED; + + *heap_block_p++ = flags; + } + /* Source file name */ - H5MM_memcpy((char *)heap_block_p, ent->source_file_name, str_size[2 * i]); - heap_block_p += str_size[2 * i]; + if (!(flags & H5O_LAYOUT_VDS_SOURCE_SAME_FILE)) { + if (flags & H5O_LAYOUT_VDS_SOURCE_FILE_SHARED) { + assert(ent->source_file_orig < i); + tmp_hsize = (hsize_t)ent->source_file_orig; + H5F_ENCODE_LENGTH(f, heap_block_p, tmp_hsize); + } + else { + H5MM_memcpy((char *)heap_block_p, ent->source_file_name, str_size[2 * i]); + heap_block_p += str_size[2 * i]; + } + } /* Source dataset name */ - H5MM_memcpy((char *)heap_block_p, ent->source_dset_name, str_size[(2 * i) + 1]); - heap_block_p += str_size[(2 * i) + 1]; + if (flags & H5O_LAYOUT_VDS_SOURCE_DSET_SHARED) { + assert(ent->source_dset_orig < i); + tmp_hsize = (hsize_t)ent->source_dset_orig; + H5F_ENCODE_LENGTH(f, heap_block_p, tmp_hsize); + } + else { + H5MM_memcpy((char *)heap_block_p, ent->source_dset_name, str_size[(2 * i) + 1]); + heap_block_p += str_size[(2 * i) + 1]; + } /* Source selection */ if (H5S_SELECT_SERIALIZE(ent->source_select, &heap_block_p) < 0) @@ -498,7 +594,7 @@ H5D__virtual_store_layout(H5F_t *f, H5O_layout_t *layout) /* Virtual selection */ if (H5S_SELECT_SERIALIZE(ent->source_dset.virtual_select, &heap_block_p) < 0) HGOTO_ERROR(H5E_OHDR, H5E_CANTCOPY, FAIL, "unable to serialize virtual selection"); - } /* end for */ + } /* Checksum */ chksum = H5_checksum_metadata(heap_block, block_size - (size_t)4, 0); @@ -507,7 +603,7 @@ H5D__virtual_store_layout(H5F_t *f, H5O_layout_t *layout) /* Insert block into global heap */ if (H5HG_insert(f, block_size, heap_block, &(virt->serial_list_hobjid)) < 0) HGOTO_ERROR(H5E_OHDR, H5E_CANTINSERT, FAIL, "unable to insert virtual dataset heap block"); - } /* end if */ + } done: heap_block = (uint8_t *)H5MM_xfree(heap_block); @@ -544,6 +640,12 @@ H5D__virtual_copy_layout(H5O_layout_t *layout) assert(layout); assert(layout->type == H5D_VIRTUAL); + /* Reset hash tables (they are owned by the original list). No need to recreate here - they are only + * needed when adding mappings, and if we add a new mapping the code in H5Pset_virtual() will rebuild + * them). */ + virt->source_file_hash_table = NULL; + virt->source_dset_hash_table = NULL; + /* Save original entry list and top-level property lists and reset in layout * so the originals aren't closed on error */ orig_source_fapl = virt->source_fapl; @@ -573,11 +675,26 @@ H5D__virtual_copy_layout(H5O_layout_t *layout) H5S_copy(orig_list[i].source_dset.virtual_select, false, true))) HGOTO_ERROR(H5E_DATASET, H5E_CANTCOPY, FAIL, "unable to copy virtual selection"); - /* Copy original source names */ - if (NULL == (ent->source_file_name = H5MM_strdup(orig_list[i].source_file_name))) - HGOTO_ERROR(H5E_DATASET, H5E_RESOURCE, FAIL, "unable to duplicate source file name"); - if (NULL == (ent->source_dset_name = H5MM_strdup(orig_list[i].source_dset_name))) - HGOTO_ERROR(H5E_DATASET, H5E_RESOURCE, FAIL, "unable to duplicate source dataset name"); + /* Copy source file name. If the original is shared, share it in the copy too. */ + ent->source_file_orig = orig_list[i].source_file_orig; + if (ent->source_file_orig == SIZE_MAX) { + /* Source file name is not shared, simply strdup to new ent */ + if (NULL == (ent->source_file_name = H5MM_strdup(orig_list[i].source_file_name))) + HGOTO_ERROR(H5E_DATASET, H5E_RESOURCE, FAIL, "unable to duplicate source file name"); + } + else + /* Source file name is shared, link to correct index in new list */ + ent->source_file_name = virt->list[ent->source_file_orig].source_file_name; + + /* Copy source dataset name. If the original is shared, share it in the copy too. */ + ent->source_dset_orig = orig_list[i].source_dset_orig; + if (ent->source_dset_orig == SIZE_MAX) { + if (NULL == (ent->source_dset_name = H5MM_strdup(orig_list[i].source_dset_name))) + HGOTO_ERROR(H5E_DATASET, H5E_RESOURCE, FAIL, "unable to duplicate source dataset name"); + } + else + /* Source dataset name is shared, link to correct index in new list */ + ent->source_dset_name = virt->list[ent->source_dset_orig].source_dset_name; /* Copy source selection */ if (NULL == (ent->source_select = H5S_copy(orig_list[i].source_select, false, true))) @@ -700,6 +817,10 @@ H5D__virtual_reset_layout(H5O_layout_t *layout) assert(layout); assert(layout->type == H5D_VIRTUAL); + /* Clear hash tables */ + HASH_CLEAR(hh_source_file, virt->source_file_hash_table); + HASH_CLEAR(hh_source_dset, virt->source_dset_hash_table); + /* Free the list entries. Note we always attempt to free everything even in * the case of a failure. Because of this, and because we free the list * afterwards, we do not need to zero out the memory in the list. */ @@ -710,8 +831,10 @@ H5D__virtual_reset_layout(H5O_layout_t *layout) HDONE_ERROR(H5E_DATASET, H5E_CANTFREE, FAIL, "unable to reset source dataset"); /* Free original source names */ - (void)H5MM_xfree(ent->source_file_name); - (void)H5MM_xfree(ent->source_dset_name); + if (ent->source_file_orig == SIZE_MAX) + (void)H5MM_xfree(ent->source_file_name); + if (ent->source_dset_orig == SIZE_MAX) + (void)H5MM_xfree(ent->source_dset_name); /* Free sub_dset */ for (j = 0; j < ent->sub_dset_nalloc; j++) diff --git a/src/H5Olayout.c b/src/H5Olayout.c index 2d612235060..95b643fc65e 100644 --- a/src/H5Olayout.c +++ b/src/H5Olayout.c @@ -560,6 +560,9 @@ H5O__layout_decode(H5F_t *f, H5O_t H5_ATTR_UNUSED *open_oh, unsigned H5_ATTR_UNU hsize_t tmp_hsize = 0; uint32_t stored_chksum; uint32_t computed_chksum; + size_t first_same_file = SIZE_MAX; + bool clear_file_hash_table = false; + bool clear_dset_hash_table = false; /* Read heap */ if (NULL == (heap_block = (uint8_t *)H5HG_read( @@ -575,10 +578,12 @@ H5O__layout_decode(H5F_t *f, H5O_t H5_ATTR_UNUSED *open_oh, unsigned H5_ATTR_UNU "ran off end of input buffer while decoding"); heap_vers = (uint8_t)*heap_block_p++; - if ((uint8_t)H5O_LAYOUT_VDS_GH_ENC_VERS != heap_vers) - HGOTO_ERROR(H5E_OHDR, H5E_VERSION, NULL, - "bad version # of encoded VDS heap information, expected %u, got %u", - (unsigned)H5O_LAYOUT_VDS_GH_ENC_VERS, (unsigned)heap_vers); + assert(H5O_LAYOUT_VDS_GH_ENC_VERS_0 == 0); + if (heap_vers > (uint8_t)H5O_LAYOUT_VDS_GH_ENC_VERS_1) + HGOTO_ERROR( + H5E_OHDR, H5E_VERSION, NULL, + "bad version # of encoded VDS heap information, expected %u or lower, got %u", + (unsigned)H5O_LAYOUT_VDS_GH_ENC_VERS_1, (unsigned)heap_vers); /* Number of entries */ if (H5_IS_BUFFER_OVERFLOW(heap_block_p, H5F_sizeof_size(f), heap_block_p_end)) @@ -602,49 +607,199 @@ H5O__layout_decode(H5F_t *f, H5O_t H5_ATTR_UNUSED *open_oh, unsigned H5_ATTR_UNU /* Decode each entry */ for (size_t i = 0; i < mesg->storage.u.virt.list_nused; i++) { + H5O_storage_virtual_ent_t + *tmp_ent; /* Temporary VDS entry pointer, for hash table lookups */ ptrdiff_t avail_buffer_space; + uint8_t flags = 0; avail_buffer_space = heap_block_p_end - heap_block_p + 1; if (avail_buffer_space <= 0) HGOTO_ERROR(H5E_OHDR, H5E_OVERFLOW, NULL, "ran off end of input buffer while decoding"); + /* Flags */ + if (heap_vers >= H5O_LAYOUT_VDS_GH_ENC_VERS_1) { + flags = *heap_block_p++; + + if (flags & ~H5O_LAYOUT_ALL_VDS_FLAGS) + HGOTO_ERROR(H5E_OHDR, H5E_BADVALUE, NULL, "bad flag value for VDS mapping"); + } + + avail_buffer_space = heap_block_p_end - heap_block_p + 1; + /* Source file name */ - tmp_size = strnlen((const char *)heap_block_p, (size_t)avail_buffer_space); - if (tmp_size == (size_t)avail_buffer_space) - HGOTO_ERROR(H5E_OHDR, H5E_OVERFLOW, NULL, + if (flags & H5O_LAYOUT_VDS_SOURCE_SAME_FILE) { + /* Source file in same file as VDS, use "." */ + if (first_same_file == SIZE_MAX) { + /* No previous instance of ".", copy "." to entry and record this instance */ + if (NULL == + (mesg->storage.u.virt.list[i].source_file_name = (char *)H5MM_malloc(2))) + HGOTO_ERROR(H5E_OHDR, H5E_CANTALLOC, NULL, + "memory allocation failed for source file string"); + mesg->storage.u.virt.list[i].source_file_name[0] = '.'; + mesg->storage.u.virt.list[i].source_file_name[1] = '\0'; + mesg->storage.u.virt.list[i].source_file_orig = SIZE_MAX; + first_same_file = i; + + /* Invalidate hash table for use after decoding since it is missing this "." + */ + clear_file_hash_table = true; + } + else { + /* Reference previous instance of "." */ + assert(first_same_file < i); + mesg->storage.u.virt.list[i].source_file_name = + mesg->storage.u.virt.list[first_same_file].source_file_name; + mesg->storage.u.virt.list[i].source_file_orig = first_same_file; + } + } + else { + if (flags & H5O_LAYOUT_VDS_SOURCE_FILE_SHARED) { + if (avail_buffer_space < H5F_SIZEOF_SIZE(f)) + HGOTO_ERROR(H5E_OHDR, H5E_OVERFLOW, NULL, + "ran off end of input buffer while decoding"); + + /* Source file is shared (stored in another entry), decode origin entry number + */ + H5F_DECODE_LENGTH(f, heap_block_p, tmp_hsize); + H5_CHECK_OVERFLOW(tmp_hsize, hsize_t, size_t); + if ((size_t)tmp_hsize >= i) + HGOTO_ERROR( + H5E_OHDR, H5E_BADVALUE, NULL, + "origin source file entry has higher index than current entry"); + mesg->storage.u.virt.list[i].source_file_orig = (size_t)tmp_hsize; + + /* Use source file name from origin entry */ + mesg->storage.u.virt.list[i].source_file_name = + mesg->storage.u.virt.list[tmp_hsize].source_file_name; + } + else { + tmp_size = strnlen((const char *)heap_block_p, (size_t)avail_buffer_space); + if (tmp_size == (size_t)avail_buffer_space) + HGOTO_ERROR( + H5E_OHDR, H5E_OVERFLOW, NULL, "ran off end of input buffer while decoding - unterminated source " "file name string"); - else - tmp_size += 1; /* Add space for NUL terminator */ + else + tmp_size += 1; /* Add space for NUL terminator */ - if (NULL == - (mesg->storage.u.virt.list[i].source_file_name = (char *)H5MM_malloc(tmp_size))) - HGOTO_ERROR(H5E_OHDR, H5E_CANTALLOC, NULL, - "unable to allocate memory for source file name"); - H5MM_memcpy(mesg->storage.u.virt.list[i].source_file_name, heap_block_p, tmp_size); - heap_block_p += tmp_size; + /* Check for source file name in hash table. While this normally shouldn't be + * necessary if it is version 1 or greater and it is at least as long as "size + * of lengths", we should still check since if we don't and it's not shared in + * the file for whatever reason it could cause the library to insert a + * duplicate key if it rebuilds the hash table. */ + tmp_ent = NULL; + if (i > 0) + HASH_FIND(hh_source_file, mesg->storage.u.virt.source_file_hash_table, + heap_block_p, tmp_size - 1, tmp_ent); + if (tmp_ent) { + /* Found source file name in previous mapping, use link to that mapping's + * source file name */ + assert(tmp_ent >= mesg->storage.u.virt.list && + tmp_ent < &mesg->storage.u.virt.list[i]); + mesg->storage.u.virt.list[i].source_file_orig = + (size_t)(tmp_ent - mesg->storage.u.virt.list); + mesg->storage.u.virt.list[i].source_file_name = tmp_ent->source_file_name; + } + else { + /* Did not find source file name, copy it to the entry and add it to the + * hash table */ + if (NULL == (mesg->storage.u.virt.list[i].source_file_name = + (char *)H5MM_malloc(tmp_size))) + HGOTO_ERROR(H5E_OHDR, H5E_CANTALLOC, NULL, + "unable to allocate memory for source file name"); + mesg->storage.u.virt.list[i].source_file_orig = SIZE_MAX; + H5MM_memcpy(mesg->storage.u.virt.list[i].source_file_name, heap_block_p, + tmp_size); + + /* Add to source file name hash table. If we eventually make the library + * resilient to repeated strings not stored shared in memory, possibly by + * permanently disabling the hash table, or marking it as needing a + * careful rebuild, we can avoid this step if the version is 1 or greater + * and the name is at least as long as "size of lengths". See comment + * above about HASH_FIND line. */ + HASH_ADD_KEYPTR(hh_source_file, + mesg->storage.u.virt.source_file_hash_table, + mesg->storage.u.virt.list[i].source_file_name, + tmp_size - 1, &(mesg->storage.u.virt.list[i])); + } + heap_block_p += tmp_size; + } + } avail_buffer_space = heap_block_p_end - heap_block_p + 1; - if (avail_buffer_space <= 0) - HGOTO_ERROR(H5E_OHDR, H5E_OVERFLOW, NULL, - "ran off end of input buffer while decoding"); /* Source dataset name */ - tmp_size = strnlen((const char *)heap_block_p, (size_t)avail_buffer_space); - if (tmp_size == (size_t)avail_buffer_space) - HGOTO_ERROR(H5E_OHDR, H5E_OVERFLOW, NULL, - "ran off end of input buffer while decoding - unterminated source " - "dataset name string"); - else - tmp_size += 1; /* Add space for NUL terminator */ + if (flags & H5O_LAYOUT_VDS_SOURCE_DSET_SHARED) { + if (avail_buffer_space < H5F_SIZEOF_SIZE(f)) + HGOTO_ERROR(H5E_OHDR, H5E_OVERFLOW, NULL, + "ran off end of input buffer while decoding"); - if (NULL == - (mesg->storage.u.virt.list[i].source_dset_name = (char *)H5MM_malloc(tmp_size))) - HGOTO_ERROR(H5E_OHDR, H5E_CANTALLOC, NULL, - "unable to allocate memory for source dataset name"); - H5MM_memcpy(mesg->storage.u.virt.list[i].source_dset_name, heap_block_p, tmp_size); - heap_block_p += tmp_size; + /* Source dataset is shared (stored in another entry), decode origin entry number + */ + H5F_DECODE_LENGTH(f, heap_block_p, tmp_hsize); + H5_CHECK_OVERFLOW(tmp_hsize, hsize_t, size_t); + if ((size_t)tmp_hsize >= i) + HGOTO_ERROR( + H5E_OHDR, H5E_BADVALUE, NULL, + "origin source dataset entry has higher index than current entry"); + mesg->storage.u.virt.list[i].source_dset_orig = (size_t)tmp_hsize; + + /* Use source dataset name from origin entry */ + mesg->storage.u.virt.list[i].source_dset_name = + mesg->storage.u.virt.list[tmp_hsize].source_dset_name; + } + else { + tmp_size = strnlen((const char *)heap_block_p, (size_t)avail_buffer_space); + if (tmp_size == (size_t)avail_buffer_space) + HGOTO_ERROR( + H5E_OHDR, H5E_OVERFLOW, NULL, + "ran off end of input buffer while decoding - unterminated source " + "dataset name string"); + else + tmp_size += 1; /* Add space for NUL terminator */ + + /* Check for source dataset name in hash table. While this normally shouldn't be + * necessary if it is version 1 or greater and it is at least as long as "size of + * lengths", we should still check since if we don't and it's not shared in the + * file for whatever reason it could cause the library to insert a duplicate key + * if it rebuilds the hash table. */ + tmp_ent = NULL; + if (i > 0) + HASH_FIND(hh_source_dset, mesg->storage.u.virt.source_dset_hash_table, + heap_block_p, tmp_size - 1, tmp_ent); + if (tmp_ent) { + /* Found source dataset name in previous mapping, use link to that mapping's + * source dataset name */ + assert(tmp_ent >= mesg->storage.u.virt.list && + tmp_ent < &mesg->storage.u.virt.list[i]); + mesg->storage.u.virt.list[i].source_dset_orig = + (size_t)(tmp_ent - mesg->storage.u.virt.list); + mesg->storage.u.virt.list[i].source_dset_name = tmp_ent->source_dset_name; + } + else { + /* Did not find source dataset name, copy it to the entry and add it to the + * hash table */ + if (NULL == (mesg->storage.u.virt.list[i].source_dset_name = + (char *)H5MM_malloc(tmp_size))) + HGOTO_ERROR(H5E_OHDR, H5E_CANTALLOC, NULL, + "unable to allocate memory for source dataset name"); + mesg->storage.u.virt.list[i].source_dset_orig = SIZE_MAX; + H5MM_memcpy(mesg->storage.u.virt.list[i].source_dset_name, heap_block_p, + tmp_size); + + /* Add to source dataset name hash table. If we eventually make the library + * resilient to repeated strings not stored shared in memory, possibly by + * permanently disabling the hash table, or marking it as needing a careful + * rebuild, we can avoid this step if the version is 1 or greater and the name + * is at least as long as "size of lengths". See comment above about HASH_FIND + * line. */ + HASH_ADD_KEYPTR(hh_source_dset, mesg->storage.u.virt.source_dset_hash_table, + mesg->storage.u.virt.list[i].source_dset_name, tmp_size - 1, + &(mesg->storage.u.virt.list[i])); + } + heap_block_p += tmp_size; + } /* Source selection */ avail_buffer_space = heap_block_p_end - heap_block_p + 1; @@ -755,6 +910,12 @@ H5O__layout_decode(H5F_t *f, H5O_t H5_ATTR_UNUSED *open_oh, unsigned H5_ATTR_UNU /* Verify that the heap block size is correct */ if ((size_t)(heap_block_p - heap_block) != block_size) HGOTO_ERROR(H5E_OHDR, H5E_BADVALUE, NULL, "incorrect heap block size"); + + /* Clear hash tables if requested */ + if (clear_file_hash_table) + HASH_CLEAR(hh_source_file, mesg->storage.u.virt.source_file_hash_table); + if (clear_dset_hash_table) + HASH_CLEAR(hh_source_dset, mesg->storage.u.virt.source_dset_hash_table); } /* end if */ /* Set the layout operations */ diff --git a/src/H5Oprivate.h b/src/H5Oprivate.h index b50f7c315be..3a1fa7c118e 100644 --- a/src/H5Oprivate.h +++ b/src/H5Oprivate.h @@ -402,8 +402,19 @@ typedef struct H5O_efl_t { #define H5O_LAYOUT_ALL_CHUNK_FLAGS \ (H5O_LAYOUT_CHUNK_DONT_FILTER_PARTIAL_BOUND_CHUNKS | H5O_LAYOUT_CHUNK_SINGLE_INDEX_WITH_FILTER) -/* Version number of encoded virtual dataset global heap blocks */ -#define H5O_LAYOUT_VDS_GH_ENC_VERS 0 +/* Initial version of encoded virtual dataset global heap blocks */ +#define H5O_LAYOUT_VDS_GH_ENC_VERS_0 0 + +/* This version added support for shared source file and dataset names, as well as not storing the source file + * name when it is "." */ +#define H5O_LAYOUT_VDS_GH_ENC_VERS_1 1 + +/* Flags for virtual dataset mappings */ +#define H5O_LAYOUT_VDS_SOURCE_FILE_SHARED 0x01 +#define H5O_LAYOUT_VDS_SOURCE_DSET_SHARED 0x02 +#define H5O_LAYOUT_VDS_SOURCE_SAME_FILE 0x04 +#define H5O_LAYOUT_ALL_VDS_FLAGS \ + (H5O_LAYOUT_VDS_SOURCE_FILE_SHARED | H5O_LAYOUT_VDS_SOURCE_DSET_SHARED | H5O_LAYOUT_VDS_SOURCE_SAME_FILE) /* Initial version of the layout information. Used when space is allocated */ #define H5O_LAYOUT_VERSION_1 1 @@ -529,7 +540,9 @@ typedef struct H5O_storage_virtual_ent_t { /* Stored */ H5O_storage_virtual_srcdset_t source_dset; /* Information about the source dataset */ char *source_file_name; /* Original (unparsed) source file name */ + size_t source_file_orig; /* Index of first entry containing source_file_name */ char *source_dset_name; /* Original (unparsed) source dataset name */ + size_t source_dset_orig; /* Index of first entry containing source_dset_name */ struct H5S_t *source_select; /* Selection in the source dataset for mapping */ /* Not stored */ @@ -559,6 +572,8 @@ typedef struct H5O_storage_virtual_ent_t { unlim_extent_virtual */ H5O_virtual_space_status_t source_space_status; /* Extent patching status of source_select */ H5O_virtual_space_status_t virtual_space_status; /* Extent patching status of virtual_select */ + UT_hash_handle hh_source_file; /* Hash handle for this entry in the source file name hash table */ + UT_hash_handle hh_source_dset; /* Hash handle for this entry in the source dataset name hash table */ } H5O_storage_virtual_ent_t; typedef struct H5O_storage_virtual_t { @@ -580,6 +595,12 @@ typedef struct H5O_storage_virtual_t { hid_t source_fapl; /* FAPL to use to open source files */ hid_t source_dapl; /* DAPL to use to open source datasets */ bool init; /* Whether all information has been completely initialized */ + H5O_storage_virtual_ent_t + *source_file_hash_table; /* Hash table of virtual entries sorted by source file name. Only the first + occurrence of each source file name is stored. */ + H5O_storage_virtual_ent_t + *source_dset_hash_table; /* Hash table of virtual entries sorted by source dataset name. Only the + first occurrence of each source dataset name is stored. */ } H5O_storage_virtual_t; typedef struct H5O_storage_t { diff --git a/src/H5Pdcpl.c b/src/H5Pdcpl.c index 483df29c803..cca1c2d11a7 100644 --- a/src/H5Pdcpl.c +++ b/src/H5Pdcpl.c @@ -89,7 +89,7 @@ { \ {HADDR_UNDEF, 0}, 0, NULL, 0, {0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, \ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}, \ - H5D_VDS_ERROR, HSIZE_UNDEF, -1, -1, false \ + H5D_VDS_ERROR, HSIZE_UNDEF, -1, -1, false, NULL, NULL \ } #define H5D_DEF_STORAGE_COMPACT \ { \ @@ -2023,12 +2023,14 @@ herr_t H5Pset_virtual(hid_t dcpl_id, hid_t vspace_id, const char *src_file_name, const char *src_dset_name, hid_t src_space_id) { - H5P_genplist_t *plist = NULL; /* Property list pointer */ - H5O_layout_t virtual_layout; /* Layout information for setting virtual info */ - H5S_t *vspace; /* Virtual dataset space selection */ - H5S_t *src_space; /* Source dataset space selection */ - H5O_storage_virtual_ent_t *old_list = NULL; /* List pointer previously on property list */ - H5O_storage_virtual_ent_t *ent = NULL; /* Convenience pointer to new VDS entry */ + H5P_genplist_t *plist = NULL; /* Property list pointer */ + H5O_layout_t virtual_layout; /* Layout information for setting virtual info */ + H5S_t *vspace; /* Virtual dataset space selection */ + H5S_t *src_space; /* Source dataset space selection */ + H5O_storage_virtual_ent_t *old_list = NULL; /* List pointer previously on property list */ + H5O_storage_virtual_ent_t *ent = NULL; /* Convenience pointer to new VDS entry */ + H5O_storage_virtual_ent_t *tmp_ent; /* Temporary VDS entry pointer, for hash table lookups */ + size_t tmp_len; /* Temporary variable holding a string length */ bool retrieved_layout = false; /* Whether the layout has been retrieved */ bool free_list = false; /* Whether to free the list of virtual entries */ herr_t ret_value = SUCCEED; /* Return value */ @@ -2078,25 +2080,95 @@ H5Pset_virtual(hid_t dcpl_id, hid_t vspace_id, const char *src_file_name, const /* Expand list if necessary */ if (virtual_layout.storage.u.virt.list_nused == virtual_layout.storage.u.virt.list_nalloc) { H5O_storage_virtual_ent_t *x; /* Pointer to the new list */ - size_t new_alloc = MAX(H5D_VIRTUAL_DEF_LIST_SIZE, virtual_layout.storage.u.virt.list_nalloc * 2); + size_t new_alloc = MAX(H5D_VIRTUAL_DEF_LIST_SIZE, virtual_layout.storage.u.virt.list_nalloc * 2); + ptrdiff_t buf_diff; /* Expand size of entry list */ if (NULL == (x = (H5O_storage_virtual_ent_t *)H5MM_realloc( virtual_layout.storage.u.virt.list, new_alloc * sizeof(H5O_storage_virtual_ent_t)))) HGOTO_ERROR(H5E_PLIST, H5E_RESOURCE, FAIL, "can't reallocate virtual dataset mapping list"); + buf_diff = (char *)x - (char *)virtual_layout.storage.u.virt.list; virtual_layout.storage.u.virt.list = x; virtual_layout.storage.u.virt.list_nalloc = new_alloc; + + /* Adjust pointers in the hash tables in case realloc moved the buffers, and hence all the elements + * and hash handles in the hash tables */ + HASH_ADJUST_PTRS(hh_source_file, virtual_layout.storage.u.virt.source_file_hash_table, buf_diff); + HASH_ADJUST_PTRS(hh_source_dset, virtual_layout.storage.u.virt.source_dset_hash_table, buf_diff); } /* end if */ - /* Add virtual dataset mapping entry */ + /* Check if we need to (re)build the hash tables */ + assert((virtual_layout.storage.u.virt.list_nused && + virtual_layout.storage.u.virt.source_file_hash_table && + virtual_layout.storage.u.virt.source_dset_hash_table) || + (!virtual_layout.storage.u.virt.source_file_hash_table && + !virtual_layout.storage.u.virt.source_dset_hash_table)); + if (virtual_layout.storage.u.virt.list_nused && !virtual_layout.storage.u.virt.source_file_hash_table) { + for (size_t i = 0; i < virtual_layout.storage.u.virt.list_nused; i++) { + if (virtual_layout.storage.u.virt.list[i].source_file_orig == SIZE_MAX) + HASH_ADD_KEYPTR(hh_source_file, virtual_layout.storage.u.virt.source_file_hash_table, + virtual_layout.storage.u.virt.list[i].source_file_name, + strlen(virtual_layout.storage.u.virt.list[i].source_file_name), + &(virtual_layout.storage.u.virt.list[i])); + if (virtual_layout.storage.u.virt.list[i].source_dset_orig == SIZE_MAX) + HASH_ADD_KEYPTR(hh_source_dset, virtual_layout.storage.u.virt.source_dset_hash_table, + virtual_layout.storage.u.virt.list[i].source_dset_name, + strlen(virtual_layout.storage.u.virt.list[i].source_dset_name), + &(virtual_layout.storage.u.virt.list[i])); + } + } + + /* + * Add virtual dataset mapping entry + */ ent = &virtual_layout.storage.u.virt.list[virtual_layout.storage.u.virt.list_nused]; memset(ent, 0, sizeof(H5O_storage_virtual_ent_t)); /* Clear before starting to set up */ + if (NULL == (ent->source_dset.virtual_select = H5S_copy(vspace, false, true))) HGOTO_ERROR(H5E_PLIST, H5E_CANTCOPY, FAIL, "unable to copy virtual selection"); - if (NULL == (ent->source_file_name = H5MM_xstrdup(src_file_name))) - HGOTO_ERROR(H5E_PLIST, H5E_RESOURCE, FAIL, "can't duplicate source file name"); - if (NULL == (ent->source_dset_name = H5MM_xstrdup(src_dset_name))) - HGOTO_ERROR(H5E_PLIST, H5E_RESOURCE, FAIL, "can't duplicate source file name"); + + /* Check for source file name in hash table */ + tmp_ent = NULL; + tmp_len = strlen(src_file_name); + if (virtual_layout.storage.u.virt.list_nused > 0) + HASH_FIND(hh_source_file, virtual_layout.storage.u.virt.source_file_hash_table, src_file_name, + tmp_len, tmp_ent); + if (tmp_ent) { + /* Found source file name in previous mapping, use link to that mapping's source file name */ + assert(tmp_ent >= virtual_layout.storage.u.virt.list && tmp_ent < ent); + ent->source_file_orig = (size_t)(tmp_ent - virtual_layout.storage.u.virt.list); + ent->source_file_name = tmp_ent->source_file_name; + } + else { + /* Did not find source file name, copy it to the entry and add it to the hash table */ + if (NULL == (ent->source_file_name = H5MM_xstrdup(src_file_name))) + HGOTO_ERROR(H5E_PLIST, H5E_RESOURCE, FAIL, "can't duplicate source file name"); + ent->source_file_orig = SIZE_MAX; + HASH_ADD_KEYPTR(hh_source_file, virtual_layout.storage.u.virt.source_file_hash_table, + ent->source_file_name, tmp_len, ent); + } + + /* Check for source dataset name in hash table */ + tmp_ent = NULL; + tmp_len = strlen(src_dset_name); + if (virtual_layout.storage.u.virt.list_nused > 0) + HASH_FIND(hh_source_dset, virtual_layout.storage.u.virt.source_dset_hash_table, src_dset_name, + tmp_len, tmp_ent); + if (tmp_ent) { + /* Found source dataset name in previous mapping, use link to that mapping's source dataset name */ + assert(tmp_ent >= virtual_layout.storage.u.virt.list && tmp_ent < ent); + ent->source_dset_orig = (size_t)(tmp_ent - virtual_layout.storage.u.virt.list); + ent->source_dset_name = tmp_ent->source_dset_name; + } + else { + /* Did not find source dataset name, copy it to the entry and add it to the hash table */ + if (NULL == (ent->source_dset_name = H5MM_xstrdup(src_dset_name))) + HGOTO_ERROR(H5E_PLIST, H5E_RESOURCE, FAIL, "can't duplicate source dataset name"); + ent->source_dset_orig = SIZE_MAX; + HASH_ADD_KEYPTR(hh_source_dset, virtual_layout.storage.u.virt.source_dset_hash_table, + ent->source_dset_name, tmp_len, ent); + } + if (NULL == (ent->source_select = H5S_copy(src_space, false, true))) HGOTO_ERROR(H5E_PLIST, H5E_CANTCOPY, FAIL, "unable to copy source selection"); if (H5D_virtual_parse_source_name(ent->source_file_name, &ent->parsed_source_file_name, @@ -2155,8 +2227,14 @@ done: if (ret_value < 0) { /* Free incomplete entry if present */ if (ent) { - ent->source_file_name = (char *)H5MM_xfree(ent->source_file_name); - ent->source_dset_name = (char *)H5MM_xfree(ent->source_dset_name); + if (ent->source_file_orig == SIZE_MAX) + ent->source_file_name = (char *)H5MM_xfree(ent->source_file_name); + else + HASH_DELETE(hh_source_file, virtual_layout.storage.u.virt.source_file_hash_table, ent); + if (ent->source_dset_orig == SIZE_MAX) + ent->source_dset_name = (char *)H5MM_xfree(ent->source_dset_name); + else + HASH_DELETE(hh_source_dset, virtual_layout.storage.u.virt.source_dset_hash_table, ent); if (ent->source_dset.virtual_select && H5S_close(ent->source_dset.virtual_select) < 0) HDONE_ERROR(H5E_DATASET, H5E_CLOSEERROR, FAIL, "unable to release virtual selection"); ent->source_dset.virtual_select = NULL; diff --git a/src/uthash.h b/src/uthash.h index 469448d83f5..2f136ab5b81 100644 --- a/src/uthash.h +++ b/src/uthash.h @@ -1112,6 +1112,37 @@ typedef unsigned char uint8_t; #define HASH_COUNT(head) HASH_CNT(hh, head) #define HASH_CNT(hh, head) ((head != NULL) ? ((head)->hh.tbl->num_items) : 0U) +/* Adjust all element and hash handle pointers by ptr_adj bytes. Does not adjust key pointers. Intended for + * the case where all elements are stored in a flat array and that array is realloced. */ +#define HASH_ADJUST_PTRS(hh, head, ptr_adj) \ + do { \ + ptrdiff_t _ptr_adj = (ptrdiff_t)(ptr_adj); \ + if (head && (_ptr_adj != 0)) { \ + unsigned _tmp_bkt; \ + (head) = (void *)((char *)head + _ptr_adj); \ + (head)->hh.tbl->tail = (UT_hash_handle *)(void *)((char *)(head)->hh.tbl->tail + _ptr_adj); \ + for (_tmp_bkt = 0; _tmp_bkt < (head)->hh.tbl->num_buckets; _tmp_bkt++) \ + if ((head)->hh.tbl->buckets[_tmp_bkt].hh_head) { \ + (head)->hh.tbl->buckets[_tmp_bkt].hh_head = \ + (UT_hash_handle *)(void *)((char *)(head)->hh.tbl->buckets[_tmp_bkt].hh_head + \ + _ptr_adj); \ + for (UT_hash_handle *_tmp_hh = (head)->hh.tbl->buckets[_tmp_bkt].hh_head; \ + _tmp_hh != NULL; _tmp_hh = _tmp_hh->hh_next) { \ + if (_tmp_hh->prev) \ + _tmp_hh->prev = (UT_hash_handle *)(void *)((char *)_tmp_hh->prev + _ptr_adj); \ + if (_tmp_hh->next) \ + _tmp_hh->next = (UT_hash_handle *)(void *)((char *)_tmp_hh->next + _ptr_adj); \ + if (_tmp_hh->hh_prev) \ + _tmp_hh->hh_prev = \ + (UT_hash_handle *)(void *)((char *)_tmp_hh->hh_prev + _ptr_adj); \ + if (_tmp_hh->hh_next) \ + _tmp_hh->hh_next = \ + (UT_hash_handle *)(void *)((char *)_tmp_hh->hh_next + _ptr_adj); \ + } \ + } \ + } \ + } while (0) + typedef struct UT_hash_bucket { struct UT_hash_handle *hh_head; unsigned count; diff --git a/test/dsets.c b/test/dsets.c index 22b02bcd826..3282fa0e946 100644 --- a/test/dsets.c +++ b/test/dsets.c @@ -78,6 +78,7 @@ static const char *FILENAME[] = {"dataset", /* 0 */ "alloc_0sized", /* 26 */ "h5s_block", /* 27 */ "h5s_plist", /* 28 */ + "vds_strings", /* 29 */ NULL}; #define OHMIN_FILENAME_A "ohdr_min_a" @@ -308,6 +309,9 @@ const char *OLD_FILENAME[] = { #define DCPL_LAYOUT_DIM2 10 #define DCPL_LAYOUT_NUM_SRC_DSETS 2 +/* Declarations for test test_vds_shared_strings */ +#define NUM_MAPPINGS_MANY 1000 + /* Local prototypes for filter functions */ static size_t filter_bogus(unsigned int flags, size_t cd_nelmts, const unsigned int *cd_values, size_t nbytes, size_t *buf_size, void **buf); @@ -16147,6 +16151,980 @@ error: return -1; } +/*------------------------------------------------------------------------- + * Function: test_vds_shared_strings + * + * Purpose: Tests VDS (Virtual Dataset) shared strings functionality. + * Verifies that string sharing works as expected and that + * the correct encoding format is used. + * + * Return: Success: 0 + * Failure: -1 + *------------------------------------------------------------------------- + */ +static int +test_vds_shared_strings(hid_t fapl) +{ + char filename[FILENAME_BUF_SIZE]; + hid_t file_id = H5I_INVALID_HID; /* File */ + hid_t dcpl_id = H5I_INVALID_HID; /* Dataset creation property list */ + hid_t src_space_id = H5I_INVALID_HID; /* Source dataspace */ + hid_t virt_space_id = H5I_INVALID_HID; /* Virtual dataspace */ + hid_t dset_id = H5I_INVALID_HID; /* Virtual dataset */ + hsize_t dims[1] = {10}; /* Dataset dimensions */ + H5O_storage_virtual_t *virt_layout = NULL; /* Virtual storage layout */ + H5D_t *dset_int = NULL; /* Internal dataset structure */ + + TESTING("VDS sharing of file/dataset names"); + + /* Set up file name */ + h5_fixname(FILENAME[29], fapl, filename, sizeof(filename)); + + /* Create source and virtual dataspaces */ + if ((src_space_id = H5Screate_simple(1, dims, NULL)) < 0) + TEST_ERROR; + if ((virt_space_id = H5Screate_simple(1, dims, NULL)) < 0) + TEST_ERROR; + + /* + * Test 1: VDS with no sharing + */ + + if ((file_id = H5Fcreate(filename, H5F_ACC_TRUNC, H5P_DEFAULT, fapl)) < 0) + TEST_ERROR; + + if ((dcpl_id = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_layout(dcpl_id, H5D_VIRTUAL) < 0) + TEST_ERROR; + + /* Add virtual mappings with completely different strings */ + if (H5Pset_virtual(dcpl_id, virt_space_id, "file1.h5", "/dataset1", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "file2.h5", "/dataset2", src_space_id) < 0) + TEST_ERROR; + + if ((dset_id = H5Dcreate2(file_id, "vds_no_share", H5T_NATIVE_INT, virt_space_id, H5P_DEFAULT, dcpl_id, + H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_file_name == virt_layout->list[1].source_file_name) { + H5_FAILED(); + puts(" Source file names are erroneously shared"); + goto error; + } + if (virt_layout->list[0].source_dset_name == virt_layout->list[1].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are erroneously shared"); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX || + virt_layout->list[1].source_file_orig != SIZE_MAX) { + H5_FAILED(); + puts(" Source file names are erroneously marked as shared"); + goto error; + } + if (virt_layout->list[0].source_dset_orig != SIZE_MAX || + virt_layout->list[1].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + puts(" Source dataset names are erroneously marked as shared"); + goto error; + } + + /* Re-open verification for test 1 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + if ((file_id = H5Fopen(filename, H5F_ACC_RDONLY, fapl)) < 0) + TEST_ERROR; + if ((dset_id = H5Dopen2(file_id, "vds_no_share", H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_file_name == virt_layout->list[1].source_file_name) { + H5_FAILED(); + puts(" Source file names are erroneously shared after re-open"); + goto error; + } + if (virt_layout->list[0].source_dset_name == virt_layout->list[1].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are erroneously shared after re-open"); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX || + virt_layout->list[1].source_file_orig != SIZE_MAX) { + H5_FAILED(); + puts(" Source file names are erroneously marked as shared after re-open"); + goto error; + } + if (virt_layout->list[0].source_dset_orig != SIZE_MAX || + virt_layout->list[1].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + puts(" Source dataset names are erroneously marked as shared after re-open"); + goto error; + } + + /* Close resources for test 1 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Pclose(dcpl_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + /* + * Test 2: VDS with shared source filenames + */ + + if ((file_id = H5Fcreate(filename, H5F_ACC_TRUNC, H5P_DEFAULT, fapl)) < 0) + TEST_ERROR; + + if ((dcpl_id = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_layout(dcpl_id, H5D_VIRTUAL) < 0) + TEST_ERROR; + + /* Add virtual mappings with repeated source file, different datasets */ + if (H5Pset_virtual(dcpl_id, virt_space_id, "shared_source.h5", "/dataset1", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "shared_source.h5", "/dataset2", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "shared_source.h5", "/dataset3", src_space_id) < 0) + TEST_ERROR; + + if ((dset_id = H5Dcreate2(file_id, "vds_shared_file", H5T_NATIVE_INT, virt_space_id, H5P_DEFAULT, dcpl_id, + H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_file_name != virt_layout->list[1].source_file_name || + virt_layout->list[0].source_file_name != virt_layout->list[2].source_file_name) { + H5_FAILED(); + puts(" Source file names are not shared"); + goto error; + } + + if (virt_layout->list[0].source_dset_name == virt_layout->list[1].source_dset_name || + virt_layout->list[0].source_dset_name == virt_layout->list[2].source_dset_name || + virt_layout->list[1].source_dset_name == virt_layout->list[2].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are erroneously shared"); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX) { + H5_FAILED(); + puts(" First source file name incorrectly marked as shared"); + goto error; + } + if (virt_layout->list[1].source_file_orig != 0 || virt_layout->list[2].source_file_orig != 0) { + H5_FAILED(); + puts(" Source file name sharing indices are incorrect"); + goto error; + } + + if (virt_layout->list[0].source_dset_orig != SIZE_MAX || + virt_layout->list[1].source_dset_orig != SIZE_MAX || + virt_layout->list[2].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + puts(" Source dataset names are erroneously marked as shared"); + goto error; + } + + /* Re-open verification for test 2 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + if ((file_id = H5Fopen(filename, H5F_ACC_RDONLY, fapl)) < 0) + TEST_ERROR; + if ((dset_id = H5Dopen2(file_id, "vds_shared_file", H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_file_name != virt_layout->list[1].source_file_name || + virt_layout->list[0].source_file_name != virt_layout->list[2].source_file_name) { + H5_FAILED(); + puts(" Source file names are not shared after re-open"); + goto error; + } + + if (virt_layout->list[0].source_dset_name == virt_layout->list[1].source_dset_name || + virt_layout->list[0].source_dset_name == virt_layout->list[2].source_dset_name || + virt_layout->list[1].source_dset_name == virt_layout->list[2].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are erroneously shared after re-open"); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX) { + H5_FAILED(); + puts(" First source file name incorrectly marked as shared after re-open"); + goto error; + } + if (virt_layout->list[1].source_file_orig != 0 || virt_layout->list[2].source_file_orig != 0) { + H5_FAILED(); + puts(" Source file name sharing indices are incorrect after re-open"); + goto error; + } + + if (virt_layout->list[0].source_dset_orig != SIZE_MAX || + virt_layout->list[1].source_dset_orig != SIZE_MAX || + virt_layout->list[2].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + puts(" Source dataset names are erroneously marked as shared after re-open"); + goto error; + } + + /* Close resources for test 2 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Pclose(dcpl_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + /* + * Test 3: VDS with shared dataset names + */ + + if ((file_id = H5Fcreate(filename, H5F_ACC_TRUNC, H5P_DEFAULT, fapl)) < 0) + TEST_ERROR; + + if ((dcpl_id = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_layout(dcpl_id, H5D_VIRTUAL) < 0) + TEST_ERROR; + + /* Add virtual mappings with different source files, same dataset */ + if (H5Pset_virtual(dcpl_id, virt_space_id, "source1.h5", "/shared_dataset", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "source2.h5", "/shared_dataset", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "source3.h5", "/shared_dataset", src_space_id) < 0) + TEST_ERROR; + + if ((dset_id = H5Dcreate2(file_id, "vds_shared_dset", H5T_NATIVE_INT, virt_space_id, H5P_DEFAULT, dcpl_id, + H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_dset_name != virt_layout->list[1].source_dset_name || + virt_layout->list[0].source_dset_name != virt_layout->list[2].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are not shared"); + goto error; + } + + if (virt_layout->list[0].source_file_name == virt_layout->list[1].source_file_name || + virt_layout->list[0].source_file_name == virt_layout->list[2].source_file_name || + virt_layout->list[1].source_file_name == virt_layout->list[2].source_file_name) { + H5_FAILED(); + puts(" Source file names are erroneously shared"); + goto error; + } + + if (virt_layout->list[0].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + puts(" First source dataset name incorrectly marked as shared"); + goto error; + } + if (virt_layout->list[1].source_dset_orig != 0 || virt_layout->list[2].source_dset_orig != 0) { + H5_FAILED(); + puts(" Source dataset name sharing indices are incorrect"); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX || + virt_layout->list[1].source_file_orig != SIZE_MAX || + virt_layout->list[2].source_file_orig != SIZE_MAX) { + H5_FAILED(); + puts(" Source file names are erroneously marked as shared"); + goto error; + } + + /* Re-open verification for test 3 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + if ((file_id = H5Fopen(filename, H5F_ACC_RDONLY, fapl)) < 0) + TEST_ERROR; + if ((dset_id = H5Dopen2(file_id, "vds_shared_dset", H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_dset_name != virt_layout->list[1].source_dset_name || + virt_layout->list[0].source_dset_name != virt_layout->list[2].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are not shared after re-open"); + goto error; + } + + if (virt_layout->list[0].source_file_name == virt_layout->list[1].source_file_name || + virt_layout->list[0].source_file_name == virt_layout->list[2].source_file_name || + virt_layout->list[1].source_file_name == virt_layout->list[2].source_file_name) { + H5_FAILED(); + puts(" Source file names are erroneously shared after re-open"); + goto error; + } + + if (virt_layout->list[0].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + puts(" First source dataset name incorrectly marked as shared after re-open"); + goto error; + } + if (virt_layout->list[1].source_dset_orig != 0 || virt_layout->list[2].source_dset_orig != 0) { + H5_FAILED(); + puts(" Source dataset name sharing indices are incorrect after re-open"); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX || + virt_layout->list[1].source_file_orig != SIZE_MAX || + virt_layout->list[2].source_file_orig != SIZE_MAX) { + H5_FAILED(); + puts(" Source file names are erroneously marked as shared after re-open"); + goto error; + } + + /* Close resources for test 3 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Pclose(dcpl_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + /* + * Test 4: VDS with both filenames and dataset names shared + */ + + if ((file_id = H5Fcreate(filename, H5F_ACC_TRUNC, H5P_DEFAULT, fapl)) < 0) + TEST_ERROR; + + if ((dcpl_id = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_layout(dcpl_id, H5D_VIRTUAL) < 0) + TEST_ERROR; + + /* Add identical virtual mappings */ + if (H5Pset_virtual(dcpl_id, virt_space_id, "shared_source.h5", "/shared_dataset", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "shared_source.h5", "/shared_dataset", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "shared_source.h5", "/shared_dataset", src_space_id) < 0) + TEST_ERROR; + + if ((dset_id = H5Dcreate2(file_id, "vds_shared_both", H5T_NATIVE_INT, virt_space_id, H5P_DEFAULT, dcpl_id, + H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_file_name != virt_layout->list[1].source_file_name || + virt_layout->list[0].source_file_name != virt_layout->list[2].source_file_name) { + H5_FAILED(); + puts(" Source file names are not shared"); + goto error; + } + if (virt_layout->list[0].source_dset_name != virt_layout->list[1].source_dset_name || + virt_layout->list[0].source_dset_name != virt_layout->list[2].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are not shared"); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX || + virt_layout->list[0].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + puts(" First entry incorrectly marked as shared"); + goto error; + } + if (virt_layout->list[1].source_file_orig != 0 || virt_layout->list[1].source_dset_orig != 0 || + virt_layout->list[2].source_file_orig != 0 || virt_layout->list[2].source_dset_orig != 0) { + H5_FAILED(); + puts(" String sharing indices are incorrect"); + goto error; + } + + /* Re-open verification for test 4 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + if ((file_id = H5Fopen(filename, H5F_ACC_RDONLY, fapl)) < 0) + TEST_ERROR; + if ((dset_id = H5Dopen2(file_id, "vds_shared_both", H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_file_name != virt_layout->list[1].source_file_name || + virt_layout->list[0].source_file_name != virt_layout->list[2].source_file_name) { + H5_FAILED(); + puts(" Source file names are not shared after re-open"); + goto error; + } + if (virt_layout->list[0].source_dset_name != virt_layout->list[1].source_dset_name || + virt_layout->list[0].source_dset_name != virt_layout->list[2].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are not shared after re-open"); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX || + virt_layout->list[0].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + puts(" First entry incorrectly marked as shared after re-open"); + goto error; + } + if (virt_layout->list[1].source_file_orig != 0 || virt_layout->list[1].source_dset_orig != 0 || + virt_layout->list[2].source_file_orig != 0 || virt_layout->list[2].source_dset_orig != 0) { + H5_FAILED(); + puts(" String sharing indices are incorrect after re-open"); + goto error; + } + + /* Close resources for test 4 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Pclose(dcpl_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + /* + * Test 5: VDS with same-file reference (".") + */ + + if ((dcpl_id = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_layout(dcpl_id, H5D_VIRTUAL) < 0) + TEST_ERROR; + + /* Add virtual mappings using "." for same file */ + if (H5Pset_virtual(dcpl_id, virt_space_id, ".", "/dataset1", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, ".", "/dataset2", src_space_id) < 0) + TEST_ERROR; + + if ((file_id = H5Fcreate(filename, H5F_ACC_TRUNC, H5P_DEFAULT, fapl)) < 0) + TEST_ERROR; + + if ((dset_id = H5Dcreate2(file_id, "vds_same_file", H5T_NATIVE_INT, virt_space_id, H5P_DEFAULT, dcpl_id, + H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_file_name != virt_layout->list[1].source_file_name) { + H5_FAILED(); + puts(" Same-file strings are not shared"); + goto error; + } + + if (strcmp(virt_layout->list[0].source_file_name, ".") != 0) { + H5_FAILED(); + printf(" Expected same-file reference '.', got '%s'\n", virt_layout->list[0].source_file_name); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX) { + H5_FAILED(); + puts(" First same-file entry incorrectly marked as shared"); + goto error; + } + if (virt_layout->list[1].source_file_orig != 0) { + H5_FAILED(); + puts(" Same-file sharing index is incorrect"); + goto error; + } + + if (virt_layout->list[0].source_dset_name == virt_layout->list[1].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are erroneously shared"); + goto error; + } + + /* Re-open verification for test 5 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + if ((file_id = H5Fopen(filename, H5F_ACC_RDONLY, fapl)) < 0) + TEST_ERROR; + if ((dset_id = H5Dopen2(file_id, "vds_same_file", H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + if (virt_layout->list[0].source_file_name != virt_layout->list[1].source_file_name) { + H5_FAILED(); + puts(" Same-file strings are not shared after re-open"); + goto error; + } + + if (strcmp(virt_layout->list[0].source_file_name, ".") != 0) { + H5_FAILED(); + printf(" Expected same-file reference '.', got '%s' after re-open\n", + virt_layout->list[0].source_file_name); + goto error; + } + + if (virt_layout->list[0].source_file_orig != SIZE_MAX) { + H5_FAILED(); + puts(" First same-file entry incorrectly marked as shared after re-open"); + goto error; + } + if (virt_layout->list[1].source_file_orig != 0) { + H5_FAILED(); + puts(" Same-file sharing index is incorrect after re-open"); + goto error; + } + + if (virt_layout->list[0].source_dset_name == virt_layout->list[1].source_dset_name) { + H5_FAILED(); + puts(" Source dataset names are erroneously shared after re-open"); + goto error; + } + + /* Clean-up for test 5 */ + if (H5Fclose(file_id) < 0) + TEST_ERROR; + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + + /* + * Test 6: VDS with unusual pattern testing robust sharing detection + * Pattern: dset1, dset2, dset1, dset3, dset2 + */ + + if ((file_id = H5Fcreate(filename, H5F_ACC_TRUNC, H5P_DEFAULT, fapl)) < 0) + TEST_ERROR; + + if ((dcpl_id = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_layout(dcpl_id, H5D_VIRTUAL) < 0) + TEST_ERROR; + + /* Add virtual mappings in unusual pattern: dset1, dset2, dset1, dset3, dset2 */ + if (H5Pset_virtual(dcpl_id, virt_space_id, "file1.h5", "/dset1", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "file2.h5", "/dset2", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "file1.h5", "/dset1", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "file3.h5", "/dset3", src_space_id) < 0) + TEST_ERROR; + if (H5Pset_virtual(dcpl_id, virt_space_id, "file2.h5", "/dset2", src_space_id) < 0) + TEST_ERROR; + + if ((dset_id = H5Dcreate2(file_id, "vds_unusual_pattern", H5T_NATIVE_INT, virt_space_id, H5P_DEFAULT, + dcpl_id, H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + /* Verify sharing pattern for files: + * [0]: file1.h5 (original) + * [1]: file2.h5 (original) + * [2]: file1.h5 (shared from [0]) + * [3]: file3.h5 (original) + * [4]: file2.h5 (shared from [1]) + */ + if (virt_layout->list[0].source_file_name != virt_layout->list[2].source_file_name) { + H5_FAILED(); + puts(" File sharing failed: entries [0] and [2] should share file1.h5"); + goto error; + } + if (virt_layout->list[1].source_file_name != virt_layout->list[4].source_file_name) { + H5_FAILED(); + puts(" File sharing failed: entries [1] and [4] should share file2.h5"); + goto error; + } + + /* Verify dataset sharing pattern: + * [0]: /dset1 (original) + * [1]: /dset2 (original) + * [2]: /dset1 (shared from [0]) + * [3]: /dset3 (original) + * [4]: /dset2 (shared from [1]) + */ + if (virt_layout->list[0].source_dset_name != virt_layout->list[2].source_dset_name) { + H5_FAILED(); + puts(" Dataset sharing failed: entries [0] and [2] should share /dset1"); + goto error; + } + if (virt_layout->list[1].source_dset_name != virt_layout->list[4].source_dset_name) { + H5_FAILED(); + puts(" Dataset sharing failed: entries [1] and [4] should share /dset2"); + goto error; + } + + /* Verify sharing indices are correct */ + if (virt_layout->list[0].source_file_orig != SIZE_MAX || + virt_layout->list[1].source_file_orig != SIZE_MAX || + virt_layout->list[3].source_file_orig != SIZE_MAX) { + H5_FAILED(); + puts(" File sharing indices: original entries incorrectly marked as shared"); + goto error; + } + if (virt_layout->list[2].source_file_orig != 0 || virt_layout->list[4].source_file_orig != 1) { + H5_FAILED(); + puts(" File sharing indices: shared entries have incorrect indices"); + goto error; + } + + if (virt_layout->list[0].source_dset_orig != SIZE_MAX || + virt_layout->list[1].source_dset_orig != SIZE_MAX || + virt_layout->list[3].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + puts(" Dataset sharing indices: original entries incorrectly marked as shared"); + goto error; + } + if (virt_layout->list[2].source_dset_orig != 0 || virt_layout->list[4].source_dset_orig != 1) { + H5_FAILED(); + puts(" Dataset sharing indices: shared entries have incorrect indices"); + goto error; + } + + /* Re-open verification for test 6 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + if ((file_id = H5Fopen(filename, H5F_ACC_RDONLY, fapl)) < 0) + TEST_ERROR; + if ((dset_id = H5Dopen2(file_id, "vds_unusual_pattern", H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + /* Re-verify sharing after re-open */ + if (virt_layout->list[0].source_file_name != virt_layout->list[2].source_file_name || + virt_layout->list[1].source_file_name != virt_layout->list[4].source_file_name) { + H5_FAILED(); + puts(" File sharing failed after re-open"); + goto error; + } + if (virt_layout->list[0].source_dset_name != virt_layout->list[2].source_dset_name || + virt_layout->list[1].source_dset_name != virt_layout->list[4].source_dset_name) { + H5_FAILED(); + puts(" Dataset sharing failed after re-open"); + goto error; + } + + /* Re-verify sharing indices after re-open */ + if (virt_layout->list[2].source_file_orig != 0 || virt_layout->list[4].source_file_orig != 1 || + virt_layout->list[2].source_dset_orig != 0 || virt_layout->list[4].source_dset_orig != 1) { + H5_FAILED(); + puts(" Sharing indices are incorrect after re-open"); + goto error; + } + + /* Clean up test 6 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Pclose(dcpl_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + /* + * Test 7: VDS with many mappings to test hash table resizing + * Creates many mappings with a mix of shared and unique strings + */ + + if ((file_id = H5Fcreate(filename, H5F_ACC_TRUNC, H5P_DEFAULT, fapl)) < 0) + TEST_ERROR; + + if ((dcpl_id = H5Pcreate(H5P_DATASET_CREATE)) < 0) + TEST_ERROR; + if (H5Pset_layout(dcpl_id, H5D_VIRTUAL) < 0) + TEST_ERROR; + + /* Create NUM_MAPPINGS_MANY mappings with a pattern that includes sharing: + * - Every 10th mapping uses "shared_file.h5" + * - Every 5th mapping uses "/shared_dataset" + * - Others use unique file/dataset names + */ + char file_name[64]; + char dset_name[64]; + int shared_file_count = 0; + int shared_dset_count = 0; + + for (int i = 0; i < NUM_MAPPINGS_MANY; i++) { + if (i % 10 == 0) { + strcpy(file_name, "shared_file.h5"); + shared_file_count++; + } + else { + snprintf(file_name, sizeof(file_name), "file_%d.h5", i); + } + + if (i % 5 == 0) { + strcpy(dset_name, "/shared_dataset"); + shared_dset_count++; + } + else { + snprintf(dset_name, sizeof(dset_name), "/dataset_%d", i); + } + + if (H5Pset_virtual(dcpl_id, virt_space_id, file_name, dset_name, src_space_id) < 0) + TEST_ERROR; + } + + if ((dset_id = H5Dcreate2(file_id, "vds_many_mappings", H5T_NATIVE_INT, virt_space_id, H5P_DEFAULT, + dcpl_id, H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + /* Verify that we have the expected number of mappings */ + if (virt_layout->list_nused != NUM_MAPPINGS_MANY) { + H5_FAILED(); + printf(" Expected %d mappings, got %zu\n", NUM_MAPPINGS_MANY, virt_layout->list_nused); + goto error; + } + + for (int i = 0; i < NUM_MAPPINGS_MANY; i++) { + /* Check file sharing */ + if (i % 10 == 0) { /* Should share "shared_file.h5" */ + if (i > 0 && virt_layout->list[0].source_file_name != virt_layout->list[i].source_file_name) { + H5_FAILED(); + printf(" File sharing failed: entry [%d] should share shared_file.h5 with entry [0]\n", i); + goto error; + } + if (i == 0) { + if (virt_layout->list[i].source_file_orig != SIZE_MAX) { + H5_FAILED(); + printf(" Entry [0] incorrectly marked as shared file (expected SIZE_MAX, got %zu)\n", + virt_layout->list[i].source_file_orig); + goto error; + } + } + else { + if (virt_layout->list[i].source_file_orig != 0) { + H5_FAILED(); + printf(" File sharing index incorrect for entry [%d]: expected 0, got %zu\n", i, + virt_layout->list[i].source_file_orig); + goto error; + } + } + } + else { /* Should not share file */ + if (virt_layout->list[i].source_file_orig != SIZE_MAX) { + H5_FAILED(); + printf(" Entry [%d] incorrectly marked as sharing file (expected SIZE_MAX, got %zu)\n", i, + virt_layout->list[i].source_file_orig); + goto error; + } + } + + /* Check dataset sharing */ + if (i % 5 == 0) { /* Should share "/shared_dataset" */ + if (i > 0 && virt_layout->list[0].source_dset_name != virt_layout->list[i].source_dset_name) { + H5_FAILED(); + printf(" Dataset sharing failed: entry [%d] should share /shared_dataset with entry [0]\n", + i); + goto error; + } + if (i == 0) { + if (virt_layout->list[i].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + printf( + " Entry [0] incorrectly marked as shared dataset (expected SIZE_MAX, got %zu)\n", + virt_layout->list[i].source_dset_orig); + goto error; + } + } + else { + if (virt_layout->list[i].source_dset_orig != 0) { + H5_FAILED(); + printf(" Dataset sharing index incorrect for entry [%d]: expected 0, got %zu\n", i, + virt_layout->list[i].source_dset_orig); + goto error; + } + } + } + else { /* Should not share dataset */ + if (virt_layout->list[i].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + printf(" Entry [%d] incorrectly marked as sharing dataset (expected SIZE_MAX, got %zu)\n", + i, virt_layout->list[i].source_dset_orig); + goto error; + } + } + } + + /* Re-open verification for test 7 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + if ((file_id = H5Fopen(filename, H5F_ACC_RDONLY, fapl)) < 0) + TEST_ERROR; + if ((dset_id = H5Dopen2(file_id, "vds_many_mappings", H5P_DEFAULT)) < 0) + TEST_ERROR; + + if ((dset_int = (H5D_t *)H5VL_object(dset_id)) == NULL) + TEST_ERROR; + virt_layout = &(dset_int->shared->layout.storage.u.virt); + + /* Full verification after re-open */ + for (int i = 0; i < NUM_MAPPINGS_MANY; i++) { + /* Check file sharing */ + if (i % 10 == 0) { /* Should share "shared_file.h5" */ + if (i > 0 && virt_layout->list[0].source_file_name != virt_layout->list[i].source_file_name) { + H5_FAILED(); + printf(" File sharing failed after re-open: entry [%d] should share shared_file.h5 with " + "entry [0]\n", + i); + goto error; + } + if (i == 0) { + if (virt_layout->list[i].source_file_orig != SIZE_MAX) { + H5_FAILED(); + printf(" Entry [0] incorrectly marked as shared file after re-open (expected " + "SIZE_MAX, got %zu)\n", + virt_layout->list[i].source_file_orig); + goto error; + } + } + else { + if (virt_layout->list[i].source_file_orig != 0) { + H5_FAILED(); + printf(" File sharing index incorrect after re-open for entry [%d]: expected 0, got " + "%zu\n", + i, virt_layout->list[i].source_file_orig); + goto error; + } + } + } + else { /* Should not share file */ + if (virt_layout->list[i].source_file_orig != SIZE_MAX) { + H5_FAILED(); + printf(" Entry [%d] incorrectly marked as sharing file after re-open (expected SIZE_MAX, " + "got %zu)\n", + i, virt_layout->list[i].source_file_orig); + goto error; + } + } + + /* Check dataset sharing */ + if (i % 5 == 0) { /* Should share "/shared_dataset" */ + if (i > 0 && virt_layout->list[0].source_dset_name != virt_layout->list[i].source_dset_name) { + H5_FAILED(); + printf(" Dataset sharing failed after re-open: entry [%d] should share /shared_dataset " + "with entry [0]\n", + i); + goto error; + } + if (i == 0) { + if (virt_layout->list[i].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + printf(" Entry [0] incorrectly marked as shared dataset after re-open (expected " + "SIZE_MAX, got %zu)\n", + virt_layout->list[i].source_dset_orig); + goto error; + } + } + else { + if (virt_layout->list[i].source_dset_orig != 0) { + H5_FAILED(); + printf(" Dataset sharing index incorrect after re-open for entry [%d]: expected 0, " + "got %zu\n", + i, virt_layout->list[i].source_dset_orig); + goto error; + } + } + } + else { /* Should not share dataset */ + if (virt_layout->list[i].source_dset_orig != SIZE_MAX) { + H5_FAILED(); + printf(" Entry [%d] incorrectly marked as sharing dataset after re-open (expected " + "SIZE_MAX, got %zu)\n", + i, virt_layout->list[i].source_dset_orig); + goto error; + } + } + } + + /* Clean up test 7 */ + if (H5Dclose(dset_id) < 0) + TEST_ERROR; + if (H5Pclose(dcpl_id) < 0) + TEST_ERROR; + if (H5Fclose(file_id) < 0) + TEST_ERROR; + + /* Clean up */ + if (H5Sclose(src_space_id) < 0) + TEST_ERROR; + if (H5Sclose(virt_space_id) < 0) + TEST_ERROR; + + PASSED(); + return SUCCEED; + +error: + H5E_BEGIN_TRY + { + H5Dclose(dset_id); + H5Pclose(dcpl_id); + H5Fclose(file_id); + H5Sclose(src_space_id); + H5Sclose(virt_space_id); + } + H5E_END_TRY; + + return FAIL; +} /* end test_vds_shared_strings() */ + /*------------------------------------------------------------------------- * Function: main * @@ -16432,6 +17410,9 @@ main(void) nerrors += (test_dcpl_layout_caching(H5D_CHUNKED) < 0 ? 1 : 0); nerrors += (test_dcpl_layout_caching(H5D_VIRTUAL) < 0 ? 1 : 0); + /* Verify that source file/dataset names are shared properly */ + nerrors += (test_vds_shared_strings(fapl) < 0 ? 1 : 0); + if (nerrors) goto error; printf("All dataset tests passed.\n");