diff --git a/src/rootfilespec/rntuple/RNTuple.py b/src/rootfilespec/rntuple/RNTuple.py index 08ef1bf..a5c9aa6 100644 --- a/src/rootfilespec/rntuple/RNTuple.py +++ b/src/rootfilespec/rntuple/RNTuple.py @@ -2,7 +2,11 @@ from collections.abc import Callable from math import ceil +from rootfilespec.bootstrap import BOOTSTRAP_CONTEXT from rootfilespec.bootstrap.RAnchor import ROOT3a3aRNTuple +from rootfilespec.bootstrap.streamedobject import Ref, read_streamed_item +from rootfilespec.bootstrap.TList import TList +from rootfilespec.bootstrap.TStreamerInfo import TStreamerInfo from rootfilespec.rntuple.envelope import RFeatureFlags from rootfilespec.rntuple.footer import FooterEnvelope from rootfilespec.rntuple.header import HeaderEnvelope @@ -15,7 +19,12 @@ ExtraTypeInformation, FieldDescription, ) -from rootfilespec.serializable import Locator, ReadBuffer, ROOTSerializable +from rootfilespec.serializable import ( + BufferContext, + Locator, + ReadBuffer, + ROOTSerializable, +) @dataclasses.dataclass @@ -129,6 +138,41 @@ def schemaDescription(self) -> SchemaDescription: self.headerEnvelope, self.footerEnvelope ) + def streamer_infos(self) -> dict[bytes, TStreamerInfo]: + """The TStreamerInfo of each class that streamed (role 0x04) fields need, by class name + + Decoded from the content of every extra type information record with content + identifier 0 (ROOT writes one, in the footer's schema extension: root-io-spec + ERRATA 10), whose fContent keeps the bytes. Records with other identifiers are + ignored, as the spec asks. The content is a TList written as if through a + pointer (RNTupleSerializer::SerializeStreamerInfos), with nothing after it. + """ + infos: dict[bytes, TStreamerInfo] = {} + for record in self.schemaDescription.extraTypeInformations: + if record.fContentIdentifier != 0: # kStreamerInfo + continue + buffer = ReadBuffer( + memoryview(record.fContent), + 0, + BOOTSTRAP_CONTEXT, + BufferContext(abspos=None), + ) + streamed, rest = read_streamed_item(buffer) + if not isinstance(streamed, TList) or rest: + msg = f"Expected the streamer info content to be one TList, got {streamed!r} and {len(rest)} more bytes" + raise ValueError(msg) + for item in streamed.items: + # A pointee comes back bare or as a Ref; always a Ref after #105 + info = item.obj if isinstance(item, Ref) else item + if not isinstance(info, TStreamerInfo): + msg = f"Expected a TStreamerInfo in the streamer info content, got {info!r}" + raise ValueError(msg) + if info.fName in infos: + msg = f"Two TStreamerInfo for class {info.fName!r} in the streamer info content" + raise ValueError(msg) + infos[info.fName] = info + return infos + # can provide helpers to get page descriptions with different filters, columns/rows/etc. def get_extended_page_descriptions( self, diff --git a/src/rootfilespec/rntuple/schema.py b/src/rootfilespec/rntuple/schema.py index 08c14f0..750de79 100644 --- a/src/rootfilespec/rntuple/schema.py +++ b/src/rootfilespec/rntuple/schema.py @@ -228,3 +228,11 @@ class ExtraTypeInformation(RecordFrame): """The version of the type for which this extra type information is provided.""" fTypeName: RNTupleString """The name of the type for which this extra type information is provided.""" + fContent: RNTupleString + """The content, laid out as an RNTuple string like the type name (root-io-spec + ERRATA 9): a 32-bit little-endian length, then the bytes. The bytes are binary, + not UTF-8 text. + + For content identifier 0 it is the ROOT-streamed TList of TStreamerInfo + of the streamed fields. ROOT writes that record in the footer's schema + extension, not the header (ERRATA 10).""" diff --git a/tests/test_extra_type_info_content.py b/tests/test_extra_type_info_content.py new file mode 100644 index 0000000..9cbc8d1 --- /dev/null +++ b/tests/test_extra_type_info_content.py @@ -0,0 +1,97 @@ +import dataclasses +from pathlib import Path + +import pytest + +from rootfilespec.bootstrap.TStreamerInfo import TStreamerElement, TStreamerInfo +from rootfilespec.reader import FileReader, open_path +from rootfilespec.rntuple.RNTuple import RNTuple + +DATA = Path(__file__).parent.parent / "reference" / "root-io-spec" / "data" / "rntuple" + + +def _rntuple(reader: FileReader) -> RNTuple: + keylist = reader.keylist() + (name,) = [n for n in keylist if keylist[n].fClassName == b"ROOT::RNTuple"] + return RNTuple.from_anchor(reader.fetch(keylist[name]), reader.fetch.buffer) + + +@pytest.mark.skipif(not DATA.exists(), reason="reference/root-io-spec not checked out") +def test_streamer_info_content(): + """Issue #118: the extra type information's content is a string after the + type name (root-io-spec ERRATA 9), in the footer's schema extension (ERRATA 10) + + rntuple/streamed.root's case.toml pins the content length 438 at offset 1236, + the streamed object's first bytes at 1240 (byte count 434 with the 0x40000000 + flag, then the new-class tag and "TList"), and the length byte 15 of the + TStreamerInfo name "RNStreamedInner" at 1319. + """ + path = DATA / "streamed.root" + raw = path.read_bytes() + with open_path(path) as reader: + rntuple = _rntuple(reader) + + assert rntuple.headerEnvelope.extraTypeInformations.items == [] + (info,) = rntuple.footerEnvelope.schemaExtension.extraTypeInformations.items + assert info.fContentIdentifier == 0 + assert info.fTypeVersion == 0 + assert info.fTypeName == b"" + assert type(info.fContent) is bytes + assert int.from_bytes(raw[1236:1240], "little") == 438 + assert info.fContent == raw[1240 : 1240 + 438] + assert info.fContent[:8] == bytes.fromhex("400001b2ffffffff") + assert info.fContent[8:13] == b"TList" + assert info.fContent[1319 - 1240] == 15 + assert info.fContent[1320 - 1240 : 1320 - 1240 + 15] == b"RNStreamedInner" + # 8 (size) + 4 (content ID) + 4 (type version) + 4 + 0 (type name) + 4 + 438 + assert info.fSize == 462 + assert info._unknown == b"" + + +def _element_names(info: TStreamerInfo) -> list[bytes]: + elements = info.fObjects.objects + assert all(isinstance(element, TStreamerElement) for element in elements) + return [ + element.fName for element in elements if isinstance(element, TStreamerElement) + ] + + +@pytest.mark.skipif(not DATA.exists(), reason="reference/root-io-spec not checked out") +def test_streamer_infos(): + """The content decodes to the TStreamerInfo of RNStreamedInner, which the + file's own StreamerInfo record also holds: the two must agree""" + with open_path(DATA / "streamed.root") as reader: + infos = _rntuple(reader).streamer_infos() + in_file = reader.streamerinfos()[b"RNStreamedInner"] + + assert list(infos) == [b"RNStreamedInner"] + info = infos[b"RNStreamedInner"] + assert info.fClassVersion == in_file.fClassVersion == 1 + assert info.fCheckSum == in_file.fCheckSum + assert _element_names(info) == _element_names(in_file) + assert len(_element_names(info)) == 3 + + +@pytest.mark.skipif(not DATA.exists(), reason="reference/root-io-spec not checked out") +def test_streamer_infos_without_streamed_fields(): + with open_path(DATA / "user-class.root") as reader: + rntuple = _rntuple(reader) + assert rntuple.schemaDescription.extraTypeInformations == [] + assert rntuple.streamer_infos() == {} + + +@pytest.mark.skipif(not DATA.exists(), reason="reference/root-io-spec not checked out") +def test_streamer_infos_ignores_other_ids_and_rejects_duplicates(): + with open_path(DATA / "streamed.root") as reader: + rntuple = _rntuple(reader) + records = rntuple.footerEnvelope.schemaExtension.extraTypeInformations.items + (info,) = records + + # An unknown content identifier is ignored (spec: forward compatibility) + records.append(dataclasses.replace(info, fContentIdentifier=1, fContent=b"?")) + assert list(rntuple.streamer_infos()) == [b"RNStreamedInner"] + + # The same class twice would be merged by name: refuse it + records.append(info) + with pytest.raises(ValueError, match="Two TStreamerInfo for class"): + rntuple.streamer_infos()