Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
46 changes: 45 additions & 1 deletion src/rootfilespec/rntuple/RNTuple.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,11 @@
from collections.abc import Callable
from math import ceil

from rootfilespec.bootstrap import BOOTSTRAP_CONTEXT
from rootfilespec.bootstrap.RAnchor import ROOT3a3aRNTuple
from rootfilespec.bootstrap.streamedobject import Ref, read_streamed_item
from rootfilespec.bootstrap.TList import TList
from rootfilespec.bootstrap.TStreamerInfo import TStreamerInfo
from rootfilespec.rntuple.envelope import RFeatureFlags
from rootfilespec.rntuple.footer import FooterEnvelope
from rootfilespec.rntuple.header import HeaderEnvelope
Expand All @@ -15,7 +19,12 @@
ExtraTypeInformation,
FieldDescription,
)
from rootfilespec.serializable import Locator, ReadBuffer, ROOTSerializable
from rootfilespec.serializable import (
BufferContext,
Locator,
ReadBuffer,
ROOTSerializable,
)


@dataclasses.dataclass
Expand Down Expand Up @@ -129,6 +138,41 @@ def schemaDescription(self) -> SchemaDescription:
self.headerEnvelope, self.footerEnvelope
)

def streamer_infos(self) -> dict[bytes, TStreamerInfo]:
"""The TStreamerInfo of each class that streamed (role 0x04) fields need, by class name

Decoded from the content of every extra type information record with content
identifier 0 (ROOT writes one, in the footer's schema extension: root-io-spec
ERRATA 10), whose fContent keeps the bytes. Records with other identifiers are
ignored, as the spec asks. The content is a TList written as if through a
pointer (RNTupleSerializer::SerializeStreamerInfos), with nothing after it.
"""
infos: dict[bytes, TStreamerInfo] = {}
for record in self.schemaDescription.extraTypeInformations:
if record.fContentIdentifier != 0: # kStreamerInfo
continue
buffer = ReadBuffer(
memoryview(record.fContent),
0,
BOOTSTRAP_CONTEXT,
BufferContext(abspos=None),
)
streamed, rest = read_streamed_item(buffer)
if not isinstance(streamed, TList) or rest:
msg = f"Expected the streamer info content to be one TList, got {streamed!r} and {len(rest)} more bytes"
raise ValueError(msg)
for item in streamed.items:
# A pointee comes back bare or as a Ref; always a Ref after #105
info = item.obj if isinstance(item, Ref) else item
if not isinstance(info, TStreamerInfo):
msg = f"Expected a TStreamerInfo in the streamer info content, got {info!r}"
raise ValueError(msg)
if info.fName in infos:
msg = f"Two TStreamerInfo for class {info.fName!r} in the streamer info content"
raise ValueError(msg)
infos[info.fName] = info
return infos

# can provide helpers to get page descriptions with different filters, columns/rows/etc.
def get_extended_page_descriptions(
self,
Expand Down
8 changes: 8 additions & 0 deletions src/rootfilespec/rntuple/schema.py
Original file line number Diff line number Diff line change
Expand Up @@ -228,3 +228,11 @@ class ExtraTypeInformation(RecordFrame):
"""The version of the type for which this extra type information is provided."""
fTypeName: RNTupleString
"""The name of the type for which this extra type information is provided."""
fContent: RNTupleString
"""The content, laid out as an RNTuple string like the type name (root-io-spec
ERRATA 9): a 32-bit little-endian length, then the bytes. The bytes are binary,
not UTF-8 text.

For content identifier 0 it is the ROOT-streamed TList of TStreamerInfo
of the streamed fields. ROOT writes that record in the footer's schema
extension, not the header (ERRATA 10)."""
97 changes: 97 additions & 0 deletions tests/test_extra_type_info_content.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,97 @@
import dataclasses
from pathlib import Path

import pytest

from rootfilespec.bootstrap.TStreamerInfo import TStreamerElement, TStreamerInfo
from rootfilespec.reader import FileReader, open_path
from rootfilespec.rntuple.RNTuple import RNTuple

DATA = Path(__file__).parent.parent / "reference" / "root-io-spec" / "data" / "rntuple"


def _rntuple(reader: FileReader) -> RNTuple:
keylist = reader.keylist()
(name,) = [n for n in keylist if keylist[n].fClassName == b"ROOT::RNTuple"]
return RNTuple.from_anchor(reader.fetch(keylist[name]), reader.fetch.buffer)


@pytest.mark.skipif(not DATA.exists(), reason="reference/root-io-spec not checked out")
def test_streamer_info_content():
"""Issue #118: the extra type information's content is a string after the
type name (root-io-spec ERRATA 9), in the footer's schema extension (ERRATA 10)

rntuple/streamed.root's case.toml pins the content length 438 at offset 1236,
the streamed object's first bytes at 1240 (byte count 434 with the 0x40000000
flag, then the new-class tag and "TList"), and the length byte 15 of the
TStreamerInfo name "RNStreamedInner" at 1319.
"""
path = DATA / "streamed.root"
raw = path.read_bytes()
with open_path(path) as reader:
rntuple = _rntuple(reader)

assert rntuple.headerEnvelope.extraTypeInformations.items == []
(info,) = rntuple.footerEnvelope.schemaExtension.extraTypeInformations.items
assert info.fContentIdentifier == 0
assert info.fTypeVersion == 0
assert info.fTypeName == b""
assert type(info.fContent) is bytes
assert int.from_bytes(raw[1236:1240], "little") == 438
assert info.fContent == raw[1240 : 1240 + 438]
assert info.fContent[:8] == bytes.fromhex("400001b2ffffffff")
assert info.fContent[8:13] == b"TList"
assert info.fContent[1319 - 1240] == 15
assert info.fContent[1320 - 1240 : 1320 - 1240 + 15] == b"RNStreamedInner"
# 8 (size) + 4 (content ID) + 4 (type version) + 4 + 0 (type name) + 4 + 438
assert info.fSize == 462
assert info._unknown == b""


def _element_names(info: TStreamerInfo) -> list[bytes]:
elements = info.fObjects.objects
assert all(isinstance(element, TStreamerElement) for element in elements)
return [
element.fName for element in elements if isinstance(element, TStreamerElement)
]


@pytest.mark.skipif(not DATA.exists(), reason="reference/root-io-spec not checked out")
def test_streamer_infos():
"""The content decodes to the TStreamerInfo of RNStreamedInner, which the
file's own StreamerInfo record also holds: the two must agree"""
with open_path(DATA / "streamed.root") as reader:
infos = _rntuple(reader).streamer_infos()
in_file = reader.streamerinfos()[b"RNStreamedInner"]

assert list(infos) == [b"RNStreamedInner"]
info = infos[b"RNStreamedInner"]
assert info.fClassVersion == in_file.fClassVersion == 1
assert info.fCheckSum == in_file.fCheckSum
assert _element_names(info) == _element_names(in_file)
assert len(_element_names(info)) == 3


@pytest.mark.skipif(not DATA.exists(), reason="reference/root-io-spec not checked out")
def test_streamer_infos_without_streamed_fields():
with open_path(DATA / "user-class.root") as reader:
rntuple = _rntuple(reader)
assert rntuple.schemaDescription.extraTypeInformations == []
assert rntuple.streamer_infos() == {}


@pytest.mark.skipif(not DATA.exists(), reason="reference/root-io-spec not checked out")
def test_streamer_infos_ignores_other_ids_and_rejects_duplicates():
with open_path(DATA / "streamed.root") as reader:
rntuple = _rntuple(reader)
records = rntuple.footerEnvelope.schemaExtension.extraTypeInformations.items
(info,) = records

# An unknown content identifier is ignored (spec: forward compatibility)
records.append(dataclasses.replace(info, fContentIdentifier=1, fContent=b"?"))
assert list(rntuple.streamer_infos()) == [b"RNStreamedInner"]

# The same class twice would be merged by name: refuse it
records.append(info)
with pytest.raises(ValueError, match="Two TStreamerInfo for class"):
rntuple.streamer_infos()
Loading