From 510e1f5d633e55c00c145bb710d373304be4c71a Mon Sep 17 00:00:00 2001 From: Jarry Shaw Date: Mon, 14 Sep 2026 23:18:44 -0400 Subject: [PATCH] ipv6: scale fragment offsets, key reassembly on the fragment ID, fix the flow label Four defects, two of which silently corrupted data while reporting success. The fragment offset was passed to the reassembler as raw wire units where it wants octets: foundation/reassembly/ip.py indexes the datagram buffer with it directly but uses FO // 8 for the received-bit table, so fragments landed at an eighth of their offsets and the completion check still passed. On ipv6.pcap the reassembled datagram was 977 octets instead of 4778, completed=True either way. _make_data now emits wire units again so Data -> make round-trips, deliberately not replicating the asymmetry ipv4.py has. scapy's toolkit needed the same scaling at its own site. Reassembly keyed on the flow label instead of the fragment identification, so id.id surfaced as 0 where the fixture carries 110308, and two concurrent fragmented datagrams sharing a label were merged into one. Fixed in the pcap, pcapng and scapy toolkits. The flow label itself was parsed from bits 8-27 rather than 12-31, so it included the last version/traffic-class nibble and dropped the label's low four bits. Worse than the issue reported: BitField uses the same namespace for packing, so pcapkit was *writing* malformed IPv6 headers - class 0x2a with label 0x12345 serialised as 62123450 instead of 62a12345. One line fixes both directions. The IPv6-Opts option registry lacked the __enum__ declaration hopopt.py has, so QuickStartOption's subclasses clobbered two of the parent registry's keys, not one as filed: Pad1 at key 0 and _SMFDPDOption at key 8. #353 and #360 interacted - a truncated label widened the collision surface of a label-keyed buffer - and keying on the identification dissolves that entirely. Four existing expectations changed because each encoded a bug, with the wire value now pinned alongside the octet value so the two cannot drift again. The toolkit fixtures also gained a fragment id distinct from their flow label, since a fixture where the two are equal proves nothing. Audited the whole schema tree for the same shapes: IPv6.hextet was the only overlapping BitField namespace and QuickStartOption the only genuine EnumSchema collision, so neither is part of a pattern. Suite 508 -> 517 passed, 315 -> 338 subtests; pylint delta exactly zero. Closes #352, closes #353, closes #360, closes #369. --- pcapkit/protocols/internet/ipv6_frag.py | 18 +- pcapkit/protocols/schema/internet/ipv6.py | 10 +- .../protocols/schema/internet/ipv6_opts.py | 7 + pcapkit/toolkit/pcap.py | 2 +- pcapkit/toolkit/pcapng.py | 2 +- pcapkit/toolkit/scapy.py | 7 +- .../internet/test_ipv6_extension_runtime.py | 6 +- .../internet/test_ipv6_extension_unit.py | 86 ++++++- .../internet/test_ipv6_reassembly_runtime.py | 222 ++++++++++++++++++ tests/protocols/internet/test_ipv6_unit.py | 101 ++++++++ tests/toolkit/test_pcap_unit.py | 19 +- tests/toolkit/test_scapy_unit.py | 13 +- 12 files changed, 477 insertions(+), 16 deletions(-) create mode 100644 tests/protocols/internet/test_ipv6_reassembly_runtime.py diff --git a/pcapkit/protocols/internet/ipv6_frag.py b/pcapkit/protocols/internet/ipv6_frag.py index 360889b27d..a8788e7018 100644 --- a/pcapkit/protocols/internet/ipv6_frag.py +++ b/pcapkit/protocols/internet/ipv6_frag.py @@ -136,9 +136,15 @@ def read(self, length: 'Optional[int]' = None, *, extension: 'bool' = False, # length = len(self) schema = self.__header__ + # NOTE: The on-wire Fragment Offset is a 13-bit count of 8-octet units + # (:rfc:`8200#section-4.5`), but ``Data_IPv6_Frag.offset`` carries octets + # so that it matches ``Data_IPv4.offset`` and can be handed straight to + # the reassembly machinery, which indexes the datagram buffer with it + # (:meth:`pcapkit.foundation.reassembly.ip.IP.reassembly`). Scale here, + # exactly as :meth:`pcapkit.protocols.internet.ipv4.IPv4.read` does. ipv6_frag = Data_IPv6_Frag( next=schema.next, - offset=schema.flags['offset'], + offset=int(schema.flags['offset']) * 8, mf=bool(schema.flags['mf']), id=schema.id, ) @@ -164,7 +170,10 @@ def make(self, next_default: Default value of next header. next_namespace: Namespace of next header. next_reversed: If the namespace of next header is reversed. - offset: Fragment offset. + offset: Fragment offset, in on-wire 8-octet units (:rfc:`8200#section-4.5`). + Note that :attr:`Data_IPv6_Frag.offset + ` is in + octets, so it must be divided by 8 before being passed here. mf: More fragments flag. id: Identification. payload: Payload of current instance. @@ -249,9 +258,12 @@ def _make_data(cls, data: 'Data_IPv6_Frag') -> 'dict[str, Any]': # type: ignore Key-value pairs for protocol construction. """ + # NOTE: ``make`` takes ``offset`` in on-wire 8-octet units, while + # ``data.offset`` is in octets (see :meth:`read`), so scale back down to + # keep the data-to-schema round trip exact. return { 'next': data.next, - 'offset': data.offset, + 'offset': data.offset // 8, 'mf': data.mf, 'id': data.id, 'payload': cls._make_payload(data), diff --git a/pcapkit/protocols/schema/internet/ipv6.py b/pcapkit/protocols/schema/internet/ipv6.py index 98e7db1346..10190a271b 100644 --- a/pcapkit/protocols/schema/internet/ipv6.py +++ b/pcapkit/protocols/schema/internet/ipv6.py @@ -38,10 +38,18 @@ class IPv6(Schema): """Header schema for IPv6 packet.""" #: Version, traffic class and flow label. + #: + #: The :class:`~pcapkit.corekit.fields.strings.BitField` namespace maps each + #: subfield to a ``(start_bit, length_in_bits)`` pair, *not* to + #: ``(start_bit, end_bit)``. Per :rfc:`8200#section-3` the first 32 bits of + #: the IPv6 header are Version (bits 0-3), Traffic Class (bits 4-11) and + #: Flow Label (bits 12-31), so the flow label starts at bit 12 -- reading it + #: as ``(8, 20)`` would overlap the low nibble of the traffic class and drop + #: the low nibble of the label. hextet: 'IPv6Hextet' = BitField(length=4, namespace={ 'version': (0, 4), 'class': (4, 8), - 'label': (8, 20), + 'label': (12, 20), }) #: Payload length. length: int = UInt16Field() diff --git a/pcapkit/protocols/schema/internet/ipv6_opts.py b/pcapkit/protocols/schema/internet/ipv6_opts.py index 3870ddc233..ee8715a69b 100644 --- a/pcapkit/protocols/schema/internet/ipv6_opts.py +++ b/pcapkit/protocols/schema/internet/ipv6_opts.py @@ -489,6 +489,13 @@ def post_process(self, packet: 'dict[str, Any]') -> 'QuickStartOption': class QuickStartOption(Option, EnumSchema[Enum_QSFunction]): """Header schema for IPv6-Opts quick start options.""" + # NOTE: This declaration scopes the QS-function registry to this class. Without + # it the subclasses below register into the *parent* ``Option`` registry, whose + # keys are IPv6-Opts option *types* -- and since ``Quick_Start_Request`` is 0 + # and ``Report_of_Approved_Rate`` is 8, they would clobber option types 0 + # (``Pad1``) and 8 (``SMF_DPD``). ``hopopt.py`` carries the same declaration. + __enum__: 'DefaultDict[Enum_QSFunction, Type[QuickStartOption]]' = collections.defaultdict(lambda: None) # type: ignore[arg-type,return-value] + #: Flags. flags: 'QuickStartFlags' = BitField(length=1, namespace={ 'func': (0, 4), diff --git a/pcapkit/toolkit/pcap.py b/pcapkit/toolkit/pcap.py index b55dfeb545..c3bacdf5cd 100644 --- a/pcapkit/toolkit/pcap.py +++ b/pcapkit/toolkit/pcap.py @@ -108,7 +108,7 @@ def ipv6_reassembly(frame: 'Frame') -> 'IP_Packet[IPv6Address] | None': bufid=( ipv6_info.src, # source IP address ipv6_info.dst, # destination IP address - ipv6_info.label, # label + ipv6_frag_info.id, # identification ipv6_frag_info.next, # next header field in IPv6 Fragment Header ), num=frame.info.number, # original packet range number diff --git a/pcapkit/toolkit/pcapng.py b/pcapkit/toolkit/pcapng.py index d36a7b8079..3f973e1519 100644 --- a/pcapkit/toolkit/pcapng.py +++ b/pcapkit/toolkit/pcapng.py @@ -112,7 +112,7 @@ def ipv6_reassembly(frame: 'PCAPNG') -> 'IP_Packet[IPv6Address] | None': bufid=( ipv6_info.src, # source IP address ipv6_info.dst, # destination IP address - ipv6_info.label, # label + ipv6_frag_info.id, # identification ipv6_frag_info.next, # next header field in IPv6 Fragment Header ), num=frame_info.number, # original packet range number diff --git a/pcapkit/toolkit/scapy.py b/pcapkit/toolkit/scapy.py index a769605bba..4fa4daa66a 100644 --- a/pcapkit/toolkit/scapy.py +++ b/pcapkit/toolkit/scapy.py @@ -190,11 +190,14 @@ def ipv6_reassembly(packet: 'Packet', *, count: 'int' = -1) -> 'IP_Packet[IPv6Ad ipaddress.ip_address(ipv6.src)), # source IP address cast('IPv6Address', ipaddress.ip_address(ipv6.dst)), # destination IP address - ipv6.fl, # label + ipv6_frag.id, # identification Enum_TransType.get(ipv6_frag.nh), # next header field in IPv6 Fragment Header ), num=count, # original packet range number - fo=ipv6_frag.offset, # fragment offset + # NOTE: Scapy reports ``IPv6ExtHdrFragment.offset`` in on-wire 8-octet + # units (:rfc:`8200#section-4.5`), but the reassembly machinery indexes + # the datagram buffer with ``fo``, so it must be scaled into octets. + fo=ipv6_frag.offset * 8, # fragment offset ihl=len(ipv6) - len(ipv6_frag), # header length, only headers before IPv6-Frag mf=bool(ipv6_frag.m), # more fragment flag tl=len(ipv6), # total length, header includes diff --git a/tests/protocols/internet/test_ipv6_extension_runtime.py b/tests/protocols/internet/test_ipv6_extension_runtime.py index bb9c513ae4..8881bd706e 100644 --- a/tests/protocols/internet/test_ipv6_extension_runtime.py +++ b/tests/protocols/internet/test_ipv6_extension_runtime.py @@ -112,7 +112,11 @@ def test_ipv6_fragment_extension_forbids_direct_payload_accessors(self) -> None: frag = list(ipv6.extension_headers.values())[0] self.assertEqual(str(frame.protochain), 'Ethernet:IPv6:IPv6-Frag:UDP:Raw') - self.assertEqual(frag.info.offset, 543) + # ``Data_IPv6_Frag.offset`` is in octets, matching ``Data_IPv4.offset``: the + # on-wire field is 543 counts of 8 octets (:rfc:`8200#section-4.5`), i.e. the + # 4344th octet of the fragmentable part, which with this fragment's 434 + # octets of payload accounts for the full 4778 octet datagram. + self.assertEqual(frag.info.offset, 4344) self.assertFalse(frag.info.mf) self.assertEqual(frag.info.id, 110308) self.assertEqual(ipv6.info.raw_len, 434) diff --git a/tests/protocols/internet/test_ipv6_extension_unit.py b/tests/protocols/internet/test_ipv6_extension_unit.py index e184c91424..f99f635fb8 100644 --- a/tests/protocols/internet/test_ipv6_extension_unit.py +++ b/tests/protocols/internet/test_ipv6_extension_unit.py @@ -111,7 +111,9 @@ def test_ipv6_frag_index_length_and_make_data(self) -> None: self.assertEqual(proto.__length_hint__(), 8) values = IPv6_Frag._make_data(data) self.assertEqual(values['next'], TransType.TCP) - self.assertEqual(values['offset'], 16) + # ``data.offset`` is in octets while ``make`` takes on-wire 8-octet units, + # so 16 octets is written back as 2 units + self.assertEqual(values['offset'], 2) self.assertEqual(values['mf'], True) self.assertEqual(values['id'], 99) self.assertIn('payload', values) @@ -198,9 +200,15 @@ def test_ipv6_frag_read_make_and_properties(self) -> None: self.assertEqual(proto.protochain, ['TCP']) data = proto.read(extension=True) self.assertEqual(data.next, TransType.TCP) - self.assertEqual(data.offset, 12) + # the schema's ``offset`` is the raw 13-bit on-wire field, a count of + # 8-octet units (:rfc:`8200#section-4.5`); ``Data_IPv6_Frag.offset`` is in + # octets, like ``Data_IPv4.offset``, so 12 units becomes 96 octets + self.assertEqual(proto.__header__.flags['offset'], 12) + self.assertEqual(data.offset, 96) self.assertTrue(data.mf) self.assertEqual(data.id, 0x12345678) + # ``_make_data`` must undo the scaling, so that data -> make round trips + self.assertEqual(IPv6_Frag._make_data(data)['offset'], 12) with mock.patch.object(IPv6_Frag, '_decode_next_layer', return_value='decoded') as decode: self.assertEqual(proto.read(length=12), 'decoded') @@ -1276,6 +1284,80 @@ def test_ipv6_route_schema_selector_and_rpl_post_process_branches(self) -> None: with_dst.post_process({'dst': ip_address('2001:db8::ffff')}) self.assertEqual([str(item) for item in with_dst.ip], ['2001:db8::1', '2001:db8::2']) + def test_option_registries_are_not_clobbered_by_a_nested_enum_registry(self) -> None: + """Option type 0 must resolve to the padding option in every module. + + An :class:`~pcapkit.protocols.schema.schema.EnumSchema` subclass that + does not declare its own ``__enum__`` shares its parent's registry, so + registering *its* subclasses writes into the parent's key space. The + quick-start options are keyed by + :class:`~pcapkit.const.ipv6.qs_function.QSFunction`, whose members are + ``0`` and ``8``, so a shared registry silently overwrites option types + ``0`` (``Pad1``) and ``8`` (``SMF_DPD``) -- and the failure is silent + because a missing ``__enum__`` is not an error, it just merges two + registries whose key spaces happen to overlap. + + :rfc:`8200#section-4.2` defines ``Pad1`` for both the Hop-by-Hop Options + and the Destination Options header, so it must resolve in both. + + """ + from pcapkit.const.ipv6.option import Option as Enum_Option + from pcapkit.const.ipv6.qs_function import QSFunction as Enum_QSFunction + from pcapkit.protocols.schema.internet import hopopt as hopopt_schema + from pcapkit.protocols.schema.internet import ipv6_opts as opts_schema + + # the two enum values that collide with option types when unscoped + self.assertEqual(Enum_QSFunction.Quick_Start_Request.value, 0) + self.assertEqual(Enum_QSFunction.Report_of_Approved_Rate.value, 8) + self.assertEqual(Enum_Option.Pad1.value, 0) + self.assertEqual(Enum_Option.PadN.value, 1) + + for module in (hopopt_schema, opts_schema): + with self.subTest(module=module.__name__.rsplit('.', 1)[-1]): + option = module.Option + quick_start = module.QuickStartOption + + # the quick-start registry must be its own, not the option one + self.assertIsNot(quick_start.__enum__, option.__enum__) + self.assertIn('__enum__', vars(quick_start)) + + # snapshot the registries: they are ``defaultdict``s, so reading a + # missing key through them would insert it + options = dict(option.registry) + functions = dict(quick_start.registry) + + # option type 0 (Pad1) and 1 (PadN) both resolve to PadOption + self.assertIs(options[Enum_Option.Pad1], module.PadOption) + self.assertIs(options[Enum_Option.PadN], module.PadOption) + # option type 8 (SMF_DPD) must not be the quick-start report either + self.assertIsNot(options[Enum_Option.SMF_DPD], module.QuickStartReportOption) + + # and the quick-start functions resolve inside their own registry + self.assertIs(functions[Enum_QSFunction.Quick_Start_Request], + module.QuickStartRequestOption) + self.assertIs(functions[Enum_QSFunction.Report_of_Approved_Rate], + module.QuickStartReportOption) + + def test_ipv6_opts_and_hopopt_option_registries_agree(self) -> None: + """The two modules describe the same option space and must map it alike. + + ``hopopt`` and ``ipv6_opts`` are near-identical by construction, so the + cheapest guard against one drifting from the other is to compare their + registries key for key. The registry collision this catches came about + from exactly one line present in one module and missing in the other. + + """ + from pcapkit.protocols.schema.internet import hopopt as hopopt_schema + from pcapkit.protocols.schema.internet import ipv6_opts as opts_schema + + hopopt_options = dict(hopopt_schema.Option.registry) + opts_options = dict(opts_schema.Option.registry) + self.assertEqual(set(hopopt_options), set(opts_options)) + + for key in sorted(hopopt_options, key=int): + with self.subTest(option=int(key)): + self.assertEqual(hopopt_options[key].__name__, opts_options[key].__name__) + if __name__ == '__main__': unittest.main() diff --git a/tests/protocols/internet/test_ipv6_reassembly_runtime.py b/tests/protocols/internet/test_ipv6_reassembly_runtime.py new file mode 100644 index 0000000000..2dc7e94c3b --- /dev/null +++ b/tests/protocols/internet/test_ipv6_reassembly_runtime.py @@ -0,0 +1,222 @@ +from __future__ import annotations + +import importlib.util +import pathlib +import struct +import tempfile +import unittest + +from tests._support import close_extractor, purge_modules + +RUNTIME_DEPS = ('tbtrim', 'aenum', 'chardet', 'dictdumper') +HAS_RUNTIME = all(importlib.util.find_spec(name) is not None for name in RUNTIME_DEPS) + +#: Ethernet header of every synthetic frame, with an IPv6 ethertype. +ETHER = bytes.fromhex('801f12c9d13d') + bytes.fromhex('000c29aab15e') + b'\x86\xdd' + +#: Link-local endpoints of the synthetic datagrams. +SRC_ADDR = bytes.fromhex('fe80000000000000a423b61d7c9270c6') +DST_ADDR = bytes.fromhex('fe80000000000000821f12fffec9d13d') + +#: Next-header value of the IPv6 Fragment header, and of the payload it carries. +NH_FRAG = 44 +NH_UDP = 17 + +#: PCAP file header: little-endian magic, version 2.4, LINKTYPE_ETHERNET. +PCAP_HEADER = struct.pack(' bytes: + """Build one Ethernet-framed IPv6 fragment. + + Args: + label: IPv6 header flow label, bits 12-31 of the first hextet. + ident: IPv6 Fragment header identification. + offset_units: Fragment offset, in the on-wire 8-octet units of + :rfc:`8200#section-4.5` -- *not* in octets. + more: More Fragments flag. + payload: The fragment's slice of the fragmentable part. + + Returns: + The complete frame, ready to be written into a PCAP file. + + """ + length = 8 + len(payload) + header = struct.pack('>IHBB', (6 << 28) | label, length, NH_FRAG, 64) + SRC_ADDR + DST_ADDR + frag = struct.pack('>BBHI', NH_UDP, 0, (offset_units << 3) | int(more), ident) + return ETHER + header + frag + payload + + +def write_pcap(path: pathlib.Path, frames: list[bytes]) -> str: + """Write ``frames`` into a PCAP file at ``path`` and return it as a :obj:`str`.""" + with open(path, 'wb') as file: + file.write(PCAP_HEADER) + for index, frame in enumerate(frames): + file.write(struct.pack(' None: + purge_modules(['pcapkit']) + self._tmpdir = tempfile.TemporaryDirectory() + self.addCleanup(self._tmpdir.cleanup) + self.tmp_path = pathlib.Path(self._tmpdir.name) + + def _reassemble(self, frames: list[bytes]) -> list: + """Extract ``frames`` with IPv6 reassembly enabled and return the datagrams.""" + import pcapkit + + capture = write_pcap(self.tmp_path / 'synthetic.pcap', frames) + extractor = pcapkit.extract(fin=capture, nofile=True, store=False, + ipv6=True, reassembly=True, reasm_strict=True) + self.addCleanup(close_extractor, extractor) + return list(extractor.reassembly.ipv6) + + @staticmethod + def _payload(datagram) -> bytes: + """Flatten a datagram payload, whether it is complete or fragmented.""" + if isinstance(datagram.payload, bytes): + return datagram.payload + return b''.join(datagram.payload) + + def test_fragment_offsets_are_octets_so_the_payload_reassembles_intact(self) -> None: + """A fragmented datagram must come back as the exact original octets. + + The on-wire Fragment Offset counts 8-octet units, while the reassembly + machinery indexes its datagram buffer in octets. Feeding the raw unit + count straight through places every fragment at an eighth of its true + offset, so the fragments overwrite one another and the datagram is + silently truncated -- and, because the received-bit table is indexed by + ``FO // 8``, still reported ``completed=True``. + + The assertion is on the payload *bytes*: a length-only check passes on a + datagram whose fragments have been scrambled into the right total size. + + """ + # a distinguishable body, so a misplaced fragment shows up as wrong bytes + # and not merely as a wrong length + body = bytes((index * 7 + 11) & 0xFF for index in range(1536)) + first, second = body[:1024], body[1024:] + self.assertEqual(len(first) % 8, 0) # fragments must be 8-octet aligned + + datagrams = self._reassemble([ + fragment(0x12345, 7001, 0, True, first), + fragment(0x12345, 7001, len(first) // 8, False, second), + ]) + + self.assertEqual(len(datagrams), 1) + datagram = datagrams[0] + self.assertTrue(datagram.completed) + self.assertEqual(datagram.index, (1, 2)) + # the payload bytes, not just their count + self.assertEqual(self._payload(datagram), body) + self.assertEqual(len(self._payload(datagram)), 1536) + # the truncation the unscaled offset produced: 128 units + 512 octets + self.assertNotEqual(len(self._payload(datagram)), 640) + + def test_reassembly_keys_on_identification_not_flow_label(self) -> None: + """Two datagrams sharing a flow label must not be merged into one. + + :rfc:`8200#section-4.5` reassembles only from fragments sharing a source + address, destination address and Fragment Identification. The flow label + carries no such guarantee -- :rfc:`6437` permits a constant or zero + label, and zero is the common case -- so keying the reassembly buffer on + it collapses concurrent datagrams into one. + + A single-datagram capture cannot show this: the consequence only appears + once two datagrams share a label, which is why the capture is built here. + + """ + label = 0x12345 + # disjoint alphabets -- every octet of ``alpha`` is even and every octet + # of ``beta`` is odd -- so that a merged payload can be caught by a census + # as well as by comparing bytes, and neither body is a run of one value + alpha = bytes((index * 2) & 0xFE for index in range(1536)) + beta = bytes(((index * 2) & 0xFE) | 1 for index in range(1536)) + self.assertNotEqual(alpha, beta) + + # interleaved A1 B1 A2 B2, both datagrams carrying the same flow label + datagrams = self._reassemble([ + fragment(label, 1001, 0, True, alpha[:1024]), + fragment(label, 2002, 0, True, beta[:1024]), + fragment(label, 1001, 128, False, alpha[1024:]), + fragment(label, 2002, 128, False, beta[1024:]), + ]) + + self.assertEqual(len(datagrams), 2) + for datagram in datagrams: + self.assertTrue(datagram.completed) + by_id = {datagram.id.id: datagram for datagram in datagrams} + + # the identification, not the flow label -- keying on the label reports + # both datagrams under ``id.id`` 0x12345 and loses one of them + self.assertEqual(sorted(by_id), [1001, 2002]) + self.assertNotIn(label, by_id) + + # payload bytes, so that a merge which happens to total the right length + # is still caught + self.assertEqual(self._payload(by_id[1001]), alpha) + self.assertEqual(self._payload(by_id[2002]), beta) + # and neither datagram may hold a single octet from the other: the merge + # this guards against delivered 552 of one body's octets and 181 of the + # other's in one payload, reported ``completed=True`` + self.assertTrue(all(octet % 2 == 0 for octet in self._payload(by_id[1001]))) + self.assertTrue(all(octet % 2 == 1 for octet in self._payload(by_id[2002]))) + + self.assertEqual(by_id[1001].index, (1, 3)) + self.assertEqual(by_id[2002].index, (2, 4)) + + def test_distinct_flow_labels_do_not_split_one_identification(self) -> None: + """Fragments of one datagram must reassemble even if their labels differ. + + The mirror image of the test above, and the case a flow-label-keyed + buffer gets wrong in the opposite direction: the flow label is not + covered by the reassembly key at all, so two fragments carrying the same + identification belong to the same datagram whatever their labels say. + + """ + body = bytes((index * 11 + 3) & 0xFF for index in range(1200)) + + datagrams = self._reassemble([ + fragment(0x11111, 3003, 0, True, body[:1024]), + fragment(0x22222, 3003, 128, False, body[1024:]), + ]) + + self.assertEqual(len(datagrams), 1) + self.assertTrue(datagrams[0].completed) + self.assertEqual(datagrams[0].id.id, 3003) + self.assertEqual(self._payload(datagrams[0]), body) + + def test_three_fragment_datagram_reassembles_in_wire_order(self) -> None: + """More than two fragments, delivered out of order. + + Two fragments can be reassembled correctly by an implementation that + merely appends in arrival order; three delivered out of order cannot. + + """ + body = bytes((index * 13 + 5) & 0xFF for index in range(2048)) + parts = [body[0:1024], body[1024:1536], body[1536:]] + + datagrams = self._reassemble([ + fragment(0, 4004, 128, True, parts[1]), # middle first + fragment(0, 4004, 192, False, parts[2]), # then the last + fragment(0, 4004, 0, True, parts[0]), # and the first last + ]) + + self.assertEqual(len(datagrams), 1) + self.assertTrue(datagrams[0].completed) + self.assertEqual(datagrams[0].id.id, 4004) + self.assertEqual(self._payload(datagrams[0]), body) + # a zero flow label is the common case and must not become the buffer key + self.assertNotEqual(datagrams[0].id.id, 0) diff --git a/tests/protocols/internet/test_ipv6_unit.py b/tests/protocols/internet/test_ipv6_unit.py index 32b1fbde74..4867c3bd31 100644 --- a/tests/protocols/internet/test_ipv6_unit.py +++ b/tests/protocols/internet/test_ipv6_unit.py @@ -289,6 +289,107 @@ def test_ipv6_remaining_decode_and_import_edges(self) -> None: proto.__proto__ = defaultdict(lambda: Raw, {custom_proto: Raw}) self.assertIsInstance(proto._import_next_layer(custom_proto, length=6), Raw) + def test_flow_label_is_read_from_bits_12_to_31(self) -> None: + """The flow label occupies bits 12-31 of the first hextet. + + Per :rfc:`8200#section-3` the IPv6 header opens with Version (4 bits), + Traffic Class (8 bits) and Flow Label (20 bits), so the label starts at + bit 12. The :class:`~pcapkit.corekit.fields.strings.BitField` namespace + spells each subfield as ``(start_bit, length_in_bits)``, so declaring the + label as ``(8, 20)`` reads bits 8-27 instead: the parsed value picks up + the low nibble of the Traffic Class and drops the low nibble of the real + label. + + A zero flow label behind a non-zero Traffic Class is the clearest case, + and also the most common one -- :rfc:`6437` makes a zero label lawful. + + """ + import io + import struct + + from pcapkit.protocols.internet.ipv6 import IPv6 + + src = bytes.fromhex('fe80000000000000a423b61d7c9270c6') + dst = bytes.fromhex('fe80000000000000821f12fffec9d13d') + + cases = ( + (0x00, 0x12345), + (0xff, 0x00000), # a zero label behind a saturated traffic class + (0xff, 0x12345), + (0x00, 0xfffff), + (0xab, 0xcdef0), + ) + for tclass, label in cases: + with self.subTest(tclass=tclass, label=label): + hextet = (6 << 28) | (tclass << 20) | label + raw = struct.pack('>IHBB', hextet, 0, 59, 64) + src + dst + info = IPv6(io.BytesIO(raw), len(raw)).info + + self.assertEqual(info.version, 6) + self.assertEqual(info['class'], tclass) + self.assertEqual(info.label, label) + + # the value the (8, 20) declaration produced: bits 8-27 rather + # than bits 12-31 of the same four octets + bits = f'{hextet:032b}' + self.assertEqual(int(bits[12:32], 2), label) + if int(bits[8:28], 2) != label: + self.assertNotEqual(info.label, int(bits[8:28], 2)) + + def test_flow_label_subfields_tile_the_hextet_without_overlap(self) -> None: + """No two subfields of the first hextet may claim the same bit. + + The ``(start, length)`` namespace is written to in declaration order, so + an overlapping declaration silently corrupts whichever subfield is + written first -- which is how the flow label came to overwrite the low + nibble of the traffic class on the way out as well as misreading it on + the way in. Asserting an exact tiling catches both directions at once. + + """ + from pcapkit.protocols.schema.internet.ipv6 import IPv6 as Schema_IPv6 + + namespace = Schema_IPv6.__fields__['hextet']._namespace + self.assertEqual(namespace, {'version': (0, 4), 'class': (4, 8), 'label': (12, 20)}) + + coverage = [0] * 32 + for start, size in namespace.values(): + for bit in range(start, start + size): + coverage[bit] += 1 + self.assertEqual(coverage, [1] * 32) + + def test_flow_label_survives_a_wire_round_trip(self) -> None: + """Building then parsing a header must preserve traffic class and label. + + ``BitField`` writes and reads through the same namespace, so a wrong bit + range round-trips cleanly whenever the subfields do not overlap and is + therefore invisible to a build-then-parse test on its own. This test + pins the *wire* bytes as well, which is what makes it a real check. + + """ + import io + + from pcapkit.protocols.internet.ipv6 import IPv6 + from pcapkit.protocols.schema.internet.ipv6 import IPv6 as Schema_IPv6 + + schema = Schema_IPv6( + hextet={'version': 6, 'class': 0x2a, 'label': 0x12345}, + length=0, + next=59, + limit=64, + src='2001:db8::1', + dst='2001:db8::2', + payload=b'', + ) + raw = bytes(schema) + + # version 6, traffic class 0x2a, flow label 0x12345 packs as 0x62a12345 + self.assertEqual(raw[:4], bytes.fromhex('62a12345')) + + info = IPv6(io.BytesIO(raw), len(raw)).info + self.assertEqual(info.version, 6) + self.assertEqual(info['class'], 0x2a) + self.assertEqual(info.label, 0x12345) + if __name__ == '__main__': unittest.main() diff --git a/tests/toolkit/test_pcap_unit.py b/tests/toolkit/test_pcap_unit.py index 3ddfa7320d..757d87bb35 100644 --- a/tests/toolkit/test_pcap_unit.py +++ b/tests/toolkit/test_pcap_unit.py @@ -62,8 +62,13 @@ def _make_ipv6(self, *, with_fragment: bool = True): from pcapkit.const.ipv6.extension_header import ExtensionHeader from pcapkit.const.reg.transtype import TransType + # NOTE: ``offset`` here stands in for ``Data_IPv6_Frag.offset``, which is in + # octets (the 8-octet on-wire units are scaled in ``IPv6_Frag.read``), so the + # toolkit passes it through to ``fo`` unscaled. ``id`` is deliberately + # different from the IPv6 header's flow label below, so that a ``bufid`` + # keyed on the wrong one of the two is visible. fragment = types.SimpleNamespace( - info=types.SimpleNamespace(next=TransType.TCP, offset=1, mf=True), + info=types.SimpleNamespace(next=TransType.TCP, offset=8, mf=True, id=4321), ) extension_headers = {ExtensionHeader.IPv6_Frag: fragment} if with_fragment else {} return types.SimpleNamespace( @@ -148,7 +153,13 @@ def test_pcap_ipv4_ipv6_tcp_and_traceflow_helpers(self) -> None: self.assertIsNotNone(v6) assert v6 is not None self.assertEqual(v6.bufid[0], ip_address('2001:db8::1')) - self.assertEqual(v6.bufid[2], 7) + # ``bufid[2]`` is the identification slot -- ``DatagramID.id`` is built from + # it -- so it must carry the IPv6 Fragment header's identification and not + # the IPv6 header's flow label, which does not identify a datagram at all + # (:rfc:`8200#section-4.5`). + self.assertEqual(v6.bufid[2], 4321) + self.assertNotEqual(v6.bufid[2], 7) + self.assertEqual(v6.fo, 8) self.assertTrue(v6.mf) self.assertEqual(v6.header, b'V' * 48) self.assertEqual(bytes(v6.payload), b'v6data') @@ -202,6 +213,10 @@ def test_pcapng_helpers_and_block_to_frame_timestamp_scaling(self) -> None: self.assertIsNotNone(v6) assert v6 is not None self.assertEqual(v6.bufid[0], ip_address('2001:db8::1')) + # the PCAPNG engine keys the buffer the same way the PCAP engine does + self.assertEqual(v6.bufid[2], 4321) + self.assertNotEqual(v6.bufid[2], 7) + self.assertEqual(v6.fo, 8) self.assertEqual(v6.header, b'V' * 48) self.assertIsNone(toolkit.ipv6_reassembly(FakeFrame({}, frame.info))) self.assertIsNone(toolkit.ipv6_reassembly( diff --git a/tests/toolkit/test_scapy_unit.py b/tests/toolkit/test_scapy_unit.py index 90c8b9955a..08925d6dc1 100644 --- a/tests/toolkit/test_scapy_unit.py +++ b/tests/toolkit/test_scapy_unit.py @@ -59,8 +59,10 @@ def _make_ipv6_fragment(self): packet = ( Ether(**self._ether_kwargs()) / + # the fragment identification is deliberately different from the flow + # label, so a ``bufid`` keyed on the wrong one of the two is visible IPv6(src='2001:db8::1', dst='2001:db8::2', fl=7) / - IPv6ExtHdrFragment(nh=6, offset=1, m=1) / + IPv6ExtHdrFragment(nh=6, offset=1, m=1, id=4321) / Raw(b'v6') ) return Ether(bytes(packet)) @@ -149,9 +151,14 @@ def test_ipv4_and_ipv6_reassembly(self) -> None: self.assertEqual(v6.num, 4) self.assertEqual(v6.bufid[0], ip_address('2001:db8::1')) self.assertEqual(v6.bufid[1], ip_address('2001:db8::2')) - self.assertEqual(v6.bufid[2], 7) + # identification, not flow label -- ``bufid[2]`` feeds ``DatagramID.id`` + self.assertEqual(v6.bufid[2], 4321) + self.assertNotEqual(v6.bufid[2], 7) self.assertEqual(v6.bufid[3].value, 6) - self.assertEqual(v6.fo, 1) + # Scapy's ``offset`` is in on-wire 8-octet units, but ``fo`` indexes the + # reassembly datagram buffer in octets, so one unit must become 8 octets + self.assertEqual(ipv6_frag.offset, 1) + self.assertEqual(v6.fo, 8) self.assertTrue(v6.mf) self.assertEqual(bytes(v6.payload), bytes(ipv6_frag.payload))