From d2f6ae5ab159974c89d6c5b80d81ab5826865ba3 Mon Sep 17 00:00:00 2001 From: Jarry Shaw Date: Mon, 14 Sep 2026 00:15:25 -0400 Subject: [PATCH 1/3] reassembly: keep TCP hole descriptors in one coordinate system Hole descriptors carried absolute sequence numbers from the toolkits, the first hole was seeded from a payload length, and submit() sliced a buffer indexed from its own initial sequence number with those bounds. On any capture with a SYN and a realistic ISN every slice landed past the end of the buffer, so an incomplete datagram was dropped silently instead of being returned with completed=False. Descriptors are now absolute and inclusive throughout, submit() is the only place they are converted into buffer offsets, and holes outside a buffer are dropped rather than indexing from its far end. A SYN no longer contributes its own sequence number to the payload, which had been prepending a NUL octet to complete datagrams. Closes #349. --- examples/generators/legacy.py | 43 ++- pcapkit/foundation/reassembly/data/tcp.py | 29 +- pcapkit/foundation/reassembly/tcp.py | 129 +++++-- pcapkit/toolkit/dpkt.py | 4 +- pcapkit/toolkit/pcap.py | 4 +- pcapkit/toolkit/pcapng.py | 4 +- pcapkit/toolkit/scapy.py | 4 +- tests/foundation/reassembly/test_tcp.py | 397 +++++++++++++++++++++- tests/toolkit/test_dpkt_unit.py | 6 +- tests/toolkit/test_pcap_unit.py | 126 ++++++- tests/toolkit/test_scapy_unit.py | 7 + 11 files changed, 671 insertions(+), 82 deletions(-) diff --git a/examples/generators/legacy.py b/examples/generators/legacy.py index 936ba13566..4913736ad3 100644 --- a/examples/generators/legacy.py +++ b/examples/generators/legacy.py @@ -58,10 +58,12 @@ data of its own. Both fixtures therefore keep each direction of each connection to a single HTTP message, and let the peer answer only with pure acknowledgements until that message is complete. -3. A datagram counts as complete when the hole descriptor list is down to two - entries or fewer, which an ordered run of segments plus a FIN or RST - achieves; an out-of-order segment adds a third entry that the missing - segment's arrival then removes. +3. A datagram counts as complete when no hole in the descriptor list falls + inside the octets that direction actually received. An ordered run of + segments leaves only the open-ended hole past the last octet, which is + outside the payload buffer and so does not count; an out-of-order segment + opens a hole in the middle, and the missing segment's arrival closes it + again. 4. The payload of a complete datagram goes to :meth:`pcapkit.protocols.transport.transport.Transport.analyze`, which picks the application protocol from the two port numbers. Port 80 is what @@ -71,17 +73,28 @@ oversight and is not: neither fixture contains a datagram that reassembles *incompletely*, even though ``test_reassembly.py`` and ``test_analyse.py`` both have a branch for one -- a payload that is a tuple of received fragments, and a -``packet`` of :data:`None`. That branch cannot be reached from a realistic -capture. ``submit`` slices the payload buffer with the bounds of each hole, but -those bounds are absolute TCP sequence numbers (``first=tcp_info.seq`` in -``pcapkit/toolkit/pcap.py``) while the buffer is indexed from the start of the -direction's data, so with any real initial sequence number every slice starts -far beyond the end of the buffer, every fragment comes out empty, and ``if -data:`` discards the datagram without a word. Measured on a probe capture with -one segment permanently missing: a realistic initial sequence number yields no -datagram for that direction at all, and only an initial sequence number of zero -produces the tuple these scripts print. A fixture cannot have both a real -handshake and that branch, so it has the real handshake. +``packet`` of :data:`None`. Nothing prevents that branch any more; it is simply +that every stream in these two captures arrives whole. Each direction here is +sent in full and every segment eventually delivered, some of them out of order +and one retransmitted, so the holes that open all close again before the FIN or +RST that submits the buffer. + +It used to be that no capture could reach that branch at all, which is how the +absence started. ``submit`` sliced the payload buffer with the bounds of each +hole, but those bounds were absolute TCP sequence numbers +(``first=tcp_info.seq`` in ``pcapkit/toolkit/pcap.py``) while the buffer is +indexed from the start of the direction's data, so with any real initial +sequence number every slice started far beyond the end of the buffer, every +fragment came out empty, and ``if data:`` discarded the datagram without a word. +That is fixed -- ``submit`` now converts each hole's absolute bounds into +offsets into the buffer it is reading, and decides completeness from whether any +hole survives that conversion -- and GitHub issue #349 records the whole of it. +Regenerating these two fixtures to carry a permanently lost segment as well was +considered and rejected: they are pinned byte-for-byte by the test suite, and a +stream with a hole in it is cheaper to build segment by segment than to read out +of a capture. ``tests/foundation/reassembly/test_tcp.py`` therefore covers the +incomplete branch directly, with realistic initial sequence numbers, while +these captures go on covering the complete one end to end. """ diff --git a/pcapkit/foundation/reassembly/data/tcp.py b/pcapkit/foundation/reassembly/data/tcp.py index 66542e108c..b06950f17a 100644 --- a/pcapkit/foundation/reassembly/data/tcp.py +++ b/pcapkit/foundation/reassembly/data/tcp.py @@ -45,9 +45,12 @@ class Packet(Info): rst: 'bool' #: Payload length, header excluded. len: 'int' - #: This sequence number. + #: Sequence number of the first octet of :attr:`payload`, i.e. the segment's + #: own sequence number. Absolute, not an offset into any payload buffer. first: 'int' - #: Next (wanted) sequence number. + #: Sequence number of the last octet of :attr:`payload`, i.e. ``first + + #: len - 1``. **Inclusive**, so a segment carrying no payload at all has + #: :attr:`last` one below :attr:`first`. last: 'int' #: Raw :obj:`bytes` type header. header: 'bytes' @@ -102,11 +105,21 @@ def __init__(self, completed: 'bool', id: 'DatagramID[_AT]', index: 'tuple[int, @info_final class HoleDescriptor(Info): - """Data model for :term:`TCP ` hole descriptor.""" + """Data model for :term:`TCP ` hole descriptor. - #: Start of hole. + Both bounds are **absolute TCP sequence numbers** and both are + **inclusive**, so a hole covers ``last - first + 1`` octets. They are not + offsets into :attr:`Fragment.raw`: the descriptor list is kept once per + buffer ID, whereas each acknowledgement number's payload buffer carries an + initial sequence number of its own, so only + :meth:`TCP.submit ` -- which + knows which buffer it is looking at -- can convert one to the other. + + """ + + #: Sequence number of the first missing octet. first: 'int' - #: Stop of hole. + #: Sequence number of the last missing octet, inclusive. last: 'int' if TYPE_CHECKING: @@ -119,7 +132,11 @@ class Fragment(Info): #: List of reassembled packets. ind: 'list[int]' - #: ISN of payload buffer. + #: Sequence number of the octet held in ``raw[0]``, i.e. the origin this + #: buffer is indexed from: ``raw[n]`` holds the octet whose sequence number + #: is ``isn + n``. Revised downwards whenever a segment turns up below the + #: data already buffered, so it is not necessarily the connection's own + #: initial sequence number. isn: 'int' #: Length of payload buffer. len: 'int' diff --git a/pcapkit/foundation/reassembly/tcp.py b/pcapkit/foundation/reassembly/tcp.py index 2134ade52f..f42528906c 100644 --- a/pcapkit/foundation/reassembly/tcp.py +++ b/pcapkit/foundation/reassembly/tcp.py @@ -42,6 +42,24 @@ class TCP(Reassembly[Packet, Datagram, BufferID, Buffer]): # Fetch result: >>> result = tcp_reassembly.datagram + Note: + There are two coordinate systems in play here, and keeping them apart + matters. The :term:`hole descriptor list ` of + :rfc:`815` is kept in **absolute TCP sequence numbers**, inclusive of + both bounds, because a hole belongs to the connection's sequence space + for that direction and not to any one payload buffer: the list is held + once per buffer ID, while each acknowledgement number gets a payload + buffer of its own with an initial sequence number of its own, and that + initial sequence number is revised whenever a segment turns up below + the data already buffered. A payload buffer, on the other hand, is + indexed from zero, such that + :attr:`buffer.raw[n] ` + holds the octet with sequence number + :attr:`buffer.isn ` + ``+ n``. :meth:`submit` is therefore the one place that converts + between the two, subtracting that buffer's initial sequence number from + each hole bound. + """ if TYPE_CHECKING: protocol: 'Type[TCP_Protocol]' @@ -77,6 +95,14 @@ def reassembly(self, info: 'Packet') -> 'None': RST = info.rst # Reset Connection Flag (Termination) SYN = info.syn # Synchronise Flag (Establishment) + # Sequence number of the first octet of this segment's payload. A SYN + # occupies a sequence number of its own (:rfc:`793`), so payload sent + # by or after a SYN starts at ``dsn + 1`` rather than at ``dsn``. + # Without this the octet the SYN spends becomes a zero byte at the head + # of the payload buffer, and every complete datagram of a connection + # whose handshake was captured comes back one octet too long. + PSN = DSN + 1 if SYN else DSN + # when SYN is set, reset buffer of existing session if SYN and BUFID in self._buffer: self._dtgram.extend( @@ -88,7 +114,11 @@ def reassembly(self, info: 'Packet') -> 'None': self._buffer[BUFID] = Buffer( hdl=[ HoleDescriptor( - first=info.len, + # everything from the octet after this segment onwards + # is still missing -- in absolute sequence numbers, so + # that the bound stays valid for every payload buffer + # under this buffer ID + first=PSN + info.len, last=sys.maxsize, ), ], @@ -98,7 +128,7 @@ def reassembly(self, info: 'Packet') -> 'None': ind=[ info.num, ], - isn=info.dsn, + isn=PSN, len=info.len, raw=info.payload, ), @@ -111,7 +141,7 @@ def reassembly(self, info: 'Packet') -> 'None': ind=[ info.num, ], - isn=info.dsn, + isn=PSN, len=info.len, raw=info.payload, ) @@ -126,18 +156,18 @@ def reassembly(self, info: 'Packet') -> 'None': # record fragment payload ISN = self._buffer[BUFID].ack[ACK].isn # Initial Sequence Number RAW = self._buffer[BUFID].ack[ACK].raw # Raw Payload Data - if DSN >= ISN: # if fragment goes after existing payload + if PSN >= ISN: # if fragment goes after existing payload LEN = self._buffer[BUFID].ack[ACK].len - GAP = DSN - (ISN + LEN) # gap length between payloads + GAP = PSN - (ISN + LEN) # gap length between payloads if GAP >= 0: # if fragment goes after existing payload RAW += bytearray(GAP) + info.payload else: # if fragment partially overlaps existing payload - RAW[DSN - ISN:DSN - ISN + info.len] = info.payload + RAW[PSN - ISN:PSN - ISN + info.len] = info.payload else: # if fragment exceeds existing payload LEN = info.len - GAP = ISN - (DSN + LEN) # gap length between payloads + GAP = ISN - (PSN + LEN) # gap length between payloads self._buffer[BUFID].ack[ACK].__update__( - isn=DSN, + isn=PSN, ) if GAP >= 0: # if fragment exceeds existing payload RAW = info.payload + bytearray(GAP) + RAW @@ -151,28 +181,36 @@ def reassembly(self, info: 'Packet') -> 'None': ) # update hole descriptor list - HDL = self._buffer[BUFID].hdl # HDL alias - for (index, hole) in enumerate(HDL): # step one - if info.first > hole.last: # step two - continue - if info.last < hole.first: # step three - continue - del HDL[index] # step four - if info.first > hole.first: # step five - new_hole = HoleDescriptor( - first=hole.first, - last=info.first - 1, - ) - HDL.insert(index, new_hole) - index += 1 - if info.last < hole.last and not FIN and not RST: # step six - new_hole = HoleDescriptor( - first=info.last + 1, - last=hole.last - ) - HDL.insert(index, new_hole) - break # step seven - #self._buffer[BUFID].hdl = HDL # update HDL + # + # A segment carrying no payload -- a bare acknowledgement, SYN, FIN + # or RST -- fills no hole, so it must not be run through the + # :rfc:`815` algorithm: its ``last`` lies one below its ``first``, + # and letting that through would split whichever hole contains it + # into two adjacent holes covering the very same octets, growing + # the list without bound on a long-lived connection. + if info.len > 0: + HDL = self._buffer[BUFID].hdl # HDL alias + for (index, hole) in enumerate(HDL): # step one + if info.first > hole.last: # step two + continue + if info.last < hole.first: # step three + continue + del HDL[index] # step four + if info.first > hole.first: # step five + new_hole = HoleDescriptor( + first=hole.first, + last=info.first - 1, + ) + HDL.insert(index, new_hole) + index += 1 + if info.last < hole.last and not FIN and not RST: # step six + new_hole = HoleDescriptor( + first=info.last + 1, + last=hole.last + ) + HDL.insert(index, new_hole) + break # step seven + #self._buffer[BUFID].hdl = HDL # update HDL # when FIN/RST is set, submit buffer of this session if FIN or RST: @@ -196,17 +234,34 @@ def submit(self, buf: 'Buffer', *, bufid: 'BufferID') -> 'list[Datagram]': # ty # check through every buffer with ACK for (ack, buffer) in buf.ack.items(): + # Translate the hole descriptor list, which is kept in absolute + # sequence numbers for the whole direction, into offsets into this + # payload buffer, which is indexed from its own initial sequence + # number. Holes lying wholly outside this buffer -- the open-ended + # one past the last octet received, and any belonging to a + # different acknowledgement number's data -- drop out here; those + # straddling an edge are clipped to it rather than being allowed to + # index from the far end of the buffer as a negative bound would. + length = len(buffer.raw) + holes = [] # type: list[tuple[int, int]] + for hole in HDL: + start = hole.first - buffer.isn # inclusive lower bound + stop = hole.last - buffer.isn + 1 # exclusive upper bound + if stop <= 0 or start >= length: + continue # hole misses this buffer + holes.append((max(start, 0), min(stop, length))) + holes.sort() + # if this buffer is not implemented # go through every hole and extract received payload - if len(HDL) > 2 and self._flag_s: + if holes and self._flag_s: data = [] # type: list[bytes] - start = stop = 0 - for hole in HDL: - stop = hole.first - byte = buffer.raw[start:stop] - start = hole.last + 1 + start = 0 + for (hole_start, hole_stop) in holes: + byte = buffer.raw[start:hole_start] if byte: # strip empty payload - data.append(byte) + data.append(bytes(byte)) + start = max(start, hole_stop) byte = buffer.raw[start:] if byte: # strip empty payload data.append(bytes(byte)) diff --git a/pcapkit/toolkit/dpkt.py b/pcapkit/toolkit/dpkt.py index 2844802d18..d3d04ebee6 100644 --- a/pcapkit/toolkit/dpkt.py +++ b/pcapkit/toolkit/dpkt.py @@ -251,8 +251,8 @@ def tcp_reassembly(packet: 'Packet', *, count: 'int' = -1) -> 'TCP_Packet | None fin=bool(int(flags[7])), # finish flag header=tcp.pack()[:tcp.__hdr_len__], # raw bytes type header payload=bytearray(tcp.pack()[tcp.__hdr_len__:]), # raw bytearray type payload - first=tcp.seq, # this sequence number - last=tcp.seq + raw_len, # next (wanted) sequence number + first=tcp.seq, # first sequence number of payload + last=tcp.seq + raw_len - 1, # last sequence number of payload len=raw_len, # payload length, header excludes ) return data diff --git a/pcapkit/toolkit/pcap.py b/pcapkit/toolkit/pcap.py index d0141af57e..b55dfeb545 100644 --- a/pcapkit/toolkit/pcap.py +++ b/pcapkit/toolkit/pcap.py @@ -163,8 +163,8 @@ def tcp_reassembly(frame: 'Frame') -> 'TCP_Packet | None': rst=tcp_info.flags.rst, # reset connection flag header=tcp.packet.header, # raw bytes type header payload=bytearray(tcp.packet.payload), # raw bytearray type payload - first=tcp_info.seq, # this sequence number - last=tcp_info.seq + raw_len, # next (wanted) sequence number + first=tcp_info.seq, # first sequence number of payload + last=tcp_info.seq + raw_len - 1, # last sequence number of payload len=raw_len, # payload length, header excludes ) return data diff --git a/pcapkit/toolkit/pcapng.py b/pcapkit/toolkit/pcapng.py index 8e6d835678..d36a7b8079 100644 --- a/pcapkit/toolkit/pcapng.py +++ b/pcapkit/toolkit/pcapng.py @@ -169,8 +169,8 @@ def tcp_reassembly(frame: 'PCAPNG') -> 'TCP_Packet | None': rst=tcp_info.flags.rst, # reset connection flag header=tcp.packet.header, # raw bytes type header payload=bytearray(tcp.packet.payload), # raw bytearray type payload - first=tcp_info.seq, # this sequence number - last=tcp_info.seq + raw_len, # next (wanted) sequence number + first=tcp_info.seq, # first sequence number of payload + last=tcp_info.seq + raw_len - 1, # last sequence number of payload len=raw_len, # payload length, header excludes ) return data diff --git a/pcapkit/toolkit/scapy.py b/pcapkit/toolkit/scapy.py index 5b578bc183..a769605bba 100644 --- a/pcapkit/toolkit/scapy.py +++ b/pcapkit/toolkit/scapy.py @@ -250,8 +250,8 @@ def tcp_reassembly(packet: 'Packet', *, count: 'int' = -1) -> 'TCP_Packet | None rst=bool(tcp.flags.R), # reset connection flag header=bytes(tcp)[:tcp.dataofs * 4], # raw bytes type header payload=bytearray(bytes(tcp.payload)), # raw bytearray type payload - first=tcp.seq, # this sequence number - last=tcp.seq + raw_len, # next (wanted) sequence number + first=tcp.seq, # first sequence number of payload + last=tcp.seq + raw_len - 1, # last sequence number of payload len=raw_len, # payload length, header excludes ) return data diff --git a/tests/foundation/reassembly/test_tcp.py b/tests/foundation/reassembly/test_tcp.py index 42d2b0f784..75f4cb28b7 100644 --- a/tests/foundation/reassembly/test_tcp.py +++ b/tests/foundation/reassembly/test_tcp.py @@ -5,11 +5,18 @@ import sys import unittest -from tests._support import purge_modules +from tests._support import close_extractor, purge_modules, sample_path RUNTIME_DEPS = ('tbtrim', 'aenum', 'chardet', 'dictdumper') HAS_RUNTIME = all(importlib.util.find_spec(name) is not None for name in RUNTIME_DEPS) +#: Initial sequence number for the coordinate-system tests below. A real +#: connection draws one at random from the whole 32-bit space; the defect those +#: tests cover is invisible at zero, where an absolute sequence number and an +#: offset into a payload buffer happen to be the same number, so none of them +#: uses zero except the one that checks the answer does not depend on the choice. +ISN = 0xC0DE1234 + @unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') class TCPReassemblyTests(unittest.TestCase): @@ -21,9 +28,20 @@ def _bufid(self): def _packet(self, *, num: int, dsn: int, ack: int = 500, payload: bytes = b'', syn: bool = False, fin: bool = False, rst: bool = False, - first: int = 0, last: int | None = None, header: bytes = b'tcp-header'): + first: int | None = None, last: int | None = None, + header: bytes = b'tcp-header'): + """Build a reassembly packet the way the engine toolkits build one. + + ``first`` defaults to ``dsn`` and ``last`` to ``dsn + len(payload) - 1`` + -- absolute sequence numbers, with ``last`` inclusive -- because that is + what every :mod:`pcapkit.toolkit` module now passes. Tests that exercise + the :rfc:`815` interval arithmetic on its own still override both. + + """ from pcapkit.foundation.reassembly.data.tcp import Packet + if first is None: + first = dsn if last is None: last = first + len(payload) - 1 return Packet(self._bufid(), dsn, ack, num, syn, fin, rst, len(payload), @@ -47,8 +65,11 @@ class TestTCP(TCP): TestTCP.register(callback_calls.append) reasm = TestTCP() - reasm(self._packet(num=1, dsn=100, payload=b'hello', syn=True, first=0, last=4)) - reasm(self._packet(num=2, dsn=105, payload=b' world', fin=True, first=5, last=10)) + # A SYN spends a sequence number of its own, so the payload it carries + # starts at 101 and ends at 105; the segment that continues the stream + # therefore opens at 106, not at 105. + reasm(self._packet(num=1, dsn=100, payload=b'hello', syn=True)) + reasm(self._packet(num=2, dsn=106, payload=b' world', fin=True)) datagram, = reasm.datagram self.assertTrue(datagram.completed) @@ -189,5 +210,373 @@ class TestTCP(TCP): bufid=bufid), []) +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TCPReassemblyCoordinateTests(unittest.TestCase): + """Regression tests for GitHub issue #349. + + TCP reassembly keeps its :rfc:`815` hole descriptor list in absolute + sequence numbers, and each payload buffer indexed from its own initial + sequence number. Mixing the two produced three separate wrong answers, none + of which the rest of the suite could see, because they all need an initial + sequence number that is not zero: + + * a stream with a gap in it yielded **no datagram at all**, silently, since + every hole bound landed far past the end of a buffer a few kilobytes long; + * a stream captured without its handshake yielded only its *first* fragment, + since the completeness test counted hole descriptors instead of asking + whether any hole fell inside the data; + * every extracted fragment came out one octet too long, since a hole bound + was computed as the sequence number *after* the segment while the + descriptor list treats its bounds as inclusive. + + A fourth, closely related error surfaced while fixing them: the sequence + number a SYN spends became a NUL octet at the head of the payload, so a + complete datagram did not compare equal to the octets that were sent. + + Every stream here is driven through :class:`~pcapkit.foundation.reassembly.tcp.TCP` + with segment descriptors built exactly as the :mod:`pcapkit.toolkit` modules + build them -- ``first`` the segment's own sequence number and ``last`` the + sequence number of its final payload octet -- so the tests pin the contract + between the toolkits and the reassembler, not just the reassembler. + + """ + + def setUp(self) -> None: + purge_modules(['pcapkit']) + + def _bufid(self): + return (ip_address('192.0.2.1'), 12345, ip_address('198.51.100.2'), 443) + + def _reassembly(self, **kwargs): + """A reassembly object whose payload analyser is a stub. + + The real one picks an application protocol from the port numbers, which + would drag HTTP parsing into tests about sequence arithmetic. + + """ + from pcapkit.foundation.reassembly.tcp import TCP + + class Analyzer: + @classmethod + def analyze(cls, ports: tuple[int, int], payload: bytes) -> bytes: + return payload + + class TestTCP(TCP): + __protocol_type__ = Analyzer + + return TestTCP(**kwargs) + + def _segment(self, *, num: int, seq: int, payload: bytes = b'', ack: int = 1000, + syn: bool = False, fin: bool = False, rst: bool = False): + """One segment, described the way every :mod:`pcapkit.toolkit` describes it.""" + from pcapkit.foundation.reassembly.data.tcp import Packet + + return Packet(self._bufid(), seq, ack, num, syn, fin, rst, len(payload), + seq, seq + len(payload) - 1, b'tcp-header', bytearray(payload)) + + def _stream(self, *, segments, isn: int = ISN, syn: bool = True, + syn_ack: int = 0, teardown: str | None = 'fin'): + """Build one direction of a connection. + + Args: + segments: ``(offset, payload)`` pairs, the offset counted from the + first octet of application data. Repeat a pair to retransmit it, + and reorder the list to deliver out of order. + isn: Initial sequence number, i.e. the sequence number of the SYN. + syn: Whether the capture caught the handshake. + syn_ack: Acknowledgement number on the SYN. A real SYN carries 0 and + so lands in a payload buffer of its own; passing the data + segments' number instead makes the SYN share their buffer, which + is what exposes the octet a SYN spends. + teardown: ``'fin'``, ``'rst'`` or :data:`None` for a stream left + open, which only :meth:`~pcapkit.foundation.reassembly.reassembly.Reassembly.fetch` + will submit. + + Returns: + The segments, in the order given, ready to be fed to the reassembler. + + """ + base = isn + 1 if syn else isn # a SYN spends a sequence number + packets = [] + if syn: + packets.append(self._segment(num=len(packets) + 1, seq=isn, syn=True, + ack=syn_ack)) + for (offset, payload) in segments: + packets.append(self._segment(num=len(packets) + 1, seq=base + offset, + payload=payload)) + if teardown is not None: + end = base + max(offset + len(payload) for (offset, payload) in segments) + packets.append(self._segment(num=len(packets) + 1, seq=end, + fin=teardown == 'fin', rst=teardown == 'rst')) + return packets + + def _run(self, packets, **kwargs): + reasm = self._reassembly(**kwargs) + for packet in packets: + reasm(packet) + return reasm + + ########################################################################## + # An incomplete datagram has to come back, not vanish. + ########################################################################## + + def test_incomplete_datagram_survives_a_realistic_initial_sequence_number(self) -> None: + """A permanently missing segment gives ``completed=False``, not silence. + + Two of the five segments never arrive. Before the fix this produced no + datagram whatsoever: the hole bounds were around 3.2 billion while the + payload buffer was 50 octets long, so every slice came back empty and + ``submit`` dropped the datagram at its ``if data:`` guard without a + word. + + """ + # 10..19 and 30..39 are lost for good + packets = self._stream(segments=[(0, b'A' * 10), (20, b'C' * 10), (40, b'E' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertFalse(datagram.completed) + self.assertIsNone(datagram.packet) + self.assertEqual(datagram.payload, (b'A' * 10, b'C' * 10, b'E' * 10)) + self.assertEqual([len(fragment) for fragment in datagram.payload], [10, 10, 10]) + self.assertEqual(datagram.index, (2, 3, 4, 5)) + self.assertEqual(datagram.header, b'tcp-header') + + def test_the_answer_does_not_depend_on_the_initial_sequence_number(self) -> None: + """The same stream reassembles the same way whatever ISN it uses. + + This is the property the defect broke: at zero the two coordinate + systems coincide and the fragments came out one octet long each too + long; anywhere else the datagram disappeared entirely. + + """ + segments = [(0, b'A' * 10), (20, b'C' * 10), (40, b'E' * 10)] + expected = (False, (b'A' * 10, b'C' * 10, b'E' * 10)) + + for isn in (0, 1, 0x1000, ISN, 0xFFFF0000): + with self.subTest(isn=isn): + reasm = self._run(self._stream(segments=segments, isn=isn)) + datagram, = reasm.datagram + self.assertEqual((datagram.completed, datagram.payload), expected) + + def test_a_capture_without_the_handshake_reports_every_fragment(self) -> None: + """A stream picked up mid-flight still reports all of its fragments. + + Before the fix this returned a *partial* answer rather than nothing -- + one fragment out of three -- because the completeness test counted hole + descriptors (``len(HDL) > 2``) rather than asking whether a hole fell + inside the received data. Without a SYN the list is one entry shorter, + so a two-gap stream slipped under the threshold. + + """ + packets = self._stream(syn=False, + segments=[(0, b'A' * 10), (20, b'C' * 10), (40, b'E' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertFalse(datagram.completed) + self.assertEqual(datagram.payload, (b'A' * 10, b'C' * 10, b'E' * 10)) + + def test_a_single_gap_is_enough_to_make_a_datagram_incomplete(self) -> None: + """One hole, not three, and the fragments either side of it come back.""" + packets = self._stream(segments=[(0, b'A' * 10), (20, b'C' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertFalse(datagram.completed) + self.assertIsNone(datagram.packet) + self.assertEqual(datagram.payload, (b'A' * 10, b'C' * 10)) + + ########################################################################## + # A complete datagram has to be the octets that were sent. + ########################################################################## + + def test_complete_datagram_reassembles_byte_exactly(self) -> None: + """The reassembled payload equals the concatenation of what was sent.""" + segments = [(0, b'A' * 10), (10, b'B' * 1448), (1458, b'C' * 7)] + reasm = self._run(self._stream(segments=segments)) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertIsInstance(datagram.payload, bytes) + self.assertEqual(datagram.payload, b''.join(payload for (_, payload) in segments)) + self.assertEqual(len(datagram.payload), 1465) + self.assertEqual(datagram.packet, datagram.payload) + + def test_a_syn_sharing_a_payload_buffer_spends_no_payload_octet(self) -> None: + """The sequence number a SYN occupies must not become a NUL octet. + + A SYN carries no payload but does consume a sequence number (:rfc:`793`), + so the first octet of application data sits at ``isn + 1``. A real SYN + acknowledges nothing and therefore gets a payload buffer of its own, + which hid this; give it the data segments' acknowledgement number -- as + the reproduction in the issue does, and as happens whenever the peer has + already sent something -- and seeding that buffer at ``isn`` instead of + ``isn + 1`` prepends a zero octet to the datagram. + + """ + sent = b'GET / HTTP/1.1\r\nHost: example.com\r\n\r\n' + reasm = self._run(self._stream(syn_ack=1000, segments=[(0, sent)])) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertEqual(datagram.payload, sent) + self.assertNotEqual(datagram.payload[:1], b'\x00') + + ########################################################################## + # Out-of-order delivery and retransmission. + ########################################################################## + + def test_out_of_order_and_retransmitted_delivery_reassembles_byte_exactly(self) -> None: + """Four segments delivered 3, 1, 3, 4, 2 still give the sent octets.""" + first, second, third, fourth = ((0, b'A' * 10), (10, b'B' * 10), + (20, b'C' * 10), (30, b'D' * 10)) + packets = self._stream(segments=[third, first, third, fourth, second]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertEqual(datagram.payload, b'A' * 10 + b'B' * 10 + b'C' * 10 + b'D' * 10) + + def test_a_hole_closes_when_the_lost_segment_is_retransmitted(self) -> None: + """A gap opened by a late segment is closed by the retransmission.""" + packets = self._stream(segments=[(0, b'A' * 10), (20, b'C' * 10), (10, b'B' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertEqual(datagram.payload, b'A' * 10 + b'B' * 10 + b'C' * 10) + + def test_out_of_order_delivery_around_a_gap_that_is_never_filled(self) -> None: + """Out of order, retransmitted, *and* one segment lost for good. + + The two fragments that survive are the contiguous runs either side of + the hole, so the third and fourth segments come back as one 20-octet + fragment rather than as two. + + """ + packets = self._stream(segments=[(20, b'C' * 10), (0, b'A' * 10), + (20, b'C' * 10), (30, b'D' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertFalse(datagram.completed) + self.assertEqual(datagram.payload, (b'A' * 10, b'C' * 10 + b'D' * 10)) + self.assertEqual([len(fragment) for fragment in datagram.payload], [10, 20]) + + ########################################################################## + # The two coordinate systems, inspected directly. + ########################################################################## + + def test_hole_descriptor_bounds_are_absolute_sequence_numbers(self) -> None: + """The hole list is in sequence space; the payload buffer is not. + + Asserted on the live buffer rather than through a datagram, because this + is the invariant the defect broke: the list was seeded with + ``first=info.len`` -- a payload length -- and then extended with absolute + sequence numbers, so one descriptor held one of each. + + """ + packets = self._stream(segments=[(0, b'A' * 10), (20, b'C' * 10)], teardown=None) + reasm = self._run(packets) + + buffer = reasm._buffer[self._bufid()] + self.assertEqual([(hole.first, hole.last) for hole in buffer.hdl], + [(ISN + 11, ISN + 20), (ISN + 31, sys.maxsize)]) + + # the SYN acknowledges nothing, so it gets a payload buffer of its own + # and the data lands in the one keyed by the data segments' ACK + fragment = buffer.ack[1000] + self.assertEqual(fragment.isn, ISN + 1) + self.assertEqual(len(fragment.raw), 30) + + # raw[n] holds the octet whose sequence number is isn + n, which is the + # conversion submit() applies -- and the hole's own octets are the ones + # the buffer never received + hole = buffer.hdl[0] + self.assertEqual(bytes(fragment.raw[hole.first - fragment.isn: + hole.last - fragment.isn + 1]), + bytes(10)) + + def test_payload_free_segments_leave_the_hole_list_alone(self) -> None: + """A bare acknowledgement fills no hole, so it must not split one. + + Its ``last`` is one below its ``first``, and running that through the + :rfc:`815` algorithm splits whichever hole contains it into two adjacent + holes covering the very same octets -- unbounded list growth on a + long-lived connection, for no change in what is missing. + + """ + base = ISN + 1 + reasm = self._run([ + self._segment(num=1, seq=ISN, syn=True, ack=0), + self._segment(num=2, seq=base, payload=b'A' * 10), + self._segment(num=3, seq=base + 20, payload=b'C' * 10), + ]) + + before = [(hole.first, hole.last) for hole in reasm._buffer[self._bufid()].hdl] + self.assertEqual(before, [(base + 10, base + 19), (base + 30, sys.maxsize)]) + + # one past the data, one *inside* the hole, and one repeated + for (num, seq) in ((4, base + 30), (5, base + 15), (6, base + 30)): + reasm(self._segment(num=num, seq=seq)) + + after = [(hole.first, hole.last) for hole in reasm._buffer[self._bufid()].hdl] + self.assertEqual(after, before) + self.assertEqual(len(reasm._buffer[self._bufid()].ack[1000].raw), 30) + + ########################################################################## + # End to end, through the extractor, on a committed capture. + ########################################################################## + + def test_sample_capture_reassembles_every_stream_byte_exactly(self) -> None: + """``test.pcap`` through :func:`~pcapkit.interface.extract`. + + The capture has out-of-order segments, a retransmission, and a segment + recovered only after a later one, all with realistic initial sequence + numbers -- so it exercises the whole path rather than the reassembler + alone. Each datagram is compared against the octets rebuilt + independently from the frames it names, by placing each segment's + payload at its own sequence number, which is what makes the comparison + a check rather than a restatement. + + """ + from pcapkit import extract + + extractor = extract(fin=sample_path('test.pcap'), fout='/tmp/out', format='tree', + store=True, nofile=True, tcp=True, reassembly=True, + reasm_strict=True) + self.addCleanup(close_extractor, extractor) + + frames = {frame.info.number: frame for frame in extractor.frame} + datagrams = extractor.reassembly.tcp + self.assertEqual(len(datagrams), 4) + + for datagram in datagrams: + with self.subTest(index=datagram.index): + segments = [] + for number in datagram.index: + tcp = frames[number]['TCP'] + payload = bytes(tcp.packet.payload) + if payload: + segments.append((tcp.info.seq, payload)) + + origin = min(seq for (seq, _) in segments) + expected = bytearray(max(seq - origin + len(payload) + for (seq, payload) in segments)) + for (seq, payload) in segments: + expected[seq - origin:seq - origin + len(payload)] = payload + + self.assertTrue(datagram.completed) + self.assertIsNotNone(datagram.packet) + self.assertEqual(datagram.payload, bytes(expected)) + + # the four messages the fixture is built around, by length and opening + self.assertEqual(sorted(len(datagram.payload) for datagram in datagrams), + [110, 269, 3642, 4587]) + self.assertEqual(sorted(bytes(datagram.payload[:4]) for datagram in datagrams), + [b'GET ', b'HTTP', b'HTTP', b'POST']) + + if __name__ == '__main__': unittest.main() diff --git a/tests/toolkit/test_dpkt_unit.py b/tests/toolkit/test_dpkt_unit.py index 9c47547d24..7324376bfc 100644 --- a/tests/toolkit/test_dpkt_unit.py +++ b/tests/toolkit/test_dpkt_unit.py @@ -153,8 +153,12 @@ def test_tcp_reassembly_and_traceflow_accept_dpkt_data_payload_tcp(self) -> None self.assertEqual(tcp.bufid[3], 80) self.assertTrue(tcp.syn) self.assertFalse(tcp.fin) + # ``last`` is the sequence number of the payload's last octet, not of + # the one after it, so ``b'data'`` at sequence number 10 ends at 13 + self.assertEqual(tcp.len, 4) self.assertEqual(tcp.first, 10) - self.assertEqual(tcp.last, 14) + self.assertEqual(tcp.last, 13) + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) flow = toolkit.tcp_traceflow(packet, 50.25, data_link=LinkType.ETHERNET, count=9) self.assertIsNotNone(flow) diff --git a/tests/toolkit/test_pcap_unit.py b/tests/toolkit/test_pcap_unit.py index 82004adf99..3ddfa7320d 100644 --- a/tests/toolkit/test_pcap_unit.py +++ b/tests/toolkit/test_pcap_unit.py @@ -78,30 +78,36 @@ def _make_ipv6(self, *, with_fragment: bool = True): ), ) - def _make_tcp(self): + def _make_tcp(self, *, seq: int = 10, payload: bytes = b'tcpdata', + syn: bool = True, fin: bool = False, rst: bool = True): return types.SimpleNamespace( info=types.SimpleNamespace( srcport=port(1234), dstport=port(80), ack=5, - seq=10, - flags=flags(syn=True, fin=False, rst=True), + seq=seq, + flags=flags(syn=syn, fin=fin, rst=rst), ), - packet=types.SimpleNamespace(header=b'T' * 20, payload=b'tcpdata'), + packet=types.SimpleNamespace(header=b'T' * 20, payload=payload), ) def _make_pcap_frame(self, *, include_tcp: bool = True, - ipv4_df: bool = False) -> FakeFrame: + ipv4_df: bool = False, seq: int = 10, + payload: bytes = b'tcpdata', number: int = 9, + syn: bool = True, fin: bool = False, + rst: bool = True) -> FakeFrame: layers = { 'IPv4': self._make_ipv4(df=ipv4_df), } layers['IP'] = layers['IPv4'] if include_tcp: - layers['TCP'] = self._make_tcp() - return FakeFrame(layers, types.SimpleNamespace(number=9, time_epoch='50.25')) + layers['TCP'] = self._make_tcp(seq=seq, payload=payload, + syn=syn, fin=fin, rst=rst) + return FakeFrame(layers, types.SimpleNamespace(number=number, time_epoch='50.25')) def _make_pcapng_frame(self, *, include_tcp: bool = True, - ipv4_df: bool = False): + ipv4_df: bool = False, seq: int = 10, + payload: bytes = b'tcpdata'): from pcapkit.const.reg.linktype import LinkType layers = { @@ -109,7 +115,7 @@ def _make_pcapng_frame(self, *, include_tcp: bool = True, } layers['IP'] = layers['IPv4'] if include_tcp: - layers['TCP'] = self._make_tcp() + layers['TCP'] = self._make_tcp(seq=seq, payload=payload) return FakeFrame( layers, types.SimpleNamespace(number=11, timestamp_epoch=60.5, @@ -159,7 +165,13 @@ def test_pcap_ipv4_ipv6_tcp_and_traceflow_helpers(self) -> None: self.assertEqual(tcp.bufid[3], 80) self.assertTrue(tcp.syn) self.assertTrue(tcp.rst) - self.assertEqual(tcp.last, 17) + # ``first``/``last`` bound the payload in absolute sequence numbers and + # ``last`` is inclusive, so the two describe exactly ``len`` octets -- + # ``seq`` is 10 and the payload is the seven octets of ``b'tcpdata'``. + self.assertEqual(tcp.len, 7) + self.assertEqual(tcp.first, 10) + self.assertEqual(tcp.last, 16) + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) self.assertIsNone(toolkit.tcp_reassembly(self._make_pcap_frame(include_tcp=False))) flow = toolkit.tcp_traceflow(frame, data_link=LinkType.ETHERNET) @@ -200,7 +212,12 @@ def test_pcapng_helpers_and_block_to_frame_timestamp_scaling(self) -> None: self.assertIsNotNone(tcp) assert tcp is not None self.assertEqual(tcp.bufid[1], 1234) - self.assertEqual(tcp.last, 17) + # same inclusive bounds as the PCAP engine above -- the two toolkits + # must agree, or a stream reassembles differently per capture format + self.assertEqual(tcp.len, 7) + self.assertEqual(tcp.first, 10) + self.assertEqual(tcp.last, 16) + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) self.assertIsNone(toolkit.tcp_reassembly(self._make_pcapng_frame(include_tcp=False))) flow = toolkit.tcp_traceflow(frame) @@ -224,6 +241,93 @@ def test_pcapng_helpers_and_block_to_frame_timestamp_scaling(self) -> None: pcap_frame_ns = toolkit.block2frame(block, nanosecond=True) self.assertEqual(pcap_frame_ns.frame_info.ts_usec, 250000000) + def test_tcp_reassembly_descriptor_bounds_are_absolute_and_inclusive(self) -> None: + """The segment descriptor handed to the reassembler (GitHub issue #349). + + ``first`` and ``last`` are absolute TCP sequence numbers bounding the + payload, and ``last`` is the sequence number of its final octet rather + than of the octet after it. Both matter: the reassembler holds its + :rfc:`815` hole descriptor list in the same absolute space and treats + both bounds as inclusive, so a descriptor that is relative, or exclusive + at the top end, silently corrupts the hole list -- and this is not + visible at a sequence number of zero, which is why a realistic one is + used here. + + Both the PCAP and the PCAP-NG toolkit are checked, and against each + other, since they feed the one reassembler and a stream must not + reassemble differently depending on the capture format it arrived in. + + """ + from pcapkit.toolkit import pcap, pcapng + + # a realistic initial sequence number, plus the 1448 octets an Ethernet + # path carries with a timestamp option in the TCP header + seq = 0xC0DE1234 + payload = bytes(range(256)) * 5 + bytes(168) + self.assertEqual(len(payload), 1448) + + for (name, toolkit, frame) in ( + ('pcap', pcap, self._make_pcap_frame(seq=seq, payload=payload)), + ('pcapng', pcapng, self._make_pcapng_frame(seq=seq, payload=payload)), + ): + with self.subTest(engine=name): + tcp = toolkit.tcp_reassembly(frame) + self.assertIsNotNone(tcp) + assert tcp is not None + + self.assertEqual(tcp.dsn, seq) + self.assertEqual(tcp.len, 1448) + self.assertEqual(tcp.first, seq) # absolute, not an offset + self.assertEqual(tcp.last, seq + 1447) # inclusive, not seq + 1448 + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) + self.assertEqual(bytes(tcp.payload), payload) + + # a segment carrying no payload bounds an empty range, i.e. ``last`` one + # below ``first``, rather than claiming the octet it does not carry + empty = pcap.tcp_reassembly(self._make_pcap_frame(seq=seq, payload=b'')) + self.assertIsNotNone(empty) + assert empty is not None + self.assertEqual(empty.len, 0) + self.assertEqual((empty.first, empty.last), (seq, seq - 1)) + + def test_tcp_reassembly_descriptor_drives_the_reassembler(self) -> None: + """The descriptor the toolkit builds reassembles to the octets sent. + + Guards the seam rather than either side of it: the toolkit and the + reassembler each looked self-consistent while disagreeing about what + ``first`` and ``last`` meant. + + """ + from pcapkit.foundation.reassembly.tcp import TCP + from pcapkit.toolkit import pcap + + class Analyzer: + @classmethod + def analyze(cls, ports: tuple[int, int], payload: bytes) -> bytes: + return payload + + class TestTCP(TCP): + __protocol_type__ = Analyzer + + seq = 0xC0DE1234 + reasm = TestTCP() + + # three data segments and then a bare RST at the sequence number after + # the data, which is what makes the reassembler submit the buffer + deliveries = ((1, 0, b'first-', False), (2, 6, b'second-', False), + (3, 13, b'third', False), (4, 18, b'', True)) + for (number, offset, payload, rst) in deliveries: + frame = self._make_pcap_frame(seq=seq + offset, payload=payload, + number=number, syn=False, rst=rst) + packet = pcap.tcp_reassembly(frame) + self.assertIsNotNone(packet) + assert packet is not None + reasm(packet) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertEqual(datagram.payload, b'first-second-third') + if __name__ == '__main__': unittest.main() diff --git a/tests/toolkit/test_scapy_unit.py b/tests/toolkit/test_scapy_unit.py index 57236d485e..90c8b9955a 100644 --- a/tests/toolkit/test_scapy_unit.py +++ b/tests/toolkit/test_scapy_unit.py @@ -178,6 +178,13 @@ def test_tcp_reassembly_and_traceflow(self) -> None: self.assertFalse(tcp.rst) self.assertEqual(tcp.header, bytes(tcp_layer)[:tcp_layer.dataofs * 4]) self.assertEqual(bytes(tcp.payload), bytes(tcp_layer[Raw])) + # ``first``/``last`` are absolute sequence numbers bounding the payload + # inclusively, so they span exactly ``len`` octets -- the same + # convention every other engine's toolkit has to use, since they all + # feed the one reassembler + self.assertEqual(tcp.first, tcp_layer.seq) + self.assertEqual(tcp.last, tcp_layer.seq + tcp.len - 1) + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) v6_tcp = toolkit.tcp_reassembly(self._make_ipv6_tcp_packet(), count=9) self.assertIsNotNone(v6_tcp) assert v6_tcp is not None From 43e991c51c4d50fd5e5107d757a8448f2275daae Mon Sep 17 00:00:00 2001 From: Jarry Shaw Date: Mon, 14 Sep 2026 00:15:25 -0400 Subject: [PATCH 2/3] tests: cover the extraction pipeline end to end The integration tier was three modules; the real end-to-end spectrums lived only as the assertion-free demonstrations in examples/legacy_smoke. Added seven modules covering report formats on disk, manual frame iteration, TCP and IP reassembly, flow tracing, engine parity, PCAP-NG including both byte orders, and the CLI as a subprocess. Assertions are on content rather than liveness: reassembled bodies match their own Content-Length, and the 331 traced flows partition all 1117 frames of http.pcap exactly once. Four tests are skipped against defects they would otherwise pin, each naming the file and line to fix. --- tests/integration/_helpers.py | 138 ++++++++ tests/integration/test_cli_subprocess.py | 151 ++++++++ tests/integration/test_engine_parity.py | 124 +++++++ tests/integration/test_frame_iteration.py | 128 +++++++ tests/integration/test_output_formats.py | 227 ++++++++++++ tests/integration/test_pcapng_end_to_end.py | 198 +++++++++++ .../integration/test_reassembly_end_to_end.py | 326 ++++++++++++++++++ .../integration/test_traceflow_end_to_end.py | 125 +++++++ 8 files changed, 1417 insertions(+) create mode 100644 tests/integration/_helpers.py create mode 100644 tests/integration/test_cli_subprocess.py create mode 100644 tests/integration/test_engine_parity.py create mode 100644 tests/integration/test_frame_iteration.py create mode 100644 tests/integration/test_output_formats.py create mode 100644 tests/integration/test_pcapng_end_to_end.py create mode 100644 tests/integration/test_reassembly_end_to_end.py create mode 100644 tests/integration/test_traceflow_end_to_end.py diff --git a/tests/integration/_helpers.py b/tests/integration/_helpers.py new file mode 100644 index 0000000000..d31499562e --- /dev/null +++ b/tests/integration/_helpers.py @@ -0,0 +1,138 @@ +# -*- coding: utf-8 -*- +"""Scaffolding shared by the end-to-end integration modules. + +The modules next to this one drive :mod:`pcapkit` the way a user does -- a +capture in, a report file or a reassembled datagram out -- so they all need the +same three things: the dependency gates that decide whether the runtime is +usable at all, a private scratch directory to write reports into, and a couple +of readers for the report formats. Those live here rather than in +:mod:`tests._support`, which is shared with the unit tiers and is deliberately +free of anything this specific. + +Nothing in this file is collected by :program:`pytest`: ``python_files`` in +:file:`pyproject.toml` is ``test_*.py``. + +""" +from __future__ import annotations + +import collections +import importlib.util +import json +import pathlib +import re +import tempfile +import unittest +import xml.etree.ElementTree as ET +from typing import TYPE_CHECKING + +from tests._support import close_extractor, purge_modules + +if TYPE_CHECKING: + from typing import Any + + from pcapkit.foundation.extraction import Extractor + +__all__ = [ + 'HAS_RUNTIME', 'HAS_DPKT', 'HAS_SCAPY', 'HAS_EMOJI', + 'EndToEndTestCase', + 'read_json', 'plist_keys', 'section_counts', 'report_stems', +] + +#: Packages :mod:`pcapkit` needs before it can parse anything at all. +RUNTIME_DEPS = ('tbtrim', 'aenum', 'chardet', 'dictdumper') +#: Whether the runtime dependencies are importable. +HAS_RUNTIME = all(importlib.util.find_spec(name) is not None for name in RUNTIME_DEPS) +#: Whether the :mod:`dpkt` extraction engine can be selected. +HAS_DPKT = importlib.util.find_spec('dpkt') is not None +#: Whether the :mod:`scapy` extraction engine can be selected. +HAS_SCAPY = importlib.util.find_spec('scapy') is not None +#: Whether the command line tool's ``cli`` extra is installed. +HAS_EMOJI = importlib.util.find_spec('emoji') is not None + + +class EndToEndTestCase(unittest.TestCase): + """Base class for the end-to-end modules. + + Gives every test a private temporary directory in :attr:`tmp_path` and an + :meth:`extract` wrapper that closes the input stream on teardown. Captures + under :file:`examples/captures/` are fixtures -- four of them are committed + -- so nothing here ever writes outside :attr:`tmp_path`. + + """ + + @classmethod + def setUpClass(cls) -> None: + """Drop the imported library so the class starts from a clean state. + + The surrounding tiers purge in :meth:`setUp`, i.e. once per test. A + fresh :mod:`pcapkit` import measures at roughly 0.45s on this machine, + which across this tier would cost more than the extractions themselves, + so the purge happens once per class instead. That is equivalent here: + every test below imports :mod:`pcapkit` inside the test method, so none + of them depends on what an earlier test left in :data:`sys.modules`. + + """ + purge_modules(['pcapkit']) + + def setUp(self) -> None: + """Hand the test a private scratch directory.""" + tmpdir = tempfile.TemporaryDirectory(prefix='pcapkit-e2e-') + self.addCleanup(tmpdir.cleanup) + self.tmp_path = pathlib.Path(tmpdir.name) + + def out(self, name: str) -> 'str': + """Absolute path to ``name`` inside this test's scratch directory.""" + return str(self.tmp_path / name) + + def extract(self, **kwargs: 'Any') -> 'Extractor': + """Run :func:`pcapkit.interface.extract`, closing the input on teardown.""" + from pcapkit.interface import extract + + extractor = extract(**kwargs) + self.addCleanup(close_extractor, extractor) + return extractor + + +def read_json(path: 'str') -> 'dict[str, Any]': + """Parse a report written with ``format='json'``.""" + with open(path, encoding='utf-8') as stream: + return json.load(stream) + + +def plist_keys(path: 'str') -> 'list[str]': + """Top-level keys of a report written with ``format='plist'``. + + Read with :mod:`xml.etree.ElementTree` rather than :mod:`plistlib`, because + the dumper emits ```` values that :mod:`plistlib` rejects; see + ``PlistRoundTripTests`` in :mod:`tests.integration.test_output_formats`. + + """ + root = ET.parse(path).getroot() + if root.tag != 'plist': + raise AssertionError(f'not a plist document: {root.tag}') + return [key.text or '' for key in root[0].findall('key')] + + +def section_counts(report: 'dict[str, Any]') -> 'collections.Counter[str]': + """Count the report's top-level sections by kind. + + ``'Frame 1'``, ``'Frame 2'`` and so on collapse to ``'Frame'``, and the + PCAP-NG block sections likewise, so a report can be described by what it + contains rather than by listing every key. + + """ + return collections.Counter(re.sub(r' \d+$', '', key) for key in report) + + +def report_stems(directory: 'str') -> 'list[str]': + """Sorted section names of the reports ``files=True`` wrote into ``directory``. + + Split at the *first* dot on purpose. Per-frame reports are named + ``f'{name}.{ext}'`` where ``ext`` already carries its own leading dot + (``pcapkit/foundation/engines/pcap.py:117`` and ``:156``, and + ``pcapkit/foundation/engines/pcapng.py:311``), so they land on disk as + ``Frame 1..json``. Taking the leading component keeps these assertions + neutral about how many dots there are, rather than pinning that defect. + + """ + return sorted(entry.name.split('.', 1)[0] for entry in pathlib.Path(directory).iterdir()) diff --git a/tests/integration/test_cli_subprocess.py b/tests/integration/test_cli_subprocess.py new file mode 100644 index 0000000000..acc22bfdec --- /dev/null +++ b/tests/integration/test_cli_subprocess.py @@ -0,0 +1,151 @@ +# -*- coding: utf-8 -*- +"""End-to-end runs of the command line tool. + +:file:`tests/cli/test_main.py` exercises :func:`pcapkit.__main__.main` with the +extractor replaced by a stub, so it checks the argument wiring and nothing else. +This module runs the real thing in a subprocess -- the same entry point the +``pcapkit-cli`` console script installs -- against a real capture, and asserts on +the exit status, what it printed, and the report it left behind. + +Every run is given a temporary working directory, so a report written to a +relative path could not touch the repository even if one of these grew a bug. + +""" +from __future__ import annotations + +import json +import pathlib +import subprocess # nosec: B404 +import sys +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_EMOJI, HAS_RUNTIME, EndToEndTestCase + +#: How long a single CLI run is allowed to take. The captures used here are two +#: to six frames, so this is a hang guard rather than a budget. +TIMEOUT = 120 + +#: Path of the installed console script, if it is on this interpreter's path. +CLI_SCRIPT = pathlib.Path(sys.executable).with_name('pcapkit-cli') + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +@unittest.skipUnless(HAS_EMOJI, "the cli extra's 'emoji' dependency is not installed") +class CommandLineTests(EndToEndTestCase): + """``python -m pcapkit``, i.e. ``pcapkit.__main__:main``.""" + + def run_cli(self, *args: 'str', expect: 'int | None' = 0) -> 'subprocess.CompletedProcess[str]': + """Run the CLI in this test's temporary directory.""" + completed = subprocess.run( # nosec: B603 + [sys.executable, '-m', 'pcapkit', *args], + cwd=str(self.tmp_path), capture_output=True, text=True, + timeout=TIMEOUT, check=False, + ) + if expect is not None: + self.assertEqual(completed.returncode, expect, + f'unexpected exit status; stderr was:\n{completed.stderr}') + return completed + + def test_version_flag_reports_the_installed_version(self) -> None: + import pcapkit + + completed = self.run_cli('-V') + + self.assertEqual(completed.stdout.strip(), pcapkit.__version__) + + def test_json_report_is_written_and_every_chain_is_printed(self) -> None: + completed = self.run_cli(sample_path('in.pcap'), '-o', 'report', '-j', '-a', '-v') + + report = self.tmp_path / 'report.json' + self.assertTrue(report.is_file()) + with report.open(encoding='utf-8') as stream: + self.assertEqual(list(json.load(stream)), [ + 'Global Header', 'Frame 1', 'Frame 2', 'Frame 3', + 'Frame 4', 'Frame 5', 'Frame 6', + ]) + + self.assertIn(sample_path('in.pcap'), completed.stdout) + self.assertIn('Frame 1: Ethernet:IPv6:IPv6_ICMP', completed.stdout) + self.assertIn('Frame 6: Ethernet:IPv4:UDP:Raw', completed.stdout) + # ``-o report`` is relative, and the run happens in the temporary + # directory, so the name the tool reports is relative too. + self.assertIn("Report file stored in 'report.json'", completed.stdout) + + def test_tree_report_is_written_without_verbose_output(self) -> None: + completed = self.run_cli(sample_path('arp.pcap'), '-o', 'report', '-t', '-a') + + report = self.tmp_path / 'report.txt' + self.assertTrue(report.is_file()) + self.assertIn('Frame 2', report.read_text(encoding='utf-8')) + self.assertEqual(completed.stdout, '') + + def test_plist_report_is_written(self) -> None: + completed = self.run_cli(sample_path('arp.pcap'), '-o', 'report', '-p', '-a') + + report = self.tmp_path / 'report.plist' + self.assertTrue(report.is_file()) + self.assertTrue(report.read_text(encoding='utf-8').startswith(' None: + self.run_cli(sample_path('arp.pcap'), '-o', 'named', '-f', 'json', '-a') + + self.assertTrue((self.tmp_path / 'named.json').is_file()) + + def test_files_flag_writes_a_directory_of_reports(self) -> None: + completed = self.run_cli(sample_path('arp.pcap'), '-o', 'frames', '-j', '-F', '-v') + + frames = self.tmp_path / 'frames' + self.assertTrue(frames.is_dir()) + self.assertEqual(sorted(entry.name.split('.', 1)[0] for entry in frames.iterdir()), + ['Frame 1', 'Frame 2', 'Global Header']) + self.assertIn('Report files stored in', completed.stdout) + + def test_engine_option_selects_an_alternative_engine(self) -> None: + self.run_cli(sample_path('in.pcap'), '-o', 'report', '-t', '-a', '-E', 'default') + + self.assertTrue((self.tmp_path / 'report.txt').is_file()) + + def test_missing_capture_fails_and_names_the_path(self) -> None: + missing = str(self.tmp_path / 'absent.pcap') + completed = self.run_cli(missing, '-o', 'report', '-j', expect=None) + + self.assertNotEqual(completed.returncode, 0) + self.assertIn('absent.pcap', completed.stderr) + self.assertFalse((self.tmp_path / 'report.json').exists()) + + def test_no_arguments_fails_with_the_usage_message(self) -> None: + completed = self.run_cli(expect=2) + + self.assertIn('pcapkit-cli', completed.stderr) + self.assertIn('input-file-name', completed.stderr) + + def test_unknown_option_fails_with_the_usage_message(self) -> None: + completed = self.run_cli('--no-such-option', expect=2) + + self.assertIn('usage: pcapkit-cli', completed.stderr) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +@unittest.skipUnless(HAS_EMOJI, "the cli extra's 'emoji' dependency is not installed") +@unittest.skipUnless(CLI_SCRIPT.is_file(), 'pcapkit-cli console script is not installed') +class ConsoleScriptTests(EndToEndTestCase): + """The installed ``pcapkit-cli`` script, rather than ``python -m pcapkit``.""" + + def test_console_script_writes_the_same_report(self) -> None: + completed = subprocess.run( # nosec: B603 + [str(CLI_SCRIPT), sample_path('arp.pcap'), '-o', 'report', '-j', '-a'], + cwd=str(self.tmp_path), capture_output=True, text=True, + timeout=TIMEOUT, check=False, + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + report = self.tmp_path / 'report.json' + self.assertTrue(report.is_file()) + with report.open(encoding='utf-8') as stream: + self.assertEqual(list(json.load(stream)), ['Global Header', 'Frame 1', 'Frame 2']) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_engine_parity.py b/tests/integration/test_engine_parity.py new file mode 100644 index 0000000000..395bab755d --- /dev/null +++ b/tests/integration/test_engine_parity.py @@ -0,0 +1,124 @@ +# -*- coding: utf-8 -*- +"""End-to-end parity between the extraction engines. + +Translates :file:`examples/legacy_smoke/test_engine.py`, which ran the same +capture through each engine and wrote four reports nobody compared, into +assertions that the engines agree. + +``pyshark`` is left out on purpose. It needs :program:`tshark`, which is not +installed here, and on Python 3.14 it fails inside its own +``get_event_loop()`` before it reads a byte; +:file:`tests/integration/test_engine_runtime.py` already pins that. The +``pipeline`` and ``server`` engines are commented out in the original script and +are not covered here either. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_DPKT, HAS_RUNTIME, HAS_SCAPY, EndToEndTestCase + +#: Captures every engine is run over, with the frame count each must report. +#: All three are small: the point is agreement, not throughput. +PARITY_CAPTURES = { + 'in.pcap': 6, + 'arp.pcap': 2, + 'tcp.pcap': 7, + 'http6.cap': 26, +} + +#: Engines to compare, with the gate that decides whether each is available. +ENGINES = ( + ('default', HAS_RUNTIME), + ('dpkt', HAS_DPKT), + ('scapy', HAS_SCAPY), +) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class EngineParityTests(EndToEndTestCase): + """The same capture through every available engine.""" + + def available(self) -> 'list[str]': + return [name for name, present in ENGINES if present] + + def test_every_engine_reports_the_same_frame_count(self) -> None: + for capture, expected in PARITY_CAPTURES.items(): + counts = {} + for engine in self.available(): + with self.subTest(capture=capture, engine=engine): + extractor = self.extract(fin=sample_path(capture), nofile=True, + store=False, engine=engine) + counts[engine] = extractor.length + self.assertEqual(extractor.length, expected) + + with self.subTest(capture=capture): + self.assertEqual(set(counts.values()), {expected}) + + def test_every_engine_writes_a_report(self) -> None: + for engine in self.available(): + with self.subTest(engine=engine): + extractor = self.extract(fin=sample_path('in.pcap'), + fout=self.out(f'{engine}-report'), + format='tree', store=False, engine=engine) + + report = self.tmp_path / f'{engine}-report.txt' + self.assertEqual(extractor.output, str(report)) + self.assertTrue(report.is_file()) + self.assertGreater(report.stat().st_size, 0) + self.assertIn('Frame 6', report.read_text(encoding='utf-8')) + + def test_engine_macro_selects_the_same_engine_as_its_name(self) -> None: + # The legacy scripts pass ``engine=pcapkit.PCAPKit`` rather than the + # string, so the macro and the name have to be interchangeable. + import pcapkit + + by_macro = self.extract(fin=sample_path('tcp.pcap'), nofile=True, store=True, + engine=pcapkit.PCAPKit) + by_name = self.extract(fin=sample_path('tcp.pcap'), nofile=True, store=True, + engine='default') + + self.assertEqual(pcapkit.PCAPKit, 'default') + self.assertEqual(by_macro.length, by_name.length) + self.assertEqual([str(frame.protochain) for frame in by_macro.frame], + [str(frame.protochain) for frame in by_name.frame]) + + @unittest.skipUnless(HAS_DPKT, 'dpkt not installed') + def test_dpkt_engine_agrees_on_the_protocol_chain(self) -> None: + # The engines hand back their own packet objects, so the chains are + # spelled differently; the toolkit is what maps one onto the other. + from pcapkit.toolkit.dpkt import packet2chain + + native = self.extract(fin=sample_path('in.pcap'), nofile=True, store=True, + engine='default') + foreign = self.extract(fin=sample_path('in.pcap'), nofile=True, store=True, + engine='dpkt') + + self.assertEqual(str(native.frame[0].protochain), 'Ethernet:IPv6:IPv6_ICMP') + self.assertEqual(packet2chain(foreign.frame[0]), 'Ethernet:IP6:ICMP6') + self.assertEqual(str(native.frame[2].protochain), 'Ethernet:IPv4:TCP') + self.assertEqual(packet2chain(foreign.frame[2]), 'Ethernet:IP:TCP') + + @unittest.skipUnless(HAS_SCAPY, 'scapy not installed') + def test_scapy_engine_returns_the_same_link_layer_bytes(self) -> None: + native = self.extract(fin=sample_path('arp.pcap'), nofile=True, store=True, + engine='default') + foreign = self.extract(fin=sample_path('arp.pcap'), nofile=True, store=True, + engine='scapy') + + self.assertEqual(len(native.frame), len(foreign.frame)) + for number, (mine, theirs) in enumerate(zip(native.frame, foreign.frame), start=1): + with self.subTest(frame=number): + # scapy does not know this capture's link type and hands back one + # opaque ``Raw`` layer holding the whole 60 octet frame, padded to + # the Ethernet minimum. Its first fourteen octets are the Ethernet + # header that the default engine parsed into a layer of its own. + captured = bytes(theirs) + self.assertEqual(len(captured), 60) + self.assertEqual(captured[:14], bytes(mine['Ethernet'].packet.header)) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_frame_iteration.py b/tests/integration/test_frame_iteration.py new file mode 100644 index 0000000000..3d5312300f --- /dev/null +++ b/tests/integration/test_frame_iteration.py @@ -0,0 +1,128 @@ +# -*- coding: utf-8 -*- +"""End-to-end frame-by-frame iteration and the extraction limits. + +Translates :file:`examples/legacy_smoke/test_http.py`, which walked a capture +with ``auto=False`` and printed the frames where ``pcapkit.HTTP in frame`` held, +into assertions about which frames those are. It reads :file:`http6.cap` (26 +frames) rather than the :file:`http.pcap` the script used, because the spectrum +is the ``in`` test and the manual walk, not the size of the capture. + +The extraction limits -- ``layer`` and ``protocol``, which the command line tool +exposes as ``-L`` and ``-P`` -- belong to the same spectrum: they are what stops +that walk short of the application layer. They do not work; see +:class:`ExtractionLimitTests`. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_RUNTIME, EndToEndTestCase + +#: Frames of :file:`http6.cap` that carry an HTTP header block: the request and +#: the response of each of the two connections. +HTTP_FRAMES = (4, 6, 19, 21) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class FrameIterationTests(EndToEndTestCase): + """``auto=False``, i.e. the caller drives the extraction.""" + + def test_manual_iteration_yields_every_frame_once_and_in_order(self) -> None: + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, store=False, + auto=False) + + numbers = [frame.info.number for frame in extractor] + + self.assertEqual(numbers, list(range(1, 27))) + self.assertEqual(extractor.length, 26) + + def test_protocol_membership_finds_the_http_frames(self) -> None: + import pcapkit + + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, store=False, + auto=False) + + found = {} + for frame in extractor: + if pcapkit.HTTP in frame: + found[frame.info.number] = str(frame.protochain) + + self.assertEqual(tuple(found), HTTP_FRAMES) + self.assertEqual(set(found.values()), {'Ethernet:IPv6:TCP:HTTP/1.1'}) + + def test_membership_is_false_for_a_protocol_the_capture_lacks(self) -> None: + import pcapkit + + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, store=True) + + self.assertFalse(any(pcapkit.UDP in frame for frame in extractor.frame)) + self.assertTrue(all(pcapkit.TCP in frame for frame in extractor.frame)) + + def test_call_form_walks_the_same_frames_as_the_iterator(self) -> None: + extractor = self.extract(fin=sample_path('arp.pcap'), nofile=True, store=False, + auto=False) + + first = next(extractor) + second = extractor() + + self.assertEqual(first.info.number, 1) + self.assertEqual(second.info.number, 2) + self.assertEqual(str(first.protochain), 'Ethernet:ARP:Raw') + # ``no_eof`` is documented as "if not raise EOFError when reach EOF", so + # raising it here is the contract rather than an accident. + with self.assertRaises(EOFError): + extractor() + + +class ExtractionLimitTests(EndToEndTestCase): + """``layer`` and ``protocol``, which stop parsing part way up the stack.""" + + @unittest.skip('blocked on the parse limit never reaching the next layer: ' + 'pcapkit/protocols/protocol.py:1157 passes it as layer=/protocol= while ' + 'pcapkit/protocols/protocol.py:514 reads _layer/_protocol') + def test_layer_and_protocol_limits_stop_the_parse(self) -> None: + """``layer`` and ``protocol`` should stop parsing where they name. + + Neither does anything at all. ``ProtocolBase.__init__`` reads the limits + from ``kwargs.pop('_layer')`` and ``kwargs.pop('_protocol')`` + (``pcapkit/protocols/protocol.py:514`` and ``:516``), but every caller + passes them under the un-prefixed names: + ``pcapkit/protocols/protocol.py:1157`` recurses with + ``layer=self._exlayer, protocol=self._exproto``, and + ``pcapkit/foundation/engines/pcap.py:146`` builds the frame with + ``layer=ext._exlyr, protocol=ext._exptl``. The keywords therefore land in + ``**kwargs`` and are dropped, ``_sigterm`` stays :data:`False` all the + way up, and the parse always runs to the top of the stack. + + Measured on frame 4 of :file:`http6.cap`, whose full chain is + ``Ethernet:IPv6:TCP:HTTP/1.1``: ``layer='internet'``, + ``layer='transport'``, ``layer='link'``, ``protocol='TCP'`` and + ``protocol='IPv6'`` every one of them yield that same full chain. + + That the mismatch is the whole story can be shown without the extractor, + by handing one protocol object each spelling:: + + >>> from pcapkit.protocols.link.ethernet import Ethernet + >>> str(Ethernet(io.BytesIO(raw), len(raw), _layer='Link').protochain) + 'Ethernet:Internet_Protocol_version_6' + >>> str(Ethernet(io.BytesIO(raw), len(raw), layer='Link').protochain) + 'Ethernet:IPv6:TCP:HTTP/1.1' + + The command line tool passes ``-L`` and ``-P`` straight through to the + same place, so ``pcapkit-cli -L internet`` is equally inert. + + """ + for limit in ({'layer': 'internet'}, {'layer': 'transport'}, {'protocol': 'TCP'}): + with self.subTest(**limit): + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, + store=True, **limit) + chain = str(extractor.frame[3].protochain) + + self.assertTrue(chain.startswith('Ethernet:IPv6')) + self.assertNotIn('HTTP', chain) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_output_formats.py b/tests/integration/test_output_formats.py new file mode 100644 index 0000000000..8c5a438f0a --- /dev/null +++ b/tests/integration/test_output_formats.py @@ -0,0 +1,227 @@ +# -*- coding: utf-8 -*- +"""End-to-end extraction into every output format. + +Translates the demonstration scripts that only ever checked that +:func:`pcapkit.interface.extract` did not raise -- +:file:`examples/legacy_smoke/test_extractor.py` (``tree``, ``json`` and +``plist`` reports), :file:`test_basic.py` (``verbose``), +:file:`test_api.py` and :file:`test_ipv6.py` (``files=True``) and +:file:`test_file.py` (an already-open binary stream) -- into assertions about +the file that actually lands on disk. + +Every report is written into the test's own temporary directory. The captures +under :file:`examples/captures/` are inputs only. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import (HAS_RUNTIME, EndToEndTestCase, plist_keys, read_json, + report_stems, section_counts) + +#: Sections a report of :file:`in.pcap` has: the global header and six frames. +IN_PCAP_SECTIONS = ['Global Header', 'Frame 1', 'Frame 2', 'Frame 3', 'Frame 4', 'Frame 5', 'Frame 6'] +#: Protocol chain of the first frame of :file:`in.pcap`. +IN_PCAP_FIRST_CHAIN = 'Ethernet:IPv6:IPv6_ICMP' + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TreeReportTests(EndToEndTestCase): + """``format='tree'``, the human-readable report.""" + + def test_tree_report_describes_the_global_header_and_every_frame(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='tree', store=False) + + self.assertEqual(extractor.length, 6) + self.assertEqual(extractor.output, self.out('report.txt')) + + report = (self.tmp_path / 'report.txt').read_text(encoding='utf-8') + self.assertIn('Global Header', report) + for number in range(1, 7): + self.assertIn(f'Frame {number}', report) + self.assertIn(IN_PCAP_FIRST_CHAIN, report) + + def test_extension_flag_decides_whether_a_suffix_is_appended(self) -> None: + appended = self.extract(fin=sample_path('arp.pcap'), fout=self.out('with-suffix'), + format='tree', store=False, extension=True) + verbatim = self.extract(fin=sample_path('arp.pcap'), fout=self.out('verbatim'), + format='tree', store=False, extension=False) + + self.assertEqual(appended.output, self.out('with-suffix.txt')) + self.assertEqual(verbatim.output, self.out('verbatim')) + self.assertTrue((self.tmp_path / 'with-suffix.txt').is_file()) + self.assertTrue((self.tmp_path / 'verbatim').is_file()) + + def test_report_directory_is_created_on_demand(self) -> None: + extractor = self.extract(fin=sample_path('arp.pcap'), fout=self.out('nested/deeper/report'), + format='tree', store=False) + + self.assertEqual(extractor.length, 2) + self.assertTrue((self.tmp_path / 'nested' / 'deeper' / 'report.txt').is_file()) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class JsonReportTests(EndToEndTestCase): + """``format='json'``, the machine-readable report.""" + + def test_json_report_parses_and_holds_one_section_per_frame(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='json', store=False) + + self.assertEqual(extractor.output, self.out('report.json')) + + report = read_json(extractor.output) + self.assertEqual(list(report), IN_PCAP_SECTIONS) + self.assertEqual(section_counts(report), {'Global Header': 1, 'Frame': 6}) + + def test_json_report_records_the_protocol_chain_of_each_frame(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(report['Frame 1']['protocols'], IN_PCAP_FIRST_CHAIN) + self.assertEqual(report['Frame 6']['protocols'], 'Ethernet:IPv4:UDP:Raw') + self.assertEqual(report['Frame 1']['number'], 1) + self.assertIn('ethernet', report['Frame 1']) + self.assertIn('ipv6', report['Frame 1']['ethernet']) + + def test_json_global_header_records_the_capture_byte_order(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(extractor.magic_number, b'\xd4\xc3\xb2\xa1') + self.assertEqual(report['Global Header']['magic_number']['byteorder'], 'little') + self.assertFalse(report['Global Header']['magic_number']['nanosecond']) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class PlistReportTests(EndToEndTestCase): + """``format='plist'``, the macOS property list report.""" + + def test_plist_report_is_well_formed_xml_with_one_key_per_frame(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='plist', store=False) + + self.assertEqual(extractor.output, self.out('report.plist')) + self.assertEqual(plist_keys(extractor.output), IN_PCAP_SECTIONS) + + +class PlistRoundTripTests(EndToEndTestCase): + """The property list report against a real property list reader. + + Kept apart from :class:`PlistReportTests` so that the skip below covers only + the reader, and the structural assertions above keep running. + + """ + + @unittest.skip('blocked on dictdumper writing values with fractional seconds, ' + 'which plistlib rejects (dictdumper/plist.py:278)') + def test_plist_report_round_trips_through_plistlib(self) -> None: + """A ``plist`` report should be readable by :func:`plistlib.load`. + + It is not. ``dictdumper/plist.py:278`` formats every timestamp as + ``'%Y-%m-%dT%H:%M:%S.%fZ'``, and a property list ```` carries no + fractional part, so :func:`plistlib.load` fails on the first frame's + ``time`` with ``AttributeError: 'NoneType' object has no attribute + 'groupdict'`` -- its date pattern simply does not match. Reproduce with:: + + >>> import dictdumper, datetime, plistlib + >>> dictdumper.PLIST('probe.plist')({'time': datetime.datetime.now()}) + >>> plistlib.load(open('probe.plist', 'rb')) + + The assertions below are what a fixed writer should satisfy, so this + test can simply be un-skipped once the dumper emits a conformant date. + + """ + import plistlib + + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='plist', store=False) + + with open(extractor.output, 'rb') as stream: + report = plistlib.load(stream) + + self.assertEqual(list(report), IN_PCAP_SECTIONS) + self.assertEqual(report['Frame 1']['protocols'], IN_PCAP_FIRST_CHAIN) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class SplitReportTests(EndToEndTestCase): + """``files=True``, one report file per frame.""" + + def test_split_json_reports_are_written_one_per_section(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('frames'), + format='json', files=True, store=False) + + self.assertEqual(extractor.output, self.out('frames')) + self.assertTrue(self.tmp_path.joinpath('frames').is_dir()) + self.assertEqual(report_stems(extractor.output), sorted(IN_PCAP_SECTIONS)) + + def test_every_split_report_is_parseable_on_its_own(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('frames'), + format='json', files=True, store=False) + + for report in sorted(self.tmp_path.joinpath('frames').iterdir()): + with self.subTest(report=report.name): + section = read_json(str(report)) + self.assertIsInstance(section, dict) + self.assertTrue(section) + + self.assertEqual(extractor.length, 6) + + def test_split_tree_reports_cover_a_sixteen_frame_capture(self) -> None: + # ``examples/legacy_smoke/test_ipv6.py`` used ipv6.pcap for exactly this, + # and at 16 frames it is still small enough to be a cheap check that the + # frame numbering does not collide once it passes single digits. + extractor = self.extract(fin=sample_path('ipv6.pcap'), fout=self.out('ipv6'), + format='tree', files=True, store=False) + + expected = ['Global Header'] + [f'Frame {number}' for number in range(1, 17)] + self.assertEqual(extractor.length, 16) + self.assertEqual(report_stems(extractor.output), sorted(expected)) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class InputAndProgressTests(EndToEndTestCase): + """How the capture gets in, and what the caller is told while it does.""" + + def test_extraction_accepts_an_already_open_binary_stream(self) -> None: + # ``examples/legacy_smoke/test_file.py``: hand ``fin`` a file object + # rather than a path. + with open(sample_path('in.pcap'), 'rb') as stream: + extractor = self.extract(fin=stream, nofile=True, store=True) + + self.assertEqual(extractor.length, 6) + self.assertEqual(extractor.input, sample_path('in.pcap')) + self.assertEqual(str(extractor.frame[0].protochain), IN_PCAP_FIRST_CHAIN) + + def test_verbose_handler_is_called_once_per_frame(self) -> None: + # ``verbose`` also takes a callable, which is the only form of it that + # can be asserted on without capturing stdout. + seen = [] # type: list[tuple[int, str]] + extractor = self.extract(fin=sample_path('in.pcap'), nofile=True, store=False, + verbose=lambda ext, frame: seen.append( + (ext.length, str(frame.protochain)))) + + self.assertEqual(extractor.length, 6) + self.assertEqual([number for number, _ in seen], [1, 2, 3, 4, 5, 6]) + self.assertEqual(seen[0][1], IN_PCAP_FIRST_CHAIN) + self.assertEqual(seen[-1][1], 'Ethernet:IPv4:UDP:Raw') + + def test_report_attributes_are_refused_when_no_report_is_written(self) -> None: + from pcapkit.utilities.exceptions import UnsupportedCall + + extractor = self.extract(fin=sample_path('arp.pcap'), nofile=True, store=False) + + with self.assertRaises(UnsupportedCall): + extractor.output # pylint: disable=pointless-statement + with self.assertRaises(UnsupportedCall): + extractor.frame # pylint: disable=pointless-statement + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_pcapng_end_to_end.py b/tests/integration/test_pcapng_end_to_end.py new file mode 100644 index 0000000000..48035754cf --- /dev/null +++ b/tests/integration/test_pcapng_end_to_end.py @@ -0,0 +1,198 @@ +# -*- coding: utf-8 -*- +"""End-to-end extraction of PCAP-NG captures. + +Translates :file:`examples/legacy_smoke/test_pcapng.py`, which dumped +:file:`dhcp.pcapng` to a tree report and checked nothing. +:file:`tests/protocols/test_pcapng_regression.py` already asserts that each +PCAP-NG fixture extracts with ``length > 0`` and ``nofile=True``; this module +pins the exact block inventory *and* dumps a real report file, which is the part +that was never covered. + +The six fixtures and what each is for are documented in the module docstring of +:file:`examples/generators/pcapng.py`, which also records the parser defects +they provoke -- those show up as log noise here and are not assertions. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import (HAS_RUNTIME, EndToEndTestCase, read_json, report_stems, + section_counts) + +#: Frame count of every PCAP-NG fixture. +PCAPNG_FRAMES = { + 'dhcp.pcapng': 4, + 'dhcp_big_endian.pcapng': 4, + 'dhcp_little_endian.pcapng': 4, + 'many_interfaces.pcapng': 64, + 'test.pcapng': 5, + 'profile.pcapng': 40, +} + +#: Block inventory of each fixture whose report can be read back. The +#: ``test.pcapng`` report cannot; see :class:`PcapngUnescapedKeyTests`. +PCAPNG_SECTIONS = { + 'dhcp.pcapng': {'Section Header': 1, 'Interface Description': 1, 'Frame': 4}, + 'dhcp_big_endian.pcapng': {'Section Header': 1, 'Interface Description': 1, 'Frame': 4}, + 'dhcp_little_endian.pcapng': {'Section Header': 1, 'Interface Description': 1, 'Frame': 4}, + 'many_interfaces.pcapng': {'Section Header': 1, 'Interface Description': 11, 'Frame': 64, + 'Name Resolution': 1, 'Interface Statistics': 11}, + 'profile.pcapng': {'Section Header': 1, 'Interface Description': 2, 'Frame': 40, + 'Interface Statistics': 2}, +} + +#: The DHCP exchange all three ``dhcp*.pcapng`` fixtures carry, as +#: ``(src, dst, srcport, dstport, udp payload length)`` per frame. +DHCP_EXCHANGE = [ + ('0.0.0.0', '255.255.255.255', 68, 67, 272), + ('192.168.0.1', '192.168.0.10', 67, 68, 300), + ('0.0.0.0', '255.255.255.255', 68, 67, 272), + ('192.168.0.1', '192.168.0.10', 67, 68, 300), +] + +#: PCAP-NG's own magic, i.e. the section header block's byte-order magic. +PCAPNG_MAGIC = b'\n\r\r\n' + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class PcapngTreeReportTests(EndToEndTestCase): + """Every fixture dumps a tree report.""" + + def test_every_fixture_dumps_a_report_with_its_expected_frame_count(self) -> None: + for capture, frames in PCAPNG_FRAMES.items(): + with self.subTest(capture=capture): + extractor = self.extract(fin=sample_path(capture), fout=self.out(capture), + format='tree', store=False) + + report = self.tmp_path / f'{capture}.txt' + self.assertEqual(extractor.length, frames) + self.assertEqual(extractor.magic_number, PCAPNG_MAGIC) + self.assertTrue(report.is_file()) + self.assertGreater(report.stat().st_size, 0) + self.assertIn('Section Header', report.read_text(encoding='utf-8')) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class PcapngJsonReportTests(EndToEndTestCase): + """The block inventory each fixture's report records.""" + + def test_reports_hold_the_expected_blocks(self) -> None: + for capture, sections in PCAPNG_SECTIONS.items(): + with self.subTest(capture=capture): + extractor = self.extract(fin=sample_path(capture), fout=self.out(capture), + format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(section_counts(report), sections) + self.assertEqual(extractor.length, PCAPNG_FRAMES[capture]) + + def test_interface_descriptions_are_reported_one_per_interface(self) -> None: + extractor = self.extract(fin=sample_path('many_interfaces.pcapng'), + fout=self.out('many'), format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(extractor.length, 64) + for number in range(1, 12): + with self.subTest(interface=number): + self.assertIn(f'Interface Description {number}', report) + self.assertNotIn('Interface Description 12', report) + + def test_split_reports_cover_every_block_not_just_the_frames(self) -> None: + extractor = self.extract(fin=sample_path('dhcp.pcapng'), fout=self.out('blocks'), + format='json', files=True, store=False) + + self.assertEqual(report_stems(extractor.output), sorted([ + 'Section Header 1', 'Interface Description 1', + 'Frame 1', 'Frame 2', 'Frame 3', 'Frame 4', + ])) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class PcapngByteOrderTests(EndToEndTestCase): + """The same DHCP exchange read out of three differently laid out files. + + :file:`dhcp.pcapng` is the committed upstream capture, ``_big_endian`` is a + big-endian section header downloaded from the Wireshark tree, and + ``_little_endian`` is synthesised. If the section header's byte-order magic + is honoured, all three yield the same four DHCP datagrams. + + """ + + def exchange(self, capture: 'str') -> 'list[tuple[str, str, int, int, int]]': + extractor = self.extract(fin=sample_path(capture), nofile=True, store=True) + self.assertEqual(extractor.length, 4) + + rows = [] + for frame in extractor.frame: + ipv4 = frame['IPv4'].info + udp = frame['UDP'] + rows.append((str(ipv4.src), str(ipv4.dst), + udp.info.srcport.port, udp.info.dstport.port, + len(bytes(udp.packet.payload)))) + return rows + + def test_all_three_layouts_yield_the_same_exchange(self) -> None: + for capture in ('dhcp.pcapng', 'dhcp_big_endian.pcapng', 'dhcp_little_endian.pcapng'): + with self.subTest(capture=capture): + self.assertEqual(self.exchange(capture), DHCP_EXCHANGE) + + def test_all_three_layouts_yield_the_same_protocol_chains(self) -> None: + chains = {} + for capture in ('dhcp.pcapng', 'dhcp_big_endian.pcapng', 'dhcp_little_endian.pcapng'): + extractor = self.extract(fin=sample_path(capture), nofile=True, store=True) + chains[capture] = [str(frame.protochain) for frame in extractor.frame] + + self.assertEqual(list(chains.values()), + [['Ethernet:IPv4:UDP:Raw'] * 4] * 3) + + +class PcapngUnescapedKeyTests(EndToEndTestCase): + """The report of :file:`test.pcapng`, which carries a decryption secrets block. + + Its tree report is fine -- :class:`PcapngTreeReportTests` covers it -- but + the ``json`` and ``plist`` reports come out unparseable, so the round trip + lives here behind a skip rather than being asserted either way. + + """ + + @unittest.skip('blocked on dictdumper interpolating mapping keys into the report without ' + 'escaping them (dictdumper/json.py:224), which the bytes-keyed TLS key log ' + 'entries of a decryption secrets block then break') + def test_json_report_of_a_decryption_secrets_block_parses(self) -> None: + """A ``json`` report should be readable by :func:`json.load`. + + For this fixture it is not, and two things stack up to make it so: + + 1. ``pcapkit/protocols/schema/misc/pcapng.py:1450`` keys the TLS key log + entries by ``bytes.fromhex(random)``, i.e. by a raw :class:`bytes` + client random rather than by its hex text, so the report's key is a + Python ``bytes`` repr: ``b' !"#$%&\\'()*+,-./0123456789:;<=>?'``. + 2. ``dictdumper/json.py:224`` writes a key as + ``'"{item}": '.format(item=item)``, with no escaping at all -- values + go through ``_encode_value`` and are escaped, keys do not. The quotes + inside that repr therefore land in the document verbatim. Reproduce + with:: + + >>> import dictdumper, json + >>> dictdumper.JSON('probe.json')({'a"b': 'c"d'}) + >>> json.load(open('probe.json')) + json.decoder.JSONDecodeError: Expecting ':' delimiter ... + + Measured on this fixture: ``json.load`` fails with ``Expecting ':' + delimiter: line 915 column 12``. The same key makes the ``plist`` report + invalid XML at line 1376, where its ``&`` is unescaped. + + """ + extractor = self.extract(fin=sample_path('test.pcapng'), fout=self.out('report'), + format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(extractor.length, 5) + self.assertEqual(section_counts(report)['Frame'], 5) + self.assertIn('Decryption Secrets 1', report) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_reassembly_end_to_end.py b/tests/integration/test_reassembly_end_to_end.py new file mode 100644 index 0000000000..0e73f64aab --- /dev/null +++ b/tests/integration/test_reassembly_end_to_end.py @@ -0,0 +1,326 @@ +# -*- coding: utf-8 -*- +"""End-to-end reassembly of TCP payloads and IP fragments. + +Translates :file:`examples/legacy_smoke/test_reassembly.py` (TCP payloads out of +:file:`test.pcap`), :file:`test_analyse.py` (the application layer found in a +reassembled datagram, out of :file:`http6.cap`), :file:`test_ip_reasm.py` and +:file:`test_ipv6_reasm.py` (IPv4 and IPv6 datagrams). Those scripts printed +their datagrams; these assert on them. + +The captures are built for this. ``test.pcap`` spreads the IPv4 response body +over four segments delivered out of order with one of them retransmitted, and +the IPv6 request body over three segments delivered in order, so a reassembled +payload that matches its own ``Content-Length`` is evidence that the ordering +and the duplicate were both handled. See the module docstring of +:file:`examples/generators/legacy.py`. + +One expectation here is skipped rather than asserted: see +:class:`IPv6FragmentPayloadTests`. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_RUNTIME, EndToEndTestCase + + +def by_direction(datagrams: 'tuple') -> 'dict[tuple[str, int, str, int], object]': + """Key TCP datagrams by their connection and direction. + + The submission order of the datagram tuple is an implementation detail of + when each direction sent its FIN or RST, so the tests below look each + datagram up by its four-tuple instead of by position. + + """ + return { + (str(datagram.id.src[0]), datagram.id.src[1], + str(datagram.id.dst[0]), datagram.id.dst[1]): datagram + for datagram in datagrams + } + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TCPReassemblyTests(EndToEndTestCase): + """TCP payload reassembly over :file:`test.pcap`.""" + + def reassemble(self) -> 'dict': + extractor = self.extract(fin=sample_path('test.pcap'), nofile=True, store=False, + tcp=True, reassembly=True, reasm_strict=True) + self.assertEqual(extractor.length, 34) + return by_direction(extractor.reassembly.tcp) + + def test_every_direction_of_both_connections_yields_one_datagram(self) -> None: + datagrams = self.reassemble() + + self.assertEqual(sorted(datagrams), sorted([ + ('10.20.30.131', 49812, '203.0.113.42', 80), + ('203.0.113.42', 80, '10.20.30.131', 49812), + ('2001:db8:2f10::131', 49814, '2001:db8:9c::7a', 80), + ('2001:db8:9c::7a', 80, '2001:db8:2f10::131', 49814), + ])) + for key, datagram in datagrams.items(): + with self.subTest(direction=key): + self.assertTrue(datagram.completed) + self.assertIsInstance(datagram.payload, bytes) + + def test_out_of_order_and_retransmitted_segments_reassemble_in_order(self) -> None: + datagram = self.reassemble()[('203.0.113.42', 80, '10.20.30.131', 49812)] + + # The four body segments arrive 3, 1, 4, 2 with the third retransmitted; + # frames 7 and 8 carry the header and the first body segment. + self.assertEqual(datagram.index, (7, 8, 10, 12, 14, 16, 18)) + + header, separator, body = datagram.payload.partition(b'\r\n\r\n') + self.assertEqual(separator, b'\r\n\r\n') + self.assertTrue(header.startswith(b'HTTP/1.1 200 OK\r\n')) + self.assertIn(b'Content-Length: 4400', header) + # The retransmission is not counted twice and no segment is missing. + self.assertEqual(len(body), 4400) + self.assertEqual(len(datagram.payload), len(header) + 4 + 4400) + self.assertTrue(body.startswith(b'')) + self.assertTrue(body.endswith(b'\n')) + + def test_request_without_a_body_reassembles_from_its_two_frames(self) -> None: + datagram = self.reassemble()[('10.20.30.131', 49812, '203.0.113.42', 80)] + + self.assertEqual(datagram.index, (5, 6)) + self.assertTrue(datagram.payload.startswith(b'GET /index.html HTTP/1.1\r\n')) + self.assertTrue(datagram.payload.endswith(b'\r\n\r\n')) + self.assertIn(b'Host: web.example.com\r\n', datagram.payload) + + def test_ipv6_connection_reassembles_a_three_segment_request_body(self) -> None: + datagrams = self.reassemble() + request = datagrams[('2001:db8:2f10::131', 49814, '2001:db8:9c::7a', 80)] + response = datagrams[('2001:db8:9c::7a', 80, '2001:db8:2f10::131', 49814)] + + self.assertEqual(request.index, (24, 25, 27, 29)) + header, _, body = request.payload.partition(b'\r\n\r\n') + self.assertTrue(header.startswith(b'POST /v1/telemetry HTTP/1.1\r\n')) + self.assertIn(b'Content-Length: 3350', header) + self.assertEqual(len(body), 3350) + self.assertTrue(body.startswith(b'{"schema":"pcapkit.example/telemetry/1"')) + + # The client aborts this connection with an RST, which submits the + # server's direction just as a FIN would. + self.assertEqual(response.index, (30, 31, 33)) + self.assertTrue(response.payload.startswith(b'HTTP/1.1 204 No Content\r\n')) + + def test_reassembly_is_refused_when_it_was_not_requested(self) -> None: + # This is the trap ``examples/legacy_smoke/test_analyse.py`` falls into: + # it asks for ``tcp=True`` but never for ``reassembly=True``. + from pcapkit.utilities.exceptions import UnsupportedCall + + extractor = self.extract(fin=sample_path('test.pcap'), nofile=True, store=False, + tcp=True, reasm_strict=True) + + with self.assertRaises(UnsupportedCall): + extractor.reassembly # pylint: disable=pointless-statement + + def test_datagrams_are_not_retained_when_storage_is_disabled(self) -> None: + from pcapkit.utilities.exceptions import UnsupportedCall + + extractor = self.extract(fin=sample_path('test.pcap'), nofile=True, store=False, + tcp=True, reassembly=True, reasm_store=False) + + with self.assertRaises(UnsupportedCall): + extractor.reassembly.tcp # pylint: disable=pointless-statement + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class ApplicationLayerAnalysisTests(EndToEndTestCase): + """The application layer :mod:`pcapkit` finds in a reassembled datagram. + + :file:`http6.cap` is two HTTP/1.1 connections over IPv6 -- a page fetch + whose response body spans three segments, and a conditional request answered + ``304 Not Modified`` with no body -- so all four datagrams analyse as HTTP. + + """ + + def analyse(self) -> 'dict': + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, store=False, + tcp=True, reassembly=True, reasm_strict=True) + self.assertEqual(extractor.length, 26) + return by_direction(extractor.reassembly.tcp) + + def test_all_four_datagrams_analyse_as_http(self) -> None: + import pcapkit + + datagrams = self.analyse() + self.assertEqual(len(datagrams), 4) + + for key, datagram in datagrams.items(): + with self.subTest(direction=key): + self.assertTrue(datagram.completed) + self.assertIsNotNone(datagram.packet) + self.assertIn(pcapkit.HTTP, datagram.packet) + self.assertEqual(datagram.packet.alias, 'HTTP/1.1') + + def test_page_fetch_is_analysed_as_a_request_and_a_response(self) -> None: + datagrams = self.analyse() + request = datagrams[('2001:db8:2f10::131', 49820, '2001:db8:9c::50', 80)] + response = datagrams[('2001:db8:9c::50', 80, '2001:db8:2f10::131', 49820)] + + self.assertEqual(request.index, (3, 4)) + self.assertEqual(request.packet.info.receipt.type.value, 'request') + self.assertEqual(request.packet.info.receipt.method.value, 'GET') + self.assertEqual(request.packet.info.receipt.uri, '/') + self.assertEqual(request.packet.info.header['Host'], 'www6.example.com') + self.assertIsNone(request.packet.info.body) + + self.assertEqual(response.index, (5, 6, 8, 10, 12)) + self.assertEqual(response.packet.info.receipt.type.value, 'response') + self.assertEqual(int(response.packet.info.receipt.status), 200) + self.assertEqual(response.packet.info.receipt.message, 'OK') + content_length = int(response.packet.info.header['Content-Length']) + self.assertEqual(len(response.packet.info.body), content_length) + + def test_conditional_request_is_analysed_as_a_bodyless_304(self) -> None: + datagrams = self.analyse() + request = datagrams[('2001:db8:2f10::131', 49822, '2001:db8:9c::50', 80)] + response = datagrams[('2001:db8:9c::50', 80, '2001:db8:2f10::131', 49822)] + + self.assertEqual(request.packet.info.receipt.uri, '/assets/site.css') + self.assertIn('If-None-Match', request.packet.info.header) + + self.assertEqual(int(response.packet.info.receipt.status), 304) + self.assertEqual(response.packet.info.receipt.message, 'Not Modified') + self.assertNotIn('Content-Length', response.packet.info.header) + self.assertIsNone(response.packet.info.body) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class IPv4ReassemblyTests(EndToEndTestCase): + """IPv4 datagram reassembly over :file:`ipv4.pcap`. + + The fixture is a multicast stream captured on the sending host, so none of + its four datagrams is fragmented; what this covers is the path where an + unfragmented datagram is submitted whole. No fixture in the repository + carries IPv4 fragments, so the multi-fragment path is not reachable from + here -- the IPv6 class below is where fragmentation is exercised. + + """ + + def test_each_unfragmented_datagram_is_submitted_whole(self) -> None: + extractor = self.extract(fin=sample_path('ipv4.pcap'), nofile=True, store=True, + ipv4=True, reassembly=True, reasm_strict=True) + datagrams = extractor.reassembly.ipv4 + + self.assertEqual(extractor.length, 4) + self.assertEqual(len(datagrams), 4) + + for number, datagram in enumerate(datagrams, start=1): + with self.subTest(datagram=number): + self.assertTrue(datagram.completed) + self.assertEqual(datagram.index, (number,)) + self.assertEqual(str(datagram.id.src), '172.31.127.230') + self.assertEqual(str(datagram.id.dst), '239.1.3.3') + self.assertEqual(int(datagram.id.proto), 17) + # The payload is the IPv4 payload of that very frame: the eight + # octet UDP header plus its 1828 octet payload. + frame_payload = bytes(extractor.frame[number - 1]['IPv4'].packet.payload) + self.assertEqual(bytes(datagram.payload), frame_payload) + self.assertEqual(len(datagram.payload), 1836) + + def test_datagram_identifications_are_distinct_and_consecutive(self) -> None: + extractor = self.extract(fin=sample_path('ipv4.pcap'), nofile=True, store=False, + ipv4=True, reassembly=True, reasm_strict=True) + + self.assertEqual([datagram.id.id for datagram in extractor.reassembly.ipv4], + [31232, 31233, 31234, 31235]) + + def test_reassembled_datagram_analyses_as_udp(self) -> None: + extractor = self.extract(fin=sample_path('ipv4.pcap'), nofile=True, store=False, + ipv4=True, reassembly=True, reasm_strict=True) + datagram = extractor.reassembly.ipv4[0] + + self.assertEqual(type(datagram.packet).__name__, 'UDP') + self.assertEqual(len(bytes(datagram.packet.packet.payload)), 1828) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class IPv6FragmentReassemblyTests(EndToEndTestCase): + """IPv6 fragment reassembly over :file:`ipv6.pcap`. + + Frames 13 to 16 are one 4778 octet UDP datagram fragmented into + 1448/1448/1448/434 octets. What is asserted here is which frames were + collected and that the datagram is reported complete -- both of which are + right. Its *contents* are not, so they live in the skipped test below. + + """ + + def reassemble(self) -> 'tuple': + extractor = self.extract(fin=sample_path('ipv6.pcap'), nofile=True, store=False, + ipv6=True, reassembly=True, reasm_strict=True) + self.assertEqual(extractor.length, 16) + return extractor.reassembly.ipv6 + + def test_only_the_fragmented_datagram_is_submitted(self) -> None: + datagrams = self.reassemble() + + # Twelve of the sixteen frames are neighbour discovery and ICMPv6 + # echoes, which carry no fragment header and are dismissed. + self.assertEqual(len(datagrams), 1) + self.assertEqual(datagrams[0].index, (13, 14, 15, 16)) + + def test_the_datagram_is_reported_complete(self) -> None: + datagram = self.reassemble()[0] + + self.assertTrue(datagram.completed) + self.assertIsInstance(datagram.payload, bytes) + self.assertEqual(str(datagram.id.src), 'fe80::a423:b61d:7c92:70c6') + self.assertEqual(str(datagram.id.dst), 'fe80::821f:12ff:fec9:d13d') + self.assertEqual(int(datagram.id.proto), 17) + + +class IPv6FragmentPayloadTests(EndToEndTestCase): + """What the reassembled IPv6 datagram should contain. + + Kept apart from :class:`IPv6FragmentReassemblyTests` so the skip covers only + the payload, and the assertions that are correct today keep running. + + """ + + @unittest.skip('blocked on IPv6 fragment offsets being used as byte offsets while the ' + 'field is in eight-octet units (pcapkit/protocols/internet/ipv6_frag.py:141)') + def test_the_four_fragments_reassemble_into_the_whole_datagram(self) -> None: + """The reassembled payload should be the four fragments, in order. + + It is not. ``pcapkit/protocols/internet/ipv4.py:274`` scales the IPv4 + fragment offset into octets (``int(schema.flags['offset']) * 8``), but + ``pcapkit/protocols/internet/ipv6_frag.py:141`` passes the IPv6 one + through unscaled (``offset=schema.flags['offset']``), and + ``pcapkit/toolkit/pcap.py:115`` then hands it to the reassembly + machinery as ``fo``, which is a byte offset. The four fragments are + therefore placed at octets 0, 181, 362 and 543 instead of 0, 1448, 2896 + and 4344, so they overwrite one another and the datagram comes out 977 + octets long -- ``543 + 434`` -- while still reporting + ``completed=True``. + + Measured on this fixture: ``len(datagram.payload)`` is 977, and the + assertion below expects 4778. + + A second defect is visible in the same call and is left alone here: + ``pcapkit/toolkit/pcap.py:111`` keys the reassembly buffer on the IPv6 + header's flow label rather than on the fragment header's identification, + so ``datagram.id.id`` is 0 where the fixture's identification is 110308. + One fragmented datagram cannot show the consequence, so no assertion is + made about it either way. + + """ + extractor = self.extract(fin=sample_path('ipv6.pcap'), nofile=True, store=True, + ipv6=True, reassembly=True, reasm_strict=True) + datagram = extractor.reassembly.ipv6[0] + + fragments = [ + bytes(extractor.frame[number - 1]['IPv6'].info.fragment.payload) + for number in (13, 14, 15, 16) + ] + self.assertEqual([len(fragment) for fragment in fragments], [1448, 1448, 1448, 434]) + self.assertEqual(len(datagram.payload), 4778) + self.assertEqual(bytes(datagram.payload), b''.join(fragments)) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_traceflow_end_to_end.py b/tests/integration/test_traceflow_end_to_end.py new file mode 100644 index 0000000000..07e9987ceb --- /dev/null +++ b/tests/integration/test_traceflow_end_to_end.py @@ -0,0 +1,125 @@ +# -*- coding: utf-8 -*- +"""End-to-end TCP flow tracing. + +Translates :file:`examples/legacy_smoke/test_trace.py`, which traced +:file:`http.pcap` and pretty-printed the index, and the ``trace=True`` half of +:file:`test_api.py`. Tracing writes one report per flow into ``trace_fout``, so +what is asserted here is the index the extractor exposes *and* that the reports +on disk hold exactly the frames the index claims. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_RUNTIME, EndToEndTestCase, read_json + +#: The four directional flows of :file:`tcp.pcap`: two concurrent SSH sessions, +#: one over IPv4 and one over IPv6, each traced per direction. +TCP_PCAP_FLOWS = { + '10.20.30.130_22-10.20.30.131_53406-1500000000.000774': (1, 7), + '10.20.30.131_53406-10.20.30.130_22-1500000000.001585': (2, 6), + 'fe80..a6.87f9.2793.16ee_51774-fe80..1ccd.7c77.bac7.46b7_22-1500000000.002433': (3,), + 'fe80..1ccd.7c77.bac7.46b7_22-fe80..a6.87f9.2793.16ee_51774-1500000000.003318': (4, 5), +} + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TraceFlowTests(EndToEndTestCase): + """Flow tracing over :file:`tcp.pcap`, which is seven frames.""" + + def trace(self, capture: 'str' = 'tcp.pcap') -> 'tuple': + extractor = self.extract(fin=sample_path(capture), nofile=True, store=False, + tcp=True, trace=True, trace_format='json', + trace_fout=self.out('trace')) + return extractor.length, extractor.trace.tcp + + def test_every_direction_of_every_connection_becomes_a_flow(self) -> None: + length, flows = self.trace() + + self.assertEqual(length, 7) + self.assertEqual({flow.label: flow.index for flow in flows}, TCP_PCAP_FLOWS) + + def test_flow_labels_carry_both_address_families(self) -> None: + _, flows = self.trace() + labels = {flow.label for flow in flows} + + # A label is ``src_port-dst_port-timestamp``, with the dots of an IPv6 + # address doubled so the label stays usable as a file name. + self.assertEqual(sum(1 for label in labels if label.startswith('10.20.30.')), 2) + self.assertEqual(sum(1 for label in labels if label.startswith('fe80..')), 2) + + def test_each_flow_is_dumped_to_its_own_report(self) -> None: + _, flows = self.trace() + + written = sorted(entry.name for entry in self.tmp_path.joinpath('trace').iterdir()) + self.assertEqual(written, sorted(f'{label}.json' for label in TCP_PCAP_FLOWS)) + + for flow in flows: + with self.subTest(flow=flow.label): + self.assertEqual(flow.fpout, self.out(f'trace/{flow.label}.json')) + self.assertTrue(self.tmp_path.joinpath('trace', f'{flow.label}.json').is_file()) + + def test_a_flow_report_holds_exactly_the_frames_its_index_names(self) -> None: + _, flows = self.trace() + + for flow in flows: + with self.subTest(flow=flow.label): + report = read_json(flow.fpout) + self.assertEqual(list(report), [f'Frame {number}' for number in flow.index]) + for number in flow.index: + self.assertEqual(report[f'Frame {number}']['number'], number) + + def test_tracing_is_refused_when_it_was_not_requested(self) -> None: + from pcapkit.utilities.exceptions import UnsupportedCall + + extractor = self.extract(fin=sample_path('tcp.pcap'), nofile=True, store=False, tcp=True) + + with self.assertRaises(UnsupportedCall): + extractor.trace # pylint: disable=pointless-statement + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TraceFlowScaleTests(EndToEndTestCase): + """Flow tracing over :file:`http.pcap`. + + This is the one place in this tier that uses the 1117 frame capture, and it + is used deliberately: it is the only fixture with hundreds of short + connections, so it is the only one that shows the tracer partitioning a + capture rather than handling a handful of flows. It costs roughly three + seconds, and ``examples/legacy_smoke/test_trace.py`` traced this same file. + + """ + + def test_every_frame_lands_in_exactly_one_flow(self) -> None: + extractor = self.extract(fin=sample_path('http.pcap'), nofile=True, store=False, + tcp=True, trace=True, trace_format='json', + trace_fout=self.out('trace')) + flows = extractor.trace.tcp + + self.assertEqual(extractor.length, 1117) + self.assertEqual(len(flows), 331) + + indexed = [number for flow in flows for number in flow.index] + self.assertEqual(len(indexed), 1117) + self.assertEqual(sorted(indexed), list(range(1, 1118))) + + def test_one_report_is_written_per_flow(self) -> None: + extractor = self.extract(fin=sample_path('http.pcap'), nofile=True, store=False, + tcp=True, trace=True, trace_format='json', + trace_fout=self.out('trace')) + flows = extractor.trace.tcp + + written = {entry.name for entry in self.tmp_path.joinpath('trace').iterdir()} + self.assertEqual(written, {f'{flow.label}.json' for flow in flows}) + self.assertEqual(len(written), 331) + + # Spot-check the longest flow rather than re-reading all 331 reports. + longest = max(flows, key=lambda flow: len(flow.index)) + report = read_json(longest.fpout) + self.assertEqual(list(report), [f'Frame {number}' for number in longest.index]) + + +if __name__ == '__main__': + unittest.main() From 379a88b2ea4ce412360389f0b09235fb6b28e020 Mon Sep 17 00:00:00 2001 From: Jarry Shaw Date: Mon, 14 Sep 2026 10:10:58 -0400 Subject: [PATCH 3/3] tests: move the fixture-dependent reassembly case out of the unit tier test_sample_capture_reassembles_every_stream_byte_exactly reads test.pcap, which examples/generators/make_samples.py builds rather than the repository carrying it, so the unit-test workflow -- which deliberately runs without generated fixtures -- failed on every Python version with FileNotFoundError. Moved to tests/foundation/reassembly/test_tcp_runtime.py, matching the convention the ignore globs already encode: fixture-dependent cases live in *_runtime.py and *_regression.py, and the integration job generates the fixtures before running them. CI selection: 318 passed, 103 subtests passed, with no fixtures present. --- tests/foundation/reassembly/test_tcp.py | 56 +------------ .../foundation/reassembly/test_tcp_runtime.py | 78 +++++++++++++++++++ 2 files changed, 82 insertions(+), 52 deletions(-) create mode 100644 tests/foundation/reassembly/test_tcp_runtime.py diff --git a/tests/foundation/reassembly/test_tcp.py b/tests/foundation/reassembly/test_tcp.py index 75f4cb28b7..5bbb16dde2 100644 --- a/tests/foundation/reassembly/test_tcp.py +++ b/tests/foundation/reassembly/test_tcp.py @@ -5,7 +5,7 @@ import sys import unittest -from tests._support import close_extractor, purge_modules, sample_path +from tests._support import purge_modules RUNTIME_DEPS = ('tbtrim', 'aenum', 'chardet', 'dictdumper') HAS_RUNTIME = all(importlib.util.find_spec(name) is not None for name in RUNTIME_DEPS) @@ -525,57 +525,9 @@ def test_payload_free_segments_leave_the_hole_list_alone(self) -> None: self.assertEqual(after, before) self.assertEqual(len(reasm._buffer[self._bufid()].ack[1000].raw), 30) - ########################################################################## - # End to end, through the extractor, on a committed capture. - ########################################################################## - - def test_sample_capture_reassembles_every_stream_byte_exactly(self) -> None: - """``test.pcap`` through :func:`~pcapkit.interface.extract`. - - The capture has out-of-order segments, a retransmission, and a segment - recovered only after a later one, all with realistic initial sequence - numbers -- so it exercises the whole path rather than the reassembler - alone. Each datagram is compared against the octets rebuilt - independently from the frames it names, by placing each segment's - payload at its own sequence number, which is what makes the comparison - a check rather than a restatement. - - """ - from pcapkit import extract - - extractor = extract(fin=sample_path('test.pcap'), fout='/tmp/out', format='tree', - store=True, nofile=True, tcp=True, reassembly=True, - reasm_strict=True) - self.addCleanup(close_extractor, extractor) - - frames = {frame.info.number: frame for frame in extractor.frame} - datagrams = extractor.reassembly.tcp - self.assertEqual(len(datagrams), 4) - - for datagram in datagrams: - with self.subTest(index=datagram.index): - segments = [] - for number in datagram.index: - tcp = frames[number]['TCP'] - payload = bytes(tcp.packet.payload) - if payload: - segments.append((tcp.info.seq, payload)) - - origin = min(seq for (seq, _) in segments) - expected = bytearray(max(seq - origin + len(payload) - for (seq, payload) in segments)) - for (seq, payload) in segments: - expected[seq - origin:seq - origin + len(payload)] = payload - - self.assertTrue(datagram.completed) - self.assertIsNotNone(datagram.packet) - self.assertEqual(datagram.payload, bytes(expected)) - - # the four messages the fixture is built around, by length and opening - self.assertEqual(sorted(len(datagram.payload) for datagram in datagrams), - [110, 269, 3642, 4587]) - self.assertEqual(sorted(bytes(datagram.payload[:4]) for datagram in datagrams), - [b'GET ', b'HTTP', b'HTTP', b'POST']) + # The end-to-end case on a generated capture lives in + # tests/foundation/reassembly/test_tcp_runtime.py, since the unit-test + # workflow runs without the generated fixtures. if __name__ == '__main__': diff --git a/tests/foundation/reassembly/test_tcp_runtime.py b/tests/foundation/reassembly/test_tcp_runtime.py new file mode 100644 index 0000000000..d544c43f6e --- /dev/null +++ b/tests/foundation/reassembly/test_tcp_runtime.py @@ -0,0 +1,78 @@ +from __future__ import annotations + +import importlib.util +import unittest + +from tests._support import close_extractor, purge_modules, sample_path + +RUNTIME_DEPS = ('tbtrim', 'aenum', 'chardet', 'dictdumper') +HAS_RUNTIME = all(importlib.util.find_spec(name) is not None for name in RUNTIME_DEPS) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TCPReassemblyRuntimeTests(unittest.TestCase): + """TCP reassembly end to end, through the extractor, on a sample capture. + + This lives in a ``*_runtime.py`` module rather than beside the unit tests + because it reads a generated capture: ``examples/captures/`` holds only a + handful of committed fixtures, and the rest are rebuilt by + ``examples/generators/make_samples.py``. The unit-test workflow skips this + file for exactly that reason, and the integration workflow generates the + fixtures before running it. + + """ + + def setUp(self) -> None: + purge_modules(['pcapkit']) + + def test_sample_capture_reassembles_every_stream_byte_exactly(self) -> None: + """``test.pcap`` through :func:`~pcapkit.interface.extract`. + + The capture has out-of-order segments, a retransmission, and a segment + recovered only after a later one, all with realistic initial sequence + numbers -- so it exercises the whole path rather than the reassembler + alone. Each datagram is compared against the octets rebuilt + independently from the frames it names, by placing each segment's + payload at its own sequence number, which is what makes the comparison + a check rather than a restatement. + + """ + from pcapkit import extract + + extractor = extract(fin=sample_path('test.pcap'), fout='/tmp/out', format='tree', + store=True, nofile=True, tcp=True, reassembly=True, + reasm_strict=True) + self.addCleanup(close_extractor, extractor) + + frames = {frame.info.number: frame for frame in extractor.frame} + datagrams = extractor.reassembly.tcp + self.assertEqual(len(datagrams), 4) + + for datagram in datagrams: + with self.subTest(index=datagram.index): + segments = [] + for number in datagram.index: + tcp = frames[number]['TCP'] + payload = bytes(tcp.packet.payload) + if payload: + segments.append((tcp.info.seq, payload)) + + origin = min(seq for (seq, _) in segments) + expected = bytearray(max(seq - origin + len(payload) + for (seq, payload) in segments)) + for (seq, payload) in segments: + expected[seq - origin:seq - origin + len(payload)] = payload + + self.assertTrue(datagram.completed) + self.assertIsNotNone(datagram.packet) + self.assertEqual(datagram.payload, bytes(expected)) + + # the four messages the fixture is built around, by length and opening + self.assertEqual(sorted(len(datagram.payload) for datagram in datagrams), + [110, 269, 3642, 4587]) + self.assertEqual(sorted(bytes(datagram.payload[:4]) for datagram in datagrams), + [b'GET ', b'HTTP', b'HTTP', b'POST']) + + +if __name__ == '__main__': + unittest.main()