diff --git a/examples/generators/legacy.py b/examples/generators/legacy.py index 936ba13566..4913736ad3 100644 --- a/examples/generators/legacy.py +++ b/examples/generators/legacy.py @@ -58,10 +58,12 @@ data of its own. Both fixtures therefore keep each direction of each connection to a single HTTP message, and let the peer answer only with pure acknowledgements until that message is complete. -3. A datagram counts as complete when the hole descriptor list is down to two - entries or fewer, which an ordered run of segments plus a FIN or RST - achieves; an out-of-order segment adds a third entry that the missing - segment's arrival then removes. +3. A datagram counts as complete when no hole in the descriptor list falls + inside the octets that direction actually received. An ordered run of + segments leaves only the open-ended hole past the last octet, which is + outside the payload buffer and so does not count; an out-of-order segment + opens a hole in the middle, and the missing segment's arrival closes it + again. 4. The payload of a complete datagram goes to :meth:`pcapkit.protocols.transport.transport.Transport.analyze`, which picks the application protocol from the two port numbers. Port 80 is what @@ -71,17 +73,28 @@ oversight and is not: neither fixture contains a datagram that reassembles *incompletely*, even though ``test_reassembly.py`` and ``test_analyse.py`` both have a branch for one -- a payload that is a tuple of received fragments, and a -``packet`` of :data:`None`. That branch cannot be reached from a realistic -capture. ``submit`` slices the payload buffer with the bounds of each hole, but -those bounds are absolute TCP sequence numbers (``first=tcp_info.seq`` in -``pcapkit/toolkit/pcap.py``) while the buffer is indexed from the start of the -direction's data, so with any real initial sequence number every slice starts -far beyond the end of the buffer, every fragment comes out empty, and ``if -data:`` discards the datagram without a word. Measured on a probe capture with -one segment permanently missing: a realistic initial sequence number yields no -datagram for that direction at all, and only an initial sequence number of zero -produces the tuple these scripts print. A fixture cannot have both a real -handshake and that branch, so it has the real handshake. +``packet`` of :data:`None`. Nothing prevents that branch any more; it is simply +that every stream in these two captures arrives whole. Each direction here is +sent in full and every segment eventually delivered, some of them out of order +and one retransmitted, so the holes that open all close again before the FIN or +RST that submits the buffer. + +It used to be that no capture could reach that branch at all, which is how the +absence started. ``submit`` sliced the payload buffer with the bounds of each +hole, but those bounds were absolute TCP sequence numbers +(``first=tcp_info.seq`` in ``pcapkit/toolkit/pcap.py``) while the buffer is +indexed from the start of the direction's data, so with any real initial +sequence number every slice started far beyond the end of the buffer, every +fragment came out empty, and ``if data:`` discarded the datagram without a word. +That is fixed -- ``submit`` now converts each hole's absolute bounds into +offsets into the buffer it is reading, and decides completeness from whether any +hole survives that conversion -- and GitHub issue #349 records the whole of it. +Regenerating these two fixtures to carry a permanently lost segment as well was +considered and rejected: they are pinned byte-for-byte by the test suite, and a +stream with a hole in it is cheaper to build segment by segment than to read out +of a capture. ``tests/foundation/reassembly/test_tcp.py`` therefore covers the +incomplete branch directly, with realistic initial sequence numbers, while +these captures go on covering the complete one end to end. """ diff --git a/pcapkit/foundation/reassembly/data/tcp.py b/pcapkit/foundation/reassembly/data/tcp.py index 66542e108c..b06950f17a 100644 --- a/pcapkit/foundation/reassembly/data/tcp.py +++ b/pcapkit/foundation/reassembly/data/tcp.py @@ -45,9 +45,12 @@ class Packet(Info): rst: 'bool' #: Payload length, header excluded. len: 'int' - #: This sequence number. + #: Sequence number of the first octet of :attr:`payload`, i.e. the segment's + #: own sequence number. Absolute, not an offset into any payload buffer. first: 'int' - #: Next (wanted) sequence number. + #: Sequence number of the last octet of :attr:`payload`, i.e. ``first + + #: len - 1``. **Inclusive**, so a segment carrying no payload at all has + #: :attr:`last` one below :attr:`first`. last: 'int' #: Raw :obj:`bytes` type header. header: 'bytes' @@ -102,11 +105,21 @@ def __init__(self, completed: 'bool', id: 'DatagramID[_AT]', index: 'tuple[int, @info_final class HoleDescriptor(Info): - """Data model for :term:`TCP ` hole descriptor.""" + """Data model for :term:`TCP ` hole descriptor. - #: Start of hole. + Both bounds are **absolute TCP sequence numbers** and both are + **inclusive**, so a hole covers ``last - first + 1`` octets. They are not + offsets into :attr:`Fragment.raw`: the descriptor list is kept once per + buffer ID, whereas each acknowledgement number's payload buffer carries an + initial sequence number of its own, so only + :meth:`TCP.submit ` -- which + knows which buffer it is looking at -- can convert one to the other. + + """ + + #: Sequence number of the first missing octet. first: 'int' - #: Stop of hole. + #: Sequence number of the last missing octet, inclusive. last: 'int' if TYPE_CHECKING: @@ -119,7 +132,11 @@ class Fragment(Info): #: List of reassembled packets. ind: 'list[int]' - #: ISN of payload buffer. + #: Sequence number of the octet held in ``raw[0]``, i.e. the origin this + #: buffer is indexed from: ``raw[n]`` holds the octet whose sequence number + #: is ``isn + n``. Revised downwards whenever a segment turns up below the + #: data already buffered, so it is not necessarily the connection's own + #: initial sequence number. isn: 'int' #: Length of payload buffer. len: 'int' diff --git a/pcapkit/foundation/reassembly/tcp.py b/pcapkit/foundation/reassembly/tcp.py index 2134ade52f..f42528906c 100644 --- a/pcapkit/foundation/reassembly/tcp.py +++ b/pcapkit/foundation/reassembly/tcp.py @@ -42,6 +42,24 @@ class TCP(Reassembly[Packet, Datagram, BufferID, Buffer]): # Fetch result: >>> result = tcp_reassembly.datagram + Note: + There are two coordinate systems in play here, and keeping them apart + matters. The :term:`hole descriptor list ` of + :rfc:`815` is kept in **absolute TCP sequence numbers**, inclusive of + both bounds, because a hole belongs to the connection's sequence space + for that direction and not to any one payload buffer: the list is held + once per buffer ID, while each acknowledgement number gets a payload + buffer of its own with an initial sequence number of its own, and that + initial sequence number is revised whenever a segment turns up below + the data already buffered. A payload buffer, on the other hand, is + indexed from zero, such that + :attr:`buffer.raw[n] ` + holds the octet with sequence number + :attr:`buffer.isn ` + ``+ n``. :meth:`submit` is therefore the one place that converts + between the two, subtracting that buffer's initial sequence number from + each hole bound. + """ if TYPE_CHECKING: protocol: 'Type[TCP_Protocol]' @@ -77,6 +95,14 @@ def reassembly(self, info: 'Packet') -> 'None': RST = info.rst # Reset Connection Flag (Termination) SYN = info.syn # Synchronise Flag (Establishment) + # Sequence number of the first octet of this segment's payload. A SYN + # occupies a sequence number of its own (:rfc:`793`), so payload sent + # by or after a SYN starts at ``dsn + 1`` rather than at ``dsn``. + # Without this the octet the SYN spends becomes a zero byte at the head + # of the payload buffer, and every complete datagram of a connection + # whose handshake was captured comes back one octet too long. + PSN = DSN + 1 if SYN else DSN + # when SYN is set, reset buffer of existing session if SYN and BUFID in self._buffer: self._dtgram.extend( @@ -88,7 +114,11 @@ def reassembly(self, info: 'Packet') -> 'None': self._buffer[BUFID] = Buffer( hdl=[ HoleDescriptor( - first=info.len, + # everything from the octet after this segment onwards + # is still missing -- in absolute sequence numbers, so + # that the bound stays valid for every payload buffer + # under this buffer ID + first=PSN + info.len, last=sys.maxsize, ), ], @@ -98,7 +128,7 @@ def reassembly(self, info: 'Packet') -> 'None': ind=[ info.num, ], - isn=info.dsn, + isn=PSN, len=info.len, raw=info.payload, ), @@ -111,7 +141,7 @@ def reassembly(self, info: 'Packet') -> 'None': ind=[ info.num, ], - isn=info.dsn, + isn=PSN, len=info.len, raw=info.payload, ) @@ -126,18 +156,18 @@ def reassembly(self, info: 'Packet') -> 'None': # record fragment payload ISN = self._buffer[BUFID].ack[ACK].isn # Initial Sequence Number RAW = self._buffer[BUFID].ack[ACK].raw # Raw Payload Data - if DSN >= ISN: # if fragment goes after existing payload + if PSN >= ISN: # if fragment goes after existing payload LEN = self._buffer[BUFID].ack[ACK].len - GAP = DSN - (ISN + LEN) # gap length between payloads + GAP = PSN - (ISN + LEN) # gap length between payloads if GAP >= 0: # if fragment goes after existing payload RAW += bytearray(GAP) + info.payload else: # if fragment partially overlaps existing payload - RAW[DSN - ISN:DSN - ISN + info.len] = info.payload + RAW[PSN - ISN:PSN - ISN + info.len] = info.payload else: # if fragment exceeds existing payload LEN = info.len - GAP = ISN - (DSN + LEN) # gap length between payloads + GAP = ISN - (PSN + LEN) # gap length between payloads self._buffer[BUFID].ack[ACK].__update__( - isn=DSN, + isn=PSN, ) if GAP >= 0: # if fragment exceeds existing payload RAW = info.payload + bytearray(GAP) + RAW @@ -151,28 +181,36 @@ def reassembly(self, info: 'Packet') -> 'None': ) # update hole descriptor list - HDL = self._buffer[BUFID].hdl # HDL alias - for (index, hole) in enumerate(HDL): # step one - if info.first > hole.last: # step two - continue - if info.last < hole.first: # step three - continue - del HDL[index] # step four - if info.first > hole.first: # step five - new_hole = HoleDescriptor( - first=hole.first, - last=info.first - 1, - ) - HDL.insert(index, new_hole) - index += 1 - if info.last < hole.last and not FIN and not RST: # step six - new_hole = HoleDescriptor( - first=info.last + 1, - last=hole.last - ) - HDL.insert(index, new_hole) - break # step seven - #self._buffer[BUFID].hdl = HDL # update HDL + # + # A segment carrying no payload -- a bare acknowledgement, SYN, FIN + # or RST -- fills no hole, so it must not be run through the + # :rfc:`815` algorithm: its ``last`` lies one below its ``first``, + # and letting that through would split whichever hole contains it + # into two adjacent holes covering the very same octets, growing + # the list without bound on a long-lived connection. + if info.len > 0: + HDL = self._buffer[BUFID].hdl # HDL alias + for (index, hole) in enumerate(HDL): # step one + if info.first > hole.last: # step two + continue + if info.last < hole.first: # step three + continue + del HDL[index] # step four + if info.first > hole.first: # step five + new_hole = HoleDescriptor( + first=hole.first, + last=info.first - 1, + ) + HDL.insert(index, new_hole) + index += 1 + if info.last < hole.last and not FIN and not RST: # step six + new_hole = HoleDescriptor( + first=info.last + 1, + last=hole.last + ) + HDL.insert(index, new_hole) + break # step seven + #self._buffer[BUFID].hdl = HDL # update HDL # when FIN/RST is set, submit buffer of this session if FIN or RST: @@ -196,17 +234,34 @@ def submit(self, buf: 'Buffer', *, bufid: 'BufferID') -> 'list[Datagram]': # ty # check through every buffer with ACK for (ack, buffer) in buf.ack.items(): + # Translate the hole descriptor list, which is kept in absolute + # sequence numbers for the whole direction, into offsets into this + # payload buffer, which is indexed from its own initial sequence + # number. Holes lying wholly outside this buffer -- the open-ended + # one past the last octet received, and any belonging to a + # different acknowledgement number's data -- drop out here; those + # straddling an edge are clipped to it rather than being allowed to + # index from the far end of the buffer as a negative bound would. + length = len(buffer.raw) + holes = [] # type: list[tuple[int, int]] + for hole in HDL: + start = hole.first - buffer.isn # inclusive lower bound + stop = hole.last - buffer.isn + 1 # exclusive upper bound + if stop <= 0 or start >= length: + continue # hole misses this buffer + holes.append((max(start, 0), min(stop, length))) + holes.sort() + # if this buffer is not implemented # go through every hole and extract received payload - if len(HDL) > 2 and self._flag_s: + if holes and self._flag_s: data = [] # type: list[bytes] - start = stop = 0 - for hole in HDL: - stop = hole.first - byte = buffer.raw[start:stop] - start = hole.last + 1 + start = 0 + for (hole_start, hole_stop) in holes: + byte = buffer.raw[start:hole_start] if byte: # strip empty payload - data.append(byte) + data.append(bytes(byte)) + start = max(start, hole_stop) byte = buffer.raw[start:] if byte: # strip empty payload data.append(bytes(byte)) diff --git a/pcapkit/toolkit/dpkt.py b/pcapkit/toolkit/dpkt.py index 2844802d18..d3d04ebee6 100644 --- a/pcapkit/toolkit/dpkt.py +++ b/pcapkit/toolkit/dpkt.py @@ -251,8 +251,8 @@ def tcp_reassembly(packet: 'Packet', *, count: 'int' = -1) -> 'TCP_Packet | None fin=bool(int(flags[7])), # finish flag header=tcp.pack()[:tcp.__hdr_len__], # raw bytes type header payload=bytearray(tcp.pack()[tcp.__hdr_len__:]), # raw bytearray type payload - first=tcp.seq, # this sequence number - last=tcp.seq + raw_len, # next (wanted) sequence number + first=tcp.seq, # first sequence number of payload + last=tcp.seq + raw_len - 1, # last sequence number of payload len=raw_len, # payload length, header excludes ) return data diff --git a/pcapkit/toolkit/pcap.py b/pcapkit/toolkit/pcap.py index d0141af57e..b55dfeb545 100644 --- a/pcapkit/toolkit/pcap.py +++ b/pcapkit/toolkit/pcap.py @@ -163,8 +163,8 @@ def tcp_reassembly(frame: 'Frame') -> 'TCP_Packet | None': rst=tcp_info.flags.rst, # reset connection flag header=tcp.packet.header, # raw bytes type header payload=bytearray(tcp.packet.payload), # raw bytearray type payload - first=tcp_info.seq, # this sequence number - last=tcp_info.seq + raw_len, # next (wanted) sequence number + first=tcp_info.seq, # first sequence number of payload + last=tcp_info.seq + raw_len - 1, # last sequence number of payload len=raw_len, # payload length, header excludes ) return data diff --git a/pcapkit/toolkit/pcapng.py b/pcapkit/toolkit/pcapng.py index 8e6d835678..d36a7b8079 100644 --- a/pcapkit/toolkit/pcapng.py +++ b/pcapkit/toolkit/pcapng.py @@ -169,8 +169,8 @@ def tcp_reassembly(frame: 'PCAPNG') -> 'TCP_Packet | None': rst=tcp_info.flags.rst, # reset connection flag header=tcp.packet.header, # raw bytes type header payload=bytearray(tcp.packet.payload), # raw bytearray type payload - first=tcp_info.seq, # this sequence number - last=tcp_info.seq + raw_len, # next (wanted) sequence number + first=tcp_info.seq, # first sequence number of payload + last=tcp_info.seq + raw_len - 1, # last sequence number of payload len=raw_len, # payload length, header excludes ) return data diff --git a/pcapkit/toolkit/scapy.py b/pcapkit/toolkit/scapy.py index 5b578bc183..a769605bba 100644 --- a/pcapkit/toolkit/scapy.py +++ b/pcapkit/toolkit/scapy.py @@ -250,8 +250,8 @@ def tcp_reassembly(packet: 'Packet', *, count: 'int' = -1) -> 'TCP_Packet | None rst=bool(tcp.flags.R), # reset connection flag header=bytes(tcp)[:tcp.dataofs * 4], # raw bytes type header payload=bytearray(bytes(tcp.payload)), # raw bytearray type payload - first=tcp.seq, # this sequence number - last=tcp.seq + raw_len, # next (wanted) sequence number + first=tcp.seq, # first sequence number of payload + last=tcp.seq + raw_len - 1, # last sequence number of payload len=raw_len, # payload length, header excludes ) return data diff --git a/tests/foundation/reassembly/test_tcp.py b/tests/foundation/reassembly/test_tcp.py index 42d2b0f784..5bbb16dde2 100644 --- a/tests/foundation/reassembly/test_tcp.py +++ b/tests/foundation/reassembly/test_tcp.py @@ -10,6 +10,13 @@ RUNTIME_DEPS = ('tbtrim', 'aenum', 'chardet', 'dictdumper') HAS_RUNTIME = all(importlib.util.find_spec(name) is not None for name in RUNTIME_DEPS) +#: Initial sequence number for the coordinate-system tests below. A real +#: connection draws one at random from the whole 32-bit space; the defect those +#: tests cover is invisible at zero, where an absolute sequence number and an +#: offset into a payload buffer happen to be the same number, so none of them +#: uses zero except the one that checks the answer does not depend on the choice. +ISN = 0xC0DE1234 + @unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') class TCPReassemblyTests(unittest.TestCase): @@ -21,9 +28,20 @@ def _bufid(self): def _packet(self, *, num: int, dsn: int, ack: int = 500, payload: bytes = b'', syn: bool = False, fin: bool = False, rst: bool = False, - first: int = 0, last: int | None = None, header: bytes = b'tcp-header'): + first: int | None = None, last: int | None = None, + header: bytes = b'tcp-header'): + """Build a reassembly packet the way the engine toolkits build one. + + ``first`` defaults to ``dsn`` and ``last`` to ``dsn + len(payload) - 1`` + -- absolute sequence numbers, with ``last`` inclusive -- because that is + what every :mod:`pcapkit.toolkit` module now passes. Tests that exercise + the :rfc:`815` interval arithmetic on its own still override both. + + """ from pcapkit.foundation.reassembly.data.tcp import Packet + if first is None: + first = dsn if last is None: last = first + len(payload) - 1 return Packet(self._bufid(), dsn, ack, num, syn, fin, rst, len(payload), @@ -47,8 +65,11 @@ class TestTCP(TCP): TestTCP.register(callback_calls.append) reasm = TestTCP() - reasm(self._packet(num=1, dsn=100, payload=b'hello', syn=True, first=0, last=4)) - reasm(self._packet(num=2, dsn=105, payload=b' world', fin=True, first=5, last=10)) + # A SYN spends a sequence number of its own, so the payload it carries + # starts at 101 and ends at 105; the segment that continues the stream + # therefore opens at 106, not at 105. + reasm(self._packet(num=1, dsn=100, payload=b'hello', syn=True)) + reasm(self._packet(num=2, dsn=106, payload=b' world', fin=True)) datagram, = reasm.datagram self.assertTrue(datagram.completed) @@ -189,5 +210,325 @@ class TestTCP(TCP): bufid=bufid), []) +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TCPReassemblyCoordinateTests(unittest.TestCase): + """Regression tests for GitHub issue #349. + + TCP reassembly keeps its :rfc:`815` hole descriptor list in absolute + sequence numbers, and each payload buffer indexed from its own initial + sequence number. Mixing the two produced three separate wrong answers, none + of which the rest of the suite could see, because they all need an initial + sequence number that is not zero: + + * a stream with a gap in it yielded **no datagram at all**, silently, since + every hole bound landed far past the end of a buffer a few kilobytes long; + * a stream captured without its handshake yielded only its *first* fragment, + since the completeness test counted hole descriptors instead of asking + whether any hole fell inside the data; + * every extracted fragment came out one octet too long, since a hole bound + was computed as the sequence number *after* the segment while the + descriptor list treats its bounds as inclusive. + + A fourth, closely related error surfaced while fixing them: the sequence + number a SYN spends became a NUL octet at the head of the payload, so a + complete datagram did not compare equal to the octets that were sent. + + Every stream here is driven through :class:`~pcapkit.foundation.reassembly.tcp.TCP` + with segment descriptors built exactly as the :mod:`pcapkit.toolkit` modules + build them -- ``first`` the segment's own sequence number and ``last`` the + sequence number of its final payload octet -- so the tests pin the contract + between the toolkits and the reassembler, not just the reassembler. + + """ + + def setUp(self) -> None: + purge_modules(['pcapkit']) + + def _bufid(self): + return (ip_address('192.0.2.1'), 12345, ip_address('198.51.100.2'), 443) + + def _reassembly(self, **kwargs): + """A reassembly object whose payload analyser is a stub. + + The real one picks an application protocol from the port numbers, which + would drag HTTP parsing into tests about sequence arithmetic. + + """ + from pcapkit.foundation.reassembly.tcp import TCP + + class Analyzer: + @classmethod + def analyze(cls, ports: tuple[int, int], payload: bytes) -> bytes: + return payload + + class TestTCP(TCP): + __protocol_type__ = Analyzer + + return TestTCP(**kwargs) + + def _segment(self, *, num: int, seq: int, payload: bytes = b'', ack: int = 1000, + syn: bool = False, fin: bool = False, rst: bool = False): + """One segment, described the way every :mod:`pcapkit.toolkit` describes it.""" + from pcapkit.foundation.reassembly.data.tcp import Packet + + return Packet(self._bufid(), seq, ack, num, syn, fin, rst, len(payload), + seq, seq + len(payload) - 1, b'tcp-header', bytearray(payload)) + + def _stream(self, *, segments, isn: int = ISN, syn: bool = True, + syn_ack: int = 0, teardown: str | None = 'fin'): + """Build one direction of a connection. + + Args: + segments: ``(offset, payload)`` pairs, the offset counted from the + first octet of application data. Repeat a pair to retransmit it, + and reorder the list to deliver out of order. + isn: Initial sequence number, i.e. the sequence number of the SYN. + syn: Whether the capture caught the handshake. + syn_ack: Acknowledgement number on the SYN. A real SYN carries 0 and + so lands in a payload buffer of its own; passing the data + segments' number instead makes the SYN share their buffer, which + is what exposes the octet a SYN spends. + teardown: ``'fin'``, ``'rst'`` or :data:`None` for a stream left + open, which only :meth:`~pcapkit.foundation.reassembly.reassembly.Reassembly.fetch` + will submit. + + Returns: + The segments, in the order given, ready to be fed to the reassembler. + + """ + base = isn + 1 if syn else isn # a SYN spends a sequence number + packets = [] + if syn: + packets.append(self._segment(num=len(packets) + 1, seq=isn, syn=True, + ack=syn_ack)) + for (offset, payload) in segments: + packets.append(self._segment(num=len(packets) + 1, seq=base + offset, + payload=payload)) + if teardown is not None: + end = base + max(offset + len(payload) for (offset, payload) in segments) + packets.append(self._segment(num=len(packets) + 1, seq=end, + fin=teardown == 'fin', rst=teardown == 'rst')) + return packets + + def _run(self, packets, **kwargs): + reasm = self._reassembly(**kwargs) + for packet in packets: + reasm(packet) + return reasm + + ########################################################################## + # An incomplete datagram has to come back, not vanish. + ########################################################################## + + def test_incomplete_datagram_survives_a_realistic_initial_sequence_number(self) -> None: + """A permanently missing segment gives ``completed=False``, not silence. + + Two of the five segments never arrive. Before the fix this produced no + datagram whatsoever: the hole bounds were around 3.2 billion while the + payload buffer was 50 octets long, so every slice came back empty and + ``submit`` dropped the datagram at its ``if data:`` guard without a + word. + + """ + # 10..19 and 30..39 are lost for good + packets = self._stream(segments=[(0, b'A' * 10), (20, b'C' * 10), (40, b'E' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertFalse(datagram.completed) + self.assertIsNone(datagram.packet) + self.assertEqual(datagram.payload, (b'A' * 10, b'C' * 10, b'E' * 10)) + self.assertEqual([len(fragment) for fragment in datagram.payload], [10, 10, 10]) + self.assertEqual(datagram.index, (2, 3, 4, 5)) + self.assertEqual(datagram.header, b'tcp-header') + + def test_the_answer_does_not_depend_on_the_initial_sequence_number(self) -> None: + """The same stream reassembles the same way whatever ISN it uses. + + This is the property the defect broke: at zero the two coordinate + systems coincide and the fragments came out one octet long each too + long; anywhere else the datagram disappeared entirely. + + """ + segments = [(0, b'A' * 10), (20, b'C' * 10), (40, b'E' * 10)] + expected = (False, (b'A' * 10, b'C' * 10, b'E' * 10)) + + for isn in (0, 1, 0x1000, ISN, 0xFFFF0000): + with self.subTest(isn=isn): + reasm = self._run(self._stream(segments=segments, isn=isn)) + datagram, = reasm.datagram + self.assertEqual((datagram.completed, datagram.payload), expected) + + def test_a_capture_without_the_handshake_reports_every_fragment(self) -> None: + """A stream picked up mid-flight still reports all of its fragments. + + Before the fix this returned a *partial* answer rather than nothing -- + one fragment out of three -- because the completeness test counted hole + descriptors (``len(HDL) > 2``) rather than asking whether a hole fell + inside the received data. Without a SYN the list is one entry shorter, + so a two-gap stream slipped under the threshold. + + """ + packets = self._stream(syn=False, + segments=[(0, b'A' * 10), (20, b'C' * 10), (40, b'E' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertFalse(datagram.completed) + self.assertEqual(datagram.payload, (b'A' * 10, b'C' * 10, b'E' * 10)) + + def test_a_single_gap_is_enough_to_make_a_datagram_incomplete(self) -> None: + """One hole, not three, and the fragments either side of it come back.""" + packets = self._stream(segments=[(0, b'A' * 10), (20, b'C' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertFalse(datagram.completed) + self.assertIsNone(datagram.packet) + self.assertEqual(datagram.payload, (b'A' * 10, b'C' * 10)) + + ########################################################################## + # A complete datagram has to be the octets that were sent. + ########################################################################## + + def test_complete_datagram_reassembles_byte_exactly(self) -> None: + """The reassembled payload equals the concatenation of what was sent.""" + segments = [(0, b'A' * 10), (10, b'B' * 1448), (1458, b'C' * 7)] + reasm = self._run(self._stream(segments=segments)) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertIsInstance(datagram.payload, bytes) + self.assertEqual(datagram.payload, b''.join(payload for (_, payload) in segments)) + self.assertEqual(len(datagram.payload), 1465) + self.assertEqual(datagram.packet, datagram.payload) + + def test_a_syn_sharing_a_payload_buffer_spends_no_payload_octet(self) -> None: + """The sequence number a SYN occupies must not become a NUL octet. + + A SYN carries no payload but does consume a sequence number (:rfc:`793`), + so the first octet of application data sits at ``isn + 1``. A real SYN + acknowledges nothing and therefore gets a payload buffer of its own, + which hid this; give it the data segments' acknowledgement number -- as + the reproduction in the issue does, and as happens whenever the peer has + already sent something -- and seeding that buffer at ``isn`` instead of + ``isn + 1`` prepends a zero octet to the datagram. + + """ + sent = b'GET / HTTP/1.1\r\nHost: example.com\r\n\r\n' + reasm = self._run(self._stream(syn_ack=1000, segments=[(0, sent)])) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertEqual(datagram.payload, sent) + self.assertNotEqual(datagram.payload[:1], b'\x00') + + ########################################################################## + # Out-of-order delivery and retransmission. + ########################################################################## + + def test_out_of_order_and_retransmitted_delivery_reassembles_byte_exactly(self) -> None: + """Four segments delivered 3, 1, 3, 4, 2 still give the sent octets.""" + first, second, third, fourth = ((0, b'A' * 10), (10, b'B' * 10), + (20, b'C' * 10), (30, b'D' * 10)) + packets = self._stream(segments=[third, first, third, fourth, second]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertEqual(datagram.payload, b'A' * 10 + b'B' * 10 + b'C' * 10 + b'D' * 10) + + def test_a_hole_closes_when_the_lost_segment_is_retransmitted(self) -> None: + """A gap opened by a late segment is closed by the retransmission.""" + packets = self._stream(segments=[(0, b'A' * 10), (20, b'C' * 10), (10, b'B' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertEqual(datagram.payload, b'A' * 10 + b'B' * 10 + b'C' * 10) + + def test_out_of_order_delivery_around_a_gap_that_is_never_filled(self) -> None: + """Out of order, retransmitted, *and* one segment lost for good. + + The two fragments that survive are the contiguous runs either side of + the hole, so the third and fourth segments come back as one 20-octet + fragment rather than as two. + + """ + packets = self._stream(segments=[(20, b'C' * 10), (0, b'A' * 10), + (20, b'C' * 10), (30, b'D' * 10)]) + reasm = self._run(packets) + + datagram, = reasm.datagram + self.assertFalse(datagram.completed) + self.assertEqual(datagram.payload, (b'A' * 10, b'C' * 10 + b'D' * 10)) + self.assertEqual([len(fragment) for fragment in datagram.payload], [10, 20]) + + ########################################################################## + # The two coordinate systems, inspected directly. + ########################################################################## + + def test_hole_descriptor_bounds_are_absolute_sequence_numbers(self) -> None: + """The hole list is in sequence space; the payload buffer is not. + + Asserted on the live buffer rather than through a datagram, because this + is the invariant the defect broke: the list was seeded with + ``first=info.len`` -- a payload length -- and then extended with absolute + sequence numbers, so one descriptor held one of each. + + """ + packets = self._stream(segments=[(0, b'A' * 10), (20, b'C' * 10)], teardown=None) + reasm = self._run(packets) + + buffer = reasm._buffer[self._bufid()] + self.assertEqual([(hole.first, hole.last) for hole in buffer.hdl], + [(ISN + 11, ISN + 20), (ISN + 31, sys.maxsize)]) + + # the SYN acknowledges nothing, so it gets a payload buffer of its own + # and the data lands in the one keyed by the data segments' ACK + fragment = buffer.ack[1000] + self.assertEqual(fragment.isn, ISN + 1) + self.assertEqual(len(fragment.raw), 30) + + # raw[n] holds the octet whose sequence number is isn + n, which is the + # conversion submit() applies -- and the hole's own octets are the ones + # the buffer never received + hole = buffer.hdl[0] + self.assertEqual(bytes(fragment.raw[hole.first - fragment.isn: + hole.last - fragment.isn + 1]), + bytes(10)) + + def test_payload_free_segments_leave_the_hole_list_alone(self) -> None: + """A bare acknowledgement fills no hole, so it must not split one. + + Its ``last`` is one below its ``first``, and running that through the + :rfc:`815` algorithm splits whichever hole contains it into two adjacent + holes covering the very same octets -- unbounded list growth on a + long-lived connection, for no change in what is missing. + + """ + base = ISN + 1 + reasm = self._run([ + self._segment(num=1, seq=ISN, syn=True, ack=0), + self._segment(num=2, seq=base, payload=b'A' * 10), + self._segment(num=3, seq=base + 20, payload=b'C' * 10), + ]) + + before = [(hole.first, hole.last) for hole in reasm._buffer[self._bufid()].hdl] + self.assertEqual(before, [(base + 10, base + 19), (base + 30, sys.maxsize)]) + + # one past the data, one *inside* the hole, and one repeated + for (num, seq) in ((4, base + 30), (5, base + 15), (6, base + 30)): + reasm(self._segment(num=num, seq=seq)) + + after = [(hole.first, hole.last) for hole in reasm._buffer[self._bufid()].hdl] + self.assertEqual(after, before) + self.assertEqual(len(reasm._buffer[self._bufid()].ack[1000].raw), 30) + + # The end-to-end case on a generated capture lives in + # tests/foundation/reassembly/test_tcp_runtime.py, since the unit-test + # workflow runs without the generated fixtures. + + if __name__ == '__main__': unittest.main() diff --git a/tests/foundation/reassembly/test_tcp_runtime.py b/tests/foundation/reassembly/test_tcp_runtime.py new file mode 100644 index 0000000000..d544c43f6e --- /dev/null +++ b/tests/foundation/reassembly/test_tcp_runtime.py @@ -0,0 +1,78 @@ +from __future__ import annotations + +import importlib.util +import unittest + +from tests._support import close_extractor, purge_modules, sample_path + +RUNTIME_DEPS = ('tbtrim', 'aenum', 'chardet', 'dictdumper') +HAS_RUNTIME = all(importlib.util.find_spec(name) is not None for name in RUNTIME_DEPS) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TCPReassemblyRuntimeTests(unittest.TestCase): + """TCP reassembly end to end, through the extractor, on a sample capture. + + This lives in a ``*_runtime.py`` module rather than beside the unit tests + because it reads a generated capture: ``examples/captures/`` holds only a + handful of committed fixtures, and the rest are rebuilt by + ``examples/generators/make_samples.py``. The unit-test workflow skips this + file for exactly that reason, and the integration workflow generates the + fixtures before running it. + + """ + + def setUp(self) -> None: + purge_modules(['pcapkit']) + + def test_sample_capture_reassembles_every_stream_byte_exactly(self) -> None: + """``test.pcap`` through :func:`~pcapkit.interface.extract`. + + The capture has out-of-order segments, a retransmission, and a segment + recovered only after a later one, all with realistic initial sequence + numbers -- so it exercises the whole path rather than the reassembler + alone. Each datagram is compared against the octets rebuilt + independently from the frames it names, by placing each segment's + payload at its own sequence number, which is what makes the comparison + a check rather than a restatement. + + """ + from pcapkit import extract + + extractor = extract(fin=sample_path('test.pcap'), fout='/tmp/out', format='tree', + store=True, nofile=True, tcp=True, reassembly=True, + reasm_strict=True) + self.addCleanup(close_extractor, extractor) + + frames = {frame.info.number: frame for frame in extractor.frame} + datagrams = extractor.reassembly.tcp + self.assertEqual(len(datagrams), 4) + + for datagram in datagrams: + with self.subTest(index=datagram.index): + segments = [] + for number in datagram.index: + tcp = frames[number]['TCP'] + payload = bytes(tcp.packet.payload) + if payload: + segments.append((tcp.info.seq, payload)) + + origin = min(seq for (seq, _) in segments) + expected = bytearray(max(seq - origin + len(payload) + for (seq, payload) in segments)) + for (seq, payload) in segments: + expected[seq - origin:seq - origin + len(payload)] = payload + + self.assertTrue(datagram.completed) + self.assertIsNotNone(datagram.packet) + self.assertEqual(datagram.payload, bytes(expected)) + + # the four messages the fixture is built around, by length and opening + self.assertEqual(sorted(len(datagram.payload) for datagram in datagrams), + [110, 269, 3642, 4587]) + self.assertEqual(sorted(bytes(datagram.payload[:4]) for datagram in datagrams), + [b'GET ', b'HTTP', b'HTTP', b'POST']) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/_helpers.py b/tests/integration/_helpers.py new file mode 100644 index 0000000000..d31499562e --- /dev/null +++ b/tests/integration/_helpers.py @@ -0,0 +1,138 @@ +# -*- coding: utf-8 -*- +"""Scaffolding shared by the end-to-end integration modules. + +The modules next to this one drive :mod:`pcapkit` the way a user does -- a +capture in, a report file or a reassembled datagram out -- so they all need the +same three things: the dependency gates that decide whether the runtime is +usable at all, a private scratch directory to write reports into, and a couple +of readers for the report formats. Those live here rather than in +:mod:`tests._support`, which is shared with the unit tiers and is deliberately +free of anything this specific. + +Nothing in this file is collected by :program:`pytest`: ``python_files`` in +:file:`pyproject.toml` is ``test_*.py``. + +""" +from __future__ import annotations + +import collections +import importlib.util +import json +import pathlib +import re +import tempfile +import unittest +import xml.etree.ElementTree as ET +from typing import TYPE_CHECKING + +from tests._support import close_extractor, purge_modules + +if TYPE_CHECKING: + from typing import Any + + from pcapkit.foundation.extraction import Extractor + +__all__ = [ + 'HAS_RUNTIME', 'HAS_DPKT', 'HAS_SCAPY', 'HAS_EMOJI', + 'EndToEndTestCase', + 'read_json', 'plist_keys', 'section_counts', 'report_stems', +] + +#: Packages :mod:`pcapkit` needs before it can parse anything at all. +RUNTIME_DEPS = ('tbtrim', 'aenum', 'chardet', 'dictdumper') +#: Whether the runtime dependencies are importable. +HAS_RUNTIME = all(importlib.util.find_spec(name) is not None for name in RUNTIME_DEPS) +#: Whether the :mod:`dpkt` extraction engine can be selected. +HAS_DPKT = importlib.util.find_spec('dpkt') is not None +#: Whether the :mod:`scapy` extraction engine can be selected. +HAS_SCAPY = importlib.util.find_spec('scapy') is not None +#: Whether the command line tool's ``cli`` extra is installed. +HAS_EMOJI = importlib.util.find_spec('emoji') is not None + + +class EndToEndTestCase(unittest.TestCase): + """Base class for the end-to-end modules. + + Gives every test a private temporary directory in :attr:`tmp_path` and an + :meth:`extract` wrapper that closes the input stream on teardown. Captures + under :file:`examples/captures/` are fixtures -- four of them are committed + -- so nothing here ever writes outside :attr:`tmp_path`. + + """ + + @classmethod + def setUpClass(cls) -> None: + """Drop the imported library so the class starts from a clean state. + + The surrounding tiers purge in :meth:`setUp`, i.e. once per test. A + fresh :mod:`pcapkit` import measures at roughly 0.45s on this machine, + which across this tier would cost more than the extractions themselves, + so the purge happens once per class instead. That is equivalent here: + every test below imports :mod:`pcapkit` inside the test method, so none + of them depends on what an earlier test left in :data:`sys.modules`. + + """ + purge_modules(['pcapkit']) + + def setUp(self) -> None: + """Hand the test a private scratch directory.""" + tmpdir = tempfile.TemporaryDirectory(prefix='pcapkit-e2e-') + self.addCleanup(tmpdir.cleanup) + self.tmp_path = pathlib.Path(tmpdir.name) + + def out(self, name: str) -> 'str': + """Absolute path to ``name`` inside this test's scratch directory.""" + return str(self.tmp_path / name) + + def extract(self, **kwargs: 'Any') -> 'Extractor': + """Run :func:`pcapkit.interface.extract`, closing the input on teardown.""" + from pcapkit.interface import extract + + extractor = extract(**kwargs) + self.addCleanup(close_extractor, extractor) + return extractor + + +def read_json(path: 'str') -> 'dict[str, Any]': + """Parse a report written with ``format='json'``.""" + with open(path, encoding='utf-8') as stream: + return json.load(stream) + + +def plist_keys(path: 'str') -> 'list[str]': + """Top-level keys of a report written with ``format='plist'``. + + Read with :mod:`xml.etree.ElementTree` rather than :mod:`plistlib`, because + the dumper emits ```` values that :mod:`plistlib` rejects; see + ``PlistRoundTripTests`` in :mod:`tests.integration.test_output_formats`. + + """ + root = ET.parse(path).getroot() + if root.tag != 'plist': + raise AssertionError(f'not a plist document: {root.tag}') + return [key.text or '' for key in root[0].findall('key')] + + +def section_counts(report: 'dict[str, Any]') -> 'collections.Counter[str]': + """Count the report's top-level sections by kind. + + ``'Frame 1'``, ``'Frame 2'`` and so on collapse to ``'Frame'``, and the + PCAP-NG block sections likewise, so a report can be described by what it + contains rather than by listing every key. + + """ + return collections.Counter(re.sub(r' \d+$', '', key) for key in report) + + +def report_stems(directory: 'str') -> 'list[str]': + """Sorted section names of the reports ``files=True`` wrote into ``directory``. + + Split at the *first* dot on purpose. Per-frame reports are named + ``f'{name}.{ext}'`` where ``ext`` already carries its own leading dot + (``pcapkit/foundation/engines/pcap.py:117`` and ``:156``, and + ``pcapkit/foundation/engines/pcapng.py:311``), so they land on disk as + ``Frame 1..json``. Taking the leading component keeps these assertions + neutral about how many dots there are, rather than pinning that defect. + + """ + return sorted(entry.name.split('.', 1)[0] for entry in pathlib.Path(directory).iterdir()) diff --git a/tests/integration/test_cli_subprocess.py b/tests/integration/test_cli_subprocess.py new file mode 100644 index 0000000000..acc22bfdec --- /dev/null +++ b/tests/integration/test_cli_subprocess.py @@ -0,0 +1,151 @@ +# -*- coding: utf-8 -*- +"""End-to-end runs of the command line tool. + +:file:`tests/cli/test_main.py` exercises :func:`pcapkit.__main__.main` with the +extractor replaced by a stub, so it checks the argument wiring and nothing else. +This module runs the real thing in a subprocess -- the same entry point the +``pcapkit-cli`` console script installs -- against a real capture, and asserts on +the exit status, what it printed, and the report it left behind. + +Every run is given a temporary working directory, so a report written to a +relative path could not touch the repository even if one of these grew a bug. + +""" +from __future__ import annotations + +import json +import pathlib +import subprocess # nosec: B404 +import sys +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_EMOJI, HAS_RUNTIME, EndToEndTestCase + +#: How long a single CLI run is allowed to take. The captures used here are two +#: to six frames, so this is a hang guard rather than a budget. +TIMEOUT = 120 + +#: Path of the installed console script, if it is on this interpreter's path. +CLI_SCRIPT = pathlib.Path(sys.executable).with_name('pcapkit-cli') + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +@unittest.skipUnless(HAS_EMOJI, "the cli extra's 'emoji' dependency is not installed") +class CommandLineTests(EndToEndTestCase): + """``python -m pcapkit``, i.e. ``pcapkit.__main__:main``.""" + + def run_cli(self, *args: 'str', expect: 'int | None' = 0) -> 'subprocess.CompletedProcess[str]': + """Run the CLI in this test's temporary directory.""" + completed = subprocess.run( # nosec: B603 + [sys.executable, '-m', 'pcapkit', *args], + cwd=str(self.tmp_path), capture_output=True, text=True, + timeout=TIMEOUT, check=False, + ) + if expect is not None: + self.assertEqual(completed.returncode, expect, + f'unexpected exit status; stderr was:\n{completed.stderr}') + return completed + + def test_version_flag_reports_the_installed_version(self) -> None: + import pcapkit + + completed = self.run_cli('-V') + + self.assertEqual(completed.stdout.strip(), pcapkit.__version__) + + def test_json_report_is_written_and_every_chain_is_printed(self) -> None: + completed = self.run_cli(sample_path('in.pcap'), '-o', 'report', '-j', '-a', '-v') + + report = self.tmp_path / 'report.json' + self.assertTrue(report.is_file()) + with report.open(encoding='utf-8') as stream: + self.assertEqual(list(json.load(stream)), [ + 'Global Header', 'Frame 1', 'Frame 2', 'Frame 3', + 'Frame 4', 'Frame 5', 'Frame 6', + ]) + + self.assertIn(sample_path('in.pcap'), completed.stdout) + self.assertIn('Frame 1: Ethernet:IPv6:IPv6_ICMP', completed.stdout) + self.assertIn('Frame 6: Ethernet:IPv4:UDP:Raw', completed.stdout) + # ``-o report`` is relative, and the run happens in the temporary + # directory, so the name the tool reports is relative too. + self.assertIn("Report file stored in 'report.json'", completed.stdout) + + def test_tree_report_is_written_without_verbose_output(self) -> None: + completed = self.run_cli(sample_path('arp.pcap'), '-o', 'report', '-t', '-a') + + report = self.tmp_path / 'report.txt' + self.assertTrue(report.is_file()) + self.assertIn('Frame 2', report.read_text(encoding='utf-8')) + self.assertEqual(completed.stdout, '') + + def test_plist_report_is_written(self) -> None: + completed = self.run_cli(sample_path('arp.pcap'), '-o', 'report', '-p', '-a') + + report = self.tmp_path / 'report.plist' + self.assertTrue(report.is_file()) + self.assertTrue(report.read_text(encoding='utf-8').startswith(' None: + self.run_cli(sample_path('arp.pcap'), '-o', 'named', '-f', 'json', '-a') + + self.assertTrue((self.tmp_path / 'named.json').is_file()) + + def test_files_flag_writes_a_directory_of_reports(self) -> None: + completed = self.run_cli(sample_path('arp.pcap'), '-o', 'frames', '-j', '-F', '-v') + + frames = self.tmp_path / 'frames' + self.assertTrue(frames.is_dir()) + self.assertEqual(sorted(entry.name.split('.', 1)[0] for entry in frames.iterdir()), + ['Frame 1', 'Frame 2', 'Global Header']) + self.assertIn('Report files stored in', completed.stdout) + + def test_engine_option_selects_an_alternative_engine(self) -> None: + self.run_cli(sample_path('in.pcap'), '-o', 'report', '-t', '-a', '-E', 'default') + + self.assertTrue((self.tmp_path / 'report.txt').is_file()) + + def test_missing_capture_fails_and_names_the_path(self) -> None: + missing = str(self.tmp_path / 'absent.pcap') + completed = self.run_cli(missing, '-o', 'report', '-j', expect=None) + + self.assertNotEqual(completed.returncode, 0) + self.assertIn('absent.pcap', completed.stderr) + self.assertFalse((self.tmp_path / 'report.json').exists()) + + def test_no_arguments_fails_with_the_usage_message(self) -> None: + completed = self.run_cli(expect=2) + + self.assertIn('pcapkit-cli', completed.stderr) + self.assertIn('input-file-name', completed.stderr) + + def test_unknown_option_fails_with_the_usage_message(self) -> None: + completed = self.run_cli('--no-such-option', expect=2) + + self.assertIn('usage: pcapkit-cli', completed.stderr) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +@unittest.skipUnless(HAS_EMOJI, "the cli extra's 'emoji' dependency is not installed") +@unittest.skipUnless(CLI_SCRIPT.is_file(), 'pcapkit-cli console script is not installed') +class ConsoleScriptTests(EndToEndTestCase): + """The installed ``pcapkit-cli`` script, rather than ``python -m pcapkit``.""" + + def test_console_script_writes_the_same_report(self) -> None: + completed = subprocess.run( # nosec: B603 + [str(CLI_SCRIPT), sample_path('arp.pcap'), '-o', 'report', '-j', '-a'], + cwd=str(self.tmp_path), capture_output=True, text=True, + timeout=TIMEOUT, check=False, + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + report = self.tmp_path / 'report.json' + self.assertTrue(report.is_file()) + with report.open(encoding='utf-8') as stream: + self.assertEqual(list(json.load(stream)), ['Global Header', 'Frame 1', 'Frame 2']) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_engine_parity.py b/tests/integration/test_engine_parity.py new file mode 100644 index 0000000000..395bab755d --- /dev/null +++ b/tests/integration/test_engine_parity.py @@ -0,0 +1,124 @@ +# -*- coding: utf-8 -*- +"""End-to-end parity between the extraction engines. + +Translates :file:`examples/legacy_smoke/test_engine.py`, which ran the same +capture through each engine and wrote four reports nobody compared, into +assertions that the engines agree. + +``pyshark`` is left out on purpose. It needs :program:`tshark`, which is not +installed here, and on Python 3.14 it fails inside its own +``get_event_loop()`` before it reads a byte; +:file:`tests/integration/test_engine_runtime.py` already pins that. The +``pipeline`` and ``server`` engines are commented out in the original script and +are not covered here either. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_DPKT, HAS_RUNTIME, HAS_SCAPY, EndToEndTestCase + +#: Captures every engine is run over, with the frame count each must report. +#: All three are small: the point is agreement, not throughput. +PARITY_CAPTURES = { + 'in.pcap': 6, + 'arp.pcap': 2, + 'tcp.pcap': 7, + 'http6.cap': 26, +} + +#: Engines to compare, with the gate that decides whether each is available. +ENGINES = ( + ('default', HAS_RUNTIME), + ('dpkt', HAS_DPKT), + ('scapy', HAS_SCAPY), +) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class EngineParityTests(EndToEndTestCase): + """The same capture through every available engine.""" + + def available(self) -> 'list[str]': + return [name for name, present in ENGINES if present] + + def test_every_engine_reports_the_same_frame_count(self) -> None: + for capture, expected in PARITY_CAPTURES.items(): + counts = {} + for engine in self.available(): + with self.subTest(capture=capture, engine=engine): + extractor = self.extract(fin=sample_path(capture), nofile=True, + store=False, engine=engine) + counts[engine] = extractor.length + self.assertEqual(extractor.length, expected) + + with self.subTest(capture=capture): + self.assertEqual(set(counts.values()), {expected}) + + def test_every_engine_writes_a_report(self) -> None: + for engine in self.available(): + with self.subTest(engine=engine): + extractor = self.extract(fin=sample_path('in.pcap'), + fout=self.out(f'{engine}-report'), + format='tree', store=False, engine=engine) + + report = self.tmp_path / f'{engine}-report.txt' + self.assertEqual(extractor.output, str(report)) + self.assertTrue(report.is_file()) + self.assertGreater(report.stat().st_size, 0) + self.assertIn('Frame 6', report.read_text(encoding='utf-8')) + + def test_engine_macro_selects_the_same_engine_as_its_name(self) -> None: + # The legacy scripts pass ``engine=pcapkit.PCAPKit`` rather than the + # string, so the macro and the name have to be interchangeable. + import pcapkit + + by_macro = self.extract(fin=sample_path('tcp.pcap'), nofile=True, store=True, + engine=pcapkit.PCAPKit) + by_name = self.extract(fin=sample_path('tcp.pcap'), nofile=True, store=True, + engine='default') + + self.assertEqual(pcapkit.PCAPKit, 'default') + self.assertEqual(by_macro.length, by_name.length) + self.assertEqual([str(frame.protochain) for frame in by_macro.frame], + [str(frame.protochain) for frame in by_name.frame]) + + @unittest.skipUnless(HAS_DPKT, 'dpkt not installed') + def test_dpkt_engine_agrees_on_the_protocol_chain(self) -> None: + # The engines hand back their own packet objects, so the chains are + # spelled differently; the toolkit is what maps one onto the other. + from pcapkit.toolkit.dpkt import packet2chain + + native = self.extract(fin=sample_path('in.pcap'), nofile=True, store=True, + engine='default') + foreign = self.extract(fin=sample_path('in.pcap'), nofile=True, store=True, + engine='dpkt') + + self.assertEqual(str(native.frame[0].protochain), 'Ethernet:IPv6:IPv6_ICMP') + self.assertEqual(packet2chain(foreign.frame[0]), 'Ethernet:IP6:ICMP6') + self.assertEqual(str(native.frame[2].protochain), 'Ethernet:IPv4:TCP') + self.assertEqual(packet2chain(foreign.frame[2]), 'Ethernet:IP:TCP') + + @unittest.skipUnless(HAS_SCAPY, 'scapy not installed') + def test_scapy_engine_returns_the_same_link_layer_bytes(self) -> None: + native = self.extract(fin=sample_path('arp.pcap'), nofile=True, store=True, + engine='default') + foreign = self.extract(fin=sample_path('arp.pcap'), nofile=True, store=True, + engine='scapy') + + self.assertEqual(len(native.frame), len(foreign.frame)) + for number, (mine, theirs) in enumerate(zip(native.frame, foreign.frame), start=1): + with self.subTest(frame=number): + # scapy does not know this capture's link type and hands back one + # opaque ``Raw`` layer holding the whole 60 octet frame, padded to + # the Ethernet minimum. Its first fourteen octets are the Ethernet + # header that the default engine parsed into a layer of its own. + captured = bytes(theirs) + self.assertEqual(len(captured), 60) + self.assertEqual(captured[:14], bytes(mine['Ethernet'].packet.header)) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_frame_iteration.py b/tests/integration/test_frame_iteration.py new file mode 100644 index 0000000000..3d5312300f --- /dev/null +++ b/tests/integration/test_frame_iteration.py @@ -0,0 +1,128 @@ +# -*- coding: utf-8 -*- +"""End-to-end frame-by-frame iteration and the extraction limits. + +Translates :file:`examples/legacy_smoke/test_http.py`, which walked a capture +with ``auto=False`` and printed the frames where ``pcapkit.HTTP in frame`` held, +into assertions about which frames those are. It reads :file:`http6.cap` (26 +frames) rather than the :file:`http.pcap` the script used, because the spectrum +is the ``in`` test and the manual walk, not the size of the capture. + +The extraction limits -- ``layer`` and ``protocol``, which the command line tool +exposes as ``-L`` and ``-P`` -- belong to the same spectrum: they are what stops +that walk short of the application layer. They do not work; see +:class:`ExtractionLimitTests`. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_RUNTIME, EndToEndTestCase + +#: Frames of :file:`http6.cap` that carry an HTTP header block: the request and +#: the response of each of the two connections. +HTTP_FRAMES = (4, 6, 19, 21) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class FrameIterationTests(EndToEndTestCase): + """``auto=False``, i.e. the caller drives the extraction.""" + + def test_manual_iteration_yields_every_frame_once_and_in_order(self) -> None: + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, store=False, + auto=False) + + numbers = [frame.info.number for frame in extractor] + + self.assertEqual(numbers, list(range(1, 27))) + self.assertEqual(extractor.length, 26) + + def test_protocol_membership_finds_the_http_frames(self) -> None: + import pcapkit + + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, store=False, + auto=False) + + found = {} + for frame in extractor: + if pcapkit.HTTP in frame: + found[frame.info.number] = str(frame.protochain) + + self.assertEqual(tuple(found), HTTP_FRAMES) + self.assertEqual(set(found.values()), {'Ethernet:IPv6:TCP:HTTP/1.1'}) + + def test_membership_is_false_for_a_protocol_the_capture_lacks(self) -> None: + import pcapkit + + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, store=True) + + self.assertFalse(any(pcapkit.UDP in frame for frame in extractor.frame)) + self.assertTrue(all(pcapkit.TCP in frame for frame in extractor.frame)) + + def test_call_form_walks_the_same_frames_as_the_iterator(self) -> None: + extractor = self.extract(fin=sample_path('arp.pcap'), nofile=True, store=False, + auto=False) + + first = next(extractor) + second = extractor() + + self.assertEqual(first.info.number, 1) + self.assertEqual(second.info.number, 2) + self.assertEqual(str(first.protochain), 'Ethernet:ARP:Raw') + # ``no_eof`` is documented as "if not raise EOFError when reach EOF", so + # raising it here is the contract rather than an accident. + with self.assertRaises(EOFError): + extractor() + + +class ExtractionLimitTests(EndToEndTestCase): + """``layer`` and ``protocol``, which stop parsing part way up the stack.""" + + @unittest.skip('blocked on the parse limit never reaching the next layer: ' + 'pcapkit/protocols/protocol.py:1157 passes it as layer=/protocol= while ' + 'pcapkit/protocols/protocol.py:514 reads _layer/_protocol') + def test_layer_and_protocol_limits_stop_the_parse(self) -> None: + """``layer`` and ``protocol`` should stop parsing where they name. + + Neither does anything at all. ``ProtocolBase.__init__`` reads the limits + from ``kwargs.pop('_layer')`` and ``kwargs.pop('_protocol')`` + (``pcapkit/protocols/protocol.py:514`` and ``:516``), but every caller + passes them under the un-prefixed names: + ``pcapkit/protocols/protocol.py:1157`` recurses with + ``layer=self._exlayer, protocol=self._exproto``, and + ``pcapkit/foundation/engines/pcap.py:146`` builds the frame with + ``layer=ext._exlyr, protocol=ext._exptl``. The keywords therefore land in + ``**kwargs`` and are dropped, ``_sigterm`` stays :data:`False` all the + way up, and the parse always runs to the top of the stack. + + Measured on frame 4 of :file:`http6.cap`, whose full chain is + ``Ethernet:IPv6:TCP:HTTP/1.1``: ``layer='internet'``, + ``layer='transport'``, ``layer='link'``, ``protocol='TCP'`` and + ``protocol='IPv6'`` every one of them yield that same full chain. + + That the mismatch is the whole story can be shown without the extractor, + by handing one protocol object each spelling:: + + >>> from pcapkit.protocols.link.ethernet import Ethernet + >>> str(Ethernet(io.BytesIO(raw), len(raw), _layer='Link').protochain) + 'Ethernet:Internet_Protocol_version_6' + >>> str(Ethernet(io.BytesIO(raw), len(raw), layer='Link').protochain) + 'Ethernet:IPv6:TCP:HTTP/1.1' + + The command line tool passes ``-L`` and ``-P`` straight through to the + same place, so ``pcapkit-cli -L internet`` is equally inert. + + """ + for limit in ({'layer': 'internet'}, {'layer': 'transport'}, {'protocol': 'TCP'}): + with self.subTest(**limit): + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, + store=True, **limit) + chain = str(extractor.frame[3].protochain) + + self.assertTrue(chain.startswith('Ethernet:IPv6')) + self.assertNotIn('HTTP', chain) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_output_formats.py b/tests/integration/test_output_formats.py new file mode 100644 index 0000000000..8c5a438f0a --- /dev/null +++ b/tests/integration/test_output_formats.py @@ -0,0 +1,227 @@ +# -*- coding: utf-8 -*- +"""End-to-end extraction into every output format. + +Translates the demonstration scripts that only ever checked that +:func:`pcapkit.interface.extract` did not raise -- +:file:`examples/legacy_smoke/test_extractor.py` (``tree``, ``json`` and +``plist`` reports), :file:`test_basic.py` (``verbose``), +:file:`test_api.py` and :file:`test_ipv6.py` (``files=True``) and +:file:`test_file.py` (an already-open binary stream) -- into assertions about +the file that actually lands on disk. + +Every report is written into the test's own temporary directory. The captures +under :file:`examples/captures/` are inputs only. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import (HAS_RUNTIME, EndToEndTestCase, plist_keys, read_json, + report_stems, section_counts) + +#: Sections a report of :file:`in.pcap` has: the global header and six frames. +IN_PCAP_SECTIONS = ['Global Header', 'Frame 1', 'Frame 2', 'Frame 3', 'Frame 4', 'Frame 5', 'Frame 6'] +#: Protocol chain of the first frame of :file:`in.pcap`. +IN_PCAP_FIRST_CHAIN = 'Ethernet:IPv6:IPv6_ICMP' + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TreeReportTests(EndToEndTestCase): + """``format='tree'``, the human-readable report.""" + + def test_tree_report_describes_the_global_header_and_every_frame(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='tree', store=False) + + self.assertEqual(extractor.length, 6) + self.assertEqual(extractor.output, self.out('report.txt')) + + report = (self.tmp_path / 'report.txt').read_text(encoding='utf-8') + self.assertIn('Global Header', report) + for number in range(1, 7): + self.assertIn(f'Frame {number}', report) + self.assertIn(IN_PCAP_FIRST_CHAIN, report) + + def test_extension_flag_decides_whether_a_suffix_is_appended(self) -> None: + appended = self.extract(fin=sample_path('arp.pcap'), fout=self.out('with-suffix'), + format='tree', store=False, extension=True) + verbatim = self.extract(fin=sample_path('arp.pcap'), fout=self.out('verbatim'), + format='tree', store=False, extension=False) + + self.assertEqual(appended.output, self.out('with-suffix.txt')) + self.assertEqual(verbatim.output, self.out('verbatim')) + self.assertTrue((self.tmp_path / 'with-suffix.txt').is_file()) + self.assertTrue((self.tmp_path / 'verbatim').is_file()) + + def test_report_directory_is_created_on_demand(self) -> None: + extractor = self.extract(fin=sample_path('arp.pcap'), fout=self.out('nested/deeper/report'), + format='tree', store=False) + + self.assertEqual(extractor.length, 2) + self.assertTrue((self.tmp_path / 'nested' / 'deeper' / 'report.txt').is_file()) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class JsonReportTests(EndToEndTestCase): + """``format='json'``, the machine-readable report.""" + + def test_json_report_parses_and_holds_one_section_per_frame(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='json', store=False) + + self.assertEqual(extractor.output, self.out('report.json')) + + report = read_json(extractor.output) + self.assertEqual(list(report), IN_PCAP_SECTIONS) + self.assertEqual(section_counts(report), {'Global Header': 1, 'Frame': 6}) + + def test_json_report_records_the_protocol_chain_of_each_frame(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(report['Frame 1']['protocols'], IN_PCAP_FIRST_CHAIN) + self.assertEqual(report['Frame 6']['protocols'], 'Ethernet:IPv4:UDP:Raw') + self.assertEqual(report['Frame 1']['number'], 1) + self.assertIn('ethernet', report['Frame 1']) + self.assertIn('ipv6', report['Frame 1']['ethernet']) + + def test_json_global_header_records_the_capture_byte_order(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(extractor.magic_number, b'\xd4\xc3\xb2\xa1') + self.assertEqual(report['Global Header']['magic_number']['byteorder'], 'little') + self.assertFalse(report['Global Header']['magic_number']['nanosecond']) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class PlistReportTests(EndToEndTestCase): + """``format='plist'``, the macOS property list report.""" + + def test_plist_report_is_well_formed_xml_with_one_key_per_frame(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='plist', store=False) + + self.assertEqual(extractor.output, self.out('report.plist')) + self.assertEqual(plist_keys(extractor.output), IN_PCAP_SECTIONS) + + +class PlistRoundTripTests(EndToEndTestCase): + """The property list report against a real property list reader. + + Kept apart from :class:`PlistReportTests` so that the skip below covers only + the reader, and the structural assertions above keep running. + + """ + + @unittest.skip('blocked on dictdumper writing values with fractional seconds, ' + 'which plistlib rejects (dictdumper/plist.py:278)') + def test_plist_report_round_trips_through_plistlib(self) -> None: + """A ``plist`` report should be readable by :func:`plistlib.load`. + + It is not. ``dictdumper/plist.py:278`` formats every timestamp as + ``'%Y-%m-%dT%H:%M:%S.%fZ'``, and a property list ```` carries no + fractional part, so :func:`plistlib.load` fails on the first frame's + ``time`` with ``AttributeError: 'NoneType' object has no attribute + 'groupdict'`` -- its date pattern simply does not match. Reproduce with:: + + >>> import dictdumper, datetime, plistlib + >>> dictdumper.PLIST('probe.plist')({'time': datetime.datetime.now()}) + >>> plistlib.load(open('probe.plist', 'rb')) + + The assertions below are what a fixed writer should satisfy, so this + test can simply be un-skipped once the dumper emits a conformant date. + + """ + import plistlib + + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('report'), + format='plist', store=False) + + with open(extractor.output, 'rb') as stream: + report = plistlib.load(stream) + + self.assertEqual(list(report), IN_PCAP_SECTIONS) + self.assertEqual(report['Frame 1']['protocols'], IN_PCAP_FIRST_CHAIN) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class SplitReportTests(EndToEndTestCase): + """``files=True``, one report file per frame.""" + + def test_split_json_reports_are_written_one_per_section(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('frames'), + format='json', files=True, store=False) + + self.assertEqual(extractor.output, self.out('frames')) + self.assertTrue(self.tmp_path.joinpath('frames').is_dir()) + self.assertEqual(report_stems(extractor.output), sorted(IN_PCAP_SECTIONS)) + + def test_every_split_report_is_parseable_on_its_own(self) -> None: + extractor = self.extract(fin=sample_path('in.pcap'), fout=self.out('frames'), + format='json', files=True, store=False) + + for report in sorted(self.tmp_path.joinpath('frames').iterdir()): + with self.subTest(report=report.name): + section = read_json(str(report)) + self.assertIsInstance(section, dict) + self.assertTrue(section) + + self.assertEqual(extractor.length, 6) + + def test_split_tree_reports_cover_a_sixteen_frame_capture(self) -> None: + # ``examples/legacy_smoke/test_ipv6.py`` used ipv6.pcap for exactly this, + # and at 16 frames it is still small enough to be a cheap check that the + # frame numbering does not collide once it passes single digits. + extractor = self.extract(fin=sample_path('ipv6.pcap'), fout=self.out('ipv6'), + format='tree', files=True, store=False) + + expected = ['Global Header'] + [f'Frame {number}' for number in range(1, 17)] + self.assertEqual(extractor.length, 16) + self.assertEqual(report_stems(extractor.output), sorted(expected)) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class InputAndProgressTests(EndToEndTestCase): + """How the capture gets in, and what the caller is told while it does.""" + + def test_extraction_accepts_an_already_open_binary_stream(self) -> None: + # ``examples/legacy_smoke/test_file.py``: hand ``fin`` a file object + # rather than a path. + with open(sample_path('in.pcap'), 'rb') as stream: + extractor = self.extract(fin=stream, nofile=True, store=True) + + self.assertEqual(extractor.length, 6) + self.assertEqual(extractor.input, sample_path('in.pcap')) + self.assertEqual(str(extractor.frame[0].protochain), IN_PCAP_FIRST_CHAIN) + + def test_verbose_handler_is_called_once_per_frame(self) -> None: + # ``verbose`` also takes a callable, which is the only form of it that + # can be asserted on without capturing stdout. + seen = [] # type: list[tuple[int, str]] + extractor = self.extract(fin=sample_path('in.pcap'), nofile=True, store=False, + verbose=lambda ext, frame: seen.append( + (ext.length, str(frame.protochain)))) + + self.assertEqual(extractor.length, 6) + self.assertEqual([number for number, _ in seen], [1, 2, 3, 4, 5, 6]) + self.assertEqual(seen[0][1], IN_PCAP_FIRST_CHAIN) + self.assertEqual(seen[-1][1], 'Ethernet:IPv4:UDP:Raw') + + def test_report_attributes_are_refused_when_no_report_is_written(self) -> None: + from pcapkit.utilities.exceptions import UnsupportedCall + + extractor = self.extract(fin=sample_path('arp.pcap'), nofile=True, store=False) + + with self.assertRaises(UnsupportedCall): + extractor.output # pylint: disable=pointless-statement + with self.assertRaises(UnsupportedCall): + extractor.frame # pylint: disable=pointless-statement + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_pcapng_end_to_end.py b/tests/integration/test_pcapng_end_to_end.py new file mode 100644 index 0000000000..48035754cf --- /dev/null +++ b/tests/integration/test_pcapng_end_to_end.py @@ -0,0 +1,198 @@ +# -*- coding: utf-8 -*- +"""End-to-end extraction of PCAP-NG captures. + +Translates :file:`examples/legacy_smoke/test_pcapng.py`, which dumped +:file:`dhcp.pcapng` to a tree report and checked nothing. +:file:`tests/protocols/test_pcapng_regression.py` already asserts that each +PCAP-NG fixture extracts with ``length > 0`` and ``nofile=True``; this module +pins the exact block inventory *and* dumps a real report file, which is the part +that was never covered. + +The six fixtures and what each is for are documented in the module docstring of +:file:`examples/generators/pcapng.py`, which also records the parser defects +they provoke -- those show up as log noise here and are not assertions. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import (HAS_RUNTIME, EndToEndTestCase, read_json, report_stems, + section_counts) + +#: Frame count of every PCAP-NG fixture. +PCAPNG_FRAMES = { + 'dhcp.pcapng': 4, + 'dhcp_big_endian.pcapng': 4, + 'dhcp_little_endian.pcapng': 4, + 'many_interfaces.pcapng': 64, + 'test.pcapng': 5, + 'profile.pcapng': 40, +} + +#: Block inventory of each fixture whose report can be read back. The +#: ``test.pcapng`` report cannot; see :class:`PcapngUnescapedKeyTests`. +PCAPNG_SECTIONS = { + 'dhcp.pcapng': {'Section Header': 1, 'Interface Description': 1, 'Frame': 4}, + 'dhcp_big_endian.pcapng': {'Section Header': 1, 'Interface Description': 1, 'Frame': 4}, + 'dhcp_little_endian.pcapng': {'Section Header': 1, 'Interface Description': 1, 'Frame': 4}, + 'many_interfaces.pcapng': {'Section Header': 1, 'Interface Description': 11, 'Frame': 64, + 'Name Resolution': 1, 'Interface Statistics': 11}, + 'profile.pcapng': {'Section Header': 1, 'Interface Description': 2, 'Frame': 40, + 'Interface Statistics': 2}, +} + +#: The DHCP exchange all three ``dhcp*.pcapng`` fixtures carry, as +#: ``(src, dst, srcport, dstport, udp payload length)`` per frame. +DHCP_EXCHANGE = [ + ('0.0.0.0', '255.255.255.255', 68, 67, 272), + ('192.168.0.1', '192.168.0.10', 67, 68, 300), + ('0.0.0.0', '255.255.255.255', 68, 67, 272), + ('192.168.0.1', '192.168.0.10', 67, 68, 300), +] + +#: PCAP-NG's own magic, i.e. the section header block's byte-order magic. +PCAPNG_MAGIC = b'\n\r\r\n' + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class PcapngTreeReportTests(EndToEndTestCase): + """Every fixture dumps a tree report.""" + + def test_every_fixture_dumps_a_report_with_its_expected_frame_count(self) -> None: + for capture, frames in PCAPNG_FRAMES.items(): + with self.subTest(capture=capture): + extractor = self.extract(fin=sample_path(capture), fout=self.out(capture), + format='tree', store=False) + + report = self.tmp_path / f'{capture}.txt' + self.assertEqual(extractor.length, frames) + self.assertEqual(extractor.magic_number, PCAPNG_MAGIC) + self.assertTrue(report.is_file()) + self.assertGreater(report.stat().st_size, 0) + self.assertIn('Section Header', report.read_text(encoding='utf-8')) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class PcapngJsonReportTests(EndToEndTestCase): + """The block inventory each fixture's report records.""" + + def test_reports_hold_the_expected_blocks(self) -> None: + for capture, sections in PCAPNG_SECTIONS.items(): + with self.subTest(capture=capture): + extractor = self.extract(fin=sample_path(capture), fout=self.out(capture), + format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(section_counts(report), sections) + self.assertEqual(extractor.length, PCAPNG_FRAMES[capture]) + + def test_interface_descriptions_are_reported_one_per_interface(self) -> None: + extractor = self.extract(fin=sample_path('many_interfaces.pcapng'), + fout=self.out('many'), format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(extractor.length, 64) + for number in range(1, 12): + with self.subTest(interface=number): + self.assertIn(f'Interface Description {number}', report) + self.assertNotIn('Interface Description 12', report) + + def test_split_reports_cover_every_block_not_just_the_frames(self) -> None: + extractor = self.extract(fin=sample_path('dhcp.pcapng'), fout=self.out('blocks'), + format='json', files=True, store=False) + + self.assertEqual(report_stems(extractor.output), sorted([ + 'Section Header 1', 'Interface Description 1', + 'Frame 1', 'Frame 2', 'Frame 3', 'Frame 4', + ])) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class PcapngByteOrderTests(EndToEndTestCase): + """The same DHCP exchange read out of three differently laid out files. + + :file:`dhcp.pcapng` is the committed upstream capture, ``_big_endian`` is a + big-endian section header downloaded from the Wireshark tree, and + ``_little_endian`` is synthesised. If the section header's byte-order magic + is honoured, all three yield the same four DHCP datagrams. + + """ + + def exchange(self, capture: 'str') -> 'list[tuple[str, str, int, int, int]]': + extractor = self.extract(fin=sample_path(capture), nofile=True, store=True) + self.assertEqual(extractor.length, 4) + + rows = [] + for frame in extractor.frame: + ipv4 = frame['IPv4'].info + udp = frame['UDP'] + rows.append((str(ipv4.src), str(ipv4.dst), + udp.info.srcport.port, udp.info.dstport.port, + len(bytes(udp.packet.payload)))) + return rows + + def test_all_three_layouts_yield_the_same_exchange(self) -> None: + for capture in ('dhcp.pcapng', 'dhcp_big_endian.pcapng', 'dhcp_little_endian.pcapng'): + with self.subTest(capture=capture): + self.assertEqual(self.exchange(capture), DHCP_EXCHANGE) + + def test_all_three_layouts_yield_the_same_protocol_chains(self) -> None: + chains = {} + for capture in ('dhcp.pcapng', 'dhcp_big_endian.pcapng', 'dhcp_little_endian.pcapng'): + extractor = self.extract(fin=sample_path(capture), nofile=True, store=True) + chains[capture] = [str(frame.protochain) for frame in extractor.frame] + + self.assertEqual(list(chains.values()), + [['Ethernet:IPv4:UDP:Raw'] * 4] * 3) + + +class PcapngUnescapedKeyTests(EndToEndTestCase): + """The report of :file:`test.pcapng`, which carries a decryption secrets block. + + Its tree report is fine -- :class:`PcapngTreeReportTests` covers it -- but + the ``json`` and ``plist`` reports come out unparseable, so the round trip + lives here behind a skip rather than being asserted either way. + + """ + + @unittest.skip('blocked on dictdumper interpolating mapping keys into the report without ' + 'escaping them (dictdumper/json.py:224), which the bytes-keyed TLS key log ' + 'entries of a decryption secrets block then break') + def test_json_report_of_a_decryption_secrets_block_parses(self) -> None: + """A ``json`` report should be readable by :func:`json.load`. + + For this fixture it is not, and two things stack up to make it so: + + 1. ``pcapkit/protocols/schema/misc/pcapng.py:1450`` keys the TLS key log + entries by ``bytes.fromhex(random)``, i.e. by a raw :class:`bytes` + client random rather than by its hex text, so the report's key is a + Python ``bytes`` repr: ``b' !"#$%&\\'()*+,-./0123456789:;<=>?'``. + 2. ``dictdumper/json.py:224`` writes a key as + ``'"{item}": '.format(item=item)``, with no escaping at all -- values + go through ``_encode_value`` and are escaped, keys do not. The quotes + inside that repr therefore land in the document verbatim. Reproduce + with:: + + >>> import dictdumper, json + >>> dictdumper.JSON('probe.json')({'a"b': 'c"d'}) + >>> json.load(open('probe.json')) + json.decoder.JSONDecodeError: Expecting ':' delimiter ... + + Measured on this fixture: ``json.load`` fails with ``Expecting ':' + delimiter: line 915 column 12``. The same key makes the ``plist`` report + invalid XML at line 1376, where its ``&`` is unescaped. + + """ + extractor = self.extract(fin=sample_path('test.pcapng'), fout=self.out('report'), + format='json', store=False) + report = read_json(extractor.output) + + self.assertEqual(extractor.length, 5) + self.assertEqual(section_counts(report)['Frame'], 5) + self.assertIn('Decryption Secrets 1', report) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_reassembly_end_to_end.py b/tests/integration/test_reassembly_end_to_end.py new file mode 100644 index 0000000000..0e73f64aab --- /dev/null +++ b/tests/integration/test_reassembly_end_to_end.py @@ -0,0 +1,326 @@ +# -*- coding: utf-8 -*- +"""End-to-end reassembly of TCP payloads and IP fragments. + +Translates :file:`examples/legacy_smoke/test_reassembly.py` (TCP payloads out of +:file:`test.pcap`), :file:`test_analyse.py` (the application layer found in a +reassembled datagram, out of :file:`http6.cap`), :file:`test_ip_reasm.py` and +:file:`test_ipv6_reasm.py` (IPv4 and IPv6 datagrams). Those scripts printed +their datagrams; these assert on them. + +The captures are built for this. ``test.pcap`` spreads the IPv4 response body +over four segments delivered out of order with one of them retransmitted, and +the IPv6 request body over three segments delivered in order, so a reassembled +payload that matches its own ``Content-Length`` is evidence that the ordering +and the duplicate were both handled. See the module docstring of +:file:`examples/generators/legacy.py`. + +One expectation here is skipped rather than asserted: see +:class:`IPv6FragmentPayloadTests`. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_RUNTIME, EndToEndTestCase + + +def by_direction(datagrams: 'tuple') -> 'dict[tuple[str, int, str, int], object]': + """Key TCP datagrams by their connection and direction. + + The submission order of the datagram tuple is an implementation detail of + when each direction sent its FIN or RST, so the tests below look each + datagram up by its four-tuple instead of by position. + + """ + return { + (str(datagram.id.src[0]), datagram.id.src[1], + str(datagram.id.dst[0]), datagram.id.dst[1]): datagram + for datagram in datagrams + } + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TCPReassemblyTests(EndToEndTestCase): + """TCP payload reassembly over :file:`test.pcap`.""" + + def reassemble(self) -> 'dict': + extractor = self.extract(fin=sample_path('test.pcap'), nofile=True, store=False, + tcp=True, reassembly=True, reasm_strict=True) + self.assertEqual(extractor.length, 34) + return by_direction(extractor.reassembly.tcp) + + def test_every_direction_of_both_connections_yields_one_datagram(self) -> None: + datagrams = self.reassemble() + + self.assertEqual(sorted(datagrams), sorted([ + ('10.20.30.131', 49812, '203.0.113.42', 80), + ('203.0.113.42', 80, '10.20.30.131', 49812), + ('2001:db8:2f10::131', 49814, '2001:db8:9c::7a', 80), + ('2001:db8:9c::7a', 80, '2001:db8:2f10::131', 49814), + ])) + for key, datagram in datagrams.items(): + with self.subTest(direction=key): + self.assertTrue(datagram.completed) + self.assertIsInstance(datagram.payload, bytes) + + def test_out_of_order_and_retransmitted_segments_reassemble_in_order(self) -> None: + datagram = self.reassemble()[('203.0.113.42', 80, '10.20.30.131', 49812)] + + # The four body segments arrive 3, 1, 4, 2 with the third retransmitted; + # frames 7 and 8 carry the header and the first body segment. + self.assertEqual(datagram.index, (7, 8, 10, 12, 14, 16, 18)) + + header, separator, body = datagram.payload.partition(b'\r\n\r\n') + self.assertEqual(separator, b'\r\n\r\n') + self.assertTrue(header.startswith(b'HTTP/1.1 200 OK\r\n')) + self.assertIn(b'Content-Length: 4400', header) + # The retransmission is not counted twice and no segment is missing. + self.assertEqual(len(body), 4400) + self.assertEqual(len(datagram.payload), len(header) + 4 + 4400) + self.assertTrue(body.startswith(b'')) + self.assertTrue(body.endswith(b'\n')) + + def test_request_without_a_body_reassembles_from_its_two_frames(self) -> None: + datagram = self.reassemble()[('10.20.30.131', 49812, '203.0.113.42', 80)] + + self.assertEqual(datagram.index, (5, 6)) + self.assertTrue(datagram.payload.startswith(b'GET /index.html HTTP/1.1\r\n')) + self.assertTrue(datagram.payload.endswith(b'\r\n\r\n')) + self.assertIn(b'Host: web.example.com\r\n', datagram.payload) + + def test_ipv6_connection_reassembles_a_three_segment_request_body(self) -> None: + datagrams = self.reassemble() + request = datagrams[('2001:db8:2f10::131', 49814, '2001:db8:9c::7a', 80)] + response = datagrams[('2001:db8:9c::7a', 80, '2001:db8:2f10::131', 49814)] + + self.assertEqual(request.index, (24, 25, 27, 29)) + header, _, body = request.payload.partition(b'\r\n\r\n') + self.assertTrue(header.startswith(b'POST /v1/telemetry HTTP/1.1\r\n')) + self.assertIn(b'Content-Length: 3350', header) + self.assertEqual(len(body), 3350) + self.assertTrue(body.startswith(b'{"schema":"pcapkit.example/telemetry/1"')) + + # The client aborts this connection with an RST, which submits the + # server's direction just as a FIN would. + self.assertEqual(response.index, (30, 31, 33)) + self.assertTrue(response.payload.startswith(b'HTTP/1.1 204 No Content\r\n')) + + def test_reassembly_is_refused_when_it_was_not_requested(self) -> None: + # This is the trap ``examples/legacy_smoke/test_analyse.py`` falls into: + # it asks for ``tcp=True`` but never for ``reassembly=True``. + from pcapkit.utilities.exceptions import UnsupportedCall + + extractor = self.extract(fin=sample_path('test.pcap'), nofile=True, store=False, + tcp=True, reasm_strict=True) + + with self.assertRaises(UnsupportedCall): + extractor.reassembly # pylint: disable=pointless-statement + + def test_datagrams_are_not_retained_when_storage_is_disabled(self) -> None: + from pcapkit.utilities.exceptions import UnsupportedCall + + extractor = self.extract(fin=sample_path('test.pcap'), nofile=True, store=False, + tcp=True, reassembly=True, reasm_store=False) + + with self.assertRaises(UnsupportedCall): + extractor.reassembly.tcp # pylint: disable=pointless-statement + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class ApplicationLayerAnalysisTests(EndToEndTestCase): + """The application layer :mod:`pcapkit` finds in a reassembled datagram. + + :file:`http6.cap` is two HTTP/1.1 connections over IPv6 -- a page fetch + whose response body spans three segments, and a conditional request answered + ``304 Not Modified`` with no body -- so all four datagrams analyse as HTTP. + + """ + + def analyse(self) -> 'dict': + extractor = self.extract(fin=sample_path('http6.cap'), nofile=True, store=False, + tcp=True, reassembly=True, reasm_strict=True) + self.assertEqual(extractor.length, 26) + return by_direction(extractor.reassembly.tcp) + + def test_all_four_datagrams_analyse_as_http(self) -> None: + import pcapkit + + datagrams = self.analyse() + self.assertEqual(len(datagrams), 4) + + for key, datagram in datagrams.items(): + with self.subTest(direction=key): + self.assertTrue(datagram.completed) + self.assertIsNotNone(datagram.packet) + self.assertIn(pcapkit.HTTP, datagram.packet) + self.assertEqual(datagram.packet.alias, 'HTTP/1.1') + + def test_page_fetch_is_analysed_as_a_request_and_a_response(self) -> None: + datagrams = self.analyse() + request = datagrams[('2001:db8:2f10::131', 49820, '2001:db8:9c::50', 80)] + response = datagrams[('2001:db8:9c::50', 80, '2001:db8:2f10::131', 49820)] + + self.assertEqual(request.index, (3, 4)) + self.assertEqual(request.packet.info.receipt.type.value, 'request') + self.assertEqual(request.packet.info.receipt.method.value, 'GET') + self.assertEqual(request.packet.info.receipt.uri, '/') + self.assertEqual(request.packet.info.header['Host'], 'www6.example.com') + self.assertIsNone(request.packet.info.body) + + self.assertEqual(response.index, (5, 6, 8, 10, 12)) + self.assertEqual(response.packet.info.receipt.type.value, 'response') + self.assertEqual(int(response.packet.info.receipt.status), 200) + self.assertEqual(response.packet.info.receipt.message, 'OK') + content_length = int(response.packet.info.header['Content-Length']) + self.assertEqual(len(response.packet.info.body), content_length) + + def test_conditional_request_is_analysed_as_a_bodyless_304(self) -> None: + datagrams = self.analyse() + request = datagrams[('2001:db8:2f10::131', 49822, '2001:db8:9c::50', 80)] + response = datagrams[('2001:db8:9c::50', 80, '2001:db8:2f10::131', 49822)] + + self.assertEqual(request.packet.info.receipt.uri, '/assets/site.css') + self.assertIn('If-None-Match', request.packet.info.header) + + self.assertEqual(int(response.packet.info.receipt.status), 304) + self.assertEqual(response.packet.info.receipt.message, 'Not Modified') + self.assertNotIn('Content-Length', response.packet.info.header) + self.assertIsNone(response.packet.info.body) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class IPv4ReassemblyTests(EndToEndTestCase): + """IPv4 datagram reassembly over :file:`ipv4.pcap`. + + The fixture is a multicast stream captured on the sending host, so none of + its four datagrams is fragmented; what this covers is the path where an + unfragmented datagram is submitted whole. No fixture in the repository + carries IPv4 fragments, so the multi-fragment path is not reachable from + here -- the IPv6 class below is where fragmentation is exercised. + + """ + + def test_each_unfragmented_datagram_is_submitted_whole(self) -> None: + extractor = self.extract(fin=sample_path('ipv4.pcap'), nofile=True, store=True, + ipv4=True, reassembly=True, reasm_strict=True) + datagrams = extractor.reassembly.ipv4 + + self.assertEqual(extractor.length, 4) + self.assertEqual(len(datagrams), 4) + + for number, datagram in enumerate(datagrams, start=1): + with self.subTest(datagram=number): + self.assertTrue(datagram.completed) + self.assertEqual(datagram.index, (number,)) + self.assertEqual(str(datagram.id.src), '172.31.127.230') + self.assertEqual(str(datagram.id.dst), '239.1.3.3') + self.assertEqual(int(datagram.id.proto), 17) + # The payload is the IPv4 payload of that very frame: the eight + # octet UDP header plus its 1828 octet payload. + frame_payload = bytes(extractor.frame[number - 1]['IPv4'].packet.payload) + self.assertEqual(bytes(datagram.payload), frame_payload) + self.assertEqual(len(datagram.payload), 1836) + + def test_datagram_identifications_are_distinct_and_consecutive(self) -> None: + extractor = self.extract(fin=sample_path('ipv4.pcap'), nofile=True, store=False, + ipv4=True, reassembly=True, reasm_strict=True) + + self.assertEqual([datagram.id.id for datagram in extractor.reassembly.ipv4], + [31232, 31233, 31234, 31235]) + + def test_reassembled_datagram_analyses_as_udp(self) -> None: + extractor = self.extract(fin=sample_path('ipv4.pcap'), nofile=True, store=False, + ipv4=True, reassembly=True, reasm_strict=True) + datagram = extractor.reassembly.ipv4[0] + + self.assertEqual(type(datagram.packet).__name__, 'UDP') + self.assertEqual(len(bytes(datagram.packet.packet.payload)), 1828) + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class IPv6FragmentReassemblyTests(EndToEndTestCase): + """IPv6 fragment reassembly over :file:`ipv6.pcap`. + + Frames 13 to 16 are one 4778 octet UDP datagram fragmented into + 1448/1448/1448/434 octets. What is asserted here is which frames were + collected and that the datagram is reported complete -- both of which are + right. Its *contents* are not, so they live in the skipped test below. + + """ + + def reassemble(self) -> 'tuple': + extractor = self.extract(fin=sample_path('ipv6.pcap'), nofile=True, store=False, + ipv6=True, reassembly=True, reasm_strict=True) + self.assertEqual(extractor.length, 16) + return extractor.reassembly.ipv6 + + def test_only_the_fragmented_datagram_is_submitted(self) -> None: + datagrams = self.reassemble() + + # Twelve of the sixteen frames are neighbour discovery and ICMPv6 + # echoes, which carry no fragment header and are dismissed. + self.assertEqual(len(datagrams), 1) + self.assertEqual(datagrams[0].index, (13, 14, 15, 16)) + + def test_the_datagram_is_reported_complete(self) -> None: + datagram = self.reassemble()[0] + + self.assertTrue(datagram.completed) + self.assertIsInstance(datagram.payload, bytes) + self.assertEqual(str(datagram.id.src), 'fe80::a423:b61d:7c92:70c6') + self.assertEqual(str(datagram.id.dst), 'fe80::821f:12ff:fec9:d13d') + self.assertEqual(int(datagram.id.proto), 17) + + +class IPv6FragmentPayloadTests(EndToEndTestCase): + """What the reassembled IPv6 datagram should contain. + + Kept apart from :class:`IPv6FragmentReassemblyTests` so the skip covers only + the payload, and the assertions that are correct today keep running. + + """ + + @unittest.skip('blocked on IPv6 fragment offsets being used as byte offsets while the ' + 'field is in eight-octet units (pcapkit/protocols/internet/ipv6_frag.py:141)') + def test_the_four_fragments_reassemble_into_the_whole_datagram(self) -> None: + """The reassembled payload should be the four fragments, in order. + + It is not. ``pcapkit/protocols/internet/ipv4.py:274`` scales the IPv4 + fragment offset into octets (``int(schema.flags['offset']) * 8``), but + ``pcapkit/protocols/internet/ipv6_frag.py:141`` passes the IPv6 one + through unscaled (``offset=schema.flags['offset']``), and + ``pcapkit/toolkit/pcap.py:115`` then hands it to the reassembly + machinery as ``fo``, which is a byte offset. The four fragments are + therefore placed at octets 0, 181, 362 and 543 instead of 0, 1448, 2896 + and 4344, so they overwrite one another and the datagram comes out 977 + octets long -- ``543 + 434`` -- while still reporting + ``completed=True``. + + Measured on this fixture: ``len(datagram.payload)`` is 977, and the + assertion below expects 4778. + + A second defect is visible in the same call and is left alone here: + ``pcapkit/toolkit/pcap.py:111`` keys the reassembly buffer on the IPv6 + header's flow label rather than on the fragment header's identification, + so ``datagram.id.id`` is 0 where the fixture's identification is 110308. + One fragmented datagram cannot show the consequence, so no assertion is + made about it either way. + + """ + extractor = self.extract(fin=sample_path('ipv6.pcap'), nofile=True, store=True, + ipv6=True, reassembly=True, reasm_strict=True) + datagram = extractor.reassembly.ipv6[0] + + fragments = [ + bytes(extractor.frame[number - 1]['IPv6'].info.fragment.payload) + for number in (13, 14, 15, 16) + ] + self.assertEqual([len(fragment) for fragment in fragments], [1448, 1448, 1448, 434]) + self.assertEqual(len(datagram.payload), 4778) + self.assertEqual(bytes(datagram.payload), b''.join(fragments)) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/integration/test_traceflow_end_to_end.py b/tests/integration/test_traceflow_end_to_end.py new file mode 100644 index 0000000000..07e9987ceb --- /dev/null +++ b/tests/integration/test_traceflow_end_to_end.py @@ -0,0 +1,125 @@ +# -*- coding: utf-8 -*- +"""End-to-end TCP flow tracing. + +Translates :file:`examples/legacy_smoke/test_trace.py`, which traced +:file:`http.pcap` and pretty-printed the index, and the ``trace=True`` half of +:file:`test_api.py`. Tracing writes one report per flow into ``trace_fout``, so +what is asserted here is the index the extractor exposes *and* that the reports +on disk hold exactly the frames the index claims. + +""" +from __future__ import annotations + +import unittest + +from tests._support import sample_path +from tests.integration._helpers import HAS_RUNTIME, EndToEndTestCase, read_json + +#: The four directional flows of :file:`tcp.pcap`: two concurrent SSH sessions, +#: one over IPv4 and one over IPv6, each traced per direction. +TCP_PCAP_FLOWS = { + '10.20.30.130_22-10.20.30.131_53406-1500000000.000774': (1, 7), + '10.20.30.131_53406-10.20.30.130_22-1500000000.001585': (2, 6), + 'fe80..a6.87f9.2793.16ee_51774-fe80..1ccd.7c77.bac7.46b7_22-1500000000.002433': (3,), + 'fe80..1ccd.7c77.bac7.46b7_22-fe80..a6.87f9.2793.16ee_51774-1500000000.003318': (4, 5), +} + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TraceFlowTests(EndToEndTestCase): + """Flow tracing over :file:`tcp.pcap`, which is seven frames.""" + + def trace(self, capture: 'str' = 'tcp.pcap') -> 'tuple': + extractor = self.extract(fin=sample_path(capture), nofile=True, store=False, + tcp=True, trace=True, trace_format='json', + trace_fout=self.out('trace')) + return extractor.length, extractor.trace.tcp + + def test_every_direction_of_every_connection_becomes_a_flow(self) -> None: + length, flows = self.trace() + + self.assertEqual(length, 7) + self.assertEqual({flow.label: flow.index for flow in flows}, TCP_PCAP_FLOWS) + + def test_flow_labels_carry_both_address_families(self) -> None: + _, flows = self.trace() + labels = {flow.label for flow in flows} + + # A label is ``src_port-dst_port-timestamp``, with the dots of an IPv6 + # address doubled so the label stays usable as a file name. + self.assertEqual(sum(1 for label in labels if label.startswith('10.20.30.')), 2) + self.assertEqual(sum(1 for label in labels if label.startswith('fe80..')), 2) + + def test_each_flow_is_dumped_to_its_own_report(self) -> None: + _, flows = self.trace() + + written = sorted(entry.name for entry in self.tmp_path.joinpath('trace').iterdir()) + self.assertEqual(written, sorted(f'{label}.json' for label in TCP_PCAP_FLOWS)) + + for flow in flows: + with self.subTest(flow=flow.label): + self.assertEqual(flow.fpout, self.out(f'trace/{flow.label}.json')) + self.assertTrue(self.tmp_path.joinpath('trace', f'{flow.label}.json').is_file()) + + def test_a_flow_report_holds_exactly_the_frames_its_index_names(self) -> None: + _, flows = self.trace() + + for flow in flows: + with self.subTest(flow=flow.label): + report = read_json(flow.fpout) + self.assertEqual(list(report), [f'Frame {number}' for number in flow.index]) + for number in flow.index: + self.assertEqual(report[f'Frame {number}']['number'], number) + + def test_tracing_is_refused_when_it_was_not_requested(self) -> None: + from pcapkit.utilities.exceptions import UnsupportedCall + + extractor = self.extract(fin=sample_path('tcp.pcap'), nofile=True, store=False, tcp=True) + + with self.assertRaises(UnsupportedCall): + extractor.trace # pylint: disable=pointless-statement + + +@unittest.skipUnless(HAS_RUNTIME, 'runtime dependencies not installed') +class TraceFlowScaleTests(EndToEndTestCase): + """Flow tracing over :file:`http.pcap`. + + This is the one place in this tier that uses the 1117 frame capture, and it + is used deliberately: it is the only fixture with hundreds of short + connections, so it is the only one that shows the tracer partitioning a + capture rather than handling a handful of flows. It costs roughly three + seconds, and ``examples/legacy_smoke/test_trace.py`` traced this same file. + + """ + + def test_every_frame_lands_in_exactly_one_flow(self) -> None: + extractor = self.extract(fin=sample_path('http.pcap'), nofile=True, store=False, + tcp=True, trace=True, trace_format='json', + trace_fout=self.out('trace')) + flows = extractor.trace.tcp + + self.assertEqual(extractor.length, 1117) + self.assertEqual(len(flows), 331) + + indexed = [number for flow in flows for number in flow.index] + self.assertEqual(len(indexed), 1117) + self.assertEqual(sorted(indexed), list(range(1, 1118))) + + def test_one_report_is_written_per_flow(self) -> None: + extractor = self.extract(fin=sample_path('http.pcap'), nofile=True, store=False, + tcp=True, trace=True, trace_format='json', + trace_fout=self.out('trace')) + flows = extractor.trace.tcp + + written = {entry.name for entry in self.tmp_path.joinpath('trace').iterdir()} + self.assertEqual(written, {f'{flow.label}.json' for flow in flows}) + self.assertEqual(len(written), 331) + + # Spot-check the longest flow rather than re-reading all 331 reports. + longest = max(flows, key=lambda flow: len(flow.index)) + report = read_json(longest.fpout) + self.assertEqual(list(report), [f'Frame {number}' for number in longest.index]) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/toolkit/test_dpkt_unit.py b/tests/toolkit/test_dpkt_unit.py index 9c47547d24..7324376bfc 100644 --- a/tests/toolkit/test_dpkt_unit.py +++ b/tests/toolkit/test_dpkt_unit.py @@ -153,8 +153,12 @@ def test_tcp_reassembly_and_traceflow_accept_dpkt_data_payload_tcp(self) -> None self.assertEqual(tcp.bufid[3], 80) self.assertTrue(tcp.syn) self.assertFalse(tcp.fin) + # ``last`` is the sequence number of the payload's last octet, not of + # the one after it, so ``b'data'`` at sequence number 10 ends at 13 + self.assertEqual(tcp.len, 4) self.assertEqual(tcp.first, 10) - self.assertEqual(tcp.last, 14) + self.assertEqual(tcp.last, 13) + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) flow = toolkit.tcp_traceflow(packet, 50.25, data_link=LinkType.ETHERNET, count=9) self.assertIsNotNone(flow) diff --git a/tests/toolkit/test_pcap_unit.py b/tests/toolkit/test_pcap_unit.py index 82004adf99..3ddfa7320d 100644 --- a/tests/toolkit/test_pcap_unit.py +++ b/tests/toolkit/test_pcap_unit.py @@ -78,30 +78,36 @@ def _make_ipv6(self, *, with_fragment: bool = True): ), ) - def _make_tcp(self): + def _make_tcp(self, *, seq: int = 10, payload: bytes = b'tcpdata', + syn: bool = True, fin: bool = False, rst: bool = True): return types.SimpleNamespace( info=types.SimpleNamespace( srcport=port(1234), dstport=port(80), ack=5, - seq=10, - flags=flags(syn=True, fin=False, rst=True), + seq=seq, + flags=flags(syn=syn, fin=fin, rst=rst), ), - packet=types.SimpleNamespace(header=b'T' * 20, payload=b'tcpdata'), + packet=types.SimpleNamespace(header=b'T' * 20, payload=payload), ) def _make_pcap_frame(self, *, include_tcp: bool = True, - ipv4_df: bool = False) -> FakeFrame: + ipv4_df: bool = False, seq: int = 10, + payload: bytes = b'tcpdata', number: int = 9, + syn: bool = True, fin: bool = False, + rst: bool = True) -> FakeFrame: layers = { 'IPv4': self._make_ipv4(df=ipv4_df), } layers['IP'] = layers['IPv4'] if include_tcp: - layers['TCP'] = self._make_tcp() - return FakeFrame(layers, types.SimpleNamespace(number=9, time_epoch='50.25')) + layers['TCP'] = self._make_tcp(seq=seq, payload=payload, + syn=syn, fin=fin, rst=rst) + return FakeFrame(layers, types.SimpleNamespace(number=number, time_epoch='50.25')) def _make_pcapng_frame(self, *, include_tcp: bool = True, - ipv4_df: bool = False): + ipv4_df: bool = False, seq: int = 10, + payload: bytes = b'tcpdata'): from pcapkit.const.reg.linktype import LinkType layers = { @@ -109,7 +115,7 @@ def _make_pcapng_frame(self, *, include_tcp: bool = True, } layers['IP'] = layers['IPv4'] if include_tcp: - layers['TCP'] = self._make_tcp() + layers['TCP'] = self._make_tcp(seq=seq, payload=payload) return FakeFrame( layers, types.SimpleNamespace(number=11, timestamp_epoch=60.5, @@ -159,7 +165,13 @@ def test_pcap_ipv4_ipv6_tcp_and_traceflow_helpers(self) -> None: self.assertEqual(tcp.bufid[3], 80) self.assertTrue(tcp.syn) self.assertTrue(tcp.rst) - self.assertEqual(tcp.last, 17) + # ``first``/``last`` bound the payload in absolute sequence numbers and + # ``last`` is inclusive, so the two describe exactly ``len`` octets -- + # ``seq`` is 10 and the payload is the seven octets of ``b'tcpdata'``. + self.assertEqual(tcp.len, 7) + self.assertEqual(tcp.first, 10) + self.assertEqual(tcp.last, 16) + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) self.assertIsNone(toolkit.tcp_reassembly(self._make_pcap_frame(include_tcp=False))) flow = toolkit.tcp_traceflow(frame, data_link=LinkType.ETHERNET) @@ -200,7 +212,12 @@ def test_pcapng_helpers_and_block_to_frame_timestamp_scaling(self) -> None: self.assertIsNotNone(tcp) assert tcp is not None self.assertEqual(tcp.bufid[1], 1234) - self.assertEqual(tcp.last, 17) + # same inclusive bounds as the PCAP engine above -- the two toolkits + # must agree, or a stream reassembles differently per capture format + self.assertEqual(tcp.len, 7) + self.assertEqual(tcp.first, 10) + self.assertEqual(tcp.last, 16) + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) self.assertIsNone(toolkit.tcp_reassembly(self._make_pcapng_frame(include_tcp=False))) flow = toolkit.tcp_traceflow(frame) @@ -224,6 +241,93 @@ def test_pcapng_helpers_and_block_to_frame_timestamp_scaling(self) -> None: pcap_frame_ns = toolkit.block2frame(block, nanosecond=True) self.assertEqual(pcap_frame_ns.frame_info.ts_usec, 250000000) + def test_tcp_reassembly_descriptor_bounds_are_absolute_and_inclusive(self) -> None: + """The segment descriptor handed to the reassembler (GitHub issue #349). + + ``first`` and ``last`` are absolute TCP sequence numbers bounding the + payload, and ``last`` is the sequence number of its final octet rather + than of the octet after it. Both matter: the reassembler holds its + :rfc:`815` hole descriptor list in the same absolute space and treats + both bounds as inclusive, so a descriptor that is relative, or exclusive + at the top end, silently corrupts the hole list -- and this is not + visible at a sequence number of zero, which is why a realistic one is + used here. + + Both the PCAP and the PCAP-NG toolkit are checked, and against each + other, since they feed the one reassembler and a stream must not + reassemble differently depending on the capture format it arrived in. + + """ + from pcapkit.toolkit import pcap, pcapng + + # a realistic initial sequence number, plus the 1448 octets an Ethernet + # path carries with a timestamp option in the TCP header + seq = 0xC0DE1234 + payload = bytes(range(256)) * 5 + bytes(168) + self.assertEqual(len(payload), 1448) + + for (name, toolkit, frame) in ( + ('pcap', pcap, self._make_pcap_frame(seq=seq, payload=payload)), + ('pcapng', pcapng, self._make_pcapng_frame(seq=seq, payload=payload)), + ): + with self.subTest(engine=name): + tcp = toolkit.tcp_reassembly(frame) + self.assertIsNotNone(tcp) + assert tcp is not None + + self.assertEqual(tcp.dsn, seq) + self.assertEqual(tcp.len, 1448) + self.assertEqual(tcp.first, seq) # absolute, not an offset + self.assertEqual(tcp.last, seq + 1447) # inclusive, not seq + 1448 + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) + self.assertEqual(bytes(tcp.payload), payload) + + # a segment carrying no payload bounds an empty range, i.e. ``last`` one + # below ``first``, rather than claiming the octet it does not carry + empty = pcap.tcp_reassembly(self._make_pcap_frame(seq=seq, payload=b'')) + self.assertIsNotNone(empty) + assert empty is not None + self.assertEqual(empty.len, 0) + self.assertEqual((empty.first, empty.last), (seq, seq - 1)) + + def test_tcp_reassembly_descriptor_drives_the_reassembler(self) -> None: + """The descriptor the toolkit builds reassembles to the octets sent. + + Guards the seam rather than either side of it: the toolkit and the + reassembler each looked self-consistent while disagreeing about what + ``first`` and ``last`` meant. + + """ + from pcapkit.foundation.reassembly.tcp import TCP + from pcapkit.toolkit import pcap + + class Analyzer: + @classmethod + def analyze(cls, ports: tuple[int, int], payload: bytes) -> bytes: + return payload + + class TestTCP(TCP): + __protocol_type__ = Analyzer + + seq = 0xC0DE1234 + reasm = TestTCP() + + # three data segments and then a bare RST at the sequence number after + # the data, which is what makes the reassembler submit the buffer + deliveries = ((1, 0, b'first-', False), (2, 6, b'second-', False), + (3, 13, b'third', False), (4, 18, b'', True)) + for (number, offset, payload, rst) in deliveries: + frame = self._make_pcap_frame(seq=seq + offset, payload=payload, + number=number, syn=False, rst=rst) + packet = pcap.tcp_reassembly(frame) + self.assertIsNotNone(packet) + assert packet is not None + reasm(packet) + + datagram, = reasm.datagram + self.assertTrue(datagram.completed) + self.assertEqual(datagram.payload, b'first-second-third') + if __name__ == '__main__': unittest.main() diff --git a/tests/toolkit/test_scapy_unit.py b/tests/toolkit/test_scapy_unit.py index 57236d485e..90c8b9955a 100644 --- a/tests/toolkit/test_scapy_unit.py +++ b/tests/toolkit/test_scapy_unit.py @@ -178,6 +178,13 @@ def test_tcp_reassembly_and_traceflow(self) -> None: self.assertFalse(tcp.rst) self.assertEqual(tcp.header, bytes(tcp_layer)[:tcp_layer.dataofs * 4]) self.assertEqual(bytes(tcp.payload), bytes(tcp_layer[Raw])) + # ``first``/``last`` are absolute sequence numbers bounding the payload + # inclusively, so they span exactly ``len`` octets -- the same + # convention every other engine's toolkit has to use, since they all + # feed the one reassembler + self.assertEqual(tcp.first, tcp_layer.seq) + self.assertEqual(tcp.last, tcp_layer.seq + tcp.len - 1) + self.assertEqual(tcp.last - tcp.first + 1, tcp.len) v6_tcp = toolkit.tcp_reassembly(self._make_ipv6_tcp_packet(), count=9) self.assertIsNotNone(v6_tcp) assert v6_tcp is not None