Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
92 changes: 88 additions & 4 deletions pcapkit/corekit/fields/collections.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
from typing import TYPE_CHECKING, Generic, TypeVar, cast

from pcapkit.corekit.fields.field import FieldBase
from pcapkit.corekit.fields.numbers import NumberField
from pcapkit.corekit.multidict import OrderedMultiDict
from pcapkit.utilities.compat import List
from pcapkit.utilities.exceptions import FieldValueError
Expand Down Expand Up @@ -178,6 +179,29 @@ class OptionField(ListField, Generic[_TS]):
This field is used to represent a list of fields, as in the case of lists of
options and/or parameters in a protocol.

Note:
:meth:`self.unpack <unpack>` selects an option's schema by reading the
``type_name`` field of ``base_schema`` **alone** off the front of the
option, instead of unpacking the whole base schema and keeping only that
one value. That is only the same read when three things hold of the type
field:

1. it is the base schema's **first** field, so that it is what sits at the
front of the option;
2. it is a :class:`~pcapkit.corekit.fields.numbers.NumberField`, so that
:meth:`Schema.unpack <pcapkit.protocols.schema.schema.Schema.unpack>`
reads it through its ordinary per-field branch;
3. its length is a fixed integer rather than a callable, so that its width
does not depend on packet data the shortcut has not read.

All of that is true of every ``OptionField`` declared in this package. A
base schema registered from outside it -- c.f.
:mod:`pcapkit.foundation.registry` -- need not satisfy it, and is **not**
rejected: such a base schema is unpacked in full, exactly as it was before
the shortcut existed. It parses correctly and pays the cost of the second
parse. Only the shortcut is withheld, so a base schema whose type field
comes second cannot be silently misread.

"""

@property
Expand Down Expand Up @@ -228,6 +252,34 @@ def __init__(self, length: 'int | Callable[[dict[str, Any]], int]' = lambda _: -
raise FieldValueError('Field <option> has no registry.')
self._registry = registry

# NOTE: Decided once, here, rather than per option in ``unpack``. See the
# docstring above for what the fast path requires and why a base schema
# that does not meet it is accommodated rather than rejected.
#
# The three conditions are exactly what makes reading the type field alone
# the same read that ``Schema.unpack`` performs for it: it must sit at the
# front of the option, ``Schema.unpack`` must handle it through its generic
# branch rather than one of the special ones (a ``NumberField`` is never
# ``PayloadField``, ``PaddingField``, ``ConditionalField``,
# ``ForwardMatchField``, ``SwitchField`` or ``OptionField``), and its width
# must not depend on packet state that the full unpack would have
# established first.
#
# Every test is written so that an unexpected base schema selects the full
# unpack instead of raising, so this cannot turn a registration that works
# today into an import-time failure.
fields = getattr(self._base_schema, '__fields__', {})
type_field = fields.get(type_name)

#: Optional[FieldBase]: Base schema's type field, when reading it on its own
#: is equivalent to unpacking the whole base schema to obtain it; otherwise
#: :data:`None`, and :meth:`self.unpack <unpack>` unpacks the base schema.
self._type_field = type_field if (
isinstance(type_field, NumberField)
and list(fields)[:1] == [type_name]
and type_field._length_callback is None # pylint: disable=protected-access
) else None

def unpack(self, buffer: 'bytes | IO[bytes]', packet: 'dict[str, Any]') -> 'list[_TS]':
"""Unpack field value from :obj:`bytes`.

Expand Down Expand Up @@ -256,15 +308,47 @@ def unpack(self, buffer: 'bytes | IO[bytes]', packet: 'dict[str, Any]') -> 'list
new_packet = packet.copy()
new_packet[self.name] = OrderedMultiDict()

# NOTE: Where it can, this reads the base schema's type field alone rather
# than the whole base schema. The option schema below re-reads the same
# octets from the rewound stream, and the base schema's result is used for
# nothing but the type code, so unpacking it in full parsed every option
# twice -- 2274 schema unpacks for 1137 options of
# ``examples/captures/profile.pcapng``.
#
# Reading the type field alone deliberately mirrors what
# :meth:`Schema.unpack <pcapkit.protocols.schema.schema.Schema.unpack>`
# does for one field -- ``field(packet)``, read ``field.length`` octets,
# then ``field.unpack(byte, packet.copy())`` -- rather than reaching for
# :func:`struct.unpack`. That keeps ``code``'s enumeration type, which both
# the ``self._eool`` comparison and the ``OrderedMultiDict`` key depend on,
# and it is what makes the two spellings equivalent rather than merely
# similar.
#
# ``self._type_field`` is :data:`None` for a base schema that the shortcut
# does not fit -- see the class docstring -- and the full unpack is used
# for it instead. Whichever way the code was read, ``consumed`` is the
# number of octets to rewind to get back to the front of the option.
type_field = self._type_field

temp = [] # type: list[_TS]
while length > 0:
# unpack option type using base schema
meta = self._base_schema.unpack(file, length, packet) # type: ignore[call-arg,misc,var-annotated]
code = cast('int', meta[self._type_name])
if type_field is None:
# unpack option type using base schema
meta = self._base_schema.unpack(file, length, packet) # type: ignore[call-arg,misc,var-annotated]
code = cast('int', meta[self._type_name])
consumed = len(meta)
else:
# unpack option type using the base schema's type field. No cast
# is needed here, unlike the branch above: the type field is known
# to be a ``NumberField``, so ``unpack`` is already typed ``int``.
field = type_field(packet)
byte = file.read(field.length)
code = field.unpack(byte, packet.copy())
consumed = len(byte)
schema = self._registry[code]

# rewind to the beginning of the option
file.seek(-len(meta), io.SEEK_CUR)
file.seek(-consumed, io.SEEK_CUR)

# unpack option using option schema
data = schema.unpack(file, length, packet) # type: ignore[call-arg,misc,var-annotated]
Expand Down
72 changes: 62 additions & 10 deletions pcapkit/dumpkit/pcap.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,12 +9,12 @@
:mod:`dictdumper`.

"""
import struct
import sys
from typing import TYPE_CHECKING

from pcapkit.dumpkit.common import DumperBase as Dumper
from pcapkit.protocols.data.misc.pcap.header import Header as Data_Header
from pcapkit.protocols.misc.pcap.frame import Frame
from pcapkit.protocols.misc.pcap.header import Header

if TYPE_CHECKING:
Expand All @@ -31,6 +31,23 @@
'PCAPIO',
]

#: Per-byte-order record header packers for the four ``uint32`` fields of a PCAP
#: record header -- ``ts_sec``, ``ts_usec``, ``incl_len``, ``orig_len``. Keyed by
#: the byte order of the global header this dumper wrote, so that the records
#: agree with the magic number a reader will find in front of them.
_RECORD_HEADER = {
'little': struct.Struct('<IIII'),
'big': struct.Struct('>IIII'),
}

#: Truncation mask for those four fields. :class:`~pcapkit.corekit.fields.numbers.UInt32Field`,
#: which used to pack them, masks to the field width in
#: :meth:`~pcapkit.corekit.fields.numbers.NumberField.pre_process` rather than
#: rejecting an out-of-range value -- a ``ts_sec`` of ``2**32 + 5`` was written as
#: ``5``. :func:`struct.pack` raises instead, so the mask is applied here to keep
#: the two spellings writing the same octets for every input.
_UINT32_MASK = 0xFFFF_FFFF


class PCAPIO(Dumper):
"""PCAP file dumper.
Expand All @@ -46,6 +63,8 @@ class PCAPIO(Dumper):
if TYPE_CHECKING:
#: PCAP file global header.
_ghdr: 'Data_Header'
#: Record header packer, in the global header's byte order.
_rechdr: 'struct.Struct'

##########################################################################
# Properties.
Expand Down Expand Up @@ -75,6 +94,11 @@ def __init__(self, fname: 'str', *, protocol: 'Enum_LinkType | StdlibIntEnum | A
"""
#: int: Frame counter.
self._fnum = 1
# NOTE: Both of these now only record how the dumper was configured -- the
# values that shape the output reach it through :meth:`self._dump_header
# <_dump_header>`'s own arguments, and are readable afterwards from
# :attr:`self._ghdr <_ghdr>`. They are kept because they are part of the
# instance surface a subclass may already read.
#: bool: Nanosecond-resolution file flag.
self._nsec = nanosecond
#: Enum_LinkType | StdlibIntEnum | AenumIntEnum | str | int: Data link type.
Expand Down Expand Up @@ -123,6 +147,13 @@ def _dump_header(self, *, protocol: 'Enum_LinkType | StdlibIntEnum | AenumIntEnu
file.write(packet)
self._ghdr = header.info

#: struct.Struct: Packer for the record header preceding each frame, in the
#: byte order of the global header just written. Taken from
#: :attr:`self._ghdr <_ghdr>` rather than from the ``byteorder`` argument
#: because :class:`~pcapkit.protocols.misc.pcap.header.Header` is what
#: validates and normalises it.
self._rechdr = _RECORD_HEADER[self._ghdr.magic_number.byteorder]

def _append_value(self, value: 'Data_Frame', file: 'IO[bytes]', name: 'str') -> 'None': # pylint: disable=unused-argument
"""Call this function to write contents.

Expand All @@ -131,14 +162,35 @@ def _append_value(self, value: 'Data_Frame', file: 'IO[bytes]', name: 'str') ->
file: output file
name: name of current content block

Notes:
A PCAP record is a 16-octet header followed by the packet octets, and
both are already in hand: the header fields are exactly
``value.frame_info`` and the octets are exactly ``value.packet``. So
this writes them directly, rather than handing them to
:class:`~pcapkit.protocols.misc.pcap.frame.Frame`, whose constructor
packs the record and then **dissects it again** through the whole
protocol stack to arrive at bytes it was given. That round trip was
about 82% of the cost of a flow-traced extraction -- ``http.pcap``,
1117 frames, best of 7: 2319 ms with the rebuild against 1263 ms
without, over a 1030 ms untraced baseline.

Dropping it is not only cheaper. The re-dissection re-emitted every
parse warning the frame had already produced once, and warned about
payloads it had no business parsing at all -- writing a 3-octet
payload raised ``SchemaWarning: packet length < 0: -3`` from a dumper
that only had to copy it.

"""
packet = Frame(
nanosecond=self._nsec,
num=self._fnum,
proto=self._link,
packet=value.packet,
header=self._ghdr,
**value.frame_info,
).data
file.write(packet)
# NOTE: The payload is read before the metadata so that a caller passing a
# mapping rather than a dissected frame -- which the flow-tracing adapters
# of several engines do -- still fails naming ``packet``, as the ``Frame``
# construction did. :mod:`pcapkit.foundation.extraction` substitutes a
# dict-capable trace format on the strength of that error.
packet = value.packet
frame_info = value.frame_info

file.write(self._rechdr.pack(frame_info.ts_sec & _UINT32_MASK,
frame_info.ts_usec & _UINT32_MASK,
frame_info.incl_len & _UINT32_MASK,
frame_info.orig_len & _UINT32_MASK) + packet)
Comment thread
JarryShaw marked this conversation as resolved.
self._fnum += 1
Loading