diff --git a/docs/source/pcapkit/utilities/chardet.rst b/docs/source/pcapkit/utilities/chardet.rst new file mode 100644 index 0000000000..29515d5cb0 --- /dev/null +++ b/docs/source/pcapkit/utilities/chardet.rst @@ -0,0 +1,17 @@ +Character Set Detection +======================= + +.. module:: pcapkit.utilities.chardet + +:mod:`pcapkit.utilities.chardet` wraps `chardet`_ with a bounded cache, for +turning the bytes of a text field into a :obj:`str`. It is shared by +:meth:`StringField.post_process +` and +:meth:`ProtocolBase.decode `, +which is why it lives here rather than beside either of them. + +.. _chardet: https://chardet.readthedocs.io + +.. autofunction:: pcapkit.utilities.chardet.detect + +.. autodata:: pcapkit.utilities.chardet.DETECT_CACHE_SIZE diff --git a/docs/source/pcapkit/utilities/index.rst b/docs/source/pcapkit/utilities/index.rst index 9a15ab3ae4..d99688d064 100644 --- a/docs/source/pcapkit/utilities/index.rst +++ b/docs/source/pcapkit/utilities/index.rst @@ -17,6 +17,7 @@ several user-refined exceptions and warnings. exceptions warnings logging + chardet Version Compatibility ===================== diff --git a/pcapkit/corekit/fields/field.py b/pcapkit/corekit/fields/field.py index 3e655e21e8..3c19b2a4e1 100644 --- a/pcapkit/corekit/fields/field.py +++ b/pcapkit/corekit/fields/field.py @@ -131,6 +131,24 @@ def __init__(self, *args: 'Any', **kwargs: 'Any') -> 'None': self._template = '0s' self._callback = lambda *_: None + def __copy__(self) -> 'Self': + """Return a shallow copy of the field. + + Every field of every protocol is copied once per packet by + :meth:`__call__`, which made the generic :func:`copy.copy` path -- via + :meth:`object.__reduce_ex__` and :func:`copy._reconstruct` -- one of the + costlier things an extraction did. This does what that path would have + done, and only that: a new instance of the same class, its + :attr:`~object.__dict__` shallow-updated from this one. + + Returns: + A new field instance sharing this one's attribute values. + + """ + new_self = self.__class__.__new__(self.__class__) + new_self.__dict__.update(self.__dict__) + return new_self + def __repr__(self) -> 'str': if not self.name.isidentifier(): return f'<{self.__class__.__name__}>' @@ -208,9 +226,14 @@ def unpack(self, buffer: 'bytes | IO[bytes]', packet: 'dict[str, Any]') -> '_T': Unpacked field value. """ + # NOTE: ``length`` recomputes struct.calcsize() on every read, so the + # three reads this method used to make were three calcsize() calls for + # one value. + length = self.length + if not isinstance(buffer, bytes): - buffer = buffer.read(self.length) - value = struct.unpack(self.template, buffer[:self.length].rjust(self.length, b'\x00'))[0] + buffer = buffer.read(length) + value = struct.unpack(self.template, buffer[:length].rjust(length, b'\x00'))[0] return self.post_process(value, packet) diff --git a/pcapkit/corekit/fields/strings.py b/pcapkit/corekit/fields/strings.py index d3199cfe57..e56a9e91c6 100644 --- a/pcapkit/corekit/fields/strings.py +++ b/pcapkit/corekit/fields/strings.py @@ -4,9 +4,8 @@ import urllib.parse as urllib_parse from typing import TYPE_CHECKING, Any, Generic, TypeVar -import chardet - from pcapkit.corekit.fields.field import Field, NoValue +from pcapkit.utilities.chardet import detect from pcapkit.utilities.compat import Dict from pcapkit.utilities.exceptions import FieldValueError @@ -168,7 +167,7 @@ def post_process(self, value: 'bytes', packet: 'dict[str, Any]') -> 'str': # py except UnicodeError: ret = urllib_parse.unquote(value.replace(b'%', rb'\x'), encoding='utf-8', errors='replace') else: - charset = self._encoding or chardet.detect(value)['encoding'] or 'utf-8' + charset = self._encoding or detect(value) try: ret = value.decode(charset, self._errors) except UnicodeError: diff --git a/pcapkit/protocols/protocol.py b/pcapkit/protocols/protocol.py index 81afa58df1..9736435233 100644 --- a/pcapkit/protocols/protocol.py +++ b/pcapkit/protocols/protocol.py @@ -26,7 +26,6 @@ from typing import TYPE_CHECKING, Any, Generic, Optional, Type, TypeVar, cast, overload import aenum -import chardet from pcapkit.corekit.context import ContextRegistry from pcapkit.corekit.module import ModuleDescriptor @@ -40,6 +39,7 @@ from pcapkit.protocols.schema.schema import Schema from pcapkit.utilities.compat import cached_property from pcapkit.utilities.decorators import beholder, seekset +from pcapkit.utilities.chardet import detect from pcapkit.utilities.exceptions import (ProtocolNotFound, ProtocolNotImplemented, RegistryError, StructError, UnsupportedCall) from pcapkit.utilities.warnings import RegistryWarning, warn @@ -331,7 +331,7 @@ def decode(byte: bytes, *, encoding: 'Optional[str]' = None, .. _chardet: https://chardet.readthedocs.io """ - charset = encoding or chardet.detect(byte)['encoding'] or 'utf-8' + charset = encoding or detect(byte) try: return byte.decode(charset, errors=errors) except UnicodeError: diff --git a/pcapkit/protocols/schema/schema.py b/pcapkit/protocols/schema/schema.py index 4effad092b..3b58f8497d 100644 --- a/pcapkit/protocols/schema/schema.py +++ b/pcapkit/protocols/schema/schema.py @@ -382,7 +382,13 @@ def __setattr__(self, name: 'str', value: '_VT') -> 'None': if name in self.__fields__: key = self.__map__.get(name, name) self.__dict__[key] = value - self.__updated__ = True + # NOTE: ``self.__updated__ = True`` would re-enter this method once + # per field assigned -- 180297 recursive calls per extraction of + # examples/captures/http.pcap -- only to miss the __fields__ test and + # fall through to object.__setattr__. ``__updated__`` is an instance + # attribute established in __new__, so the direct store is the same + # write with none of the round trip. + self.__dict__['__updated__'] = True return return super().__setattr__(name, value) @@ -615,24 +621,29 @@ def unpack(cls, data: 'bytes | IO[bytes]', for field in self.__fields__.values(): field = field(packet) + # NOTE: ``Field.length`` recomputes struct.calcsize() on every read, + # so it is read once per field here rather than at each use. if isinstance(field, PayloadField): - payload_length = field.length or cast('int', packet['__length__']) + length = field.length + payload_length = length or cast('int', packet['__length__']) payload = data.read(payload_length) self.__buffer__[field.name] = payload - packet['__length__'] -= field.length + packet['__length__'] -= length packet[field.name] = payload setattr(self, field.name, payload) continue if isinstance(field, PaddingField): - byte = data.read(field.length) + length = field.length + + byte = data.read(length) self.__buffer__[field.name] = byte packet[field.name] = byte - packet['__length__'] -= field.length + packet['__length__'] -= length setattr(self, field.name, byte) continue @@ -645,7 +656,9 @@ def unpack(cls, data: 'bytes | IO[bytes]', continue field = field.field(packet) - byte = data.read(field.length) + length = field.length + + byte = data.read(length) self.__buffer__[field.name] = byte value = field.unpack(byte, packet.copy()) @@ -657,18 +670,18 @@ def unpack(cls, data: 'bytes | IO[bytes]', packet['__option_padding__'] = field.option_padding if isinstance(field, ForwardMatchField): - data.seek(-field.length, io.SEEK_CUR) + data.seek(-length, io.SEEK_CUR) elif isinstance(field, OptionField) and field.option_padding > 0: # the option list ended before the declared field length was # exhausted; give the unconsumed remainder back to ``data`` # so that the following fields can read it data.seek(-field.option_padding, io.SEEK_CUR) - consumed = field.length - field.option_padding + consumed = length - field.option_padding self.__buffer__[field.name] = byte[:consumed] packet['__length__'] -= consumed else: - packet['__length__'] -= field.length + packet['__length__'] -= length if packet['__length__'] < 0: warn(f'packet length < 0: {packet["__length__"]}', diff --git a/pcapkit/utilities/__init__.py b/pcapkit/utilities/__init__.py index 2086907841..8e86915243 100644 --- a/pcapkit/utilities/__init__.py +++ b/pcapkit/utilities/__init__.py @@ -12,10 +12,11 @@ several user-refined exceptions and warnings. """ +from pcapkit.utilities.chardet import detect from pcapkit.utilities.decorators import beholder, prepare, seekset from pcapkit.utilities.exceptions import stacklevel from pcapkit.utilities.logging import configure, ensure_output, get_logger, logger, reset from pcapkit.utilities.warnings import warn __all__ = ['logger', 'get_logger', 'configure', 'reset', 'ensure_output', - 'warn', 'stacklevel'] + 'warn', 'stacklevel', 'detect'] diff --git a/pcapkit/utilities/chardet.py b/pcapkit/utilities/chardet.py new file mode 100644 index 0000000000..3a8018c42e --- /dev/null +++ b/pcapkit/utilities/chardet.py @@ -0,0 +1,67 @@ +# -*- coding: utf-8 -*- +"""Character Set Detection +============================= + +.. module:: pcapkit.utilities.chardet + +:mod:`pcapkit.utilities.chardet` wraps `chardet`_ with a bounded cache, for +turning the bytes of a text field into a :obj:`str`. + +.. _chardet: https://chardet.readthedocs.io + +""" + +import functools + +import chardet + +__all__ = ['detect'] + +#: How many distinct bytestrings :func:`detect` will remember. Bounded so that a +#: capture full of never-repeating text cannot retain all of it. +DETECT_CACHE_SIZE = 1024 + + +@functools.lru_cache(maxsize=DETECT_CACHE_SIZE) +def detect(value: 'bytes') -> 'str': + """Detect the character set of ``value``. + + :func:`chardet.detect` is a pure function of the bytes handed to it, and the + single most expensive step in turning a text field into a :obj:`str`. The + strings a capture presents repeat heavily -- an HTTP-heavy capture asked for + the encoding of ``b'Connection'`` once per message and got the same answer + every time -- so the verdict is memoised rather than recomputed. The result is + by construction the one :func:`chardet.detect` would have returned. + + Note: + The cache is bounded by entry *count*, not by size, and it holds the bytes + it was keyed on: :func:`functools.lru_cache` caches an argument's *hash* + but still keeps the argument, since a :obj:`dict` needs the key to settle + equality on a hash collision. Measured, feeding 20 distinct 1 MB values + retains 19.1 MB. :data:`DETECT_CACHE_SIZE` therefore caps the entries + rather than the footprint, which matters because + :meth:`ProtocolBase.decode + ` is public and a caller + may hand it a whole payload. Use :meth:`detect.cache_clear + ` to release it in a long-running + process. + + Two alternatives were tried and rejected. Keying on a *prefix* is + unsound, since :func:`chardet.detect` is statistical over the whole + sequence: an ASCII header followed by a UTF-8, Latin-1 or CP1251 body is + detected as ``ascii`` from its first 256 octets and correctly otherwise, + three disagreements in six realistic cases. Keying on a digest bounds the + footprint exactly and was measured retaining 0.0 MB for the same 19 MB of + input, but it cannot be expressed with :func:`~functools.lru_cache` -- + which keys on what it is passed -- and hand-rolling the eviction was + judged not worth the six lines. + + Args: + value: Bytestring whose encoding is to be detected. + + Returns: + Name of the detected encoding, or ``'utf-8'`` where detection declines to + name one. + + """ + return chardet.detect(value)['encoding'] or 'utf-8'