diff --git a/Makefile b/Makefile index cc85fdd1a0..70bde8b0c1 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: bootstrap setup dist release docs samples test test-all coverage +.PHONY: bootstrap setup dist release docs samples test test-all coverage bench bench-quick bench-test export PIPENV_VENV_IN_PROJECT=1 export PIPENV_CACHE_DIR ?= $(CURDIR)/.pipenv-cache @@ -87,6 +87,31 @@ coverage: samples pipenv run coverage run -m pytest -q pipenv run coverage report +# The engine speed table in README.rst -- every supported Python version, one +# image each, measured in containers so the host does not affect the result. Needs +# docker, and nothing else: deliberately not run through pipenv, since the whole +# point is that the measuring environments are the pinned ones inside the images +# rather than whatever is installed here. +# +# Expect around two hours at the defaults: five interpreters, seven environments, +# and almost all of the measuring time is pyshark, which spawns a tshark process +# per extraction. Cut it with --engines, --pythons, or --quick. +bench: + examples/benchmark/run.sh + +# A smoke check that the harness works end to end, not a measurement -- two +# interpreters and the two cheapest engines, which is enough to exercise both +# virtualenvs, the matrix loop and both emitted tables. The full matrix at --quick +# would still build five images, and building is most of a quick run's cost. +bench-quick: + examples/benchmark/run.sh --quick --pythons 3.11,3.12 --engines default,dpkt + +# The harness's own tests: ratio arithmetic, environment stitching, the per-version +# grid, and that the emitted reStructuredText parses under plain docutils. No +# docker needed. +bench-test: + pipenv run python -m pytest -q examples/benchmark/test_harness.py + docs: PCAPKIT_SPHINX=1 pipenv run $(MAKE) -C docs html diff --git a/examples/benchmark/Dockerfile b/examples/benchmark/Dockerfile new file mode 100644 index 0000000000..fb8ca327af --- /dev/null +++ b/examples/benchmark/Dockerfile @@ -0,0 +1,213 @@ +# syntax=docker/dockerfile:1 +# +# The benchmark image: one interpreter, one or two virtualenvs, every engine that +# the interpreter can hold. +# +# One image per Python version, selected by `PYTHON_IMAGE`. The versions and their +# digests live in `python-images.txt`, which is the matrix `run.sh` reads; the +# reasoning behind pinning by digest rather than by tag is there too, next to the +# digests it applies to, rather than duplicated here. The default below is 3.11 +# because it is the *last* interpreter on which every engine can run -- 3.10 is the +# only other one -- which makes it the right thing to get from a bare `docker build` +# with no `--build-arg`. It has to stay equal to the 3.11 row of the pins file, and a +# self-test asserts that it does: one digest written in two places is one that drifts. +# +# Two virtualenvs where two are needed, one where they are not. `pypcap` and +# `pcap-ct` both install a top-level `pcap` module, and with both present +# `import pcap` resolves to `pcap-ct` -- upstream's extension module is shadowed and +# unreachable -- so no single environment can hold every engine. On 3.12 and newer, +# though, `pypcap` cannot be installed at all, and a second virtualenv there would +# be minutes of compiling to produce nothing. `WITH_PYPCAP=0` skips it and records +# why, so the report can say the engine is unsupported on this interpreter rather +# than that it failed here. +ARG PYTHON_IMAGE=python:3.11-slim-bookworm@sha256:528257d48c1da0dcecc2e725d1ae34498d60c965f1241e39cd6a85a8859bdf84 +FROM ${PYTHON_IMAGE} + +# Redeclared: an ARG consumed by `FROM` is out of scope for the build stage, so +# without this the reference the image was built from could not be recorded in it. +ARG PYTHON_IMAGE + +# Whether to build the `pypcap` virtualenv. See the header, and `python-images.txt` +# for which interpreters can hold it. +ARG WITH_PYPCAP=1 + +# Recorded in the report so a table can be traced back to the code that produced +# it. `pcapkit` is installed from the working tree, not from PyPI, so its git +# revision is the only version number that means anything. +ARG PCAPKIT_REVISION=unknown + +ENV PIP_DISABLE_PIP_VERSION_CHECK=1 \ + PIP_NO_CACHE_DIR=1 \ + PIP_ROOT_USER_ACTION=ignore \ + PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + PCAPKIT_REVISION=${PCAPKIT_REVISION} \ + PCAPKIT_BASE_IMAGE=${PYTHON_IMAGE} \ + VENV_PYPCAP=/opt/venv-pypcap \ + VENV_PCAP_CT=/opt/venv-pcap_ct + +# Three things beyond the base image, each needed by exactly one engine: +# +# build-essential + libpcap-dev -- `pypcap` is an sdist with a C extension, and +# its setup.py hard-fails unless it can find both `pcap.h` and an +# unversioned `libpcap.so`. The runtime-only libpcap0.8 package is not +# enough for it, though it would be enough for `pcap-ct`, which resolves the +# versioned soname through ctypes instead. +# tshark -- `pyshark` does no parsing of its own; it shells out to Wireshark's +# command-line tool once per extraction, so without the binary the Python +# package is inert. Preseeded to *not* install setuid dumpcap: this image +# only ever reads a file from disk, and live capture privileges would be a +# liability for nothing. +# +# Installed unconditionally, including on the interpreters that get no `pypcap` +# virtualenv and therefore never invoke a compiler. That is deliberate: the +# per-version table's columns are compared with each other in absolute +# milliseconds, so every image has to differ from the others in the interpreter and +# nothing else. Installing a smaller apt set on 3.12+ would save a little build +# time and buy a difference between columns that could not be attributed. +# +# Debian package versions are deliberately not pinned -- the archive drops old +# versions, so a pinned apt line breaks rather than reproduces. The resolved +# `tshark` and libpcap versions are recorded in the report instead, which is the +# honest form of reproducibility here. +RUN set -eux; \ + export DEBIAN_FRONTEND=noninteractive; \ + echo 'wireshark-common wireshark-common/install-setuid boolean false' | debconf-set-selections; \ + apt-get update; \ + apt-get install --yes --no-install-recommends \ + build-essential \ + libpcap-dev \ + tshark; \ + rm -rf /var/lib/apt/lists/*; \ + tshark --version | head -n 1 + +WORKDIR /src + +# Copied file by file rather than as a whole tree, so that editing the harness +# does not invalidate the layer that compiles `pypcap`, and so that nothing from +# the host's checkout (a virtualenv, build output, captures generated locally) +# can leak into the image and change what is measured. +COPY examples/benchmark/requirements-common.txt \ + examples/benchmark/requirements-pypcap.txt \ + examples/benchmark/requirements-pcap_ct.txt \ + ./examples/benchmark/ + +# The shared engines first, into every environment this image will hold. Everything +# here is a wheel on every architecture and every interpreter in the matrix -- +# checked on 3.10, 3.12, 3.13 and 3.14 as well as 3.11 -- so a failure is a real +# failure and stays fatal. +# +# `$VENV_PYPCAP` is created only where it can be filled. Nothing downstream is told +# which of the two happened: each step asks the filesystem instead, so the image +# cannot end up claiming an environment it does not have. +RUN set -eux; \ + mkdir -p /opt/locks /opt/install-failures /opt/not-attempted; \ + venvs="$VENV_PCAP_CT"; \ + if [ "$WITH_PYPCAP" = 1 ]; then venvs="$venvs $VENV_PYPCAP"; fi; \ + for venv in $venvs; do \ + python -m venv "$venv"; \ + "$venv/bin/pip" install --quiet --upgrade pip setuptools wheel; \ + "$venv/bin/pip" install --quiet --requirement examples/benchmark/requirements-common.txt; \ + done + +# `pcap-ct` and `libpcap` are py3-none-any wheels with nothing to compile, so this +# too is fatal on failure. +RUN set -eux; \ + "$VENV_PCAP_CT/bin/pip" install --quiet --requirement examples/benchmark/requirements-pcap_ct.txt; \ + "$VENV_PCAP_CT/bin/pip" freeze > /opt/locks/pcap_ct.txt + +# `pypcap` is the only package in the set that compiles, which makes it the only +# one whose install can fail for reasons outside these pins -- a toolchain change, +# a libpcap header that moved, an architecture whose library path its setup.py does +# not name. So its failure is recorded rather than fatal. +# +# The alternative is worse: a fatal failure here means `run.sh` produces no table at +# all, and the six engines that do work go unreported because the seventh did not +# build. Recording it keeps the promise that a missing engine is an explained row +# rather than an absent one -- benchmark.py reads this file and reports the build +# error as the reason `pypcap` was not measured. +# +# On an interpreter with no `pypcap` virtualenv the failure is not a failure but a +# ceiling, and the two must not be reported as the same thing: a note in +# `/opt/not-attempted/` says the engine cannot be installed on this interpreter at +# all, which is a fact about `pypcap` and 3.12+ rather than about this build. +RUN set -eux; \ + if [ ! -d "$VENV_PYPCAP" ]; then \ + { echo 'pypcap is not installable on this interpreter, so it was not built into this'; \ + echo 'image: pypcap 1.3.0 ships a pcap.c pre-generated by Cython 0.29.32, which does'; \ + echo 'not compile against the Python 3.12+ C API. The engine is unsupported here'; \ + echo 'rather than broken here.'; } > /opt/not-attempted/pypcap.txt; \ + cat /opt/not-attempted/pypcap.txt >&2; \ + elif "$VENV_PYPCAP/bin/pip" install --requirement examples/benchmark/requirements-pypcap.txt \ + > /tmp/pypcap-install.log 2>&1; then \ + "$VENV_PYPCAP/bin/python" -c 'import pcap; print("pypcap", pcap.__version__)'; \ + "$VENV_PYPCAP/bin/pip" freeze > /opt/locks/pypcap.txt; \ + else \ + echo 'WARNING: pypcap failed to install; it will be reported as not measured.' >&2; \ + { echo 'pypcap failed to build in this image, so it could not be measured.'; \ + echo 'The tail of the pip output was:'; \ + tail -n 12 /tmp/pypcap-install.log; } > /opt/install-failures/pypcap.txt; \ + cat /opt/install-failures/pypcap.txt >&2; \ + "$VENV_PYPCAP/bin/pip" freeze > /opt/locks/pypcap.txt; \ + fi; \ + rm -f /tmp/pypcap-install.log + +# `pcapkit` last, because it is the thing under test and changes on every commit. +COPY pyproject.toml setup.py MANIFEST.in README.rst ./ +COPY pcapkit ./pcapkit + +# --no-deps so the pinned requirements above stay authoritative: `pcapkit` +# declares unpinned ranges, and letting them resolve here would quietly move the +# baseline the whole report is normalised against. +RUN set -eux; \ + for venv in /opt/venv-*; do \ + "$venv/bin/pip" install --quiet --no-deps .; \ + "$venv/bin/python" -c 'import pcapkit; print(pcapkit.__version__)'; \ + done + +# What this image can measure, as three whitespace-separated fields per line: the +# environment label the report will use, the virtualenv to run it in, and the lock +# file recording what that virtualenv resolved to. +# +# Written here rather than assembled by `entrypoint.sh` from build arguments, +# because the label has to name the interpreter that actually ran. Asking the +# interpreter for its own version makes that unfalsifiable; deriving it from a +# `--build-arg` would let a mislabelled build put 3.11 numbers in the 3.12 column, +# which is a silent error of exactly the kind the rest of this harness refuses to +# leave open. The `pypcap` line is likewise conditioned on the virtualenv existing +# rather than on `WITH_PYPCAP`. +RUN set -eux; \ + series="$(python -c 'import sys; print("%d.%d" % sys.version_info[:2])')"; \ + { printf '%s-pcap_ct %s /opt/locks/pcap_ct.txt\n' "$series" "$VENV_PCAP_CT"; \ + if [ -d "$VENV_PYPCAP" ]; then \ + printf '%s-pypcap %s /opt/locks/pypcap.txt\n' "$series" "$VENV_PYPCAP"; \ + fi; } > /opt/environments; \ + cat /opt/environments + +COPY examples/captures/in.pcap ./examples/captures/in.pcap +COPY examples/benchmark/benchmark.py \ + examples/benchmark/report.py \ + examples/benchmark/entrypoint.sh \ + ./examples/benchmark/ + +# Run as an ordinary user. `tshark` prints a "Running as user root ... could be +# dangerous" banner on every start, and `pyshark` starts it once per extraction -- +# a thousand times per repeat -- so this is not only hygiene, it keeps that +# warning out of the measured path. A real HOME is needed because `pyshark` +# resolves its config through appdirs. +# +# `/in` is where the reporting pass receives the JSON documents from the other +# versions' containers, since one interpreter has to report on a matrix none of them +# measured alone. It is a plain directory rather than a second volume: `docker cp` +# into a stopped container is what fills it, and keeping inputs and outputs in +# separate directories means the reporting pass cannot mistake a file it was handed +# for one it produced. +RUN set -eux; \ + useradd --create-home --shell /bin/bash bench; \ + mkdir -p /out /in; \ + chown bench:bench /out /in +USER bench +ENV HOME=/home/bench + +VOLUME ["/out"] +ENTRYPOINT ["/src/examples/benchmark/entrypoint.sh"] diff --git a/examples/benchmark/README.rst b/examples/benchmark/README.rst new file mode 100644 index 0000000000..70e41516c2 --- /dev/null +++ b/examples/benchmark/README.rst @@ -0,0 +1,532 @@ +====================== +Engine Benchmark Suite +====================== + +A containerised benchmark that produces the engine speed table in the project +``README.rst`` -- the whole table, every supported Python version, in one run. One +command, on any machine with Docker, and the output is two paste-ready +reStructuredText snippets: + +.. code-block:: shell + + examples/benchmark/run.sh + +It exists because that table could not previously be completed on **any** single +machine, and because the numbers in it were not reproducible. Both problems are the +same problem -- the environment was never pinned and could not hold every engine at +once -- so the fix is to define the environment precisely and ship it. Once the +environment is a container, the number of them is a loop, which is what makes a +complete table a single command rather than five hand-run measurements pasted +together. + +The matrix +---------- + +**One image per Python version**, five of them by default, each pinned by digest in +`python-images.txt `_ -- which is the matrix, and is a table +rather than logic so that a reader can check it. Inside each image, **one or two +virtualenvs**, so an "environment" here is a *(Python version, pcap provider)* +pair: + +.. list-table:: + :header-rows: 1 + + * - Python + - environments + - why + * - 3.10, 3.11 + - ``-pypcap`` and ``-pcap_ct`` + - both ``pcap`` providers install, and they are mutually exclusive + * - 3.12, 3.13, 3.14 + - ``-pcap_ct`` only + - ``pypcap`` cannot be installed at all, so a second virtualenv would compile + for minutes to produce nothing + * - 3.15 + - opt-in, ``-pcap_ct`` only + - no released image; see below + +Seven environments for a default run, and a version named with ``--pythons`` that is +not in the file is a typo rather than a column. + +Why this cannot be one virtualenv, or one Python +------------------------------------------------ + +Two hard constraints shape everything here. Neither is a ``pcapkit`` defect and +neither can be worked around from inside ``pcapkit``: + +1. **pypcap and pcap-ct are mutually exclusive.** Both distributions install + a top-level module named ``pcap``. With both present, ``import pcap`` resolves to + ``pcap-ct``'s package directory and upstream's extension module is shadowed and + unreachable, so ``engine='pypcap'`` cannot run at all. No install order changes + that. **So no single environment can hold all seven engines**, and an image that + can hold ``pypcap`` carries two virtualenvs -- ``/opt/venv-pypcap`` and + ``/opt/venv-pcap_ct``, identical but for which ``pcap`` provider they have. +2. **3.11 is the** *last* **interpreter on which every engine runs**, and 3.10 is the + other one. ``pypcap`` 1.3.0 ships a ``pcap.c`` pre-generated by Cython 0.29.32 + that will not compile against the 3.12+ C API; ``pypcapfile`` 0.12.0 imports + ``imp``, removed in 3.12; ``pyshark`` 0.6 calls + ``asyncio.get_event_loop_policy().get_event_loop()``, which raises from 3.14. + **That is now a fact the table reports rather than a constraint on the run**: each + engine's ceiling shows up as ``--`` cells with the reason, in the columns past it. + +The ``default`` engine -- ``pcapkit``'s own parser, also spelled ``pcapkit`` -- is +present in every virtualenv of every image, and is what joins all of them into one +table. See "How the runs are joined" below. + +Two tables, because there are two questions +------------------------------------------- + +The run emits both, and neither replaces the other: + +.. list-table:: + :header-rows: 1 + + * - File + - Question it answers + - Arithmetic + * - ``table-versions.rst`` + - how does each engine do on each interpreter? + - **absolute** milliseconds per packet, one column per Python version + * - ``table.rst`` + - which engine is faster, on any machine? + - **ratios** against ``default``, pooled across the whole matrix + +**The per-version table is absolute, and must be.** Every column of it came out of +one run on one host, differing in the interpreter and nothing else, which is +precisely the condition under which absolute times are comparable -- with each other. +A ratio there would be worse than redundant: ratios are normalised *within* one +environment, so dividing across two interpreters yields a number that is neither +engine's speed nor the interpreter's. The snippet says so in its own prose, because +the person reading the table in ``README.rst`` later is not the person who read this +file. + +**The ratio table is normalised to the default engine measured in the same +virtualenv on the same pass.** An absolute timing describes the machine it was taken +on at least as much as it describes the engine, so a table of absolute numbers cannot +be reproduced across machines -- which is exactly the state the old table was in. A +ratio divides the machine out, and pooling it across every interpreter is what makes +it an answer about engines rather than about one interpreter. The observed range then +includes any disagreement *between* interpreters, which is real uncertainty about the +engine and is not something to average away; the per-version table is where it gets +broken out. + +One absolute figure survives inside the ratio snippet, as a footnote: the ``default`` +engine's own milliseconds per packet. That is what lets a reader convert the ratios +back into times on the machine the run happened on. + +Both snippets carry their own ``Test Results`` heading -- the per-version one takes +the name, since it is what ``README.rst``'s **Test Results** section is, and the ratio +one is ``Test Results (Relative)`` so that pasting both does not produce two sections +with one name. The ``Test Environment`` block, which serves both, is in +``table.rst``. + +**The run-to-run spread is reported too.** A single pass cannot tell a real difference +between two engines from the machine having been briefly busy, so the whole engine set +is measured several times (``--repeats``, default 3) and every row carries the range +actually observed. Rows whose ranges overlap are marked ``‡``: this run does not +establish which of them is faster, and printing their medians as though it did would +be inventing precision. The footnote states the median and the widest spread seen, so +a reader can judge any gap in the table for themselves. + +The property that matters most +------------------------------ + +**An engine that cannot run does not raise.** ``Extractor.run`` emits an +``EngineWarning`` and falls back to ``pcapkit``'s own parser, so an extraction that +appears to succeed can be timing something entirely different from what was asked +for. A benchmark that does not check this will happily report the default engine's +speed three times over under three different names, and nothing in the output would +look wrong. + +So the harness checks ``type(extractor.engine).__engine_name__`` on **every single +extraction**, not just the first, against a list of expected drivers derived from +``Extractor.__engine__`` rather than written out by hand. A run whose driver ever +disagrees raises and is discarded; no number is reported for it. This is asserted, not +assumed, and it is the single most important correctness property of the suite. + +An engine that genuinely cannot run here is recorded as ``*not measured*`` with the +reason the engine itself gives through ``unsupported_reason()`` -- never omitted, and +never as a zero. + +One flaky engine must not cost the others their rows +---------------------------------------------------- + +This is not hypothetical; it is what the first full run of this harness did. +``pyshark`` spawns a ``tshark`` process per extraction, so a default run spawns **six +thousand** of them, and one of them died -- ``TSharkCrashException ... retcode: 255``, +on the third pass. The exception propagated, ``benchmark.py`` exited, no JSON was +written, and the run produced **no table at all**. Six working engines went unreported +because the seventh hiccupped once. + +So failure is contained at two levels, and counted at both: + +- **A failed extraction** is discarded and the pass carries on, up to + ``TOLERATED_FAILURES`` (5) per engine per pass. A crashed extraction is a + *non-measurement*, not a slow measurement, so discarding its sample is correct -- + what would not be honest is discarding it silently, so the count and the first + reason are reported. +- **A failed pass** costs that engine that pass, not the run. The row is built from + the passes that did work, its ``Passes`` count shows fewer, and it is marked ``¶`` + with what was lost. If *every* pass failed, the engine becomes ``*not measured*`` + with the exception as its reason. + +Two things are never tolerated, because tolerating them would mean reporting a number +that is not true: the escalated ``EngineWarning`` that means the engine fell back to +``pcapkit``'s parser, and a capture that yielded no packets. Both end the run. + +The same principle covers the build. ``pypcap`` is the only engine in the set that +compiles, and therefore the only one whose *install* can fail for reasons these pins +do not control. A fatal failure there would mean no table at all, so the Dockerfile +records the pip error to ``/opt/install-failures/pypcap.txt`` and carries on; +``benchmark.py`` reads it and reports the compiler error as the reason ``pypcap`` was +not measured. Everything else in the image is a wheel on every architecture *and every +interpreter in the matrix* -- checked on 3.10, 3.12, 3.13 and 3.14 as well as 3.11 -- +so a failure installing it is a real error and stays fatal. + +A missing column is held to the same standard as a missing row +-------------------------------------------------------------- + +With five images there are five chances for something outside these pins to go wrong: +a base image withdrawn, a wheel that stops publishing for one interpreter, a daemon +that runs out of disk halfway through. **None of those ends the run**, and none of +them silently narrows the table: + +- ``run.sh`` catches the failure per version, writes the reason to + ``out/missing/.txt``, and carries on to the next one. That is the same + mechanism ``/opt/install-failures/`` uses for ``pypcap``, one level up. +- the reporting pass turns each of those files into a column of ``--`` **with the + reason attached** -- in the per-version table's notes, and in the ``Test + Environment`` block as a ``Python 3.15 (not measured)`` row, so neither snippet + reads as a complete matrix when it is not. +- only a run in which *every* version failed exits without a table, since then there + is genuinely nothing to report. + +**The exit status says which happened**, because a scheduled invocation reads nothing +else: ``0`` when every requested version was measured, ``3`` **when the tables were +written but a version was lost**, ``1`` when no table was produced at all, and ``2`` +for bad arguments. A ``make bench`` that measured one column out of five is not a +failure -- the tables are there -- but it is not success either, and it should not be +indistinguishable from it. + +**pypcap on 3.12+ is deliberately not reported this way.** It did not fail; it +cannot be installed, which is a fact about the interpreter rather than about this +build. The image records that in ``/opt/not-attempted/pypcap.txt``, and that note +*replaces* the engine's own ``unsupported_reason()`` -- which on those images says the +installed ``pcap`` module is ``pcap-ct``, true but a description of the consequence. +The reader is owed the cause. + +How the runs are joined +----------------------- + +Every virtualenv measures the **whole** engine list, not just the engine unique to +it. That is deliberate: ``pcap_ct`` in the ``pypcap`` environment comes back +unmeasured with the mutual exclusion stated by the engine itself, which is better +documentation of the constraint than anything this README could assert. An engine +measured in *any* environment gets a row; one that could not be measured anywhere gets +the reason, attributed per environment when they disagree. + +The two tables join the environments differently, and the difference is the point: + +- **the per-version table pools within one interpreter and never across two.** An + engine measured in both virtualenvs of 3.11 contributes both readings to the 3.11 + cell, since they differ only in which ``pcap`` provider was installed alongside. A + cell is ``--`` only when *no* environment on that interpreter measured it. +- **the ratio table pools everything**, because each ratio was normalised in its own + environment before pooling and is therefore already machine-independent. + +A pass whose ``default`` reading is missing or zero is dropped from the ratio table -- +there is nothing to divide by -- but kept in the per-version table, where the +millisecond figure is complete on its own. Losing a real measurement to a rule that +does not apply to it would be the wrong kind of caution. + +The text report additionally prints each shared engine's median *per environment*: if +two environments disagree about ``dpkt``, the join on ``default`` is not doing its +job, and that is the line that shows it. With seven environments that line is also the +quickest way to see whether an engine behaves differently on one interpreter. + +Departures from the legacy methodology +-------------------------------------- + +The shape comes from ``examples/legacy_smoke/test_time.py`` so the figures stay +comparable in kind with what the project has already published: 1,000 timed +extractions of ``examples/captures/in.pcap`` per engine, timed with +``time.perf_counter_ns``, first sample discarded as a warm-up, reported as +milliseconds per packet. What changed, and why: + +.. list-table:: + :header-rows: 1 + + * - Change + - Why + * - The whole set is measured ``--repeats`` times, not once + - one pass gives no way to tell a real gap from noise, which is most of what was + wrong with the old table + * - Every Python version is measured in one run, not one per sitting + - the legacy script measures whichever interpreter invokes it, so a complete + table was five separate sittings hand-assembled; a column measured last week + is not comparable in absolute terms with one measured today, which is the + property the per-version table depends on + * - A second table reports ratios, with one absolute footnote + - absolute times are not comparable across machines, and "which engine is + faster" is a question about engines rather than about this host + * - The driver is asserted on every extraction + - the legacy script preflights once; an engine that stops being itself mid-run + would go unnoticed + * - Progress is printed to stderr every 100 rounds, not to stdout every round + - stdout carries the JSON document, and a flushed write per round is jitter + bought for nothing + * - Warnings other than ``EngineWarning`` are silenced during timing + - ``ExtractionWarning: EOF reached`` fires on every extraction of this capture; + ``EngineWarning`` is escalated to an exception so a mid-run fallback cannot be + missed + * - Engines run inside the repeat loop, not outside + - a machine that slows down partway then affects every engine in a pass about + equally, which is what lets the within-pass ratio cancel the drift + +Not changed: the capture, the round count, the clock, the warm-up discard, and the +``store=False, nofile=True`` extraction arguments. + +Apple Silicon, arm64 and emulation +---------------------------------- + +``run.sh`` builds for **this machine's architecture by default**, which on Apple +Silicon means native ``linux/arm64`` images. That is the setting that produces usable +numbers, and it should not need overriding: + +- Every base image digest in ``python-images.txt`` is a multi-arch OCI *index* that + includes ``linux/arm64/v8``, so the pins resolve natively rather than forcing + emulation. +- ``pcap-ct`` and ``libpcap`` publish ``py3-none-any`` wheels -- arch-independent, + nothing to compile. (``libpcap`` does vendor an ``aarch64`` ``libpcap.so`` under + ``libpcap/_platform/``, but its published config sets ``LIBPCAP = None`` and falls + through to ``find_library("pcap")``, so the image deliberately uses Debian's libpcap + instead, which is closer to what a user gets.) +- ``dpkt``, ``scapy``, ``pyshark`` and ``pypcapfile`` are pure Python. ``lxml``, + ``pyshark``'s only compiled dependency, publishes manylinux wheels for aarch64 on + every interpreter in the matrix (cp310 through cp314), so no column needs a compiler + for it. +- ``pypcap`` is the only thing that compiles, and nothing in it is + architecture-specific: its pre-generated ``pcap.c`` contains no arch guards or + inline assembly, and its ``setup.py`` branches on ``sys.maxsize > 2**32`` rather + than on the machine. Its library-search list does omit ``aarch64-linux-gnu`` -- it + names only the x86 multiarch paths -- but the list ends in an empty prefix that makes + it walk ``/usr`` recursively, which finds + ``/usr/lib/aarch64-linux-gnu/libpcap.so``. The consequence on arm64 is a slower + build, not a failed one. + +**If you do pass** ``--platform linux/amd64`` **on an Apple Silicon machine, the run +happens under QEMU emulation.** Numbers taken that way are not comparable with native +ones, and the ratios are only as trustworthy as the emulator is uniform across the +very different work each engine does -- which for a C extension against a ``ctypes`` +binding against a subprocess spawn is not an assumption worth making. ``run.sh`` +detects the mismatch, warns on stderr, **and threads the warning into the emitted +table**, because the person reading the table later is not the person who saw the +stderr. + +And if ``pypcap`` does fail to build on arm64 despite the above, the run still +produces a table: the row says ``*not measured*`` and carries the compiler error, per +the section above. That is the honest outcome, and it is better than the two +alternatives -- no table, or an ``amd64`` run under emulation whose numbers cannot be +compared with anything. + +Reproducibility: what is pinned and what is not +----------------------------------------------- + +Pinned exactly: + +- every base image, **by digest**, in ``python-images.txt`` rather than by tag: + ``3.11-slim-bookworm`` and its four siblings are mutable and rebuilt whenever Debian + or CPython ships a patch, so a tag pin is a pin in name only. They are all + **slim-bookworm**, which is a measurement decision rather than a packaging one -- + the per-version table's columns are compared with each other, so everything but the + interpreter is held constant, down to the same ``tshark`` and the same libpcap; +- every engine package, in ``requirements-common.txt``, ``requirements-pypcap.txt`` + and ``requirements-pcap_ct.txt``. Note that ``pcap-ct`` and ``libpcap`` are + pre-release-only projects, so ``pip install pcap-ct`` with no version resolves to + nothing at all -- an exact pre-release pin is honoured without ``--pre``, which is + why the versions are spelled out; +- ``pcapkit``'s own runtime dependencies, because they are on the baseline's hot path + and the baseline is what every ratio divides by. + +Not pinned, deliberately: + +- **Debian package versions** (``tshark``, ``libpcap-dev``). The archive drops old + versions, so a pinned ``apt-get`` line breaks rather than reproduces. The resolved + ``tshark`` and libpcap versions are *recorded in the report* instead, which is the + honest form of reproducibility here. The libpcap version is read through ``ctypes`` + from the library ``find_library("pcap")`` actually resolves -- neither ``pypcap`` + nor ``pcap-ct`` exposes a ``lib_version`` attribute, so reading one off the module + silently yields nothing. +- **The transitive closure** beyond the direct pins. It is frozen at build time inside + each image and copied out beside the results, one file per environment, so + ``pip freeze`` for 3.11's ``pypcap`` virtualenv does not overwrite 3.12's. +- **pcapkit itself**, which is installed from the working tree rather than from + PyPI. Its git revision is recorded instead, with a ``-dirty`` suffix when the tree + has uncommitted changes under ``pcapkit/`` -- a revision that does not say so is a + claim about code that was not measured. + +Output +------ + +Everything lands in ``--out`` (default ``examples/benchmark/out/``, which is +untracked): + +.. list-table:: + :header-rows: 1 + + * - File + - What it is + * - ``table-versions.rst`` + - absolute ms/packet per engine and Python version -- ``README.rst``'s **Test + Results** section, verbatim + * - ``table.rst`` + - ``Test Environment`` plus the machine-independent ratio table + * - ``-.json`` + - the raw per-environment measurements, every pass, e.g. ``3.11-pypcap.json`` + * - ``-.txt`` + - ``pip freeze`` of that virtualenv as built + * - ``missing/.txt`` + - why a version produced no measurements, when one did not + +``run.sh`` clears the previous run's ``*.json``, ``*.rst``, ``*.txt`` and +``missing/`` before it starts, **and that is not tidiness.** The reporting pass reports +on every document it is handed, so a document left over from a run on another +``pcapkit`` revision would fill a column of this run's table and pool into this run's +ratios, while the provenance block -- which reports one revision -- would not show the +disagreement. A stale table would likewise make "no table was produced" +unfalsifiable. + +``out/`` is already covered by the repository's ``.gitignore``, so nothing here needs +a new ignore rule. + +The report prints an image reference **per Python version**, the ``pcapkit`` +revision, every interpreter measured, the architecture, the capture and its digest, +the iteration count, and every engine package's resolved version per environment. It +prints **nothing that identifies the host** -- no hostname, no kernel, no CPU model. +Containerising the benchmark is what makes the host irrelevant; printing host details +would undo that, and would leak machine identity into a public README. + +How long it takes +----------------- + +**A full default matrix is on the order of two hours**, and it splits roughly in two: + +- **build: 15-25 minutes.** Five images, each installing the same pinned set, plus a + ``pypcap`` C extension compiled twice (3.10 and 3.11). Docker's layer cache does not + help across versions -- a different base image is a different layer chain -- so this + cost is per column. Measured on this suite: a 3.12 image, which compiles nothing, + takes about a minute and a half; a 3.11 image, which does, takes about three. +- **measure: 1.5-2 hours.** Seven environments x 1,000 rounds x 3 passes, and **almost + all of it is** ``pyshark``, which spawns a ``tshark`` process per extraction and is + two to three orders of magnitude slower than anything else in the table. It runs in + six of the seven environments (everything but 3.14, where it cannot run at all). + +Three ways to cut it, in descending order of how much they save: + +.. code-block:: shell + + # drop pyshark and the run is minutes rather than hours + examples/benchmark/run.sh --engines default,dpkt,scapy,pypcap,pcap_ct,pypcapfile + + # fewer columns: one version is two environments at most + examples/benchmark/run.sh --pythons 3.11 + + # a smoke check that the harness works, not a measurement + examples/benchmark/run.sh --quick --pythons 3.11,3.12 --engines default,dpkt + +``--quick`` is 50 rounds over 2 passes. It is enough to confirm every engine runs and +both tables render; it is not enough to publish, and the spread it reports will say +so. + +Options +------- + +.. code-block:: text + + --rounds N timed extractions per engine per pass (default 1000) + --repeats N passes over the whole engine set (default 3) + --engines LIST comma-separated; must include 'default', which every ratio uses + --pythons LIST comma-separated Python versions (default: every 'default'-tier + row of python-images.txt, i.e. 3.10 through 3.14) + --quick --rounds 50 --repeats 2 + --platform P docker platform; defaults to this machine's architecture + --out DIR where to write results (default examples/benchmark/out) + --no-build reuse the existing images instead of rebuilding + +From the repository root: ``make bench`` runs ``run.sh`` with no arguments, +``make bench-quick`` runs a reduced smoke matrix, and ``make bench-test`` runs the +self-tests below. + +Python 3.15 +----------- + +Opt-in, and allowed to fail -- matching the repository's CI, which already treats 3.15 +as an allowed-to-fail leg. There is no released ``python:3.15-slim-bookworm``; the +pins file names a ``3.15-rc`` digest, which freezes one particular release candidate +and will name an older one as time passes. Measure it with: + +.. code-block:: shell + + examples/benchmark/run.sh --pythons 3.10,3.11,3.12,3.13,3.14,3.15 + +If that image cannot be built or the run fails in it, the result is a ``3.15`` column +of ``--`` with the reason attached, not a missing column and not an aborted run. + +As it happens the pinned rc **does** build and measure: every pinned requirement +installs on it, and ``default``, ``dpkt``, ``scapy`` and ``pcap_ct`` all produced +figures. ``pypcap``, ``pypcapfile`` and ``pyshark`` come back unmeasured with the same +reasons they give on 3.14. That is a fact about one release candidate on one day +rather than a promise about 3.15, which is why the row stays opt-in. + +Self-tests +---------- + +.. code-block:: shell + + python -m pytest examples/benchmark/test_harness.py + +No docker needed. These cover the parts of a benchmark that *can* be checked without +a stopwatch, which is more than it sounds: that a ratio is taken against the right +baseline, that machine drift cancels, that repeats are paired one to one rather than +averaged first, that the environments stitch together correctly, that an unmeasured +engine keeps its reason, that the overlap marking neither over- nor under-claims, and +that the emitted reStructuredText parses under **plain docutils** with no warnings and +uses no Sphinx-only roles. That last one matters because GitHub renders +``README.rst`` with docutils, where a ``:mod:`` role comes out as a visible error +block while looking perfectly fine in the project's own Sphinx docs. + +The matrix adds its own statements to check, since most of the ways a version column +can be wrong are silent: that a document lands in the column its *interpreter* names +rather than the one its label claims; that ``3.9`` sorts before ``3.10``; that a cell +pools the two virtualenvs of one interpreter and never divides across two of them; +that a cell measured in one virtualenv is not reported as a gap; that a version which +produced nothing is a marked column of ``--`` with a reason rather than an absent +column, and does not mark the engines for it; and that the absolute table contains no +ratio anywhere, ``pcapkit`` included. + +They also cover the driver assertion, including that a measurement whose engine fell +back to ``pcapkit``'s parser refuses to report a number. + +Troubleshooting +--------------- + +- ``docker is installed but not responding`` -- start Docker Desktop and wait for it + to report "running". +- The ``pypcap`` layer takes minutes on arm64 -- expected, and explained above: + its ``setup.py`` walks ``/usr`` recursively to find a library its search list does + not name. It is built for 3.10 and 3.11 only, so this cost is paid twice, not five + times. +- ``pyshark`` reports not measured, "the tshark binary ... is not installed" -- + the image installs ``tshark``, so this means the apt install was skipped or failed; + rebuild without ``--no-build``. +- A row you expected is ``*not measured*`` -- read the reason. It comes from the + engine's own ``unsupported_reason()``, which names the actual cause rather than + guessing at one -- except for ``pypcap`` on 3.12+, where the interpreter ceiling is + reported instead. +- A whole column is ``--`` -- read the note under the table, and + ``out/missing/.txt``. The version was attempted; the build or the run + failed, and the reason is recorded rather than the column dropped. +- ``'3.1' is not in python-images.txt`` -- ``--pythons`` takes versions the pins + file knows. An unknown one is fatal rather than reported as a missing column, since + a column saying "3.1 could not be measured" would be an honest-looking answer to a + question nobody asked. diff --git a/examples/benchmark/benchmark.py b/examples/benchmark/benchmark.py new file mode 100644 index 0000000000..c04b6e59c0 --- /dev/null +++ b/examples/benchmark/benchmark.py @@ -0,0 +1,784 @@ +# -*- coding: utf-8 -*- +"""Time ``pcapkit.extract`` on each engine available in *this* environment. + +This is the measuring half of the benchmark suite; :mod:`report` is the reporting +half. One invocation covers one Python environment and writes a JSON document +describing what it measured. The suite runs it many times over, because an +environment here is a *(Python version, ``pcap`` distribution)* pair and the matrix +covers several of each: + +* several interpreters, one image apiece, because the point of the exercise is a + table with a column per Python version; +* one or two virtualenvs per interpreter, because ``pypcap`` and ``pcap_ct`` both + provide the top-level :mod:`pcap` module and cannot be installed side by side, so + no single environment can hold every engine -- and only one virtualenv where + ``pypcap`` cannot be installed at all. + +:mod:`report` stitches the runs together on the ``default`` engine, which is present +in every one of them. + +Methodology is inherited from :file:`examples/legacy_smoke/test_time.py` so the +figures stay comparable in kind with what the project has already published: +:data:`ROUNDS` timed extractions of the same capture per engine, timed with +:func:`time.perf_counter_ns`, first sample discarded as a warm-up, reported as +milliseconds per packet. The deliberate departures are listed in +:file:`README.rst` under "Departures from the legacy methodology". + +Two properties matter more than the numbers themselves. + +**The engine that ran is asserted, not assumed.** A missing or unusable engine +does not raise: :meth:`Extractor.run +` emits an +:class:`~pcapkit.utilities.warnings.EngineWarning` and falls back to ``pcapkit``'s +own parser, so an extraction that appears to succeed can be timing something else +entirely. Every single extraction therefore has its driver checked against +:func:`expected_drivers`, and a run whose driver ever disagrees is discarded +rather than reported -- see :func:`measure`. + +**An engine that cannot run is recorded, not omitted.** :func:`preflight` asks the +engine's own ``unsupported_reason()``, since that is the check :meth:`Extractor.run +` itself consults and it names the +actual cause. Ahead of it sits one thing the engine cannot know -- whether the image +even tried to install it, which on 3.12 and newer it does not for ``pypcap``; see +:func:`_not_attempted`. Either way the engine is reported with +``status='unmeasured'`` and a reason, so a gap in the table is visibly a gap with an +explanation rather than a missing row or a zero. + +""" + +import argparse +import hashlib +import importlib.metadata +import json +import os +import platform +import statistics +import subprocess # nosec: B404 +import sys +import time +import warnings +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from typing import Any, Optional + + from pcapkit.foundation.extraction import Extractor + +__all__ = ['ENGINES', 'ROUNDS', 'REPEATS', 'expected_drivers', 'preflight', 'measure', 'run'] + +#: Every engine ``pcapkit.extract(engine=...)`` accepts, in the order the suite +#: exercises them. Matches :data:`examples.legacy_smoke._engine_support.ENGINES`. +#: ``'default'`` is ``pcapkit``'s own parser and is the cross-environment +#: baseline, so it is always measured first and never dropped from the list. +ENGINES = ('default', 'dpkt', 'scapy', 'pypcap', 'pcap_ct', 'pypcapfile', 'pyshark') + +#: Timed extractions per engine per repeat. The first is discarded as a warm-up +#: round, matching the legacy script, so a repeat reports the mean of +#: ``ROUNDS - 1`` samples. +ROUNDS = 1_000 + +#: How many times the whole engine set is measured. More than one because a +#: single pass cannot tell a real gap between two engines from the machine having +#: been busy; :mod:`report` turns the spread across repeats into the noise floor. +REPEATS = 3 + +#: Failed extractions to discard per engine per pass before giving up on it. +#: Small on purpose: this is for a flaky *external process*, not for an engine that +#: does not work. ``pyshark`` starts a :program:`tshark` per extraction, and one +#: crashing in a few thousand spawns is a fact of that design rather than a result; +#: five is enough to survive it and far too few to hide an engine that fails +#: routinely. +TOLERATED_FAILURES = 5 + +#: Distributions whose resolved version belongs in the report. Keyed by the name +#: ``pip`` knows, which is not always the name that gets imported -- ``pypcapfile`` +#: imports as ``pcapfile``, ``pcap-ct`` and ``pypcap`` both as ``pcap``. Asked of +#: the installed metadata rather than of the modules for exactly that reason: the +#: metadata can see a distribution whose module the import system did not resolve +#: to, which is the ``pypcap``/``pcap-ct`` collision this suite exists to work +#: around. +DISTRIBUTIONS = ( + 'pypcapkit', 'dpkt', 'scapy', 'pyshark', 'pypcapfile', 'pypcap', 'pcap-ct', + 'libpcap', 'lxml', +) + +#: Directory where the image records a package whose install was allowed to fail +#: rather than abort the whole build. Overridable so the harness is still runnable +#: outside the container, where the directory does not exist and its absence simply +#: means nothing was recorded. +INSTALL_FAILURES = os.environ.get('BENCH_INSTALL_FAILURES', '/opt/install-failures') + +#: Directory where the image records a package it did not even try to install, +#: because the interpreter cannot hold it. Separate from :data:`INSTALL_FAILURES` +#: because the two are different findings that a single directory would flatten into +#: one: ``pypcap`` on 3.12 is not a build this image got wrong, it is an engine the +#: interpreter rules out, and reporting the second as the first sends the next reader +#: looking for a compiler problem that does not exist. Overridable for the same +#: reason as above. +NOT_ATTEMPTED = os.environ.get('BENCH_NOT_ATTEMPTED', '/opt/not-attempted') + + +def _distribution_versions() -> 'dict[str, Optional[str]]': + """Resolved version of every distribution in :data:`DISTRIBUTIONS`. + + Returns: + Mapping of distribution name to its installed version, or :data:`None` + where it is not installed here. The absent ones are kept rather than + dropped, because "``pypcap`` is not in this environment" is a fact the + report needs in order to explain why a row came from the other one. + + """ + versions = {} # type: dict[str, Optional[str]] + for name in DISTRIBUTIONS: + try: + versions[name] = importlib.metadata.version(name) + except importlib.metadata.PackageNotFoundError: + versions[name] = None + return versions + + +def _tshark_version() -> 'Optional[str]': + """First line of ``tshark --version``, or :data:`None` if it is not there. + + ``pyshark`` shells out to Wireshark's :program:`tshark` for all of its + parsing, so the binary's version is as much a part of a ``pyshark`` figure as + the Python package's is. + + """ + try: + completed = subprocess.run( # nosec: B603, B607 + ['tshark', '--version'], capture_output=True, text=True, timeout=30, check=False, + ) + except (OSError, subprocess.SubprocessError): + return None + if completed.returncode != 0: + return None + return completed.stdout.splitlines()[0].strip() if completed.stdout else None + + +def _libpcap_version() -> 'Optional[str]': + """What :manpage:`libpcap(3)` the ``pcap`` engines will actually map. + + Asked of the library through :mod:`ctypes` rather than of the :mod:`pcap` + module, because **neither** distribution exposes it: measured in this image, + ``pypcap`` 1.3.0 and ``pcap-ct`` 1.3.0b3 both lack a ``lib_version`` attribute + entirely, so reading one off the module silently yields nothing. + ``pcap_lib_version()`` is part of libpcap's own ABI and is always there. + + :func:`ctypes.util.find_library` is used deliberately rather than a fixed + soname: it is the *same* resolution ``pcap-ct`` performs, since the ``libpcap`` + distribution ships ``LIBPCAP = None`` in its config and falls through to it. + So this reports the library that will really be loaded, which is a property of + the environment rather than of any pinned package version -- and therefore + cannot be inferred from the pins and has to be asked at run time. + + Returns: + The library's version banner, or :data:`None` when no system libpcap can + be found or read. + + """ + import ctypes # pylint: disable=import-outside-toplevel + import ctypes.util # pylint: disable=import-outside-toplevel + + located = ctypes.util.find_library('pcap') + if located is None: + return None + try: + library = ctypes.CDLL(located) + library.pcap_lib_version.restype = ctypes.c_char_p + banner = library.pcap_lib_version() + except (OSError, AttributeError): + # Found but unloadable, or loadable but not libpcap. Neither is worth + # failing a benchmark over; the report simply omits the line. + return None + if banner is None: + return None + return banner.decode(errors='replace') if isinstance(banner, bytes) else str(banner) + + +def expected_drivers(engine: 'str') -> 'tuple[str, ...]': + """The ``__engine_name__`` values that mean *engine* really ran. + + Derived from the registry rather than from a table written out by hand, so it + cannot drift out of step with the engines: a new engine registered through + :func:`pcapkit.foundation.registry.foundation.register_extractor_engine` is + covered automatically, and a renamed ``__engine_name__`` cannot silently turn + the assertion into a no-op. + + Args: + engine: Engine name, as passed to ``pcapkit.extract``. + + Returns: + One name for a third-party engine. Two for ``'default'``/``'pcapkit'``, + which pick their parser from the file's magic number and are therefore + correct as either. + + Raises: + KeyError: If *engine* is not a registered engine name. Deliberately fatal: + ``Extractor`` would only warn and fall back, and a typo that silently + benchmarks the default engine under another engine's name is precisely + the failure this function exists to prevent. + + """ + from pcapkit.foundation.engines.pcap import PCAP # pylint: disable=import-outside-toplevel + from pcapkit.foundation.engines.pcapng import PCAPNG # pylint: disable=import-outside-toplevel + from pcapkit.foundation.extraction import Extractor # pylint: disable=import-outside-toplevel + + if engine in ('default', 'pcapkit'): + return (PCAP.name, PCAPNG.name) + + registered = Extractor.__engine__[engine] + # A registry entry is either the class or a lazy ``ModuleDescriptor``; the + # latter resolves the class on attribute access. + klass = getattr(registered, 'klass', registered) + return (klass.name,) + + +def _declared_reason(engine: 'str') -> 'Optional[str]': + """The engine's own ``unsupported_reason()``, if it declares one. + + This is the same preflight :meth:`Extractor.run + ` consults, and it is the only + thing that names the real cause -- a Python ceiling, a missing + :program:`tshark`, a missing :manpage:`libpcap(3)`, the wrong ``pcap`` + distribution installed. Without it the only available explanation is "its + package is not installed", which is wrong in every case where the package is + installed and unusable. + + Args: + engine: Engine name, as passed to ``pcapkit.extract``. + + Returns: + The engine's reason, or :data:`None` for ``'default'``, for an engine that + declares no limitation, or for anything that goes wrong while asking. + + """ + from pcapkit.foundation.extraction import Extractor # pylint: disable=import-outside-toplevel + + if engine in ('default', 'pcapkit'): + return None + registered = Extractor.__engine__.get(engine) + if registered is None: + return f'{engine} is not a registered extraction engine' + try: + klass = getattr(registered, 'klass', registered) + return klass.unsupported_reason() + except Exception as exc: # pylint: disable=broad-except # noqa: BLE001 + # Resolving a lazy descriptor imports the engine module, which can fail on + # its own. That is still an answer -- just not one the engine phrased. + return f'{type(exc).__name__}: {exc}' + + +def _extract(engine: 'str', capture: 'str') -> 'Extractor': + """One extraction, with the arguments the legacy timing script used. + + ``store=False`` and ``nofile=True`` keep the measurement on parsing rather + than on accumulating frames in memory or serialising them to disk. + + Args: + engine: Engine name. + capture: Path to the capture file. + + Returns: + The finished extractor. + + """ + import pcapkit # pylint: disable=import-outside-toplevel + + return pcapkit.extract(fin=capture, store=False, nofile=True, verbose=False, + engine=engine) # type: ignore[arg-type] + + +def _unavailable(exc: 'Exception') -> 'Optional[str]': + """Explain *exc* if it means the engine cannot run here. + + A last resort behind :func:`_declared_reason`, for the failures no engine + declares in advance. + + Args: + exc: Exception the engine raised. + + Returns: + A reason to record the engine as unmeasured, or :data:`None` if *exc* is a + real failure the caller should re-raise. + + """ + # pcapkit raises its own ModuleNotFound, which derives from ImportError, so + # one check covers both it and a plain missing import. + if isinstance(exc, ImportError): + return f'{exc.name or exc} is not installed' + # Matched by name rather than imported, since pyshark may be the thing missing. + if type(exc).__name__ == 'TSharkNotFoundException': + return 'the tshark binary from Wireshark is not installed' + if isinstance(exc, RuntimeError) and 'event loop' in str(exc): + version = '.'.join(str(part) for part in sys.version_info[:3]) + return (f'{exc} -- pyshark asks for an implicit asyncio event loop, which ' + f'Python {version} no longer provides') + # ``pcap-ct`` with no system libpcap raises OSError from its loader, which is + # not an ImportError and so reaches here. + if isinstance(exc, OSError) and 'libpcap' in str(exc): + return f'{exc} -- no system libpcap for the `pcap` module to load' + return None + + +def _recorded_note(directory: 'str', engine: 'str', separator: 'str') -> 'Optional[str]': + """One line of whatever *directory* records about *engine*. + + Args: + directory: Directory the image writes its notes into. + engine: Engine name, as passed to ``pcapkit.extract``. + separator: What to join the recorded lines with. Not a detail: pip output is + a sequence of distinct records and needs a visible separator to stay + legible once flattened, while a note written as prose is one sentence + wrapped for the Dockerfile and reads as gibberish if pipes are inserted + at its wrap points. + + Returns: + The note, flattened, or :data:`None` when there is none. + + """ + path = os.path.join(directory, f'{engine}.txt') + try: + with open(path, encoding='utf-8', errors='replace') as file: + recorded = file.read() + except OSError: + return None + # Collapsed to one line and capped: this ends up in a table cell and a bullet in + # the emitted RST, where a dozen lines of pip output would be unreadable. The + # full text stays in the image for anyone who needs it. + flattened = separator.join(line.strip() for line in recorded.splitlines() if line.strip()) + return flattened[:400] + (' ...' if len(flattened) > 400 else '') or None + + +def _install_failure(engine: 'str') -> 'Optional[str]': + """What the image recorded about this engine's package failing to install. + + The Dockerfile lets exactly one install fail without aborting the build -- + ``pypcap``, the only engine in the set that compiles and therefore the only one + whose install can fail for reasons the pins do not control. It writes the pip + error to a file, and this is what turns that file into the reason the engine was + not measured. Without it the reason would be the generic "its package is not + installed", which is true but says nothing about *why*, and the build error is + the whole diagnosis. + + Args: + engine: Engine name, as passed to ``pcapkit.extract``. + + Returns: + A one-line summary of the recorded failure, or :data:`None` when none was + recorded -- which is the normal case, including outside the container. + + """ + return _recorded_note(INSTALL_FAILURES, engine, ' | ') + + +def _not_attempted(engine: 'str') -> 'Optional[str]': + """What the image recorded about not trying to install this engine at all. + + The matrix runs one image per Python version, and ``pypcap`` cannot be installed + on 3.12 or newer -- its pre-generated ``pcap.c`` does not compile against that C + API. Building a ``pypcap`` virtualenv there would be minutes of compiling to + produce nothing, so the image does not, and records that instead. + + That distinction has to survive into the report. Left to the engine's own + :meth:`unsupported_reason`, the answer on a 3.12 image is "the installed ``pcap`` + module is ``pcap-ct``, not ``pypcap``" -- correct, and a description of the + consequence rather than the cause. The reader is owed the cause, which is a fact + about the interpreter, not about how this image happened to be assembled. + + Args: + engine: Engine name, as passed to ``pcapkit.extract``. + + Returns: + A one-line summary of the recorded note, or :data:`None` when there is none + -- the normal case for every engine that was installed. + + """ + return _recorded_note(NOT_ATTEMPTED, engine, ' ') + + +def preflight(engine: 'str', capture: 'str') -> 'tuple[Optional[str], Optional[str]]': + """Try *engine* once and decide whether it is worth timing. + + Done before the timed rounds rather than during them: a failure partway + through a thousand extractions throws the whole repeat away, and an engine + this environment does not have is not a measurement at all. + + Three sources are consulted, in this order, and the order is the point: what the + image recorded about not installing the engine at all (:func:`_not_attempted`), + then the engine's own ``unsupported_reason()``, then an actual attempt. The first + wins outright where it exists, since an engine the interpreter rules out cannot + describe its own ceiling -- it can only report what it finds in the environment + that ceiling produced. + + Args: + engine: Engine name to try. + capture: Capture file to read. + + Returns: + ``(reason, driver)``. *reason* is :data:`None` when the engine works and + the driver it used is returned alongside; otherwise *reason* explains why + it cannot be measured here and *driver* is whatever ran instead, which is + :data:`None` when nothing did. + + Raises: + Exception: Whatever the engine raised, when that is a real failure rather + than the engine being unavailable in this environment. + + """ + # An engine the interpreter rules out is answered here and nowhere else. The + # recorded note *replaces* the engine's own reason rather than being appended to + # it, which is the opposite of how a build failure is handled below, and + # deliberately: the engine can only describe what it finds in this environment, + # and on an interpreter where the package cannot be installed at all that + # description is a symptom. Nothing is lost by dropping it, because it is + # derivable from the note -- whereas the note is not derivable from it. + not_attempted = _not_attempted(engine) + if not_attempted is not None: + return not_attempted, None + + # Computed up front and appended to whichever reason comes back, because a + # recorded build failure explains every one of them: an engine whose package + # never installed will report "not installed", and the interesting part is the + # compiler error behind that. + build_failure = _install_failure(engine) + + def _reason(text: 'str') -> 'str': + """Attach the recorded build failure to *text*, when there is one.""" + return f'{text} -- {build_failure}' if build_failure else text + + reason = _declared_reason(engine) + if reason is not None: + return _reason(reason), None + + try: + extraction = _extract(engine, capture) + except Exception as exc: # pylint: disable=broad-except + reason = _unavailable(exc) + if reason is None: + raise + return _reason(reason), None + + driver = type(extraction.engine).__engine_name__ + if driver not in expected_drivers(engine): + # The quiet failure: ``Extractor`` warned and fell back, so the extraction + # succeeded and reported a frame count that has nothing to do with the + # engine asked for. + return _reason(f'its package is not installed -- pcapkit fell back to its own ' + f'{driver} parser, so timing it would measure the wrong thing'), driver + return None, driver + + +def measure(engine: 'str', capture: 'str', rounds: 'int', + tolerate: 'int' = TOLERATED_FAILURES) -> 'dict[str, Any]': + """Time *rounds* extractions of *capture* on *engine*. + + The driver is checked on **every** extraction, not just on the first. Checking + once would leave the run open to an engine that starts as itself and stops + being itself partway through -- and since the fallback only warns, nothing + else in the stack would object. The check is deliberately outside the timed + bracket so it cannot bias the figure. + + A small number of failed extractions is tolerated, and this is not + fastidiousness. ``pyshark`` spawns a :program:`tshark` process per extraction, + so a default run spawns six thousand of them, and measured here: one of them + crashed with ``TSharkCrashException ... retcode: 255`` partway through the + third pass and took the entire run with it -- six working engines went + unreported because the seventh hiccupped once. A crashed extraction is a + *non-measurement* rather than a slow measurement, so discarding its sample is + correct; what would not be honest is discarding it silently, which is why the + count and the reasons are returned and reported. + + Args: + engine: Engine name to time. + capture: Capture file to read. + rounds: Timed extractions to collect. The first is discarded as a warm-up. + tolerate: How many failed extractions to discard before giving up on this + pass. + + Returns: + A mapping with the mean milliseconds per packet, the totals it was derived + from, the driver that ran throughout, and anything that was discarded. + + Raises: + RuntimeError: If the driver ever disagreed with :func:`expected_drivers`, + or the capture yielded no packets. Fatal rather than warned about, + because the alternative is publishing a number for an engine that did + not produce it. + Exception: Whatever the engine raised, once more than *tolerate* + extractions have failed. The caller decides whether that costs the + engine its row or the whole run -- see :func:`run`. + + """ + wanted = expected_drivers(engine) + samples = [] # type: list[float] + drivers = set() # type: set[str] + discarded = [] # type: list[str] + length = 0 + + # ``EngineWarning`` is escalated to an exception for the whole loop, so a + # mid-run fallback surfaces here rather than on stderr where it would scroll + # past. Everything else is silenced: ``ExtractionWarning: EOF reached`` fires + # on every extraction of a truncated capture and its only effect on a + # benchmark is noise. + from pcapkit.utilities.warnings import EngineWarning # pylint: disable=import-outside-toplevel + + with warnings.catch_warnings(): + warnings.simplefilter('ignore') + warnings.simplefilter('error', EngineWarning) + + while len(samples) < rounds: + try: + # NOTE: perf_counter_ns is monotonic. time_ns is wall clock and can + # step backwards under an NTP adjustment, giving a negative delta. + now = time.perf_counter_ns() + extraction = _extract(engine, capture) + delta = time.perf_counter_ns() - now + except EngineWarning: + # Never tolerated: this is the escalated fallback warning, i.e. the + # engine stopped being itself. Discarding it would turn the harness's + # central assertion into a shrug. + raise + except Exception as exc: # pylint: disable=broad-except + discarded.append(f'{type(exc).__name__}: {exc}') + if len(discarded) > tolerate: + raise + continue + + samples.append(float(delta)) + drivers.add(type(extraction.engine).__engine_name__) + length = extraction.length + + if len(samples) % 100 == 1: + # Progress goes to stderr at a hundredth of the legacy script's + # rate: stdout carries the JSON document, and a flushed write per + # round is jitter bought for nothing. + print(f' {engine}: round {len(samples)}/{rounds}', + end='\r', file=sys.stderr, flush=True) + + unexpected = drivers - set(wanted) + if unexpected: + raise RuntimeError( + f'engine {engine!r} was expected to run as {"/".join(wanted)} but ' + f'{"/".join(sorted(drivers))} ran; refusing to report the measurement' + ) + if length <= 0: + raise RuntimeError(f'engine {engine!r} reported {length} packets; nothing to divide by') + + if discarded: + print(f' {engine}: discarded {len(discarded)} failed extraction(s): ' + f'{discarded[0]}', file=sys.stderr, flush=True) + + samples.pop(0) # discard the warm-up round, as the legacy script does + mean_ns = statistics.mean(samples) + return { + 'driver': drivers.pop(), + 'discarded': discarded, + 'packets': length, + 'rounds': rounds, + 'timed_samples': len(samples), + 'mean_ns_per_extraction': mean_ns, + 'ms_per_packet': mean_ns / length / 1_000_000, + } + + +def _capture_facts(capture: 'str') -> 'dict[str, Any]': + """Identify the capture, so a figure cannot be silently taken on another one. + + Args: + capture: Path to the capture file. + + Returns: + Its basename, size, and SHA-256 digest. + + """ + with open(capture, 'rb') as file: + payload = file.read() + return { + 'name': os.path.basename(capture), + 'bytes': len(payload), + 'sha256': hashlib.sha256(payload).hexdigest(), + } + + +def run(capture: 'str', engines: 'tuple[str, ...]', rounds: 'int', repeats: 'int', + label: 'str', tolerate: 'int' = TOLERATED_FAILURES) -> 'dict[str, Any]': + """Measure every engine in *engines*, *repeats* times over. + + The loop is engines-inside-repeats rather than the other way round. A machine + that gets slower halfway through then affects every engine in a repeat about + equally, which is what lets :mod:`report` cancel the drift by taking each + engine's ratio against the ``default`` engine *from the same repeat*. + + Args: + capture: Capture file to read. + engines: Engine names to measure. + rounds: Timed extractions per engine per repeat. + repeats: How many times to measure the whole set. + label: Name of this environment, used by :mod:`report` to say where a row + came from. + tolerate: Failed extractions to discard per engine per pass before that + pass is written off. See :func:`measure`. + + Returns: + A JSON-serialisable document describing the environment and every + measurement in it. Engines that could not be measured are included, with + the reason -- never dropped. + + """ + facts = _capture_facts(capture) + + # Preflight once, up front. An engine ruled out here is ruled out for every + # repeat, and reporting the reason once beats reporting it three times. + status = {} # type: dict[str, Optional[str]] + for engine in engines: + reason, driver = preflight(engine, capture) + status[engine] = reason + if reason is None: + print(f'{label}: {engine} runs as {driver}', file=sys.stderr, flush=True) + else: + print(f'{label}: {engine} not measured -- {reason}', file=sys.stderr, flush=True) + + measurements = {engine: [] for engine in engines if status[engine] is None + } # type: dict[str, list[dict[str, Any]]] + drivers = {} # type: dict[str, str] + packets = {} # type: dict[str, int] + failures = {engine: [] for engine in engines} # type: dict[str, list[str]] + + for repeat in range(repeats): + print(f'{label}: repeat {repeat + 1}/{repeats}', file=sys.stderr, flush=True) + for engine in engines: + if status[engine] is not None: + continue + try: + result = measure(engine, capture, rounds, tolerate) + except RuntimeError: + # The harness's own assertions -- wrong driver, no packets. These + # mean a reported number would be a lie, so they end the run rather + # than costing one engine its row. + raise + except Exception as exc: # pylint: disable=broad-except + # A third-party engine gave up partway through this pass. Measured: + # `tshark` crashed on the third pass of a real run and, before this + # branch existed, took the six working engines down with it. The + # pass is lost; the run is not, and the loss is recorded rather than + # quietly rounded away. + failures[engine].append(f'pass {repeat + 1}: {type(exc).__name__}: {exc}') + print(f'{label}: {engine} failed on pass {repeat + 1} -- ' + f'{type(exc).__name__}: {exc}', file=sys.stderr, flush=True) + continue + + measurements[engine].append({ + 'repeat': repeat, + 'ms_per_packet': result['ms_per_packet'], + 'mean_ns_per_extraction': result['mean_ns_per_extraction'], + 'timed_samples': result['timed_samples'], + 'discarded': result['discarded'], + }) + drivers[engine] = result['driver'] + packets[engine] = result['packets'] + + results = [] # type: list[dict[str, Any]] + for engine in engines: + reason = status[engine] + collected = measurements.get(engine, []) + if reason is None and not collected: + # It preflighted cleanly and then failed every pass. Unmeasured, with + # what went wrong -- not a missing row, and not a zero. + reason = ('every timed pass failed: ' + + '; '.join(failures[engine])) if failures[engine] else \ + 'no pass produced a measurement' + results.append({ + 'engine': engine, + 'status': 'unmeasured' if reason is not None else 'measured', + 'reason': reason, + 'driver': drivers.get(engine), + 'packets': packets.get(engine), + 'repeats': collected, + 'failures': failures[engine], + }) + + return { + 'schema': 1, + 'environment': label, + # `pcapkit` is installed from the working tree rather than from PyPI, so + # its declared version does not distinguish one commit from another. The + # revision the image was built from is set by the Dockerfile and is the + # only thing that does. + 'pcapkit_revision': os.environ.get('PCAPKIT_REVISION') or None, + # Which image produced this document. The matrix runs one image per Python + # version, so a report covering several of them cannot name a single image the + # way a one-interpreter run could: the reference has to travel with the + # measurement rather than be passed to the report alongside it. + 'image': os.environ.get('BENCH_IMAGE') or None, + # The base image, by digest, as `python-images.txt` pinned it. Recorded here + # and deliberately not rendered: the JSON is a published artefact and is where + # provenance this fine-grained belongs, whereas a row per version in the + # report's provenance block would push the facts a reader does need off the + # top of the table. It is also recoverable from the pins file at the + # `pcapkit` revision above, which is what makes leaving it out of the prose + # safe rather than lossy. + 'base_image': os.environ.get('PCAPKIT_BASE_IMAGE') or None, + 'python': platform.python_version(), + 'implementation': platform.python_implementation(), + 'machine': platform.machine(), + 'capture': facts, + 'rounds': rounds, + 'repeats': repeats, + 'packages': _distribution_versions(), + 'tshark': _tshark_version(), + 'libpcap': _libpcap_version(), + 'results': results, + } + + +def main(argv: 'Optional[list[str]]' = None) -> 'int': + """Command line entry point. + + Args: + argv: Argument list, defaulting to :data:`sys.argv`. + + Returns: + Process exit status. + + """ + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument('--capture', required=True, help='capture file to extract') + parser.add_argument('--label', required=True, + help='name of this environment, e.g. the virtualenv it runs in') + parser.add_argument('--out', required=True, help='where to write the JSON document') + parser.add_argument('--rounds', type=int, default=ROUNDS, + help=f'timed extractions per engine per pass (default {ROUNDS})') + parser.add_argument('--repeats', type=int, default=REPEATS, + help=f'passes over the whole engine set (default {REPEATS})') + parser.add_argument('--engines', default=','.join(ENGINES), + help='comma-separated engines to measure') + parser.add_argument('--tolerate', type=int, default=TOLERATED_FAILURES, + help=('failed extractions to discard per engine per pass before ' + f'giving up on it (default {TOLERATED_FAILURES})')) + args = parser.parse_args(argv) + + if args.rounds < 2: + parser.error('--rounds must be at least 2; the first round is a discarded warm-up') + if args.repeats < 1: + parser.error('--repeats must be at least 1') + if args.tolerate < 0: + parser.error('--tolerate cannot be negative') + + engines = tuple(name.strip() for name in args.engines.split(',') if name.strip()) + if 'default' not in engines: + # Every ratio is taken against ``default`` in the same environment, and it + # is the only engine present in all of them, so the report cannot stitch + # anything together without it. + parser.error("--engines must include 'default'; it is the baseline every ratio uses") + + document = run(args.capture, engines, args.rounds, args.repeats, args.label, args.tolerate) + with open(args.out, 'w', encoding='utf-8') as file: + json.dump(document, file, indent=2, sort_keys=True) + file.write('\n') + print(f'{args.label}: wrote {args.out}', file=sys.stderr, flush=True) + return 0 + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/examples/benchmark/entrypoint.sh b/examples/benchmark/entrypoint.sh new file mode 100755 index 0000000000..07ecc0cd2e --- /dev/null +++ b/examples/benchmark/entrypoint.sh @@ -0,0 +1,119 @@ +#!/usr/bin/env bash +# +# Inside the image, in one of two modes. +# +# `BENCH_MODE=measure` (the default) times every environment this image holds and +# writes one JSON document per environment. Which environments those are is read +# from `/opt/environments`, written at build time by the interpreter itself -- so +# this script never has to work out whether it is on a version that can hold +# `pypcap`, and cannot get the answer wrong. +# +# The split into two virtualenvs, where there are two, is forced. `pypcap` and +# `pcap-ct` both install a top-level `pcap` module, and with both present +# `import pcap` resolves to `pcap-ct`, leaving upstream's extension module shadowed +# and unreachable -- so no single environment can hold every engine. The `default` +# engine is in all of them, and is what report.py normalises against to join them. +# +# `BENCH_MODE=report` reports on documents produced by *other* containers, handed in +# through `/in`. The matrix spans several interpreters and no single one of them +# measured all of it, so the reporting pass cannot be the tail of a measuring run +# the way it was when there was one image. It runs here rather than on the host +# because the host is not required to have a Python at all -- docker is the only +# thing `run.sh` insists on, and report.py is standard library only, so the +# interpreter it happens to run under does not affect what it says. +set -euo pipefail + +CAPTURE="${BENCH_CAPTURE:-/src/examples/captures/in.pcap}" +ROUNDS="${BENCH_ROUNDS:-1000}" +REPEATS="${BENCH_REPEATS:-3}" +ENGINES="${BENCH_ENGINES:-default,dpkt,scapy,pypcap,pcap_ct,pypcapfile,pyshark}" +EMULATED="${BENCH_EMULATED:-}" +MODE="${BENCH_MODE:-measure}" +OUT="${BENCH_OUT:-/out}" +IN="${BENCH_IN:-/in}" + +mkdir -p "$OUT" + +if [ "$MODE" = measure ]; then + # Every environment runs the *whole* engine list, not just the engine unique to + # it. The engines it cannot run then come back with the reason it cannot -- which + # for `pcap_ct` in the `pypcap` environment is the mutual exclusion itself, stated + # by the engine rather than asserted by this script. + # + # Read on file descriptor 3 rather than on stdin. `pyshark` spawns a `tshark` + # per extraction and those children inherit this shell's stdin, so a loop fed + # through stdin is a loop whose remaining lines a subprocess is free to swallow + # -- and the symptom would be an environment silently never measured rather than + # an error. + while read -r label venv lock <&3; do + [ -n "$label" ] || continue + + echo "==> measuring in the ${label} environment" >&2 + "$venv/bin/python" /src/examples/benchmark/benchmark.py \ + --capture "$CAPTURE" \ + --label "$label" \ + --out "$OUT/$label.json" \ + --rounds "$ROUNDS" \ + --repeats "$REPEATS" \ + --engines "$ENGINES" + + # The locks record the resolved transitive closure, which the pinned + # requirements files only partly fix. Copied out under the environment's own + # label rather than the virtualenv's, because the matrix has one lock per + # (version, `pcap` provider) pair and `pypcap.txt` from five images would be + # four files overwriting each other. + cp -f "$lock" "$OUT/$label.txt" 2>/dev/null || true + done 3< /opt/environments + exit 0 +fi + +if [ "$MODE" != report ]; then + echo "entrypoint.sh: unknown BENCH_MODE '${MODE}'; expected 'measure' or 'report'." >&2 + exit 2 +fi + +# `[ -e "$path" ] || continue` rather than `shopt -s nullglob`, so an empty directory +# is handled by the loop that reads it instead of by a shell option set a screen +# earlier. +documents=() +for path in "$IN"/*.json; do + [ -e "$path" ] || continue + documents+=("$path") +done + +if [ "${#documents[@]}" -eq 0 ]; then + echo "entrypoint.sh: no benchmark documents in ${IN}; nothing to report on." >&2 + exit 1 +fi + +# A version whose image could not be built or run leaves a note here rather than +# leaving its column out. `run.sh` writes them; the file name is the Python series +# and the contents are the reason, flattened to one line because it becomes a +# bullet in the emitted markup. +report_args=() +for path in "$IN"/missing/*.txt; do + [ -e "$path" ] || continue + series="$(basename "$path" .txt)" + reason="$(tr '\n' ' ' < "$path")" + report_args+=(--missing "${series}=${reason}") +done + +# Built with `if` rather than `[ -n "$X" ] && args+=(...)`: under `set -e` a top-level +# AND-list whose test fails is itself a failed command, so the short form would exit +# the script whenever the variable happened to be empty -- which for a native run, the +# common case, is always. +if [ -n "$EMULATED" ]; then + report_args+=(--emulated "$EMULATED") +fi + +# `${report_args[@]+"${report_args[@]}"}` rather than a bare `"${report_args[@]}"`: +# expanding an empty array under `set -u` is an error before bash 4.4, and a native run +# with no missing versions leaves it empty, which is the common case. The image ships +# bash 5, so this is belt and braces -- but the alternative is a line that is only +# correct because of a fact about the base image that nothing here states. +echo >&2 +exec python /src/examples/benchmark/report.py \ + "${documents[@]}" \ + --rst-out "$OUT/table.rst" \ + --versions-rst-out "$OUT/table-versions.rst" \ + ${report_args[@]+"${report_args[@]}"} diff --git a/examples/benchmark/python-images.txt b/examples/benchmark/python-images.txt new file mode 100644 index 0000000000..990c53336c --- /dev/null +++ b/examples/benchmark/python-images.txt @@ -0,0 +1,74 @@ +# The base image for every interpreter in the matrix, and what can be built on it. +# +# This file is the matrix. `run.sh` reads it to decide which versions to run and +# what to build for each, and it is deliberately a table a person can read rather +# than a `case` statement buried in the script: the digests below were each chosen +# by resolving a tag once, and a reader is entitled to see that they were chosen +# rather than defaulted into. +# +# Fields, whitespace-separated: +# +# 1. Python series, as `pcapkit.extract`'s users would name it. +# 2. Base image, pinned **by digest**. A tag pin is not a pin: `3.11-slim-bookworm` +# is mutable and the image behind it is rebuilt whenever Debian or CPython +# ships a patch, so a run six months from now would silently measure a +# different interpreter and a different libpcap. Each digest is the multi-arch +# OCI *index* rather than a per-platform manifest, so it resolves natively on +# both linux/amd64 and linux/arm64 and the pin does not force emulation. +# 3. `pypcap` or `no-pypcap` -- whether this interpreter gets a second virtualenv +# holding upstream PyPCAP. See below. +# 4. `default` or `opt-in` -- whether `run.sh` measures it when given no +# `--pythons`. +# +# Every image is **slim-bookworm**, not a mixture. That is a measurement decision +# rather than a packaging one: the per-version table reports absolute milliseconds +# and its columns are meant to be compared with each other, so everything except +# the interpreter is held constant -- the same Debian, hence the same `tshark`, the +# same libpcap, and the same compiler behind `pypcap`. Mixing bookworm and trixie +# across columns would leave a difference between two columns unattributable to +# either the interpreter or the system libraries. +# +# ## Which engines run on which interpreter +# +# Measured, not assumed. `default` (pcapkit's own parser), `dpkt` and `scapy` run +# everywhere. The other four have ceilings, and only the first of them is a +# *build* problem, which is why only it needs a column here: +# +# | engine | interpreters | why it stops | +# |--------------|-------------------|-------------------------------------------| +# | `pypcap` | 3.11 and older | ships a `pcap.c` pre-generated by Cython | +# | | | 0.29.32, which does not compile against | +# | | | the 3.12+ C API. Cannot be *installed*. | +# | `pypcapfile` | 3.11 and older | 0.12.0's linklayer imports `imp`, removed | +# | | | in 3.12. Installs; fails on import. | +# | `pyshark` | 3.13 and older | 0.6 wants an implicit asyncio event loop, | +# | | | which 3.14 refuses to create. Installs. | +# | `pcap_ct` | 3.10+ | the whole matrix, in practice. | +# +# `pypcapfile` and `pyshark` install on every interpreter and fail at run time, and +# `pcapkit`'s own engines already say so through `unsupported_reason()` -- so the +# report gets the real cause from the engine and this file has nothing to add. +# `pypcap` is different: its install is what fails, so on 3.12+ a `pypcap` +# virtualenv would be several minutes of compiling to produce nothing. Those rows +# say `no-pypcap`, the image skips that virtualenv, and it records why so the report +# can say the engine is unsupported on the interpreter rather than that it failed +# here. +# +# ## 3.15 +# +# Opt-in, and allowed to fail. There is no released `python:3.15-slim-bookworm`; +# `3.15-rc-slim-bookworm` is a release candidate and the tag moves with every rc, so +# the digest below freezes one particular candidate and will name an older rc as +# time passes -- bump it deliberately, or drop the row when 3.15 ships. The +# repository's CI already treats 3.15 as an allowed-to-fail leg, and this matches +# that: `run.sh --pythons ...,3.15` measures it, a plain `run.sh` does not, and a +# version that cannot be built is reported as a missing column with the reason +# rather than dropped. +# +# version base image pypcap tier +3.10 python:3.10-slim-bookworm@sha256:68d914ec641a0b69267ce65184d000a2bc3a9ee2590ab702b82250ab2385735a pypcap default +3.11 python:3.11-slim-bookworm@sha256:528257d48c1da0dcecc2e725d1ae34498d60c965f1241e39cd6a85a8859bdf84 pypcap default +3.12 python:3.12-slim-bookworm@sha256:782412e85d0f0984994c290652577d4018aff08145c85b262bb63dc0c7522254 no-pypcap default +3.13 python:3.13-slim-bookworm@sha256:ed86c82274b3c69b52fb5820f358f0bd7df0b603332063cb5c6e32bd220c3e6e no-pypcap default +3.14 python:3.14-slim-bookworm@sha256:9ab8d9c8514b44f90cf0029dd42fdd7e9e211e639c8b995304cc04568dee900f no-pypcap default +3.15 python:3.15-rc-slim-bookworm@sha256:984bee69a9c4abf834e9c458a0d81ef10332a3c9d3a8666e566f46cc23935599 no-pypcap opt-in diff --git a/examples/benchmark/report.py b/examples/benchmark/report.py new file mode 100644 index 0000000000..f5b3e2cece --- /dev/null +++ b/examples/benchmark/report.py @@ -0,0 +1,1377 @@ +# -*- coding: utf-8 -*- +"""Turn benchmark JSON documents into paste-ready reStructuredText tables. + +The first line above is deliberately free of Sphinx roles: :func:`main` hands it to +:mod:`argparse` as the command's description, where ``:mod:`benchmark``` would be +shown to the user verbatim. + +:mod:`benchmark` measures one Python environment. This module stitches all of them +together and emits **two** tables, because the run answers two questions that want +different arithmetic: + +**Which engine is faster?** :func:`render_rst` -- ratios against the ``default`` +engine. An absolute figure describes the machine it was taken on at least as much as +it describes the engine, so a table of them cannot be reproduced and cannot be +compared with anything. A ratio against the ``default`` engine *measured in the same +environment on the same repeat* divides the machine out. It is also what makes +mutually exclusive environments joinable at all: ``pypcap`` and ``pcap_ct`` both +provide the top-level :mod:`pcap` module and cannot coexist, and ``default`` is the +engine present in every environment. One absolute number survives, as a footnote: +the baseline's own milliseconds per packet, which is what a reader needs to convert +the ratios back. + +**How does each engine do on each interpreter?** :func:`render_versions_rst` -- +absolute milliseconds per packet, one column per Python version. Here the machine +must *not* be divided out, and a ratio would be actively wrong: ratios are +normalised within one environment, so dividing across two interpreters would report +a number that is neither engine's speed nor the interpreter's. Every column of that +table was measured in one run on one host with everything but the interpreter held +constant, which is exactly the condition under which absolute times are comparable +-- with each other, and with nothing on any other machine. + +So the two tables are not two formattings of one result and neither replaces the +other. The ratio table pools every interpreter and answers a question about engines; +the per-version table separates them and answers a question about interpreters. + +**Why the spread is reported.** A single pass cannot distinguish a real difference +between two engines from the machine having been briefly busy. So the whole set is +measured several times and each row carries the range actually observed. Rows +whose ranges overlap are marked: this run does not establish which of them is +faster, and reporting their medians as though it did would be inventing precision. + +The emitted markup is plain reStructuredText for **docutils**, which is what +renders the project's README on GitHub. No Sphinx-only roles (``:mod:``, +``:func:``, ``:manpage:``) appear in it -- only double-backtick literals -- because +docutils does not know them and renders them as errors. + +""" + +import argparse +import json +import math +import statistics +import sys +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from typing import Any, Iterable, Optional, Sequence + +__all__ = ['Row', 'absolutes_by_version', 'baseline_absolutes', 'collect', 'overlapping', + 'python_series', 'render_rst', 'render_text', 'render_versions_rst', + 'unmeasured_by_version', 'version_columns'] + +#: Engine name reported for ``pcapkit``'s own parser. ``pcapkit.extract`` accepts +#: ``'default'``; the README and the docs call it ``pcapkit``, and that is the +#: spelling the table uses. +BASELINE = 'default' + +#: How the baseline is spelled in the report. +BASELINE_LABEL = 'pcapkit' + +#: Marks a row whose observed range overlaps another row's, i.e. a gap this run +#: does not resolve. A double dagger rather than a footnote reference, so the +#: emitted snippet cannot collide with footnote numbers already in the README. +OVERLAP_MARK = '‡' + +#: Marks a row that was not measured, and points at the note carrying the reason. +UNMEASURED_MARK = '†' + +#: Marks a measured row that lost a pass, or had extractions discarded. The number +#: is still reported -- what was collected is real -- but it rests on less evidence +#: than the rows beside it, and that has to be visible rather than inferable from +#: the pass count. +PARTIAL_MARK = '¶' + +#: Characters that begin reStructuredText inline markup, and therefore have to be +#: neutralised in any text this module did not write itself. See :func:`_escape`. +#: The backslash is first because the loop that applies these would otherwise escape +#: the backslashes it had just inserted. +RST_SPECIAL = ('\\', '`', '*', '_', '|', '[', ']') + + +def _escape(text: 'str') -> 'str': + """Neutralise reStructuredText markup in text that came from elsewhere. + + Every reason in this report is written by something else -- an engine's + ``unsupported_reason()``, an exception's ``str()``, the tail of a pip or gcc log + that the image recorded -- and all of it lands in emitted markup as prose. None of + those sources knows it is writing reStructuredText. + + Measured, with the same ``halt_level=2`` docutils settings the suite's own tests + use, on strings a compiler produces without trying: + + * ``gcc: error: **kwargs handling broke the build`` -- *inline strong start-string + without end-string*; + * ``cannot find `pcap.h`` -- GNU diagnostics quote like that, and an unbalanced + backtick is *inline interpreted text start-string without end-string*; + * ``imports imp_ which was removed`` -- a trailing underscore is a reference, so + *unknown target name: "imp"*; + * anything ending ``::`` -- *literal block expected; none found*. + + Each of those turns the snippet this module promises can be pasted into + :file:`README.rst` verbatim into a visible error block on GitHub, and the failure + is invisible here because the harness's own fixtures are all well-behaved English. + + Args: + text: Prose from outside this module. + + Returns: + The same prose, with markup characters escaped. A backslash escape in + reStructuredText is removed on rendering, so the reader sees the original. + + """ + for char in RST_SPECIAL: + text = text.replace(char, '\\' + char) + # Not in the loop above, because a colon is only dangerous at the end of a + # paragraph -- and every reason here is its own single-line paragraph -- where + # ``::`` promises a literal block that the next line is not. Escaping every colon + # instead would put a backslash into the middle of most of these sentences for + # nothing. + if text.endswith(':'): + text = text[:-1] + '\\:' + return text + + +class Row: + """One engine's line in the table. + + Args: + engine: Engine name as ``pcapkit.extract`` spells it. + label: How the engine is written in the table. + ratios: Every ratio observed for it, one per (environment, repeat) pair. + absolutes: The matching milliseconds per packet, for the human report. + reasons: Why it could not be measured, keyed by environment. Empty when it + was measured somewhere. + environments: Environments that contributed a measurement. + failures: Passes that failed outright, attributed to their environment. + discarded: How many individual extractions were discarded as failures. + + """ + + def __init__(self, engine: 'str', label: 'str', ratios: 'list[float]', + absolutes: 'list[float]', reasons: 'dict[str, str]', + environments: 'list[str]', failures: 'Optional[list[str]]' = None, + discarded: 'int' = 0) -> 'None': + self.engine = engine + self.label = label + self.ratios = ratios + self.absolutes = absolutes + self.reasons = reasons + self.environments = environments + self.failures = failures or [] + self.discarded = discarded + + @property + def partial(self) -> 'bool': + """Whether this row lost a pass or discarded extractions.""" + return bool(self.failures) or self.discarded > 0 + + @property + def measured(self) -> 'bool': + """Whether any environment produced a number for this engine.""" + return bool(self.ratios) + + @property + def median(self) -> 'float': + """Median of the observed ratios. + + The median rather than the mean: a repeat that landed on a scheduling + hiccup is an outlier in one direction only, and the mean follows it. + + """ + return statistics.median(self.ratios) + + @property + def low(self) -> 'float': + """Smallest ratio observed.""" + return min(self.ratios) + + @property + def high(self) -> 'float': + """Largest ratio observed.""" + return max(self.ratios) + + @property + def spread(self) -> 'float': + """Observed range as a fraction of the median. + + Returns: + ``(high - low) / median``, or ``0.0`` from a single sample -- where the + honest answer is "not observed", which :func:`render_text` says in + words rather than encoding as a number here. + + """ + if len(self.ratios) < 2 or self.median == 0: + return 0.0 + return (self.high - self.low) / self.median + + @property + def reason(self) -> 'str': + """One phrase explaining why the engine was not measured. + + Environments usually agree, and when they do not the disagreement is the + interesting part -- ``pypcap`` is unavailable in the ``pcap_ct`` + environment for a completely different reason than on an interpreter too + new for it -- so every distinct reason is kept, attributed. + + """ + distinct = sorted(set(self.reasons.values())) + if len(distinct) == 1: + return distinct[0] + return '; '.join(f'in {env}, {self.reasons[env]}' for env in sorted(self.reasons)) + + +def _significant(value: 'float', digits: 'int' = 3) -> 'str': + """Format *value* to *digits* significant figures, without exponent notation. + + Ratios in this table span three orders of magnitude -- ``pyshark`` is a + subprocess spawn per extraction, ``dpkt`` is not -- so a fixed number of + decimal places is either noise at one end or a rounded-away difference at the + other. + + Implemented from the magnitude rather than with ``%g``, which reaches for an + exponent outside roughly ``1e-5``..``1e6``: ``5e-09`` is correct and unreadable + in a document that otherwise reads as prose, and a table of ratios should never + make a reader parse scientific notation to see which row is faster. + + Args: + value: Number to format. + digits: Significant figures to keep. + + Returns: + The formatted number, in plain decimal. + + """ + if value == 0: + return '0' + magnitude = math.floor(math.log10(abs(value))) + # Never negative: a large number keeps its integer part in full rather than + # being rounded to the requested significance, since '1235' misleads nobody and + # '1.23e+03' does. + decimals = max(0, digits - 1 - magnitude) + return f'{value:.{decimals}f}' + + +def _range(low: 'float', high: 'float') -> 'str': + """Format an observed range so it does not read as a range of one value. + + At three significant figures a genuinely tight range collapses: ``pypcap`` was + measured at ``0.183-0.183``, which looks like a formatting bug and tells the + reader nothing. So the precision is raised until the two ends differ, and if + they still do not, the value is shown once -- "the range is narrower than the + precision shown" rather than a fake interval. + + Args: + low: Smallest value observed. + high: Largest value observed. + + Returns: + Either ``low-high`` or a single value. + + """ + for digits in (3, 4, 5): + left, right = _significant(low, digits), _significant(high, digits) + if left != right: + return f'{left}-{right}' + return _significant(low) + + +def _fixed(value: 'float', decimals: 'int' = 4) -> 'str': + """Format *value* to a fixed number of decimal places. + + Used only by the per-version table, and deliberately not :func:`_significant`, + which every other number in the report goes through. A column of figures is read + down rather than one at a time, and significant figures give each row a different + number of decimal places -- ``0.01700`` above ``14.74`` -- which is precisely the + layout that makes a column impossible to scan. Fixed places line the decimal + points up, and they are also what the table this replaces in + :file:`README.rst` has always used, so a regenerated table is a diff of the + numbers rather than a diff of the formatting. + + Args: + value: Number to format. + decimals: Decimal places to keep. + + Returns: + The formatted number. + + """ + return f'{value:.{decimals}f}' + + +def python_series(document: 'dict[str, Any]') -> 'str': + """The ``major.minor`` Python version a document was measured on. + + Read from the interpreter's own :func:`platform.python_version`, recorded at + measuring time, rather than parsed out of the environment label. The label is a + name the runner chose and the recorded version is what actually ran, so they are + not the same kind of fact -- and if they ever disagreed, putting a document in the + column its label claimed would be exactly the silent mislabelling the rest of this + harness refuses to leave open. + + Args: + document: One :mod:`benchmark` document. + + Returns: + The series, e.g. ``'3.11'``. + + """ + parts = document['python'].split('.') + return '.'.join(parts[:2]) if len(parts) >= 2 else document['python'] + + +def _series_key(series: 'str') -> 'tuple[tuple[int, ...], str]': + """Sort key that orders ``3.9`` before ``3.10`` rather than after it. + + The whole point: these are dotted numbers pretending to be strings, and every + string sort of them is wrong from the moment a minor version reaches double + digits. Non-numeric components fall back to the raw string, which keeps the sort + total rather than raising on something unexpected. + + Args: + series: A version, e.g. ``'3.10'`` or ``'3.10.19'``. + + Returns: + A key that sorts numerically. + + """ + return tuple(int(part) for part in series.split('.') if part.isdigit()), series + + +def _ordered(documents: 'Sequence[dict[str, Any]]') -> 'list[dict[str, Any]]': + """The documents in reporting order: by interpreter, then by environment name. + + Presentation only -- :func:`collect` keeps the order it was given, since the order + ratios are pooled in does not change them. This exists so that what the report + *says* does not depend on the order the JSON files happened to arrive in, which + for a matrix of nine environments is otherwise a diff between two runs of the same + thing. + + Args: + documents: One :mod:`benchmark` document per environment. + + Returns: + The same documents, ordered. + + """ + return sorted(documents, + key=lambda document: (_series_key(python_series(document)), + document['environment'])) + + +def version_columns(documents: 'Sequence[dict[str, Any]]', + missing: 'Sequence[tuple[str, str]]' = ()) -> 'list[str]': + """Every Python version the run covers, oldest first. + + Includes the versions in *missing*, which produced no documents at all. A version + whose image could not be built is a column of ``--`` with a reason, not an absent + column: leaving it out would turn "we could not measure this" into "we did not ask + about this", and the reader has no way to tell those apart from a table alone. + + Args: + documents: One :mod:`benchmark` document per environment. + missing: ``(version, reason)`` pairs for versions that produced nothing. + + Returns: + The versions, in numeric order. + + """ + series = {python_series(document) for document in documents} + series.update(version for version, _ in missing) + return sorted(series, key=_series_key) + + +def absolutes_by_version(documents: 'Sequence[dict[str, Any]]' + ) -> 'dict[str, dict[str, list[float]]]': + """Milliseconds per packet, pooled by engine and Python version. + + An engine measured in both virtualenvs of one interpreter contributes to that + interpreter's cell from each, since the two are measuring the same engine on the + same Python and the pooled median is the better figure. + + Note what is *not* required here, in contrast to :func:`collect`: a baseline. A + pass whose ``default`` reading is missing or zero cannot yield a ratio and is + dropped there, but its own millisecond figure is complete and is kept here. + Discarding it would lose a real measurement to a rule that does not apply to it. + + Args: + documents: One :mod:`benchmark` document per environment. + + Returns: + ``{engine: {series: [ms_per_packet, ...]}}``, holding only what was measured. + + """ + grid = {} # type: dict[str, dict[str, list[float]]] + for document in documents: + series = python_series(document) + for entry in document['results']: + if entry['status'] != 'measured': + continue + values = [sample['ms_per_packet'] for sample in entry['repeats']] + if not values: + continue + grid.setdefault(entry['engine'], {}).setdefault(series, []).extend(values) + return grid + + +def unmeasured_by_version(documents: 'Sequence[dict[str, Any]]' + ) -> 'dict[str, dict[str, str]]': + """Why each engine has no figure for each Python version it has none for. + + Only cells that really are empty get a reason. An engine unmeasured in the + ``pypcap`` environment and measured in the ``pcap_ct`` one on the same interpreter + has a figure for that interpreter, and reporting the mutual exclusion as though it + were a gap would explain a cell that is not there -- which is how a table acquires + footnotes that contradict it. + + Args: + documents: One :mod:`benchmark` document per environment. + + Returns: + ``{engine: {series: reason}}``. Where the environments of one interpreter + disagree, each reason is attributed to its environment, since a disagreement + is more informative than either half of it. + + """ + measured = absolutes_by_version(documents) + reasons = {} # type: dict[str, dict[str, dict[str, str]]] + for document in documents: + series = python_series(document) + for entry in document['results']: + if entry['status'] == 'measured': + continue + engine = entry['engine'] + if series in measured.get(engine, {}): + continue + reasons.setdefault(engine, {}).setdefault(series, {})[document['environment']] = \ + entry['reason'] or 'no reason recorded' + + collapsed = {} # type: dict[str, dict[str, str]] + for engine, by_series in reasons.items(): + for series, by_environment in by_series.items(): + distinct = sorted(set(by_environment.values())) + if len(distinct) == 1: + collapsed.setdefault(engine, {})[series] = distinct[0] + else: + collapsed.setdefault(engine, {})[series] = '; '.join( + f'in {environment}, {by_environment[environment]}' + for environment in sorted(by_environment) + ) + return collapsed + + +def _usable_baseline(value: 'Optional[float]') -> 'bool': + """Whether a baseline reading can normalise the pass it was taken alongside. + + Args: + value: The baseline's milliseconds per packet, or :data:`None` when the + baseline did not run on that repeat at all. + + Returns: + Whether dividing by it is meaningful. + + The two rejected cases are different problems that happen to share a fix. + :data:`None` means no baseline was recorded for the repeat, so there is + nothing to normalise against; ``0.0`` is a reading that *was* taken and + cannot be divided by. Testing the value's truthiness alone would conflate + them, and a zero baseline is a broken run worth telling apart from an absent + one even though both drop the sample. + + """ + return value is not None and value > 0 + + +def collect(documents: 'Sequence[dict[str, Any]]') -> 'list[Row]': + """Build the table's rows from the per-environment documents. + + Each ratio is taken **within** one environment and one repeat, against that + environment's own ``default`` measurement from that same repeat. Doing it any + other way -- pooling the baselines first, or dividing by the other + environment's baseline -- would put the drift the repeats exist to cancel + straight back into the answer. + + An engine present in several environments contributes a ratio from each. That + is deliberate: those ratios are already machine-normalised, so pooling them + gives a better estimate, and the extra spread it exposes is a real part of the + uncertainty rather than something to hide by picking one environment. + + Args: + documents: One :mod:`benchmark` document per environment. + + Returns: + One :class:`Row` per engine, ordered fastest first, with unmeasured + engines last. + + Raises: + ValueError: If *documents* is empty, or if an environment has no usable + baseline -- without one, nothing in it can be normalised, and a table + built from the remainder would silently be missing engines. + + """ + if not documents: + raise ValueError('no benchmark documents to report on') + + ratios = {} # type: dict[str, list[float]] + absolutes = {} # type: dict[str, list[float]] + reasons = {} # type: dict[str, dict[str, str]] + environments = {} # type: dict[str, list[str]] + failures = {} # type: dict[str, list[str]] + discarded = {} # type: dict[str, int] + order = [] # type: list[str] + + for document in documents: + env = document['environment'] + results = {entry['engine']: entry for entry in document['results']} + + base = results.get(BASELINE) + if base is None or base['status'] != 'measured': + raise ValueError( + f'environment {env!r} has no {BASELINE!r} measurement, so its ' + f'engines cannot be normalised' + ) + # Keyed by repeat index, so an engine is only ever divided by the baseline + # it actually ran alongside. + base_by_repeat = {sample['repeat']: sample['ms_per_packet'] + for sample in base['repeats']} + + for entry in document['results']: + engine = entry['engine'] + if engine not in order: + order.append(engine) + ratios.setdefault(engine, []) + absolutes.setdefault(engine, []) + reasons.setdefault(engine, {}) + environments.setdefault(engine, []) + failures.setdefault(engine, []) + discarded.setdefault(engine, 0) + + # Attributed to the environment: "the third pass failed" is far less + # useful than knowing which of the two environments it failed in. + for failure in entry.get('failures') or []: + failures[engine].append(f'in {env}, {failure}') + + if entry['status'] != 'measured': + reasons[engine][env] = entry['reason'] or 'no reason recorded' + continue + + contributed = False + for sample in entry['repeats']: + divisor = base_by_repeat.get(sample['repeat']) + if not _usable_baseline(divisor): + # No usable baseline for this repeat: the pair is unusable + # rather than approximable, so it is dropped instead of being + # divided by a baseline from a different repeat. + continue + ratios[engine].append(sample['ms_per_packet'] / divisor) + absolutes[engine].append(sample['ms_per_packet']) + discarded[engine] += len(sample.get('discarded') or []) + contributed = True + if contributed: + # Keyed on a ratio having survived, not on ``entry['repeats']`` + # being non-empty: passes whose baseline was missing contribute + # nothing, and counting their environment would claim a + # measurement the cross-environment check cannot then show. + environments[engine].append(env) + + rows = [ + Row(engine, + BASELINE_LABEL if engine == BASELINE else engine, + ratios[engine], absolutes[engine], reasons[engine], environments[engine], + failures[engine], discarded[engine]) + for engine in order + ] + # Fastest first, unmeasured last. The baseline lands wherever its ratio of 1.0 + # puts it, which is the honest place for it. + rows.sort(key=lambda row: (not row.measured, row.median if row.measured else 0.0)) + return rows + + +def overlapping(rows: 'Iterable[Row]') -> 'set[str]': + """Which measured rows have ranges that overlap another row's. + + Two engines whose observed ranges overlap were not separated by this run, and + saying which is faster on the strength of their medians would be reading + precision into noise. Compared pairwise across the whole table rather than + only between neighbours, since a row with a wide range can straddle several + tighter ones. + + Args: + rows: The table's rows. + + Returns: + Engine names to mark. A row measured only once has no range and cannot be + separated from anything, so it is marked too. + + """ + measured = [row for row in rows if row.measured] + marked = set() # type: set[str] + for row in measured: + if len(row.ratios) < 2: + marked.add(row.engine) + for index, left in enumerate(measured): + for right in measured[index + 1:]: + if left.low <= right.high and right.low <= left.high: + marked.add(left.engine) + marked.add(right.engine) + return marked + + +def baseline_absolutes(documents: 'Sequence[dict[str, Any]]') -> 'list[float]': + """Every ``default`` milliseconds-per-packet figure across the documents. + + This is the one absolute number the report keeps, and it is what lets a reader + turn the ratios back into times on the machine the run happened on. + + Args: + documents: One :mod:`benchmark` document per environment. + + Returns: + The figures, in the order encountered. + + """ + values = [] # type: list[float] + for document in documents: + for entry in document['results']: + if entry['engine'] == BASELINE and entry['status'] == 'measured': + values.extend(sample['ms_per_packet'] for sample in entry['repeats']) + return values + + +def _simple_table(headers: 'Sequence[str]', body: 'Sequence[Sequence[str]]', + numeric: 'bool' = False) -> 'list[str]': + """Render a reStructuredText simple table. + + Column widths come from the widest cell, and the rule lines are built to + match. Built here rather than with a library so the harness stays + dependency-light, and because a simple table is three lines of layout. + + Args: + headers: Column headings. + body: Rows of cells, all the same length as *headers*. + numeric: Right-align every column but the first, for a table of figures. + Decimal points then line up down each column, which is the difference + between a table that can be scanned and one that has to be read. The + first column stays left-aligned whatever this says, and not only for + looks: in a simple table an indented cell in the first column is how + docutils spells "this line continues the row above", so right-aligning + it would silently merge rows. + + Returns: + The table's lines, without a trailing blank. + + """ + widths = [len(header) for header in headers] + for row in body: + for index, cell in enumerate(row): + widths[index] = max(widths[index], len(cell)) + + if numeric and len(widths) > 1: + # One width for every figure column, not a width per column. Columns sized + # individually come out ragged -- the ``pyshark`` row makes one column two + # characters wider than its neighbours -- and a table of like quantities that + # is ragged reads as though the columns held different kinds of thing. + uniform = max(widths[1:]) + widths = [widths[0]] + [uniform] * (len(widths) - 1) + + def line(cells: 'Sequence[str]', align_numeric: 'bool') -> 'str': + padded = [cell.rjust(widths[index]) if align_numeric and index else + cell.ljust(widths[index]) for index, cell in enumerate(cells)] + # rstrip: trailing padding on the last column is legal but noisy in a diff. + return ' '.join(padded).rstrip() + + rule = ' '.join('=' * width for width in widths) + # The header stays left-aligned even in a numeric table, matching the hand-written + # table this one replaces: a right-aligned ``3.10`` sits over the tail of its + # column and reads as though it belonged to the column before it. + lines = [rule, line(headers, False), rule] + lines.extend(line(row, numeric) for row in body) + lines.append(rule) + return lines + + +def _image_pairs(documents: 'Sequence[dict[str, Any]]', + image: 'Optional[str]') -> 'list[tuple[str, str]]': + """The image each interpreter's figures came out of, as label/value pairs. + + One row when every document shares an image, which is what a single-version run + produces; one row per version otherwise. A matrix has no single image to name, and + the reference is the only thing that ties a column of the per-version table to a + specific build of a specific interpreter -- so it is reported per version rather + than collapsed into "several". + + Args: + documents: One :mod:`benchmark` document per environment. + image: Fallback reference, for documents that carry none of their own. + + Returns: + Ordered label/value pairs, empty when no reference is known at all. + + """ + seen = [] # type: list[tuple[str, str]] + for document in _ordered(documents): + reference = document.get('image') or image + if not reference: + continue + pair = (python_series(document), reference) + if pair not in seen: + seen.append(pair) + if not seen: + return [] + if len({reference for _, reference in seen}) == 1: + return [('Image', f'``{seen[0][1]}``')] + return [(f'Image ({series})', f'``{reference}``') for series, reference in seen] + + +def _provenance(documents: 'Sequence[dict[str, Any]]', image: 'Optional[str]', + emulated: 'Optional[str]') -> 'list[tuple[str, str]]': + """The facts that make a run reproducible, as label/value pairs. + + Deliberately excludes anything identifying the host it ran on -- no hostname, + no kernel, no CPU model. Containerising the benchmark is what makes the host + irrelevant, and printing host details would undo that while also making the + output awkward to paste into a public README. The architecture *is* included, + because it changes the numbers and identifies nothing. + + Args: + documents: One :mod:`benchmark` document per environment. + image: Image reference or digest the run happened in, for documents that do + not carry their own. + emulated: Description of the emulation in play, or :data:`None` when the + run was native. + + Returns: + Ordered label/value pairs. Versions the run could not measure are *not* here; + see :func:`_missing_pairs` for why they are assembled by the caller. + + """ + ordered = _ordered(documents) + first = ordered[0] + capture = first['capture'] + pairs = _image_pairs(documents, image) + if first.get('pcapkit_revision'): + pairs.append(('``pcapkit`` revision', f"``{first['pcapkit_revision']}``")) + # Every interpreter measured, at patch-level precision. The matrix's whole subject + # is the difference between these, so naming only the first would describe one + # column and imply it stood for all of them. + implementations = '/'.join(sorted({document['implementation'] for document in ordered})) + versions = sorted({document['python'] for document in ordered}, key=_series_key) + pairs.append(('Python', f"{implementations} {', '.join(versions)}")) + pairs.append(('Architecture', f"``{first['machine']}``")) + if emulated: + # Raw, like a missing-version reason and for the same reason: this value is also + # consumed by the plain-text report, so escaping it here would put backslashes in + # front of the operator. The two RST renderers escape it at their own boundary. + pairs.append(('Emulation', emulated)) + pairs.append(('Capture', f"``{capture['name']}`` -- {capture['bytes']} bytes, " + f"SHA-256 ``{capture['sha256'][:16]}...``")) + # Every engine on every interpreter should see the same number of frames in the same + # capture, so this row is normally one number. A disagreement is reported rather than + # dropped: with one interpreter it was barely possible and omitting it cost nothing, + # but two CPython versions extracting different counts from one file would be the + # most interesting thing in the report, and the previous behaviour was to hide + # exactly that by printing no row at all. + packets = {entry['packets'] for document in ordered for entry in document['results'] + if entry['packets']} + if len(packets) == 1: + pairs.append(('Packets per extraction', str(packets.pop()))) + elif packets: + by_series = {} # type: dict[str, set[int]] + for document in ordered: + for entry in document['results']: + if entry['packets']: + by_series.setdefault(python_series(document), set()).add(entry['packets']) + detail = ', '.join( + f"{series} {'/'.join(str(count) for count in sorted(by_series[series]))}" + for series in sorted(by_series, key=_series_key)) + pairs.append(('Packets per extraction', + f'**disagreed across the run** -- {detail}. One capture should yield ' + f'one frame count everywhere; treat every figure below as suspect ' + f'until this is explained.')) + pairs.append(('Iterations', f"{first['rounds']} timed extractions per engine per pass, " + f"the first discarded as a warm-up")) + # "pass" throughout rather than "repeat", so this row and the table's "Samples" + # column are counting the same unit. They are not the same *number*, and cannot be: + # a pass happens once per environment, so the table's count is this figure times the + # number of environments -- 3 passes over 7 environments is 21 samples per engine. + # The column was called "Passes" while there were two environments and one + # interpreter, where the difference was easy to overlook; across a matrix it reads as + # a contradiction, hence the rename. + environments = ', '.join(f"``{document['environment']}``" for document in ordered) + noun = 'environment' if len(ordered) == 1 else 'environments' + pairs.append(('Passes', f"{first['repeats']} over the whole engine set, " + f"in {len(ordered)} {noun}: {environments}")) + tsharks = sorted({document['tshark'] for document in ordered if document.get('tshark')}) + if tsharks: + pairs.append(('tshark', ', '.join(f'``{value}``' for value in tsharks))) + libpcaps = sorted({document['libpcap'] for document in ordered if document.get('libpcap')}) + if libpcaps: + pairs.append(('libpcap', ', '.join(f'``{value}``' for value in libpcaps))) + return pairs + + +def _missing_pairs(documents: 'Sequence[dict[str, Any]]', + missing: 'Sequence[tuple[str, str]]') -> 'list[tuple[str, str]]': + """The versions the run could not measure, as label/value pairs. + + A version that was asked for and could not be measured belongs in the record of + what the run covered. Without it the provenance block reads as the complete matrix, + and a reader comparing two runs would see one silently narrower than the other. + + Kept out of :func:`_provenance` for one specific reason: the reason string is the + only value in that block this module did not write itself, so it has to be escaped + before it enters markup and must *not* be escaped in the plain-text report. A value + whose correct form depends on where it is going cannot come out of the function both + destinations share. + + Args: + documents: One :mod:`benchmark` document per environment. + missing: ``(version, reason)`` pairs for versions whose build or run failed. + + Returns: + Ordered label/value pairs, the reason unescaped. + + """ + measured_series = {python_series(document) for document in documents} + pairs = [] # type: list[tuple[str, str]] + for version, reason in sorted(missing, key=lambda pair: _series_key(pair[0])): + # A version that produced some documents before failing is labelled + # differently, since "not measured" would contradict the figures it did + # contribute -- and those figures are printed a few lines further down. + state = 'partly measured' if version in measured_series else 'not measured' + pairs.append((f'Python {version} ({state})', reason)) + return pairs + + +def _package_table(documents: 'Sequence[dict[str, Any]]') -> 'list[str]': + """A simple table of every engine package's resolved version, per environment. + + Args: + documents: One :mod:`benchmark` document per environment. + + Returns: + The table's lines. + + """ + ordered = _ordered(documents) + environments = [document['environment'] for document in ordered] + names = [] # type: list[str] + for document in ordered: + for name in document['packages']: + if name not in names: + names.append(name) + + body = [] # type: list[list[str]] + for name in names: + cells = [f'``{name}``'] + for document in ordered: + version = document['packages'].get(name) + # An absent package is a fact about the environment, not a blank: it is + # why ``pypcap`` and ``pcap_ct`` need two environments in the first place. + cells.append(version if version else '*absent*') + if all(cell == '*absent*' for cell in cells[1:]): + continue + body.append(cells) + return _simple_table(['Package', *environments], body) + + +def render_versions_rst(rows: 'Sequence[Row]', documents: 'Sequence[dict[str, Any]]', + missing: 'Sequence[tuple[str, str]]' = (), + emulated: 'Optional[str]' = None) -> 'str': + """Render the per-version table of absolute milliseconds per packet. + + This is the snippet that replaces :file:`README.rst`'s **Test Results** section + outright: engines down the side, Python versions across the top, ``--`` for a cell + that could not be measured. It is self-contained -- heading, prose, table, notes -- + because it is pasted as a unit, and a note explaining a gap is no use in a + different file from the gap. + + Milliseconds, not ratios, and the prose says so. Two interpreters measured in one + run on one host differ in the interpreter and nothing else, which is the only + condition under which absolute times mean anything; a ratio would be worse than + redundant here, since ratios are normalised inside one environment and dividing + across two would produce a number describing neither. + + Args: + rows: The table's rows, from :func:`collect`. Used for its ordering, so that + this table and the ratio table list the engines in the same order and a + reader can carry their eye from one to the other. + documents: One :mod:`benchmark` document per environment. + missing: ``(version, reason)`` pairs for versions whose build or run failed. + A version that produced nothing at all becomes a column of ``--``; one + that failed after writing some of its environments keeps the figures it + managed and is labelled as incomplete rather than as unmeasured. + emulated: Description of the emulation in play, or :data:`None`. + + Returns: + reStructuredText, ready to paste. + + """ + first = _ordered(documents)[0] + columns = version_columns(documents, missing) + grid = absolutes_by_version(documents) + reasons = unmeasured_by_version(documents) + + # A version can be both measured and reported as failed: `run.sh` collects whatever + # documents a container wrote before it died, so a 3.11 whose second virtualenv + # crashed arrives with real figures *and* a note. The two cases need telling apart, + # because "not measured at all" is false of the second one -- and its column, having + # data, does have gaps worth attributing to the engines that left them. + measured_series = {python_series(document) for document in documents} + absent = {version: reason for version, reason in missing + if version not in measured_series} + partial = {version: reason for version, reason in missing + if version in measured_series} + + lines = [] # type: list[str] + lines.append('Test Results') + lines.append('~~~~~~~~~~~~') + lines.append('') + lines.append(f"Measured with ``examples/benchmark/run.sh``: {first['rounds']} timed") + lines.append('extractions of ``examples/captures/in.pcap`` per engine and Python version.') + lines.append('The first extraction is discarded as a warm-up. Values are milliseconds per') + lines.append(f"packet, the median of {first['repeats']} passes.") + lines.append('') + lines.append('All columns come from one run on one machine, differing in the interpreter') + lines.append('and nothing else, so they may be compared with each other. They may not be') + lines.append('compared with figures from another machine: absolute times are a property of') + lines.append('the host as much as of the engine.') + lines.append('') + + body = [] # type: list[list[str]] + for row in rows: + by_series = grid.get(row.engine, {}) + gaps = [series for series in columns + if series not in by_series and series not in absent] + # Marked for a gap in a column that produced something only. A blank under a + # version whose image never built is explained by the column's own note, and + # tagging the engine for it would blame the engine for someone else's failure. + name = f'``{row.label}``' + (f' {UNMEASURED_MARK}' if gaps else '') + cells = [name] + for series in columns: + values = by_series.get(series) + cells.append(_fixed(statistics.median(values)) if values else '--') + body.append(cells) + + # Marked when there is a note about the column itself, whether that is "nothing was + # measured here" or "not everything was". A note nothing in the table points at is a + # note a reader has no reason to look for. + headers = ['Engine'] + for series in columns: + noted = series in absent or series in partial + headers.append(f'{series} {UNMEASURED_MARK}' if noted else series) + lines.extend(_simple_table(headers, body, numeric=True)) + lines.append('') + + # The owner's own sentence about the old hand-maintained table, kept word for word: + # it is the right warning, and a regenerated table that rewords it would show up as + # a diff in the prose every time the numbers changed. + lines.append('The unavailable cells were attempted. They are not zeroes and must not be') + lines.append('compared with a measured row.') + lines.append('') + + notes = [] # type: list[str] + for version, reason in sorted(absent.items(), key=lambda pair: _series_key(pair[0])): + notes.append(f'* Python {version} -- not measured at all: {_escape(reason)}') + for version, reason in sorted(partial.items(), key=lambda pair: _series_key(pair[0])): + notes.append(f'* Python {version} -- measured, but the run did not complete, so this ' + f'column may be missing engines that would otherwise have a figure: ' + f'{_escape(reason)}') + for row in rows: + by_reason = {} # type: dict[str, list[str]] + for series, reason in sorted(reasons.get(row.engine, {}).items(), key=lambda pair: + _series_key(pair[0])): + if series in absent: + continue + by_reason.setdefault(reason, []).append(series) + # Grouped by reason rather than one bullet per cell: the usual case is one + # sentence true of three consecutive versions, and repeating it three times + # makes the notes longer than the table they annotate. + for reason, series_list in by_reason.items(): + notes.append(f'* ``{row.label}`` on {", ".join(series_list)} -- {_escape(reason)}') + + if notes: + lines.append(f'``{UNMEASURED_MARK}`` not measured, or not measured in full, for the') + lines.append('reason recorded at the time:') + lines.append('') + lines.extend(notes) + lines.append('') + + if emulated: + # `run.sh` composes this sentence itself today, which makes it safe today. It + # still arrives through `--emulated` on the command line, so it is outside text by + # every test that matters, and the one channel of it left unescaped would be the + # one nobody thought about. + lines.append(f'**{_escape(emulated)}** Timings taken under emulation are not ' + f'comparable with') + lines.append('native ones, and an absolute figure taken that way describes the emulator') + lines.append('as much as the interpreter. Re-run natively before publishing.') + lines.append('') + + return '\n'.join(lines).rstrip() + '\n' + + +def render_rst(rows: 'Sequence[Row]', documents: 'Sequence[dict[str, Any]]', + image: 'Optional[str]' = None, + emulated: 'Optional[str]' = None, + missing: 'Sequence[tuple[str, str]]' = ()) -> 'str': + """Render the whole snippet: provenance, table, and notes. + + Args: + rows: The table's rows, from :func:`collect`. + documents: One :mod:`benchmark` document per environment. + image: Image reference or digest the run happened in. + emulated: Description of the emulation in play, or :data:`None`. + missing: ``(version, reason)`` pairs for Python versions that produced no + measurements, recorded in the provenance block so this snippet does not + read as the complete matrix when it is not. + + Returns: + reStructuredText, ready to paste. + + """ + marked = overlapping(rows) + # Escaped once, here at the boundary where this snippet stops being data and starts + # being markup, and used for both places it appears below. See the note in + # :func:`_provenance` for why that function is handed the escaped form rather than + # doing this itself. + emulated_rst = _escape(emulated) if emulated else emulated + lines = [] # type: list[str] + + lines.append('Test Environment') + lines.append('~~~~~~~~~~~~~~~~') + lines.append('') + lines.append('.. list-table::') + lines.append('') + for label, value in _provenance(documents, image, emulated_rst): + lines.append(f' * - {label}') + lines.append(f' - {value}') + for label, value in _missing_pairs(documents, missing): + lines.append(f' * - {label}') + lines.append(f' - {_escape(value)}') + lines.append('') + lines.append('Resolved package versions:') + lines.append('') + lines.extend(_package_table(documents)) + lines.append('') + + # A heading of its own rather than "Test Results", which is the per-version + # absolute table's -- both snippets go into the same README, and two sections + # with one name is a document where a reader cannot say which table a sentence is + # about. + heading = 'Test Results (Relative)' + lines.append(heading) + lines.append('~' * len(heading)) + lines.append('') + lines.append(f'Times relative to ``{BASELINE_LABEL}``, which is 1 by definition; lower is') + lines.append('faster. Each ratio is taken against the baseline measured in the same') + lines.append('environment on the same pass, so the machine cancels out. "Observed range" is') + lines.append('the smallest and largest ratio actually seen across the passes.') + lines.append('') + if len({python_series(document) for document in documents}) > 1: + # Said explicitly, because it is the one thing about this table that a reader + # would otherwise get wrong: it is not "the ratio on some interpreter", it is + # every interpreter's ratio pooled. Each one was normalised in its own + # environment before pooling, so the pooling is sound -- and the extra width it + # puts into the observed range is a real disagreement between interpreters + # rather than noise to be averaged away. + lines.append('Ratios are pooled across every environment in the run, which means across') + lines.append('every Python version measured as well as both ``pcap`` providers. The') + lines.append('observed range therefore includes any disagreement between interpreters') + lines.append('about an engine; the per-version table is where that is broken out.') + lines.append('') + + body = [] # type: list[list[str]] + for row in rows: + name = f'``{row.label}``' + if row.measured: + if row.engine == BASELINE: + relative, observed = '1', '*baseline*' + else: + relative = _significant(row.median) + observed = (_range(row.low, row.high) + if len(row.ratios) > 1 else '*single pass*') + if row.engine in marked: + name += f' {OVERLAP_MARK}' + if row.partial: + name += f' {PARTIAL_MARK}' + body.append([name, relative, observed, str(len(row.ratios))]) + else: + body.append([f'{name} {UNMEASURED_MARK}', '*not measured*', '--', '0']) + + lines.extend(_simple_table(['Engine', 'Relative time', 'Observed range', 'Samples'], body)) + lines.append('') + + absolutes = baseline_absolutes(documents) + if absolutes: + median = statistics.median(absolutes) + lines.append(f'The single absolute figure, for converting the ratios back: ``{BASELINE_LABEL}``') + lines.append(f'itself ran at **{_significant(median, 4)} ms per packet** on this run') + if len(absolutes) > 1: + lines.append(f'({_range(min(absolutes), max(absolutes))} ' + f'across {len(absolutes)} passes). Absolute times are a property of the') + else: + lines.append('(one pass only). Absolute times are a property of the') + lines.append('machine as much as of the engine, which is why every other figure here is a') + lines.append('ratio.') + lines.append('') + + # The spread is reported whether or not any row is marked -- it is what tells a + # reader which gaps in the table are large enough to mean anything. The marker's + # legend, by contrast, is only emitted when a row actually carries it: an + # explanation of a symbol that does not appear reads as though the run failed to + # separate rows it in fact separated cleanly. + spreads = [row.spread for row in rows if row.measured and len(row.ratios) > 1] + if spreads: + lines.append(f'Run-to-run spread: the median row varied by ' + f'{_significant(statistics.median(spreads) * 100, 2)}% of its ratio across') + lines.append(f'the passes and the widest by {_significant(max(spreads) * 100, 2)}%. ' + f'A gap between two rows that is') + lines.append('smaller than that is noise, not a result.') + lines.append('') + + if marked: + if spreads: + lines.append(f'``{OVERLAP_MARK}`` this run does not separate these rows: their observed') + lines.append('ranges overlap, so the order their medians suggest is not established.') + else: + lines.append(f'``{OVERLAP_MARK}`` measured on a single pass, so no run-to-run spread was') + lines.append('observed and no gap between rows is established. Re-run with more passes.') + lines.append('') + + partial = [row for row in rows if row.measured and row.partial] + if partial: + lines.append(f'``{PARTIAL_MARK}`` measured, but on less evidence than the rows beside it.') + lines.append('What was collected is real; what was lost is named here rather than left to') + lines.append('be inferred from the pass count:') + lines.append('') + for row in partial: + detail = [] # type: list[str] + if row.discarded: + detail.append(f'{row.discarded} individual extraction(s) failed and were discarded') + detail.extend(_escape(failure) for failure in row.failures) + lines.append(f'* ``{row.label}`` -- ' + '; '.join(detail)) + lines.append('') + + unmeasured = [row for row in rows if not row.measured] + if unmeasured: + lines.append(f'``{UNMEASURED_MARK}`` not measured, for the reason the engine itself gives:') + lines.append('') + for row in unmeasured: + lines.append(f'* ``{row.label}`` -- {_escape(row.reason)}') + lines.append('') + + if emulated: + lines.append(f'**{emulated_rst}** Timings taken under emulation are not comparable ' + f'with native') + lines.append('ones, and the ratios are only as trustworthy as the emulator is uniform') + lines.append('across the work each engine does. Re-run natively before publishing.') + lines.append('') + + return '\n'.join(lines).rstrip() + '\n' + + +def render_text(rows: 'Sequence[Row]', documents: 'Sequence[dict[str, Any]]', + image: 'Optional[str]' = None, + emulated: 'Optional[str]' = None, + missing: 'Sequence[tuple[str, str]]' = ()) -> 'str': + """Render the operator-facing summary that precedes the RST snippets. + + Carries more than the tables do: the absolute milliseconds behind each ratio, + and the per-environment medians for engines measured in more than one, which + is the cross-check that the stitching worked. If two environments disagree + about ``dpkt``, the join on ``default`` is not doing its job and no amount of + tidy formatting downstream would reveal it. + + Args: + rows: The table's rows, from :func:`collect`. + documents: One :mod:`benchmark` document per environment. + image: Image reference or digest the run happened in. + emulated: Description of the emulation in play, or :data:`None`. + missing: ``(version, reason)`` pairs for Python versions that produced no + measurements. + + Returns: + Plain text. + + """ + marked = overlapping(rows) + lines = [] # type: list[str] + lines.append('=' * 78) + lines.append('pcapkit engine benchmark') + lines.append('=' * 78) + for label, value in _provenance(documents, image, emulated): + lines.append(f'{label + ":":26} {value.replace("``", "")}') + for label, value in _missing_pairs(documents, missing): + lines.append(f'{label + ":":26} {value}') + lines.append('') + + # Absolute milliseconds per interpreter, which the ratio table below cannot show + # and is the whole reason the run covers several Python versions. Printed here as + # well as emitted as markup so that the operator watching the run sees the answer + # without opening a file. + columns = version_columns(documents, missing) + if len(columns) > 1: + grid = absolutes_by_version(documents) + lines.append('ms per packet by Python version (absolute; comparable across columns,') + lines.append('since one host measured them all, and with no other machine):') + lines.append(f'{"engine":14} ' + ' '.join(f'{series:>9}' for series in columns)) + lines.append('-' * 78) + for row in rows: + by_series = grid.get(row.engine, {}) + cells = [] # type: list[str] + for series in columns: + values = by_series.get(series) + cells.append(f'{_fixed(statistics.median(values)):>9}' if values + else f'{"--":>9}') + lines.append(f'{row.label:14} ' + ' '.join(cells)) + lines.append('') + # No per-version failure lines here. The provenance block above already names + # every version the run could not measure and why, in the same output a few + # lines up, and a second rendering of the same fact managed to disagree with the + # first: it called a partly measured version "not measured" while its figures + # were printed in the grid directly above. + + lines.append(f'{"engine":14} {"relative":>10} {"range":>19} {"ms/packet":>12} samples') + lines.append('-' * 78) + for row in rows: + if not row.measured: + lines.append(f'{row.label:14} {"not measured":>10} {row.reason}') + continue + span = _range(row.low, row.high) if len(row.ratios) > 1 else 'single pass' + absolute = _significant(statistics.median(row.absolutes), 4) + flag = f' {OVERLAP_MARK}' if row.engine in marked else '' + if row.partial: + flag += f' {PARTIAL_MARK}' + lines.append(f'{row.label:14} {_significant(row.median):>10} {span:>19} ' + f'{absolute:>12} {len(row.ratios)}{flag}') + lines.append('') + + for row in rows: + if row.measured and row.partial: + detail = [] # type: list[str] + if row.discarded: + detail.append(f'{row.discarded} extraction(s) discarded') + detail.extend(row.failures) + lines.append(f'{PARTIAL_MARK} {row.label}: ' + '; '.join(detail)) + + shared = [row for row in rows if row.measured and len(row.environments) > 1] + if shared: + lines.append('cross-environment check (engines measured in more than one environment;') + lines.append('the two should agree, since each ratio is normalised in its own):') + for row in shared: + per_env = [] # type: list[str] + for env in row.environments: + values = _ratios_for(documents, row.engine, env) + if values: + per_env.append(f'{env}={_significant(statistics.median(values))}') + lines.append(f' {row.label:14} ' + ' '.join(per_env)) + lines.append('') + + if marked: + lines.append(f'{OVERLAP_MARK} ranges overlap another row: this run does not establish the order.') + lines.append('') + if emulated: + lines.append(f'WARNING: {emulated}') + lines.append('Numbers taken under emulation are not comparable with native ones.') + lines.append('') + return '\n'.join(lines) + + +def _ratios_for(documents: 'Sequence[dict[str, Any]]', engine: 'str', + environment: 'str') -> 'list[float]': + """Ratios contributed by one engine in one environment. + + Args: + documents: One :mod:`benchmark` document per environment. + engine: Engine name. + environment: Environment label. + + Returns: + The ratios, or an empty list when that pairing produced none. + + """ + for document in documents: + if document['environment'] != environment: + continue + results = {entry['engine']: entry for entry in document['results']} + base = results.get(BASELINE) + entry = results.get(engine) + if base is None or entry is None or base['status'] != 'measured': + return [] + by_repeat = {sample['repeat']: sample['ms_per_packet'] for sample in base['repeats']} + return [sample['ms_per_packet'] / by_repeat[sample['repeat']] + for sample in entry['repeats'] + if _usable_baseline(by_repeat.get(sample['repeat']))] + return [] + + +def main(argv: 'Optional[list[str]]' = None) -> 'int': + """Command line entry point. + + Args: + argv: Argument list, defaulting to :data:`sys.argv`. + + Returns: + Process exit status. + + """ + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument('json', nargs='+', help='benchmark JSON documents, one per environment') + parser.add_argument('--image', default=None, + help='image reference or digest the run happened in') + parser.add_argument('--emulated', default=None, + help='describe the emulation in play; omit for a native run') + parser.add_argument('--missing', action='append', default=[], metavar='VERSION=REASON', + help=('a Python version that produced no measurements, and why; ' + 'repeatable. It becomes a column of "--" with the reason ' + 'attached, rather than an absent column')) + parser.add_argument('--rst-out', default=None, + help='also write just the ratio RST snippet here') + parser.add_argument('--versions-rst-out', default=None, + help='also write just the per-version absolute RST snippet here') + args = parser.parse_args(argv) + + missing = [] # type: list[tuple[str, str]] + for entry in args.missing: + version, separator, reason = entry.partition('=') + # Rejected rather than guessed at: a bare version with no reason would render a + # column of gaps whose note said nothing, which is the outcome --missing exists + # to prevent. + if not separator or not version.strip() or not reason.strip(): + parser.error(f'--missing wants VERSION=REASON, not {entry!r}') + # Two reasons for one version have no coherent rendering -- the provenance block + # would list both and the table's notes only the last -- and there is no honest + # way to pick. `run.sh` writes one note file per version so it cannot happen from + # there; anyone driving report.py by hand is told rather than shown half of it. + if version.strip() in {existing for existing, _ in missing}: + parser.error(f'--missing was given twice for Python {version.strip()}') + missing.append((version.strip(), reason.strip())) + + documents = [] # type: list[dict[str, Any]] + for path in args.json: + with open(path, encoding='utf-8') as file: + documents.append(json.load(file)) + + rows = collect(documents) + print(render_text(rows, documents, args.image, args.emulated, missing)) + versions_snippet = render_versions_rst(rows, documents, missing, args.emulated) + snippet = render_rst(rows, documents, args.image, args.emulated, missing) + + print('-' * 78) + print('reStructuredText below, ready to paste into README.rst') + print('-' * 78) + print() + print(versions_snippet) + print(snippet) + + if args.versions_rst_out: + with open(args.versions_rst_out, 'w', encoding='utf-8') as file: + file.write(versions_snippet) + if args.rst_out: + with open(args.rst_out, 'w', encoding='utf-8') as file: + file.write(snippet) + return 0 + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/examples/benchmark/requirements-common.txt b/examples/benchmark/requirements-common.txt new file mode 100644 index 0000000000..d0050fa974 --- /dev/null +++ b/examples/benchmark/requirements-common.txt @@ -0,0 +1,30 @@ +# Everything present in *both* benchmark environments. +# +# Pinned exactly, because an unpinned harness produces numbers nobody can +# reproduce next month. The engine packages are the point of the pin; ``pcapkit`` +# itself is deliberately absent, since it is installed from the working tree and +# its "version" is the git revision the image was built from. +# +# `pcapkit`'s own runtime dependencies. Pinned as well: they are on the default +# engine's hot path, so a different `chardet` is a different baseline, and the +# baseline is what every ratio in the report divides by. +dictdumper==0.8.4.post6 +chardet==7.6.0 +aenum==3.1.17 +tbtrim==0.3.1 + +# Engines available in every environment. Only `pypcap` and `pcap_ct` have to be +# separated -- see requirements-pypcap.txt for why. +dpkt==1.9.8 +scapy==2.7.0 +pypcapfile==0.12.0 +pyshark==0.6 + +# `pyshark`'s dependencies. Pinned mostly for `lxml`, which is the only compiled +# package in the set and therefore the only one whose wheel availability can turn +# a build into a compile -- it publishes cp311 manylinux wheels for both aarch64 +# and x86_64, so neither architecture needs a compiler for it. +lxml==6.1.3 +termcolor==3.3.0 +appdirs==1.4.4 +packaging==26.3 diff --git a/examples/benchmark/requirements-pcap_ct.txt b/examples/benchmark/requirements-pcap_ct.txt new file mode 100644 index 0000000000..3110e3f02b --- /dev/null +++ b/examples/benchmark/requirements-pcap_ct.txt @@ -0,0 +1,21 @@ +# The `pcap_ct` environment: pcap-ct, plus everything shared. +# +# Separated from `pypcap` for the reason given in requirements-pypcap.txt: both +# provide the top-level `pcap` module and `pcap-ct` wins the import. +-r requirements-common.txt + +# Both of these are pre-release-only projects -- every release ever published is a +# beta -- so `pip install pcap-ct` with no version resolves to nothing at all. An +# *exact* pre-release pin is honoured without `--pre`, which is why the versions +# are spelled out here rather than left to a range plus a flag. +# +# Both publish `py3-none-any` wheels, so neither needs a compiler on any +# architecture. They do need a system libpcap at *run* time: `libpcap`'s published +# libpcap.cfg reads `LIBPCAP = None`, which sends its loader to +# ctypes.util.find_library("pcap") and so to the host's libpcap.so.1. The wheel +# does vendor its own copies (including an aarch64 one) under +# `libpcap/_platform/`, but only `LIBPCAP = tcpdump` selects them, and this harness +# deliberately does not -- measuring against the distribution's libpcap is closer +# to what a user gets. +pcap-ct==1.3.0b3 +libpcap==1.11.0b29 diff --git a/examples/benchmark/requirements-pypcap.txt b/examples/benchmark/requirements-pypcap.txt new file mode 100644 index 0000000000..fb0409cec3 --- /dev/null +++ b/examples/benchmark/requirements-pypcap.txt @@ -0,0 +1,16 @@ +# The `pypcap` environment: upstream PyPCAP, plus everything shared. +# +# It has to be its own environment. `pypcap` and `pcap-ct` both install a +# top-level `pcap` module, and with both present `import pcap` resolves to +# `pcap-ct`'s package directory -- upstream's extension module is shadowed and +# unreachable, so `engine='pypcap'` cannot run at all. Neither ordering nor +# install order changes that, which is why there are two environments rather than +# one. +-r requirements-common.txt + +# sdist only: no wheels for any version. Its `pcap.c` was pre-generated by Cython +# 0.29.32 (so Cython is *not* a build requirement, but the C API it was generated +# against caps this at Python 3.11), and its setup.py refuses to build without +# `pcap.h` and an unversioned `libpcap.so` -- hence build-essential and +# libpcap-dev in the Dockerfile. Nothing in it is architecture-specific. +pypcap==1.3.0 diff --git a/examples/benchmark/run.sh b/examples/benchmark/run.sh new file mode 100755 index 0000000000..9184595b69 --- /dev/null +++ b/examples/benchmark/run.sh @@ -0,0 +1,484 @@ +#!/usr/bin/env bash +# +# One command: build a benchmark image per Python version, run them all, print a +# paste-ready RST table for every version at once. +# +# examples/benchmark/run.sh +# +# Everything that decides what the numbers mean lives in the images -- the +# interpreters, the pinned engine versions, the capture -- so that the host running +# this script is irrelevant to the result. That is the whole reason for +# containerising a benchmark, and it is why nothing here reports the host's name, +# kernel or CPU. The one host fact that does matter is the architecture, because +# asking for an image that does not match it means running under emulation, and +# emulated timings are not comparable with native ones. This script detects that +# and says so, in its own output. +# +# The versions, their base image digests, and which of them can hold `pypcap` come +# from `python-images.txt`, which is a table rather than logic precisely so that a +# reader can check it. Nothing about the matrix is hard-coded here. +# +# **A version that cannot be built or run does not end the run.** The suite's +# standing promise is that a missing engine is an explained row rather than an +# absent one; a missing *column* is held to the same standard. The reason is written +# where the reporting pass will read it, so it reaches the table rather than +# scrolling past in this script's stderr. +# +# Written for bash 3.2, which is what macOS ships as /bin/bash: no associative +# arrays, and every array expansion guarded so `set -u` does not trip over an +# empty one. +set -euo pipefail + +# Assigned before being made readonly, rather than in one statement: `readonly x=$(...)` +# succeeds even when the command substitution fails, so `set -e` would not catch a +# `cd` into a directory that is not there and the script would carry on with a +# wrong path. +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +readonly SCRIPT_DIR +REPO_ROOT="$(cd "${SCRIPT_DIR}/../.." && pwd)" +readonly REPO_ROOT +readonly PINS_FILE="${SCRIPT_DIR}/python-images.txt" +readonly IMAGE_PREFIX="pcapkit-benchmark" + +ROUNDS=1000 +REPEATS=3 +ENGINES='default,dpkt,scapy,pypcap,pcap_ct,pypcapfile,pyshark' +CAPTURE='/src/examples/captures/in.pcap' +OUT_DIR="${SCRIPT_DIR}/out" +PLATFORM='' +PYTHONS='' +DO_BUILD=1 + +usage() { + cat <<'USAGE' +usage: run.sh [options] + +Builds one benchmark image per Python version and runs them all, printing a +summary and two paste-ready reStructuredText tables for README.rst: absolute +milliseconds per (engine, Python version), and engine ratios pooled across the +whole matrix. + +options: + --rounds N timed extractions per engine per pass (default 1000, matching + the methodology the published figures were taken with) + --repeats N passes over the whole engine set (default 3; more than one is + what makes the run-to-run spread observable, and the spread is + what says which gaps in the table are real) + --engines LIST comma-separated engines to measure; must include 'default', + which every ratio is taken against + --pythons LIST comma-separated Python versions, e.g. 3.11,3.12. Defaults to + every version python-images.txt marks 'default'; versions it + marks 'opt-in' are measured only when named here + --quick --rounds 50 --repeats 2, for checking the harness works + without waiting for a real measurement + --platform P docker platform, e.g. linux/arm64 or linux/amd64. Defaults to + this machine's architecture, which is the only setting that + produces comparable numbers + --out DIR where to write the JSON, the tables and the dependency locks + (default examples/benchmark/out) + --no-build reuse existing images instead of rebuilding + -h, --help this message + +A full matrix takes a long while: five versions, up to two virtualenvs each, and +almost all of the measuring time is pyshark, which spawns a tshark process per +extraction. Cut it down with --pythons for fewer columns, --engines to drop +pyshark, or --quick to check the harness rather than measure anything. + +exit status: + 0 every requested version was measured, and both tables were written + 3 the tables were written, but at least one version could not be measured + in full; its reason is in the tables and in /missing/ + 1 no table was produced + 2 the arguments were wrong; nothing was built or measured +USAGE +} + +while [ $# -gt 0 ]; do + case "$1" in + --rounds) ROUNDS="${2:?--rounds needs a value}"; shift 2 ;; + --repeats) REPEATS="${2:?--repeats needs a value}"; shift 2 ;; + --engines) ENGINES="${2:?--engines needs a value}"; shift 2 ;; + --pythons) PYTHONS="${2:?--pythons needs a value}"; shift 2 ;; + --platform) PLATFORM="${2:?--platform needs a value}"; shift 2 ;; + --out) OUT_DIR="${2:?--out needs a value}"; shift 2 ;; + --capture) CAPTURE="${2:?--capture needs a value}"; shift 2 ;; + --quick) ROUNDS=50; REPEATS=2; shift ;; + --no-build) DO_BUILD=0; shift ;; + -h|--help) usage; exit 0 ;; + *) echo "run.sh: unknown option '$1'" >&2; usage >&2; exit 2 ;; + esac +done + +if ! command -v docker >/dev/null 2>&1; then + echo 'run.sh: docker is not on PATH.' >&2 + echo ' On macOS, install Docker Desktop and make sure it is running.' >&2 + exit 1 +fi +if ! docker info >/dev/null 2>&1; then + echo 'run.sh: docker is installed but not responding.' >&2 + echo ' Start Docker Desktop and wait for it to report "running", then retry.' >&2 + exit 1 +fi +if [ ! -f "${PINS_FILE}" ]; then + echo "run.sh: ${PINS_FILE} is missing; it is the version matrix." >&2 + exit 1 +fi + +# The pins file read through awk rather than parsed in bash. Comment lines and the +# blank lines between them are skipped by requiring four fields, which every real +# row has and no comment line does. +# +# `$1 "" == version ""` rather than `$1 == version`, and this is not noise. awk +# compares two operands *numerically* when both look like numbers, and these look +# exactly like numbers: measured, `--pythons 3.1` matched the `3.10` row, because +# 3.1 == 3.10 is true of the numbers and false of the strings. It built an image +# tagged `py3.1` from 3.10's base and started measuring it -- a typo silently +# becoming a run, which is the class of error the rest of this harness exists to +# refuse. Concatenating an empty string makes both operands strings, so the +# comparison is the one intended. +pin_field() { + awk -v version="$1" -v field="$2" \ + 'substr($1, 1, 1) != "#" && NF >= 4 && $1 "" == version "" { print $field; exit }' \ + "${PINS_FILE}" +} + +# Versions measured when --pythons is not given. Reading the tier from the file +# keeps "which versions does a plain run cover" a property of the matrix rather +# than of this script, so adding a version is a one-line change in one place. +default_versions() { + awk 'substr($1, 1, 1) != "#" && NF >= 4 && $4 == "default" { printf "%s ", $1 }' \ + "${PINS_FILE}" +} + +known_versions() { + awk 'substr($1, 1, 1) != "#" && NF >= 4 { printf "%s ", $1 }' "${PINS_FILE}" +} + +if [ -z "${PYTHONS}" ]; then + VERSIONS="$(default_versions)" +else + VERSIONS="$(echo "${PYTHONS}" | tr ',' ' ')" +fi + +if [ -z "${VERSIONS// /}" ]; then + echo 'run.sh: no Python versions to measure.' >&2 + exit 2 +fi + +# An unknown version is a typo, and a typo is fatal here rather than reported as a +# missing column. A column that says "3.1 could not be measured" would be an +# honest-looking answer to a question nobody asked. +# +# Repeats are dropped in the same pass. `--pythons 3.11,3.11` would otherwise give both +# iterations the same container name, so the second `docker run` fails on the collision +# and the version gets written off as a gap -- a spurious failure reported for a version +# that in fact measured perfectly well the first time round. +WANTED='' +for version in ${VERSIONS}; do + if [ -z "$(pin_field "${version}" 1)" ]; then + echo "run.sh: '${version}' is not in ${PINS_FILE}." >&2 + echo " Known versions: $(known_versions)" >&2 + exit 2 + fi + case " ${WANTED} " in + *" ${version} "*) ;; + *) WANTED="${WANTED} ${version}" ;; + esac +done +VERSIONS="${WANTED}" + +# `default` is what every ratio in the report is taken against and the only engine +# present in every environment, so `collect` refuses to build a table without it. +# Caught here rather than there: without this the whole matrix builds and measures +# first, and the failure arrives as a traceback from inside the reporting container +# minutes -- or hours -- after the mistake was made. +case ",${ENGINES}," in + *,default,*) ;; + *) echo "run.sh: --engines must include 'default'; it is the baseline every ratio uses." >&2 + exit 2 ;; +esac + +# What this machine is, and hence what it can run without emulation. `uname -m` +# says arm64 on Apple Silicon and aarch64 on Linux ARM; both mean linux/arm64 to +# docker. +HOST_ARCH="$(uname -m)" +case "${HOST_ARCH}" in + arm64|aarch64) HOST_PLATFORM='linux/arm64' ;; + x86_64|amd64) HOST_PLATFORM='linux/amd64' ;; + *) HOST_PLATFORM='' ;; +esac + +if [ -z "${PLATFORM}" ]; then + if [ -z "${HOST_PLATFORM}" ]; then + echo "run.sh: cannot map '${HOST_ARCH}' onto a docker platform; pass --platform." >&2 + exit 1 + fi + PLATFORM="${HOST_PLATFORM}" +fi + +# Emulation is the one thing that silently invalidates the whole table, so it is +# detected here and threaded all the way into the report rather than left as a +# remark in this script's output that a reader of the table would never see. +EMULATED='' +if [ -n "${HOST_PLATFORM}" ] && [ "${PLATFORM}" != "${HOST_PLATFORM}" ]; then + EMULATED="Measured under emulation: a ${PLATFORM} image on a ${HOST_PLATFORM} host." + echo "run.sh: WARNING -- ${EMULATED}" >&2 + echo ' Emulated timings are not comparable with native ones, and the ratios are' >&2 + echo ' only as trustworthy as the emulator is uniform across the work each engine' >&2 + echo ' does. Prefer the native platform unless an engine cannot build on it.' >&2 +fi + +# Recorded in the report. `pcapkit` is installed from the working tree rather than +# from PyPI, so the git revision is its only meaningful version -- and a dirty tree +# has to say so, or the revision is a claim about code that was not measured. +REVISION='unknown' +if command -v git >/dev/null 2>&1 && git -C "${REPO_ROOT}" rev-parse --git-dir >/dev/null 2>&1; then + REVISION="$(git -C "${REPO_ROOT}" rev-parse --short HEAD)" + if ! git -C "${REPO_ROOT}" diff --quiet HEAD -- pcapkit; then + REVISION="${REVISION}-dirty" + fi +fi + +mkdir -p "${OUT_DIR}" "${OUT_DIR}/missing" + +# A previous run's output must not be inherited, and this is not tidiness. The +# reporting pass reports on every document it is handed, so a `3.11-pypcap.json` +# left over from a run on another `pcapkit` revision would be pooled into this run's +# ratios and would fill a column of this run's table -- and the provenance block, +# which reports one revision, would not show the disagreement. A stale note is the +# same problem in the other direction: a version that failed yesterday and builds +# today would still be reported as a missing column, citing a failure nobody could +# reproduce. And a stale table is what makes "no table was produced" below +# unfalsifiable. +# +# Removed by extension rather than with `rm -rf "${OUT_DIR}"`, because --out points +# wherever the caller says and deleting a directory this script did not create is +# not its business. +rm -f "${OUT_DIR}/missing/"*.txt "${OUT_DIR}"/*.json "${OUT_DIR}"/*.rst "${OUT_DIR}"/*.txt + +# Results come out with `docker cp` rather than through a bind mount. A mount puts +# the container's uid up against the host directory's ownership, which differs +# between Docker Desktop's file sharing and a plain Linux daemon; copying from a +# stopped container behaves the same everywhere. Containers are therefore not +# `--rm`, and the trap is what stops them leaking one per version. `-v` goes with +# the removal because /out is a declared VOLUME, so each container brings an +# anonymous volume that outlives it otherwise. +CONTAINERS='' +cleanup() { + for name in ${CONTAINERS}; do + docker rm --force --volumes "${name}" >/dev/null 2>&1 || true + done +} +trap cleanup EXIT INT TERM + +# Recorded rather than announced: this is the note the reporting pass turns into a +# bullet under the table, so the person reading the table months later sees the same +# reason as the person who watched the run. +# +# "a gap" rather than "not measured", because the two callers below that record a note +# *after* copying results out may well have collected an environment or two first, and +# the reporting pass tells those cases apart -- a column with figures in it and a note +# attached is partly measured, not unmeasured. +GAPS=0 +record_missing() { + local version="$1" + local reason="$2" + printf '%s\n' "${reason}" > "${OUT_DIR}/missing/${version}.txt" + GAPS=$((GAPS + 1)) + echo "run.sh: ${version} will be reported as a gap in the table -- ${reason}" >&2 +} + +REPORT_IMAGE='' + +for version in ${VERSIONS}; do + base_image="$(pin_field "${version}" 2)" + pypcap_flag="$(pin_field "${version}" 3)" + image_tag="${IMAGE_PREFIX}:py${version}" + + if [ "${pypcap_flag}" = 'pypcap' ]; then + with_pypcap=1 + else + with_pypcap=0 + fi + + echo >&2 + echo "==> Python ${version}: ${image_tag}" >&2 + + if [ "${DO_BUILD}" -eq 1 ]; then + echo "==> building for ${PLATFORM} (pcapkit ${REVISION}, base ${base_image})" >&2 + # The one place a build failure is caught rather than propagated. Five + # versions means five chances for a base image to have gone missing or a + # wheel to have stopped publishing for one interpreter, and none of those is + # a reason to abandon the four columns that would have worked. + if ! docker build \ + --platform "${PLATFORM}" \ + --file "${SCRIPT_DIR}/Dockerfile" \ + --tag "${image_tag}" \ + --build-arg "PYTHON_IMAGE=${base_image}" \ + --build-arg "WITH_PYPCAP=${with_pypcap}" \ + --build-arg "PCAPKIT_REVISION=${REVISION}" \ + "${REPO_ROOT}"; then + record_missing "${version}" \ + "the ${image_tag} image failed to build from ${base_image}; see the build log" + continue + fi + elif ! docker image inspect "${image_tag}" >/dev/null 2>&1; then + record_missing "${version}" \ + "--no-build was given and ${image_tag} does not exist, so nothing was measured" + continue + fi + + # The architecture of the image that exists is checked against the one asked for. + # Without this, a tag left over from a build for the other platform sends + # `docker run --platform` off to the registry for an image that was never + # published -- and the resulting "pull access denied" says nothing at all about + # the real problem. Only the os/arch pair is compared, so `linux/arm64/v8` and + # `linux/arm64` are not treated as a mismatch. + # Guarded, unlike every other command substitution in this script: an unguarded + # one here would take `set -e` and the whole matrix down with it, which is exactly + # the behaviour the per-version failure handling exists to prevent. Everywhere else + # a failing substitution really should be fatal; inside this loop nothing should be. + if ! image_platform="$(docker image inspect --format '{{.Os}}/{{.Architecture}}' "${image_tag}")"; then + record_missing "${version}" \ + "${image_tag} could not be inspected, so nothing was measured for this version" + continue + fi + wanted_platform="$(echo "${PLATFORM}" | cut -d/ -f1,2)" + if [ "${image_platform}" != "${wanted_platform}" ]; then + record_missing "${version}" \ + "${image_tag} is a ${image_platform} image but ${wanted_platform} was requested; rebuild it" + continue + fi + + # The image ID is the reference the report quotes. A locally built image has no + # registry digest to name, and the ID is a digest of its config, so it identifies + # the exact image these numbers came from -- which is what the reader needs. + if ! image_id="$(docker image inspect --format '{{.Id}}' "${image_tag}")"; then + record_missing "${version}" \ + "${image_tag} could not be inspected, so nothing was measured for this version" + continue + fi + + if [ -z "${REPORT_IMAGE}" ]; then + # Whichever version's image is usable first does the reporting. report.py is + # standard library only and its arithmetic does not depend on the interpreter, + # so this is a choice of "one that exists" rather than a choice that matters -- + # but it is made deterministically all the same, so two runs of the same matrix + # report under the same interpreter. Recorded before the measuring run rather + # than after it, because a version whose *run* failed still leaves an image + # perfectly capable of reporting on whatever the others managed. + REPORT_IMAGE="${image_tag}" + fi + + container="${IMAGE_PREFIX}-${version}-$$" + CONTAINERS="${CONTAINERS} ${container}" + + echo "==> running ${ROUNDS} rounds x ${REPEATS} passes" >&2 + if ! docker run \ + --name "${container}" \ + --platform "${PLATFORM}" \ + --env "BENCH_ROUNDS=${ROUNDS}" \ + --env "BENCH_REPEATS=${REPEATS}" \ + --env "BENCH_ENGINES=${ENGINES}" \ + --env "BENCH_CAPTURE=${CAPTURE}" \ + --env "BENCH_IMAGE=${image_tag} (${image_id})" \ + "${image_tag}"; then + # A run that died partway may still have written a document or two, so its + # output is collected before the version is written off -- an environment + # that finished is a column cell that can be reported. + docker cp "${container}:/out/." "${OUT_DIR}/" >/dev/null 2>&1 || true + record_missing "${version}" \ + "measuring in ${image_tag} exited non-zero; any environment that finished first is still reported" + continue + fi + + if ! docker cp "${container}:/out/." "${OUT_DIR}/" >/dev/null 2>&1; then + record_missing "${version}" \ + "the measurements could not be copied out of ${container}" + continue + fi +done + +# Whether there is anything to report is decided by counting the documents that +# landed, not by counting the versions that finished cleanly. A container that died +# after writing its first environment's JSON still contributed a measurement, and +# refusing to report it would throw away real work while a note explaining the +# failure sat next to it. +DOCUMENTS=0 +for path in "${OUT_DIR}"/*.json; do + [ -e "${path}" ] || continue + DOCUMENTS=$((DOCUMENTS + 1)) +done + +echo >&2 +if [ "${DOCUMENTS}" -eq 0 ] || [ -z "${REPORT_IMAGE}" ]; then + echo 'run.sh: no environment produced a measurement; there is nothing to report.' >&2 + echo ' The reasons are in:' >&2 + ls -1 "${OUT_DIR}/missing" >&2 2>/dev/null || true + exit 1 +fi + +# The reporting pass. It runs inside an image rather than on the host because the +# host is not required to have a Python -- docker is the only thing this script +# insists on -- and because a pinned interpreter is one less thing that can differ +# between two runs of the same matrix. +# +# Inputs go in through /in and results come back out of /out, both with `docker cp`, +# for the same reason the measuring containers avoid a bind mount: ownership of a +# host directory behaves differently under Docker Desktop's file sharing than under +# a plain Linux daemon, and copying to and from a stopped container behaves the same +# everywhere. +REPORTER="${IMAGE_PREFIX}-report-$$" +CONTAINERS="${CONTAINERS} ${REPORTER}" + +echo "==> reporting on ${DOCUMENTS} environment(s) in ${REPORT_IMAGE}" >&2 +docker create \ + --name "${REPORTER}" \ + --platform "${PLATFORM}" \ + --env 'BENCH_MODE=report' \ + --env "BENCH_EMULATED=${EMULATED}" \ + "${REPORT_IMAGE}" >/dev/null + +status=0 +if ! docker cp "${OUT_DIR}/." "${REPORTER}:/in/" >/dev/null 2>&1; then + echo "run.sh: could not hand the measurements to the reporting container." >&2 + exit 1 +fi +# `|| status=$?` rather than a bare invocation: the report's output is still worth +# collecting when it fails partway, and `set -e` would abandon it. +docker start --attach "${REPORTER}" || status=$? +docker cp "${REPORTER}:/out/." "${OUT_DIR}/" >/dev/null 2>&1 || status=1 + +# What actually landed is checked rather than announced. `docker cp` of an empty +# directory succeeds, so an earlier version of this reported "wrote ... table.rst" +# after a run that crashed before writing anything -- which is worse than saying +# nothing, because it sends the reader looking for a file that is not there. +echo >&2 +if [ -f "${OUT_DIR}/table-versions.rst" ] && [ -f "${OUT_DIR}/table.rst" ]; then + echo "==> wrote ${OUT_DIR}/table-versions.rst -- absolute milliseconds per engine" >&2 + echo " and Python version, ready to paste into README.rst's Test Results," >&2 + echo " and ${OUT_DIR}/table.rst -- the machine-independent ratio view," >&2 + echo ' alongside the raw JSON and the per-environment locks' >&2 + + # A run that lost a column produced everything it could and is still not a run that + # went as asked, so it says so in the exit status as well as in the tables. Nothing + # was aborted -- the requirement is that a failing version must not end the run, not + # that it must be indistinguishable from success -- and an exit code is the only part + # of this a scheduled invocation reads. A distinct code rather than 1, so "some + # columns are missing, the tables are written" can be told from "there is no table". + if [ "${GAPS}" -gt 0 ] && [ "${status}" -eq 0 ]; then + echo >&2 + echo "run.sh: ${GAPS} of the requested Python version(s) could not be measured in" >&2 + echo ' full; the tables were still written, with the reasons in them and in' >&2 + echo " ${OUT_DIR}/missing/. Exiting 3 to say so." >&2 + status=3 + fi +else + echo "run.sh: no table was produced; ${OUT_DIR} holds whatever the run got to." >&2 + ls -1 "${OUT_DIR}" >&2 2>/dev/null || true + if [ "${status}" -eq 0 ]; then + status=1 + fi +fi + +exit "${status}" diff --git a/examples/benchmark/test_harness.py b/examples/benchmark/test_harness.py new file mode 100644 index 0000000000..f0ae156b75 --- /dev/null +++ b/examples/benchmark/test_harness.py @@ -0,0 +1,1534 @@ +# -*- coding: utf-8 -*- +"""Self-tests for the benchmark harness's arithmetic and its emitted markup. + +A benchmark's output is not self-checking. A ratio computed against the wrong +baseline, a row sorted into the wrong place, or a table that renders as an error +block on GitHub all look exactly like a successful run, so the parts that can be +tested without a stopwatch are tested here: + +* the ratio arithmetic, including that it cancels machine drift, which is the + entire justification for reporting ratios at all; +* the stitching of several environments into one table, which is what makes the + mutually exclusive ``pypcap`` and ``pcap_ct`` reportable together; +* the per-version grid, where the arithmetic is the opposite -- absolute + milliseconds, pooled across the virtualenvs of one interpreter and never divided + across two of them -- and where a version that could not be measured has to come + out as an explained column of ``--`` rather than as an absent column; +* the overlap marking, which is what stops the report claiming a gap it did not + measure; +* that the emitted reStructuredText parses under **plain docutils** with no + errors and uses no Sphinx-only roles -- the README is rendered by docutils on + GitHub, where ``:mod:`` and friends come out as visible errors. + +Run with ``python -m pytest examples/benchmark/test_harness.py``. Nothing here +needs docker, and only the driver-assertion tests need ``pcapkit`` importable. + +""" + +import json +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +import report # noqa: E402 pylint: disable=wrong-import-position + +#: Roles Sphinx defines and docutils does not. Their appearance in the emitted +#: snippet is the failure this list exists to catch: docutils renders an unknown +#: role as an error block, so a table carrying one is visibly broken on GitHub +#: while looking perfectly fine in the project's own HTML docs. +SPHINX_ONLY_ROLES = (':mod:', ':func:', ':class:', ':meth:', ':attr:', ':data:', + ':manpage:', ':envvar:', ':program:', ':doc:', ':exc:', ':obj:', + ':term:', ':ref:', ':file:', ':c:func:') + + +def document(environment, measurements, unmeasured=None, repeats=2, packages=None, + failures=None, discarded=None, python='3.11.14', image=None, + tshark='TShark (Wireshark) 4.0.17'): + """Build a :mod:`benchmark`-shaped document from plain numbers. + + Args: + environment: Environment label. + measurements: Mapping of engine name to a list of milliseconds-per-packet + figures, one per repeat. + unmeasured: Mapping of engine name to the reason it was not measured. + repeats: How many repeats the figures represent. + packages: Resolved package versions, with :data:`None` for a package this + environment does not have. Defaults to the real shape of the ``pypcap`` + environment, i.e. ``pypcap`` present and ``pcap-ct`` absent. + failures: Mapping of engine name to passes that failed outright. + discarded: Mapping of engine name to how many extractions were discarded + per pass. + python: Interpreter version, which is what decides the column a document + lands in -- see :func:`report.python_series`. + image: Reference of the image this was measured in. The matrix runs one per + Python version, so this travels with the document rather than being + passed to the report alongside it. + tshark: Resolved ``tshark`` banner, or :data:`None` where the binary is + absent. + + Returns: + A document :mod:`report` can consume. + + """ + if packages is None: + packages = {'dpkt': '1.9.8', 'pypcap': '1.3.0', 'pcap-ct': None} + failures = failures or {} + discarded = discarded or {} + results = [] + for engine, values in measurements.items(): + results.append({ + 'engine': engine, + 'status': 'measured', + 'reason': None, + 'driver': engine.upper(), + 'packets': 6, + 'repeats': [{'repeat': index, 'ms_per_packet': value, + 'mean_ns_per_extraction': value * 6 * 1e6, + 'timed_samples': 999, + 'discarded': ['TSharkCrashException: boom'] * discarded.get(engine, 0)} + for index, value in enumerate(values)], + 'failures': failures.get(engine, []), + }) + for engine, reason in (unmeasured or {}).items(): + results.append({ + 'engine': engine, 'status': 'unmeasured', 'reason': reason, + 'driver': None, 'packets': None, 'repeats': [], + 'failures': failures.get(engine, []), + }) + return { + 'schema': 1, + 'environment': environment, + 'pcapkit_revision': 'abc1234', + 'image': image, + 'base_image': 'python:3.11-slim-bookworm@sha256:5282', + 'python': python, + 'implementation': 'CPython', + 'machine': 'aarch64', + 'capture': {'name': 'in.pcap', 'bytes': 605, 'sha256': 'a' * 64}, + 'rounds': 1000, + 'repeats': repeats, + 'packages': packages, + 'tshark': tshark, + 'libpcap': 'libpcap version 1.10.3', + 'results': results, + } + + +class TestRatios: + """The arithmetic that turns milliseconds into comparable numbers.""" + + def test_ratio_is_taken_against_the_same_environment(self): + """An engine's ratio divides by its own environment's baseline.""" + docs = [document('pypcap', {'default': [0.2, 0.2], 'dpkt': [0.02, 0.02]})] + rows = {row.engine: row for row in report.collect(docs)} + assert rows['default'].median == pytest.approx(1.0) + assert rows['dpkt'].median == pytest.approx(0.1) + + def test_machine_drift_cancels(self): + """A pass on a machine twice as slow yields the same ratio. + + This is the property that justifies the whole design. The second repeat + below is uniformly 2x slower -- every engine and the baseline alike -- and + must not move the reported ratio at all. + + """ + docs = [document('pypcap', {'default': [0.2, 0.4], 'dpkt': [0.02, 0.04]})] + rows = {row.engine: row for row in report.collect(docs)} + assert rows['dpkt'].ratios == pytest.approx([0.1, 0.1]) + assert rows['dpkt'].low == pytest.approx(rows['dpkt'].high) + assert rows['dpkt'].spread == pytest.approx(0.0) + + def test_ratio_pairs_repeats_not_means(self): + """Repeats are paired one to one, never averaged first. + + Averaging the baseline before dividing would let a slow pass on one engine + be cancelled by a fast pass on another, which is exactly the error the + per-repeat pairing exists to avoid. + + """ + # Baseline doubles on the second pass; dpkt does not. The honest answer is + # two very different ratios, not one tidy average. + docs = [document('pypcap', {'default': [0.2, 0.4], 'dpkt': [0.02, 0.02]})] + rows = {row.engine: row for row in report.collect(docs)} + assert rows['dpkt'].ratios == pytest.approx([0.1, 0.05]) + + def test_repeat_without_a_baseline_is_dropped(self): + """A repeat with no baseline contributes nothing rather than an estimate.""" + docs = [document('pypcap', {'default': [0.2], 'dpkt': [0.02, 0.02]})] + rows = {row.engine: row for row in report.collect(docs)} + assert rows['dpkt'].ratios == pytest.approx([0.1]) + + def test_a_zero_baseline_is_dropped_rather_than_divided_by(self): + """A baseline of ``0.0`` drops the pair instead of raising. + + A zero baseline is a broken run, not an absent one, but it is just as + undividable -- so the sample goes the same way a missing baseline's does, + and the engine is not credited with a ratio it never had. + + """ + docs = [document('pypcap', {'default': [0.0, 0.2], 'dpkt': [0.02, 0.02]})] + rows = {row.engine: row for row in report.collect(docs)} + assert rows['dpkt'].ratios == pytest.approx([0.1]) + + def test_an_environment_contributing_no_ratio_is_not_counted(self): + """An environment only counts once a ratio has actually survived it. + + The engine ran, so its repeats are non-empty, but every one of them lost + its baseline -- so the environment contributed nothing and must not reach + the cross-environment check, which would then advertise an agreement it + has only one side of. + + """ + docs = [ + document('pypcap', {'default': [0.2, 0.2], 'dpkt': [0.02, 0.02]}), + # ``dpkt`` runs a second pass here that the baseline never reaches. + document('pcap-ct', {'default': [0.2], 'dpkt': [0.02, 0.02]}, + packages={'dpkt': '1.9.8', 'pypcap': None, 'pcap-ct': '1.3.0b3'}), + ] + rows = {row.engine: row for row in report.collect(docs)} + assert rows['dpkt'].environments == ['pypcap', 'pcap-ct'] + + # And with the baseline missing outright, the environment drops away. + docs[1]['results'] = [entry for entry in docs[1]['results'] + if entry['engine'] != 'dpkt'] + docs[1]['results'].append({ + 'engine': 'dpkt', 'status': 'measured', 'reason': None, + 'driver': 'DPKT', 'packets': 6, + 'repeats': [{'repeat': 7, 'ms_per_packet': 0.02, + 'mean_ns_per_extraction': 0.12e6, 'timed_samples': 999, + 'discarded': []}], + 'failures': [], + }) + rows = {row.engine: row for row in report.collect(docs)} + assert rows['dpkt'].environments == ['pypcap'] + + def test_environment_without_a_baseline_is_fatal(self): + """Nothing in an environment is reportable without its baseline.""" + docs = [document('pypcap', {'dpkt': [0.02, 0.02]})] + with pytest.raises(ValueError, match="no 'default' measurement"): + report.collect(docs) + + def test_no_documents_is_fatal(self): + """An empty run is an error, not an empty table.""" + with pytest.raises(ValueError, match='no benchmark documents'): + report.collect([]) + + +class TestStitching: + """Joining the mutually exclusive environments on the shared baseline.""" + + def _two_environments(self): + """Two environments, each with the ``pcap`` provider the other cannot have.""" + return [ + document('pypcap', + {'default': [0.20, 0.20], 'dpkt': [0.020, 0.020], 'pypcap': [0.010, 0.010]}, + {'pcap_ct': 'the installed `pcap` module is pypcap, not pcap-ct'}), + # Deliberately a slower machine reading, to prove the join normalises. + document('pcap_ct', + {'default': [0.40, 0.40], 'dpkt': [0.040, 0.040], 'pcap_ct': [0.060, 0.060]}, + {'pypcap': 'the installed `pcap` module is pcap-ct, not pypcap'}), + ] + + def test_exclusive_engines_both_appear(self): + """Both ``pypcap`` and ``pcap_ct`` land in one table.""" + rows = {row.engine: row for row in report.collect(self._two_environments())} + assert rows['pypcap'].measured + assert rows['pcap_ct'].measured + assert rows['pypcap'].median == pytest.approx(0.05) + assert rows['pcap_ct'].median == pytest.approx(0.15) + + def test_shared_engine_pools_both_environments(self): + """An engine in both environments contributes a ratio from each.""" + rows = {row.engine: row for row in report.collect(self._two_environments())} + assert len(rows['dpkt'].ratios) == 4 + assert rows['dpkt'].environments == ['pypcap', 'pcap_ct'] + # 0.020/0.20 and 0.040/0.40 are the same ratio on machines 2x apart. + assert rows['dpkt'].ratios == pytest.approx([0.1, 0.1, 0.1, 0.1]) + + def test_measured_somewhere_beats_unmeasured_elsewhere(self): + """Being unavailable in one environment does not blank the row.""" + rows = {row.engine: row for row in report.collect(self._two_environments())} + assert rows['pypcap'].measured + # The reason from the other environment is still recorded, because the + # mutual exclusion is a fact worth keeping even once the row is filled in. + assert 'pcap-ct, not pypcap' in rows['pypcap'].reasons['pcap_ct'] + + def test_unmeasured_everywhere_keeps_its_reason(self): + """A row nothing could measure says why, rather than showing a zero.""" + docs = [ + document('pypcap', {'default': [0.2, 0.2]}, {'pyshark': 'no tshark binary'}), + document('pcap_ct', {'default': [0.2, 0.2]}, {'pyshark': 'no tshark binary'}), + ] + rows = {row.engine: row for row in report.collect(docs)} + assert not rows['pyshark'].measured + assert rows['pyshark'].reason == 'no tshark binary' + + def test_disagreeing_reasons_are_both_kept(self): + """Two environments unavailable for different reasons report both.""" + docs = [ + document('pypcap', {'default': [0.2, 0.2]}, {'pyshark': 'no tshark binary'}), + document('pcap_ct', {'default': [0.2, 0.2]}, {'pyshark': 'python too new'}), + ] + rows = {row.engine: row for row in report.collect(docs)} + assert 'in pcap_ct, python too new' in rows['pyshark'].reason + assert 'in pypcap, no tshark binary' in rows['pyshark'].reason + + def test_ordering_is_fastest_first_unmeasured_last(self): + """The table reads top to bottom as fastest to slowest, then the gaps.""" + docs = [document('pypcap', + {'default': [0.2, 0.2], 'dpkt': [0.02, 0.02], 'pyshark': [24.0, 24.0]}, + {'pypcapfile': 'python too new'})] + assert [row.engine for row in report.collect(docs)] == \ + ['dpkt', 'default', 'pyshark', 'pypcapfile'] + + +class TestOverlap: + """Marking the gaps this run does not actually establish.""" + + def test_overlapping_ranges_are_marked(self): + """Two engines whose observed ranges cross are not separated.""" + docs = [document('pypcap', {'default': [0.20, 0.20], + 'dpkt': [0.020, 0.030], + 'scapy': [0.025, 0.035]})] + marked = report.overlapping(report.collect(docs)) + assert marked == {'dpkt', 'scapy'} + + def test_separated_ranges_are_not_marked(self): + """A gap wider than the noise is reported as a gap.""" + docs = [document('pypcap', {'default': [0.20, 0.20], + 'dpkt': [0.020, 0.021], + 'scapy': [0.090, 0.091]})] + assert report.overlapping(report.collect(docs)) == set() + + def test_a_single_pass_establishes_nothing(self): + """One pass gives no range, so no row can be separated from another.""" + docs = [document('pypcap', {'default': [0.20], 'dpkt': [0.020]}, repeats=1)] + marked = report.overlapping(report.collect(docs)) + assert marked == {'default', 'dpkt'} + + def test_a_wide_row_straddles_several_narrow_ones(self): + """Overlap is checked pairwise across the table, not only between neighbours.""" + docs = [document('pypcap', {'default': [0.20, 0.20], + 'dpkt': [0.010, 0.010], + 'scapy': [0.005, 0.100], + 'pypcapfile': [0.050, 0.050]})] + marked = report.overlapping(report.collect(docs)) + assert marked == {'dpkt', 'scapy', 'pypcapfile'} + + +class TestFormatting: + """Number formatting, which spans three orders of magnitude in one table.""" + + @pytest.mark.parametrize(('value', 'expected'), [ + (1.0, '1.00'), + (0.052134, '0.0521'), + (0.4587, '0.459'), + (123.456, '123'), + # Past three digits the integer part is kept in full rather than rounded to + # significance: '1234' misleads nobody, and '1.23e+03' in a prose table does. + # The .5 rounds to even, which is what Python's formatting does. + (1234.5, '1234'), + (1235.5, '1236'), + ]) + def test_significant_figures(self, value, expected): + """Three significant figures, so neither end of the table is rounded away.""" + assert report._significant(value) == expected # pylint: disable=protected-access + + def test_zero_is_not_an_exponent(self): + """Zero formats as zero rather than as scientific notation.""" + assert report._significant(0.0) == '0' # pylint: disable=protected-access + + def test_tiny_values_avoid_exponent_notation(self): + """A pathological measurement must not put ``5e-09`` into readable prose.""" + assert 'e' not in report._significant(5e-9) # pylint: disable=protected-access + + def test_a_tight_range_does_not_read_as_a_range_of_one_value(self): + """Precision rises until the ends differ, rather than printing ``x-x``. + + Measured: ``pypcap`` came back as ``0.183-0.183`` at three significant + figures, which looks like a formatting bug and conveys nothing. + + """ + assert report._range(0.18321, 0.18349) == '0.1832-0.1835' # pylint: disable=protected-access + assert report._range(0.0612, 0.0620) == '0.0612-0.0620' # pylint: disable=protected-access + + def test_an_identical_range_shows_one_value(self): + """When the ends really are equal, one value -- never a fake interval.""" + assert report._range(1.0, 1.0) == '1.00' # pylint: disable=protected-access + + +class TestRenderedMarkup: + """What actually gets pasted into the README.""" + + def _snippet(self, emulated=None): + """Render a full snippet from a two-environment run.""" + docs = [ + document('pypcap', + {'default': [0.200, 0.204], 'dpkt': [0.0104, 0.0106], + 'scapy': [0.0917, 0.0921], 'pypcap': [0.0061, 0.0063], + 'pyshark': [24.7, 24.9]}, + {'pcap_ct': 'the installed `pcap` module is pypcap, not pcap-ct'}), + document('pcap_ct', + {'default': [0.201, 0.203], 'dpkt': [0.0105, 0.0107], + 'scapy': [0.0918, 0.0925], 'pcap_ct': [0.0078, 0.0081], + 'pyshark': [24.6, 25.0]}, + {'pypcap': 'the installed `pcap` module is pcap-ct, not pypcap', + 'pypcapfile': 'pypcapfile does not support Python 3.12'}, + # The mutual exclusion, as the package table actually sees it: + # each environment has exactly one `pcap` provider. + packages={'dpkt': '1.9.8', 'pypcap': None, 'pcap-ct': '1.3.0b3'}), + ] + rows = report.collect(docs) + return report.render_rst(rows, docs, image='pcapkit-benchmark:local (sha256:dead)', + emulated=emulated), docs, rows + + def test_no_sphinx_only_roles(self): + """Only literals, because docutils renders unknown roles as errors.""" + snippet, _, _ = self._snippet() + for role in SPHINX_ONLY_ROLES: + assert role not in snippet, f'{role} is Sphinx-only and breaks on GitHub' + + def test_parses_under_plain_docutils(self): + """The snippet renders cleanly with the parser GitHub actually uses.""" + docutils_core = pytest.importorskip('docutils.core') + from docutils.utils import SystemMessage # pylint: disable=import-outside-toplevel + + snippet, _, _ = self._snippet() + messages = [] + try: + docutils_core.publish_doctree( + snippet, + settings_overrides={ + # halt_level 2 turns a warning into an exception, which is the + # only way to make a malformed table fail a test rather than + # quietly render as a literal block. + 'halt_level': 2, 'report_level': 2, 'warning_stream': messages, + 'input_encoding': 'unicode', 'output_encoding': 'unicode', + }, + ) + except SystemMessage as exc: # pragma: no cover - only on a real failure + pytest.fail(f'docutils rejected the snippet: {exc}\n\n{snippet}') + assert not messages, f'docutils warned: {messages}\n\n{snippet}' + + def test_table_has_a_row_per_engine(self): + """Every engine appears, measured or not -- no row is dropped.""" + snippet, _, rows = self._snippet() + for row in rows: + assert f'``{row.label}``' in snippet + + def test_unmeasured_row_carries_its_reason(self): + """A gap is an explained gap, not a blank and not a zero.""" + snippet, _, _ = self._snippet() + assert '*not measured*' in snippet + assert 'pypcapfile does not support Python 3.12' in snippet + + def test_baseline_absolute_is_a_footnote_not_a_column(self): + """One absolute number survives, and it is the baseline's own.""" + snippet, docs, _ = self._snippet() + assert 'ms per packet' in snippet + assert report.baseline_absolutes(docs) + # The table itself must not carry absolute times: they describe the machine + # at least as much as the engine, and a column of them invites comparison + # across machines, which is exactly what is not valid. + table = snippet.split('Test Results')[1] + assert 'ms per packet' in table # the footnote lives below the table + assert table.count('ms per packet') == 1 + + def test_spread_is_reported(self): + """The reader is told how much the run wobbled.""" + snippet, _, _ = self._snippet() + assert 'Run-to-run spread' in snippet + + def test_overlap_legend_only_appears_when_a_row_carries_it(self): + """No legend for a marker that is not in the table. + + Explaining a symbol nothing is tagged with reads as though the run failed to + separate rows it in fact separated cleanly -- which is the opposite of the + marker's purpose. + + """ + clean = [document('pypcap', {'default': [0.200, 0.201], + 'dpkt': [0.0104, 0.0105], + 'scapy': [0.0917, 0.0918]})] + snippet = report.render_rst(report.collect(clean), clean) + assert report.OVERLAP_MARK not in snippet + assert 'does not separate these rows' not in snippet + # ...but the spread is still reported, because it is what makes the gaps + # above interpretable at all. + assert 'Run-to-run spread' in snippet + + muddy = [document('pypcap', {'default': [0.200, 0.201], + 'dpkt': [0.020, 0.030], + 'scapy': [0.025, 0.035]})] + snippet = report.render_rst(report.collect(muddy), muddy) + assert report.OVERLAP_MARK in snippet + assert 'does not separate these rows' in snippet + + def test_provenance_is_present_and_host_free(self): + """Self-documenting, without identifying the machine it ran on.""" + snippet, _, _ = self._snippet() + for expected in ('Image', 'Python', 'Architecture', 'Capture', 'Iterations', + 'Passes', 'CPython 3.11.14', 'in.pcap', '1000', 'abc1234'): + assert expected in snippet + # The containerised design is what makes the host irrelevant; printing host + # details would undo it, and would leak machine identity into a public README. + import platform as host_platform # pylint: disable=import-outside-toplevel + for leak in (host_platform.node(), host_platform.release(), host_platform.version()): + if leak: + assert leak not in snippet + + def test_package_versions_are_reported(self): + """Every engine package's resolved version, per environment.""" + snippet, _, _ = self._snippet() + assert '``dpkt``' in snippet + assert '1.9.8' in snippet + # A package absent from one environment is stated, not left blank: it is the + # reason two environments exist. + assert '*absent*' in snippet + + def test_emulation_warning_reaches_the_snippet(self): + """Emulation is recorded in the table, not only in the script's stderr.""" + snippet, _, _ = self._snippet(emulated='Measured under emulation: linux/amd64 on arm64.') + assert 'emulation' in snippet.lower() + assert 'not comparable with native' in snippet + + def test_native_run_says_nothing_about_emulation(self): + """No scary caveat on a run that did not need one.""" + snippet, _, _ = self._snippet() + assert 'emulation' not in snippet.lower() + + +class TestTextReport: + """The operator-facing summary, which carries more than the table does.""" + + def test_cross_environment_check_is_shown(self): + """Shared engines are reported per environment, so a bad join is visible.""" + docs = [ + document('pypcap', {'default': [0.20, 0.20], 'dpkt': [0.020, 0.020]}), + document('pcap_ct', {'default': [0.40, 0.40], 'dpkt': [0.040, 0.040]}), + ] + text = report.render_text(report.collect(docs), docs) + assert 'cross-environment check' in text + assert 'pypcap=0.1' in text + assert 'pcap_ct=0.1' in text + + def test_absolute_times_are_in_the_text_report(self): + """Milliseconds stay available to the operator, out of the published table.""" + docs = [document('pypcap', {'default': [0.20, 0.20], 'dpkt': [0.020, 0.020]})] + text = report.render_text(report.collect(docs), docs) + assert 'ms/packet' in text + assert '0.02' in text + + +class TestDriverAssertion: + """The harness's single most important correctness property.""" + + def test_expected_drivers_come_from_the_registry(self): + """Derived, not hand-written, so a renamed engine cannot slip past.""" + benchmark = pytest.importorskip('benchmark') + assert benchmark.expected_drivers('dpkt') == ('DPKT',) + assert benchmark.expected_drivers('scapy') == ('Scapy',) + assert benchmark.expected_drivers('pypcap') == ('PyPCAP',) + assert benchmark.expected_drivers('pcap_ct') == ('PCAP_CT',) + + def test_default_accepts_either_builtin_parser(self): + """``default`` picks its parser from the magic number, so either is correct.""" + benchmark = pytest.importorskip('benchmark') + assert set(benchmark.expected_drivers('default')) == {'PCAP', 'PCAP-NG'} + assert benchmark.expected_drivers('pcapkit') == benchmark.expected_drivers('default') + + def test_an_unknown_engine_is_fatal(self): + """A typo must not silently benchmark the default engine under another name.""" + benchmark = pytest.importorskip('benchmark') + with pytest.raises(KeyError): + benchmark.expected_drivers('not-an-engine') + + def test_a_wrong_driver_refuses_to_report(self, monkeypatch, tmp_path): + """A measurement whose engine fell back is discarded, not published. + + The fallback only warns, so nothing else in the stack objects. This is the + check that turns "the extraction succeeded" into "the right engine ran". + + """ + benchmark = pytest.importorskip('benchmark') + + class _FellBack: + """Stands in for pcapkit's own parser answering for another engine.""" + __engine_name__ = 'PCAP' + + class _Extraction: + """Minimal stand-in for an Extractor.""" + engine = _FellBack() + length = 6 + + monkeypatch.setattr(benchmark, '_extract', lambda engine, capture: _Extraction()) + with pytest.raises(RuntimeError, match='refusing to report'): + benchmark.measure('dpkt', str(tmp_path / 'unused.pcap'), rounds=3) + + def test_zero_packets_refuses_to_divide(self, monkeypatch, tmp_path): + """A capture that yielded nothing is an error, not a division by zero.""" + benchmark = pytest.importorskip('benchmark') + + class _Empty: + """An extraction that parsed no frames at all.""" + engine = type('_D', (), {'__engine_name__': 'DPKT'})() + length = 0 + + monkeypatch.setattr(benchmark, '_extract', lambda engine, capture: _Empty()) + with pytest.raises(RuntimeError, match='nothing to divide by'): + benchmark.measure('dpkt', str(tmp_path / 'unused.pcap'), rounds=3) + + +class TestFlakyEngines: + """One flaky engine must not cost the others their rows. + + Not hypothetical. On a real 1,000-round run, ``tshark`` crashed on the third + pass -- ``TSharkCrashException ... retcode: 255``, after roughly two thousand + process spawns -- and the run produced no table at all. Six working engines went + unreported because the seventh hiccupped once. + + """ + + def test_a_lost_pass_still_reports_the_engine(self): + """A row built from two passes instead of three is still a row.""" + docs = [document('pypcap', + {'default': [0.20, 0.20, 0.20], 'pyshark': [24.0, 24.4]}, + repeats=3, + failures={'pyshark': ['pass 3: TSharkCrashException: retcode 255']})] + rows = {row.engine: row for row in report.collect(docs)} + assert rows['pyshark'].measured + assert len(rows['pyshark'].ratios) == 2 + assert rows['pyshark'].partial + + def test_a_lost_pass_is_marked_and_explained(self): + """The reader is told what was lost, not left to count passes.""" + docs = [document('pypcap', + {'default': [0.20, 0.20, 0.20], 'pyshark': [24.0, 24.4]}, + repeats=3, + failures={'pyshark': ['pass 3: TSharkCrashException: retcode 255']})] + snippet = report.render_rst(report.collect(docs), docs) + assert report.PARTIAL_MARK in snippet + assert 'TSharkCrashException' in snippet + assert 'in pypcap, pass 3' in snippet + + def test_discarded_extractions_are_counted(self): + """Discarded extractions are reported, not silently dropped.""" + docs = [document('pypcap', {'default': [0.20, 0.20], 'pyshark': [24.0, 24.4]}, + discarded={'pyshark': 2})] + rows = {row.engine: row for row in report.collect(docs)} + assert rows['pyshark'].discarded == 4 # two per pass, two passes + snippet = report.render_rst(report.collect(docs), docs) + assert '4 individual extraction(s) failed and were discarded' in snippet + + def test_a_clean_row_is_not_marked_partial(self): + """No marker on a row that lost nothing.""" + docs = [document('pypcap', {'default': [0.20, 0.20], 'dpkt': [0.020, 0.020]})] + rows = {row.engine: row for row in report.collect(docs)} + assert not rows['dpkt'].partial + assert report.PARTIAL_MARK not in report.render_rst(report.collect(docs), docs) + + def test_every_pass_failing_makes_the_row_unmeasured(self): + """An engine that never produced a number reports why, not a zero.""" + docs = [document('pypcap', {'default': [0.20, 0.20]}, + {'pyshark': 'every timed pass failed: pass 1: TSharkCrashException'})] + rows = {row.engine: row for row in report.collect(docs)} + assert not rows['pyshark'].measured + assert 'TSharkCrashException' in rows['pyshark'].reason + + def test_measure_tolerates_a_bounded_number_of_failures(self, monkeypatch, tmp_path): + """A flaky extraction is discarded and the pass carries on.""" + benchmark = pytest.importorskip('benchmark') + + class _Extraction: + """A successful extraction.""" + engine = type('_D', (), {'__engine_name__': 'DPKT'})() + length = 6 + + calls = {'n': 0} + + def _flaky(engine, capture): + """Fail on the third call only, as a crashing subprocess would.""" + calls['n'] += 1 + if calls['n'] == 3: + raise OSError('tshark went away') + return _Extraction() + + monkeypatch.setattr(benchmark, '_extract', _flaky) + result = benchmark.measure('dpkt', str(tmp_path / 'unused.pcap'), rounds=5) + # Five good samples were still collected, and the failure is on the record. + assert result['timed_samples'] == 4 # 5 collected, warm-up discarded + assert result['discarded'] == ['OSError: tshark went away'] + + def test_measure_gives_up_past_the_tolerance(self, monkeypatch, tmp_path): + """Persistent failure is a failure, not something to retry forever.""" + benchmark = pytest.importorskip('benchmark') + + def _broken(engine, capture): + """Always fail, as a genuinely unusable engine would.""" + raise OSError('tshark is gone') + + monkeypatch.setattr(benchmark, '_extract', _broken) + with pytest.raises(OSError, match='tshark is gone'): + benchmark.measure('dpkt', str(tmp_path / 'unused.pcap'), rounds=5, tolerate=2) + + def test_a_fallback_is_never_tolerated(self, monkeypatch, tmp_path): + """The escalated EngineWarning is fatal however lenient the tolerance. + + Discarding it would turn the harness's central assertion into a shrug: the + engine has stopped being itself, and no number should come out of that. + + """ + benchmark = pytest.importorskip('benchmark') + from pcapkit.utilities.warnings import EngineWarning + + def _falls_back(engine, capture): + """Warn exactly as Extractor.run does when it falls back.""" + import warnings as warnings_module + warnings_module.warn('engine DPKT is not installed', EngineWarning) + raise AssertionError('unreachable: the warning is escalated to an error') + + monkeypatch.setattr(benchmark, '_extract', _falls_back) + with pytest.raises(EngineWarning): + benchmark.measure('dpkt', str(tmp_path / 'unused.pcap'), rounds=5, tolerate=100) + + +class TestInstallFailure: + """A package that failed to build explains itself in the table.""" + + def test_a_recorded_build_failure_becomes_the_reason(self, monkeypatch, tmp_path): + """The compiler error is the diagnosis, not 'its package is not installed'.""" + benchmark = pytest.importorskip('benchmark') + (tmp_path / 'pypcap.txt').write_text( + 'pypcap failed to build in this image, so it could not be measured.\n' + "fatal error: pcap.h: No such file or directory\n", + encoding='utf-8') + monkeypatch.setattr(benchmark, 'INSTALL_FAILURES', str(tmp_path)) + recorded = benchmark._install_failure('pypcap') # pylint: disable=protected-access + assert 'pcap.h: No such file or directory' in recorded + # Collapsed to one line, since it lands in a table cell and an RST bullet. + assert '\n' not in recorded + + def test_no_recorded_failure_is_not_an_error(self, monkeypatch, tmp_path): + """The normal case, including outside the container, is simply nothing.""" + benchmark = pytest.importorskip('benchmark') + monkeypatch.setattr(benchmark, 'INSTALL_FAILURES', str(tmp_path / 'absent')) + assert benchmark._install_failure('pypcap') is None # pylint: disable=protected-access + + +def matrix(versions=('3.10', '3.11', '3.12', '3.13', '3.14'), missing_pyshark=('3.14',), + missing_pypcapfile=('3.12', '3.13', '3.14'), pypcap_on=('3.10', '3.11')): + """A whole matrix run, shaped the way `run.sh` actually produces one. + + One document per (Python version, ``pcap`` provider) pair, with the real + engine-support ceilings: two virtualenvs where ``pypcap`` can be installed and one + where it cannot, ``pypcapfile`` gone from 3.12, ``pyshark`` gone from 3.14. + + Args: + versions: Python series to include. + missing_pyshark: Series where ``pyshark`` cannot run. + missing_pypcapfile: Series where ``pypcapfile`` cannot run. + pypcap_on: Series that get a second, ``pypcap`` virtualenv. + + Returns: + A list of documents :mod:`report` can consume. + + """ + documents = [] + for index, series in enumerate(versions): + # Deliberately drifting per version, so a test cannot pass by accident on + # figures that are all the same number. + base = 0.20 + index * 0.01 + shared = {'default': [base, base * 1.01], + 'dpkt': [base / 20, base / 20 * 1.02], + 'scapy': [base / 7, base / 7 * 1.01]} + unmeasured = {} + if series in missing_pyshark: + unmeasured['pyshark'] = f'pyshark does not support Python {series}' + else: + shared['pyshark'] = [base * 70, base * 71] + if series in missing_pypcapfile: + unmeasured['pypcapfile'] = f'pypcapfile does not support Python {series}' + else: + shared['pypcapfile'] = [base / 15, base / 15 * 1.01] + + ct_unmeasured = dict(unmeasured) + ct_unmeasured['pypcap'] = ( + 'the installed `pcap` module is pcap-ct, not pypcap' if series in pypcap_on else + 'pypcap is not installable on this interpreter, so it was not built into this image' + ) + documents.append(document( + f'{series}-pcap_ct', dict(shared, pcap_ct=[base / 25, base / 25 * 1.01]), + ct_unmeasured, python=f'{series}.7', + image=f'pcapkit-benchmark:py{series} (sha256:{index}{index})', + packages={'dpkt': '1.9.8', 'pypcap': None, 'pcap-ct': '1.3.0b3'})) + + if series in pypcap_on: + documents.append(document( + f'{series}-pypcap', dict(shared, pypcap=[base / 32, base / 32 * 1.01]), + dict(unmeasured, pcap_ct='the installed `pcap` module is pypcap, not pcap-ct'), + python=f'{series}.7', + image=f'pcapkit-benchmark:py{series} (sha256:{index}{index})')) + return documents + + +class TestVersionSeries: + """Which column a document lands in, and in what order the columns go.""" + + def test_series_comes_from_the_interpreter_not_the_label(self): + """The recorded version wins over the environment's name. + + These are not the same kind of fact. The label is a name the runner chose + when it started the container; the version is what the interpreter answered + once it was running. If they ever disagree -- a mislabelled build, a + hand-edited document -- filing the numbers under the label would put one + interpreter's figures in another's column, silently, which is the whole class + of error this harness is built to refuse. + + """ + doc = document('3.12-pcap_ct', {'default': [0.2]}, python='3.11.14') + assert report.python_series(doc) == '3.11' + + def test_columns_are_ordered_numerically(self): + """3.9 comes before 3.10, which no string sort of these gets right.""" + docs = [document('3.9-pcap_ct', {'default': [0.2]}, python='3.9.18'), + document('3.10-pcap_ct', {'default': [0.2]}, python='3.10.19'), + document('3.11-pcap_ct', {'default': [0.2]}, python='3.11.14')] + assert report.version_columns(docs) == ['3.9', '3.10', '3.11'] + + def test_two_virtualenvs_of_one_version_are_one_column(self): + """A column is a Python version, not an environment.""" + docs = matrix(versions=('3.11',)) + assert len(docs) == 2 + assert report.version_columns(docs) == ['3.11'] + + def test_a_version_that_produced_nothing_still_gets_a_column(self): + """A version that could not be built is a gap, not a question never asked. + + Dropping the column would make "we could not measure this" indistinguishable + from "this was not part of the run", and a reader has no way to tell those + apart from the table alone. + + """ + docs = matrix(versions=('3.11',)) + assert report.version_columns(docs, [('3.15', 'no image')]) == ['3.11', '3.15'] + + +class TestAbsolutesByVersion: + """The per-version grid, which is milliseconds rather than ratios.""" + + def test_a_cell_pools_both_virtualenvs_of_that_interpreter(self): + """Both environments measure the same engine on the same Python. + + So both readings belong in that interpreter's cell: they differ only in which + ``pcap`` provider happened to be installed alongside, which is nothing to do + with the engine being timed. + + """ + docs = matrix(versions=('3.11',)) + grid = report.absolutes_by_version(docs) + # `dpkt` is in both environments, twice each; `pypcap` only in the one. + assert len(grid['dpkt']['3.11']) == 4 + assert len(grid['pypcap']['3.11']) == 2 + + def test_a_pass_without_a_baseline_still_has_an_absolute_figure(self): + """A missing baseline costs a ratio, not a measurement. + + :func:`report.collect` must drop such a pass, since there is nothing to divide + by. The absolute figure is complete on its own, and discarding it here would + lose a real measurement to a rule that does not apply to it. + + """ + docs = [document('3.11-pcap_ct', {'default': [0.2], 'dpkt': [0.02, 0.03]})] + rows = {row.engine: row for row in report.collect(docs)} + assert len(rows['dpkt'].ratios) == 1 + + grid = report.absolutes_by_version(docs) + assert grid['dpkt']['3.11'] == pytest.approx([0.02, 0.03]) + + def test_an_engine_measured_nowhere_has_no_cell(self): + """An engine no interpreter ran contributes nothing to the grid.""" + docs = matrix(versions=('3.12',)) + assert '3.12' not in report.absolutes_by_version(docs).get('pypcap', {}) + + +class TestUnmeasuredByVersion: + """Explaining the empty cells, and only the empty ones.""" + + def test_measured_in_one_virtualenv_is_not_a_gap(self): + """``pypcap`` has a figure for 3.11 even though one environment lacks it. + + The mutual exclusion is real and is reported in the ratio table's per-row + reasons, but it is not a gap in *this* table: the cell is filled. A footnote + explaining a cell that is not empty is a footnote that contradicts the table + it annotates. + + """ + docs = matrix(versions=('3.11',)) + reasons = report.unmeasured_by_version(docs) + assert 'pypcap' not in reasons + assert 'pcap_ct' not in reasons + + def test_a_genuinely_empty_cell_keeps_its_reason(self): + """Where no environment measured it, the reason survives.""" + docs = matrix(versions=('3.12',)) + reasons = report.unmeasured_by_version(docs) + assert 'not installable on this interpreter' in reasons['pypcap']['3.12'] + assert 'pypcapfile does not support Python 3.12' in reasons['pypcapfile']['3.12'] + + def test_environments_disagreeing_about_one_cell_report_both(self): + """Two reasons for one empty cell are both kept, attributed. + + A disagreement is the informative case -- an engine unavailable in the + ``pypcap`` environment because of the ``pcap`` collision is a different fact + from the same engine unavailable because ``tshark`` is missing -- so picking + one of them would throw away the half that explains the other. + + """ + docs = [document('3.11-pypcap', {'default': [0.2]}, + {'pyshark': 'no tshark binary'}), + document('3.11-pcap_ct', {'default': [0.2]}, + {'pyshark': 'python too new'})] + reason = report.unmeasured_by_version(docs)['pyshark']['3.11'] + assert 'in 3.11-pypcap, no tshark binary' in reason + assert 'in 3.11-pcap_ct, python too new' in reason + + +class TestVersionsMarkup: + """The per-version table as it will be pasted into README.rst.""" + + def _snippet(self, missing=(), emulated=None, **kwargs): + """Render the per-version snippet from a whole matrix run.""" + docs = matrix(**kwargs) + rows = report.collect(docs) + return report.render_versions_rst(rows, docs, missing, emulated), docs, rows + + def test_no_sphinx_only_roles(self): + """Only literals, because docutils renders unknown roles as errors.""" + snippet, _, _ = self._snippet() + for role in SPHINX_ONLY_ROLES: + assert role not in snippet, f'{role} is Sphinx-only and breaks on GitHub' + + def test_parses_under_plain_docutils(self): + """The snippet renders cleanly with the parser GitHub actually uses. + + A simple table is the format most easily broken by a cell that outgrows its + column rule, and this table's cells are generated from measurements -- so the + width that works today is not evidence about the width a slower engine + produces tomorrow. + + """ + docutils_core = pytest.importorskip('docutils.core') + from docutils.utils import SystemMessage # pylint: disable=import-outside-toplevel + + snippet, _, _ = self._snippet(missing=[('3.15', 'the image failed to build')]) + messages = [] + try: + docutils_core.publish_doctree( + snippet, + settings_overrides={ + 'halt_level': 2, 'report_level': 2, 'warning_stream': messages, + 'input_encoding': 'unicode', 'output_encoding': 'unicode', + }, + ) + except SystemMessage as exc: # pragma: no cover - only on a real failure + pytest.fail(f'docutils rejected the snippet: {exc}\n\n{snippet}') + assert not messages, f'docutils warned: {messages}\n\n{snippet}' + + def test_the_figures_are_absolute_and_never_a_ratio(self): + """Milliseconds, said in words, and no ratio anywhere in the snippet. + + Ratios are normalised inside one environment, so a ratio between two columns + of this table would describe neither interpreter. The baseline row is the test + that this has not been confused: in the ratio table ``pcapkit`` is 1 by + definition, and here it must carry its own measured milliseconds like every + other row. + + """ + snippet, _, _ = self._snippet(versions=('3.11',)) + assert 'milliseconds per' in snippet + assert 'ratio' not in snippet.lower() + assert 'relative' not in snippet.lower() + assert '*baseline*' not in snippet + + pcapkit_row = [line for line in snippet.splitlines() + if line.startswith('``pcapkit``')][0] + # 0.20 and 0.202 from each of the two virtualenvs, so the median of the four + # pooled readings -- and emphatically not the 1 the ratio table gives it. + assert '0.2010' in pcapkit_row + + def test_columns_are_labelled_comparable_with_each_other_only(self): + """The one caveat absolute figures need, in the snippet rather than beside it.""" + snippet, _, _ = self._snippet() + assert 'compared with each other' in snippet + assert 'may not be' in snippet + assert 'another machine' in snippet + + def test_an_unmeasured_cell_is_a_dash(self): + """Never a zero, and never an absent row.""" + snippet, _, _ = self._snippet() + pypcap_row = [line for line in snippet.splitlines() + if line.startswith('``pypcap``')][0] + # Measured on 3.10 and 3.11, unmeasurable on the three newer interpreters. + assert pypcap_row.count('--') == 3 + + def test_every_engine_has_a_row_whatever_it_managed(self): + """A row per engine, in the same order as the ratio table. + + The two tables are read together, and an engine that moves between them costs + the reader the ability to carry their eye from one to the other. + + """ + snippet, _, rows = self._snippet() + positions = [snippet.index(f'``{row.label}``') for row in rows] + assert positions == sorted(positions) + assert len(positions) == len(rows) + + def test_an_engine_with_a_gap_is_marked_and_explained(self): + """The mark points at a reason, and the reason is the engine's own.""" + snippet, _, _ = self._snippet() + assert f'``pypcapfile`` {report.UNMEASURED_MARK}' in snippet + assert 'pypcapfile does not support Python 3.12' in snippet + + def test_a_fully_measured_engine_is_not_marked(self): + """No mark on a row with nothing to explain.""" + snippet, _, _ = self._snippet() + assert f'``dpkt`` {report.UNMEASURED_MARK}' not in snippet + + def test_one_reason_covering_several_versions_is_stated_once(self): + """Notes are grouped by reason, not one bullet per empty cell. + + With five columns and seven engines the ungrouped form runs to more lines than + the table it annotates, and reads as though each repetition were a separate + finding. + + """ + snippet, _, _ = self._snippet() + note = [line for line in snippet.splitlines() + if line.startswith('* ``pypcap``')] + assert len(note) == 1 + assert '3.12, 3.13, 3.14' in note[0] + + def test_a_missing_version_is_a_marked_column_of_dashes(self): + """A version that produced nothing says so in the table and in a note.""" + snippet, _, _ = self._snippet( + missing=[('3.15', 'the py3.15 image failed to build')]) + assert f'3.15 {report.UNMEASURED_MARK}' in snippet + assert '* Python 3.15 -- not measured at all: the py3.15 image failed to build' in snippet + + def test_a_missing_version_does_not_blame_the_engines(self): + """No engine is marked for a column that never ran. + + The column's own note explains every blank in it, and tagging each engine as + well would attribute someone else's build failure to seven engines that were + never given the chance to fail. + + """ + snippet, _, rows = self._snippet(versions=('3.11',), + missing=[('3.15', 'the image failed to build')]) + assert f'3.15 {report.UNMEASURED_MARK}' in snippet + # Every engine, not a sample of two: on 3.11 all seven are measurable, so the + # only mark in the whole table should be the one on the 3.15 column heading. + for row in rows: + assert f'``{row.label}`` {report.UNMEASURED_MARK}' not in snippet + assert snippet.count(report.UNMEASURED_MARK) == 2 # the heading, and its legend + + def test_a_partly_measured_version_is_not_called_unmeasured(self): + """A version can have both figures and a failure note, and both are true. + + `run.sh` copies out whatever a container wrote before it died, so a 3.11 whose + second virtualenv crashed arrives with real measurements *and* a recorded + reason. Calling that column "not measured at all" would contradict the numbers + printed in it, and suppressing the engines' own gap reasons -- as a column with + nothing in it rightly does -- would leave those gaps unexplained. + + """ + snippet, _, _ = self._snippet( + versions=('3.11', '3.12'), + missing=[('3.11', 'the 3.11-pypcap container exited non-zero')]) + assert 'not measured at all' not in snippet + assert 'the run did not complete' in snippet + # The column keeps its figures... + dpkt_row = [line for line in snippet.splitlines() if line.startswith('``dpkt``')][0] + assert dpkt_row.count('--') == 0 + # ...and an engine with a real gap in it is still marked and still explained. + assert f'``pypcapfile`` {report.UNMEASURED_MARK}' in snippet + assert 'pypcapfile does not support Python 3.12' in snippet + + def test_emulation_warning_reaches_the_snippet(self): + """An absolute figure taken under emulation describes the emulator.""" + snippet, _, _ = self._snippet( + emulated='Measured under emulation: linux/amd64 on arm64.') + assert 'emulation' in snippet.lower() + assert 'not comparable with' in snippet + + def test_figure_columns_all_share_one_width(self): + """A ragged table of like quantities reads as unlike quantities.""" + snippet, _, _ = self._snippet() + rule = [line for line in snippet.splitlines() if line.startswith('====')][0] + widths = [len(run) for run in rule.split()] + assert len(set(widths[1:])) == 1 + + +class TestMatrixProvenance: + """What the Test Environment block says once there is more than one interpreter.""" + + def test_every_interpreter_is_named(self): + """Naming only the first would describe one column and imply it stood for all.""" + docs = matrix() + snippet = report.render_rst(report.collect(docs), docs) + for version in ('3.10.7', '3.11.7', '3.12.7', '3.13.7', '3.14.7'): + assert version in snippet + + def test_each_version_names_the_image_it_came_from(self): + """The image is the only thing tying a column to a specific build.""" + docs = matrix(versions=('3.11', '3.12')) + snippet = report.render_rst(report.collect(docs), docs) + assert 'Image (3.11)' in snippet + assert 'Image (3.12)' in snippet + + def test_a_single_interpreter_still_reports_one_image(self): + """One version, one image, one row -- not a row labelled with its own version.""" + docs = matrix(versions=('3.11',)) + snippet = report.render_rst(report.collect(docs), docs) + assert 'Image (3.11)' not in snippet + assert 'pcapkit-benchmark:py3.11' in snippet + + def test_distinct_tshark_versions_are_all_reported(self): + """``pyshark``'s figures are as much about tshark as about the package. + + Every image in the matrix is built on the same Debian precisely so that this + line has one entry, which makes a second entry the signal that something in + the base images has drifted apart -- so all of them are reported rather than + just the first. + + """ + docs = [document('3.11-pcap_ct', {'default': [0.2]}, python='3.11.14', + tshark='TShark (Wireshark) 4.0.17'), + document('3.12-pcap_ct', {'default': [0.2]}, python='3.12.12', + tshark='TShark (Wireshark) 4.2.2')] + snippet = report.render_rst(report.collect(docs), docs) + assert '4.0.17' in snippet + assert '4.2.2' in snippet + + def test_a_missing_version_is_recorded_in_the_provenance(self): + """The ratio table pools the interpreters, so it has to say which ones.""" + docs = matrix(versions=('3.11',)) + snippet = report.render_rst(report.collect(docs), docs, + missing=[('3.15', 'the image failed to build')]) + assert 'Python 3.15 (not measured)' in snippet + assert 'the image failed to build' in snippet + + def test_a_partly_measured_version_says_partly(self): + """A version that contributed figures before failing is not "not measured".""" + docs = matrix(versions=('3.11',)) + snippet = report.render_rst(report.collect(docs), docs, + missing=[('3.11', 'one container exited non-zero')]) + assert 'Python 3.11 (partly measured)' in snippet + + def test_pooling_across_interpreters_is_declared(self): + """A ratio pooled over five interpreters must not read as one interpreter's.""" + docs = matrix() + snippet = report.render_rst(report.collect(docs), docs) + assert 'pooled across every environment' in snippet + + # ...and not claimed on a run that had only one interpreter to pool. + single = matrix(versions=('3.11',)) + assert 'pooled across every environment' not in \ + report.render_rst(report.collect(single), single) + + def test_a_packet_count_disagreement_is_reported_not_dropped(self): + """One capture must yield one frame count on every interpreter. + + If it does not, that is the most important thing in the report -- and the + previous behaviour was to print no row at all, which hid precisely the case the + row exists to establish. With one interpreter the disagreement was barely + possible; across a matrix it is a real failure mode. + + """ + docs = matrix(versions=('3.11', '3.12')) + for entry in docs[-1]['results']: + entry['packets'] = 5 if entry['packets'] else entry['packets'] + snippet = report.render_rst(report.collect(docs), docs) + assert 'Packets per extraction' in snippet + assert 'disagreed across the run' in snippet + assert '3.11 6' in snippet + assert '3.12 5' in snippet + + def test_an_agreeing_packet_count_is_just_the_number(self): + """No alarm on the normal case.""" + docs = matrix(versions=('3.11', '3.12')) + snippet = report.render_rst(report.collect(docs), docs) + assert 'disagreed' not in snippet + assert 'Packets per extraction' in snippet + + def test_the_sample_column_is_not_called_passes(self): + """A pass happens once per environment, so samples outnumber passes. + + The provenance block says "3 passes ... in 7 environments" and the table's own + count is the product of the two. Calling both "Passes" read as a contradiction + as soon as there was more than one interpreter. + + """ + docs = matrix(versions=('3.11', '3.12')) + snippet = report.render_rst(report.collect(docs), docs) + table = snippet.split('Test Results (Relative)')[1] + assert 'Samples' in table + assert 'Passes' not in table + # ...while the provenance block above still counts passes, in passes. + assert 'Passes' in snippet.split('Test Results (Relative)')[0] + + def test_the_two_tables_do_not_share_a_heading(self): + """Two sections named "Test Results" in one README is one too many. + + Both snippets are pasted into the same document, and a reader then has no way + to say which table a sentence underneath is about. + + """ + docs = matrix(versions=('3.11',)) + rows = report.collect(docs) + assert 'Test Results (Relative)' in report.render_rst(rows, docs) + assert 'Test Results (Relative)' not in report.render_versions_rst(rows, docs) + + +class TestPinsFile: + """`python-images.txt` is the whole matrix, and nothing else validates it. + + `run.sh` reads it with awk and would happily act on a malformed row -- a missing + field silently becomes a row awk skips, so the version quietly stops being measured + with no error anywhere. These are the assertions that would otherwise only be made + by a benchmark run that takes two hours to reach them. + + """ + + #: Field meanings, matching the header comment in the file itself. + FIELDS = ('version', 'image', 'pypcap', 'tier') + + def _rows(self): + """The real rows, filtered exactly as `run.sh`'s awk does.""" + path = Path(__file__).resolve().parent / 'python-images.txt' + rows = [] + for line in path.read_text(encoding='utf-8').splitlines(): + fields = line.split() + if not fields or fields[0].startswith('#') or len(fields) < len(self.FIELDS): + continue + rows.append(dict(zip(self.FIELDS, fields))) + return rows + + def test_every_row_has_every_field(self): + """A short row is one awk skips, which is a version silently not measured.""" + rows = self._rows() + assert rows, 'no rows parsed at all; run.sh would have nothing to measure' + for row in rows: + assert all(row[field] for field in self.FIELDS), row + + def test_every_base_image_is_pinned_by_digest(self): + """A tag pin is not a pin -- the file's own header says so. + + `3.11-slim-bookworm` is rebuilt whenever Debian or CPython ships a patch, so a + row that lost its digest would keep working and quietly measure a different + interpreter than the one the last run measured. + + """ + for row in self._rows(): + assert '@sha256:' in row['image'], row + assert len(row['image'].split('@sha256:')[1]) == 64, row + + def test_the_pypcap_and_tier_columns_use_the_documented_words(self): + """`run.sh` compares these literally, so a synonym is a silent behaviour change.""" + for row in self._rows(): + assert row['pypcap'] in ('pypcap', 'no-pypcap'), row + assert row['tier'] in ('default', 'opt-in'), row + + def test_a_default_run_measures_something(self): + """`run.sh` with no arguments has to have a matrix to run.""" + assert [row['version'] for row in self._rows() if row['tier'] == 'default'] + + def test_pypcap_is_offered_only_where_it_can_be_installed(self): + """The ceiling is 3.11: building that virtualenv on 3.12+ compiles for nothing.""" + for row in self._rows(): + expected = 'pypcap' if tuple(int(part) for part in row['version'].split('.')) \ + <= (3, 11) else 'no-pypcap' + assert row['pypcap'] == expected, row + + def test_the_dockerfile_default_matches_the_pins_file(self): + """The 3.11 digest is written in two places, so it can drift in one. + + The Dockerfile needs a usable default so that a bare ``docker build`` works, and + `run.sh`'s header claims nothing about the matrix is hard-coded elsewhere. Both + are reasonable; together they are a duplicated pin, and this is the only thing + that would notice them disagreeing. + + """ + pinned = {row['version']: row['image'] for row in self._rows()}['3.11'] + dockerfile = (Path(__file__).resolve().parent / 'Dockerfile').read_text(encoding='utf-8') + default = [line.split('=', 1)[1].strip() for line in dockerfile.splitlines() + if line.startswith('ARG PYTHON_IMAGE=')] + assert default == [pinned] + + +class TestHostileReasons: + """Reasons are written by compilers and exceptions, not by this harness. + + Every reason in the report comes from somewhere that has never heard of + reStructuredText: an engine's ``unsupported_reason()``, an exception's ``str()``, + or the tail of a pip or gcc log that the Dockerfile recorded because a `pypcap` + build was allowed to fail. All of it is interpolated into emitted markup as prose, + and the suite's other fixtures are all well-behaved English -- which is exactly why + this went unnoticed until a hostile string was tried. + + """ + + #: Strings a compiler or a Python traceback produces without trying, each of which + #: broke the emitted snippet under plain docutils before the escaping went in. + HOSTILE = ( + # `**kwargs` in a gcc diagnostic: inline strong start-string without end-string. + 'gcc: error: **kwargs handling broke the build', + # GNU tools quote like `this', so the backtick never balances. + "pip said: cannot find `pcap.h", + # A trailing underscore is a reference: unknown target name. + 'linklayer imports imp_ which was removed in 3.12', + # A paragraph ending in `::` promises a literal block that never arrives. + 'the build died here::', + # And the rest of the inline markup characters, together. + 'a *very* |odd| [1] c:\\path\\to\\thing reason', + ) + + def _documents(self, reason): + """A two-interpreter run in which every reason is *reason*.""" + return [ + document('3.11-pcap_ct', {'default': [0.2, 0.2], 'dpkt': [0.02, 0.02]}, + {'pypcap': reason}, failures={'dpkt': [reason]}), + document('3.12-pcap_ct', {'default': [0.2, 0.2], 'dpkt': [0.02, 0.02]}, + {'pypcap': reason}, python='3.12.12'), + ] + + @pytest.mark.parametrize('reason', HOSTILE) + def test_both_snippets_still_parse(self, reason): + """A gcc diagnostic in a reason must not break the pasted table. + + The promise is that these snippets go into README.rst verbatim, and GitHub + renders that with docutils -- where each of these strings produces a visible + error block instead of the table. + + """ + docutils_core = pytest.importorskip('docutils.core') + from docutils.utils import SystemMessage # pylint: disable=import-outside-toplevel + + docs = self._documents(reason) + rows = report.collect(docs) + for snippet in (report.render_versions_rst(rows, docs, [('3.15', reason)]), + report.render_rst(rows, docs, missing=[('3.15', reason)])): + messages = [] + try: + docutils_core.publish_doctree( + snippet, + settings_overrides={ + 'halt_level': 2, 'report_level': 2, 'warning_stream': messages, + 'input_encoding': 'unicode', 'output_encoding': 'unicode', + }, + ) + except SystemMessage as exc: # pragma: no cover - only on a real failure + pytest.fail(f'docutils rejected a reason it should have survived: ' + f'{exc}\n\n{snippet}') + assert not messages, f'docutils warned: {messages}\n\n{snippet}' + + def test_the_reader_still_sees_the_original_text(self): + """Escaping must not be censoring: the rendered document says what gcc said. + + A backslash escape is removed when reStructuredText is rendered, so this holds + without the reason being rewritten -- which matters because the reason is + diagnostic output and a paraphrase of it is worth nothing. + + """ + docutils_core = pytest.importorskip('docutils.core') + + reason = self.HOSTILE[0] + docs = self._documents(reason) + snippet = report.render_versions_rst(report.collect(docs), docs) + rendered = docutils_core.publish_doctree( + snippet, + settings_overrides={'input_encoding': 'unicode', 'output_encoding': 'unicode', + 'report_level': 5}, + ).astext() + assert reason in rendered + + def test_the_plain_text_report_is_not_escaped(self): + """Backslashes belong in markup, not in the operator's terminal.""" + reason = 'gcc: error: **kwargs handling broke the build' + docs = self._documents(reason) + text = report.render_text(report.collect(docs), docs, missing=[('3.15', reason)]) + assert reason in text + assert '\\*' not in text + + def test_the_emulation_note_is_escaped_too(self): + """`--emulated` is outside text as much as any reason is. + + `run.sh` composes that sentence itself, which makes it safe in practice and is + exactly why it was the last channel left unescaped. It still arrives on the + command line, and it reaches markup in two places -- a provenance row and a bold + paragraph -- so a value with a stray ``*`` in it would break the snippet from a + direction nobody was watching. + + """ + docutils_core = pytest.importorskip('docutils.core') + from docutils.utils import SystemMessage # pylint: disable=import-outside-toplevel + + emulated = 'Measured under **emulation**: a linux/amd64 image on an `arm64 host.' + docs = self._documents('an ordinary reason') + rows = report.collect(docs) + for snippet in (report.render_versions_rst(rows, docs, (), emulated), + report.render_rst(rows, docs, emulated=emulated)): + messages = [] + try: + docutils_core.publish_doctree( + snippet, + settings_overrides={ + 'halt_level': 2, 'report_level': 2, 'warning_stream': messages, + 'input_encoding': 'unicode', 'output_encoding': 'unicode', + }, + ) + except SystemMessage as exc: # pragma: no cover - only on a real failure + pytest.fail(f'docutils rejected the emulation note: {exc}\n\n{snippet}') + assert not messages, f'docutils warned: {messages}\n\n{snippet}' + + # ...and the operator still sees it as it was written. + assert emulated in report.render_text(rows, docs, emulated=emulated) + + +class TestFixedFormatting: + """Formatting for a column read downwards rather than a value read alone.""" + + @pytest.mark.parametrize(('value', 'expected'), [ + (0.017, '0.0170'), + (14.7434, '14.7434'), + (0.2516, '0.2516'), + ]) + def test_four_decimal_places_whatever_the_magnitude(self, value, expected): + """Decimal points line up, which is what makes a column scannable. + + Significant figures -- which every other number in the report uses -- would + give ``0.01700`` and ``14.74`` different numbers of decimal places in the same + column. Four places is also what the hand-maintained table in README.rst has + always used, so a regenerated table diffs its numbers rather than its layout. + + """ + assert report._fixed(value) == expected # pylint: disable=protected-access + + +class TestMissingArgument: + """The interface `run.sh` uses to get a build failure into the table.""" + + def _one_document(self, tmp_path): + """Write a minimal single-environment document to disk.""" + path = tmp_path / '3.11-pcap_ct.json' + path.write_text(json.dumps(document('3.11-pcap_ct', {'default': [0.2, 0.2]})), + encoding='utf-8') + return path + + def test_a_reason_becomes_a_column_and_a_note(self, tmp_path, capsys): + """What run.sh records is what the reader sees.""" + path = self._one_document(tmp_path) + assert report.main([str(path), '--missing', '3.15=no released image exists', + '--versions-rst-out', str(tmp_path / 'versions.rst')]) == 0 + snippet = (tmp_path / 'versions.rst').read_text(encoding='utf-8') + assert '3.15' in snippet + assert 'no released image exists' in snippet + + def test_a_version_without_a_reason_is_rejected(self, tmp_path): + """A gap whose note says nothing is the outcome --missing exists to prevent.""" + path = self._one_document(tmp_path) + with pytest.raises(SystemExit): + report.main([str(path), '--missing', '3.15']) + with pytest.raises(SystemExit): + report.main([str(path), '--missing', '3.15=']) + + def test_two_reasons_for_one_version_are_rejected(self, tmp_path): + """There is no honest rendering of two reasons for one column. + + The provenance block would list both and the table's notes only the last, so the + same run would report the gap differently in two places. `run.sh` writes one note + file per version and cannot produce this; a hand-driven invocation is told. + + """ + path = self._one_document(tmp_path) + with pytest.raises(SystemExit): + report.main([str(path), '--missing', '3.15=first', '--missing', '3.15=second']) + + +class TestNotAttempted: + """An engine the interpreter rules out, told apart from one that failed here.""" + + def test_the_recorded_note_is_the_reason(self, monkeypatch, tmp_path): + """The image says why it did not try, and that is what reaches the table.""" + benchmark = pytest.importorskip('benchmark') + (tmp_path / 'pypcap.txt').write_text( + 'pypcap is not installable on this interpreter, so it was not built into\n' + 'this image: its pcap.c does not compile against the 3.12+ C API.\n', + encoding='utf-8') + monkeypatch.setattr(benchmark, 'NOT_ATTEMPTED', str(tmp_path)) + recorded = benchmark._not_attempted('pypcap') # pylint: disable=protected-access + assert 'not installable on this interpreter' in recorded + # Collapsed to one line, since it lands in a table cell and an RST bullet. + assert '\n' not in recorded + + def test_it_replaces_the_engine_s_own_reason(self, monkeypatch, tmp_path): + """The cause wins over the symptom. + + On a 3.12 image the engine's own ``unsupported_reason()`` says the installed + ``pcap`` module is ``pcap-ct`` -- true, and a description of how this image was + assembled rather than of why it had to be. The reader is owed the interpreter + ceiling, and stacking both would bury it behind the consequence. + + """ + benchmark = pytest.importorskip('benchmark') + (tmp_path / 'pypcap.txt').write_text( + 'pypcap is not installable on this interpreter.\n', encoding='utf-8') + monkeypatch.setattr(benchmark, 'NOT_ATTEMPTED', str(tmp_path)) + monkeypatch.setattr(benchmark, '_declared_reason', + lambda engine: 'the installed `pcap` module is pcap-ct, not pypcap') + + reason, driver = benchmark.preflight('pypcap', str(tmp_path / 'unused.pcap')) + assert reason == 'pypcap is not installable on this interpreter.' + assert 'pcap-ct' not in reason + assert driver is None + + def test_an_engine_that_was_attempted_records_nothing(self, monkeypatch, tmp_path): + """The normal case, for every engine the image did install.""" + benchmark = pytest.importorskip('benchmark') + monkeypatch.setattr(benchmark, 'NOT_ATTEMPTED', str(tmp_path / 'absent')) + assert benchmark._not_attempted('dpkt') is None # pylint: disable=protected-access + + def test_prose_is_rejoined_as_prose_and_log_output_is_not(self, monkeypatch, tmp_path): + """A wrapped sentence comes back as a sentence, pip output keeps its records. + + Both notes are flattened to one line, because both end up in a table cell and + an RST bullet -- but they are not the same kind of text. Measured: the + interpreter-ceiling note joined with the pip separator read ``not built into + this | image: pypcap 1.3.0 ships ...``, which is the Dockerfile's line wrapping + leaking into a published table. + + """ + benchmark = pytest.importorskip('benchmark') + (tmp_path / 'pypcap.txt').write_text('one sentence wrapped\nacross two lines.\n', + encoding='utf-8') + monkeypatch.setattr(benchmark, 'NOT_ATTEMPTED', str(tmp_path)) + monkeypatch.setattr(benchmark, 'INSTALL_FAILURES', str(tmp_path)) + assert benchmark._not_attempted('pypcap') == 'one sentence wrapped across two lines.' + assert benchmark._install_failure('pypcap') == \ + 'one sentence wrapped | across two lines.' + + +class TestRoundTrip: + """The JSON contract between the two halves of the harness.""" + + def test_a_real_document_survives_json(self, tmp_path): + """What benchmark.py writes is what report.py reads.""" + docs = [document('pypcap', {'default': [0.2, 0.2], 'dpkt': [0.02, 0.02]})] + path = tmp_path / 'pypcap.json' + path.write_text(json.dumps(docs[0]), encoding='utf-8') + reloaded = json.loads(path.read_text(encoding='utf-8')) + rows = {row.engine: row for row in report.collect([reloaded])} + assert rows['dpkt'].median == pytest.approx(0.1)