diff --git a/pcapkit/const/reg/linktype.py b/pcapkit/const/reg/linktype.py
index b5418eb9d1..52d7fc0565 100644
--- a/pcapkit/const/reg/linktype.py
+++ b/pcapkit/const/reg/linktype.py
@@ -743,6 +743,10 @@ class LinkType(IntEnum):
#: header prepended to a standard DLT\_DECT\_NR MAC-layer frame.
DECT_NR_TAP = 304
+ # The following legacy alias(es) are emitted here, after every
+ # current entry above, instead of at their original numeric
+ # position -- see pcapkit.vendor.reg.linktype.LinkType.process
+ # for why.
#: [``DLT_IPMB_LINUX``] Legacy names (do not use) for Linux I2C below.
IPMB_LINUX = 209
diff --git a/pcapkit/vendor/reg/linktype.py b/pcapkit/vendor/reg/linktype.py
index 0e9c5ab915..1944fcd206 100644
--- a/pcapkit/vendor/reg/linktype.py
+++ b/pcapkit/vendor/reg/linktype.py
@@ -67,6 +67,58 @@ def process(self, data: 'list[Tag]') -> 'tuple[list[str], list[str]]':
miss = [
"return extend_enum(cls, 'Unassigned_%d' % value, value)",
]
+
+ # Value-aware guard for the ``sink`` decision below (GitHub issue #852,
+ # a follow-up to #848/#844). A row's notes mentioning "legacy" is
+ # tcpdump's only textual signal for *which* member of a duplicated
+ # value is the deprecated one, but that word can appear in a row's
+ # notes for an unrelated reason -- e.g. referencing some other,
+ # differently-valued code as "the legacy DLT_FOO". Computing which
+ # values are genuinely duplicated in this table upfront, independent
+ # of row order, stops the sink from firing on a value that has no
+ # duplicate to disambiguate in the first place. It does not, on its
+ # own, guarantee which member of a *genuine* duplicate pair is
+ # correct if tcpdump's own wording is attached to the wrong one of
+ # the two -- that residual case still relies on the wording being
+ # trustworthy.
+ #
+ # A range row (``DLT_USER0``..``DLT_USER15``) is never itself routed
+ # through ``sink`` (see item 2 in #852), but its *expanded* values
+ # still have to count here: a single-value row can legitimately
+ # duplicate one of the values a range expands to, and missing that
+ # would silently under-count the duplicate the same way the
+ # word-only rule did before #852.
+ #
+ # The single-value branch accepts whatever ``int()`` accepts --
+ # a leading sign, PEP 515 underscores -- rather than the narrower
+ # ``str.isdigit()``, so it recognises exactly the same values the
+ # main loop's own ``int(temp)`` below does. A signed or underscored
+ # value is not reachable in tcpdump's live table today (every
+ # committed member matches a plain unsigned decimal), but a
+ # narrower recognition set here than in the main loop would be the
+ # same class of silent under-count this pre-scan exists to close
+ # for ranges: a legacy row's value would go uncounted, so its alias
+ # would stop being sunk without anything failing loudly.
+ def _expand(temp: 'str') -> 'list[int]':
+ try:
+ return [int(temp)]
+ except ValueError:
+ pass
+ if '–' in temp: # en dash, not a hyphen -- see the ``except ValueError`` below
+ try:
+ start, stop = map(int, temp.split('–'))
+ except ValueError:
+ return []
+ return list(range(start, stop + 1))
+ return []
+
+ value_counts = collections.Counter(
+ value
+ for temp in (content.select('td.number')[0].text.strip() for content in data)
+ for value in _expand(temp)
+ ) # type: Counter[int]
+ dup_values = {value for value, count in value_counts.items() if count > 1}
+
for content in data:
name = content.select('td.symbol')[0].text.strip()[9:].strip()
temp = content.select('td.number')[0].text.strip()
@@ -76,17 +128,8 @@ def process(self, data: 'list[Tag]') -> 'tuple[list[str], list[str]]':
if not name:
name = desc[4:]
- # tcpdump's table lists a handful of rows -- today, only
- # ``DLT_IPMB_LINUX`` -- as a legacy alias sharing its value with a
- # later, current row (``DLT_I2C_LINUX``); its note column says so
- # verbatim ("Legacy names (do not use) ..."). Sink those into a
- # separate bucket appended after every current entry, so the
- # *current* name -- not the legacy one -- is the first member
- # defined for that value and therefore wins ``LinkType(value).name``.
- sink = legacy if 'legacy' in cmmt.lower() else enum
-
try:
- code, _ = temp, int(temp)
+ code, code_int = temp, int(temp)
if not name:
name = f'Unassigned_{code}'
@@ -100,6 +143,13 @@ def process(self, data: 'list[Tag]') -> 'tuple[list[str], list[str]]':
# sufs = f"\n{' '*80}{sufs}"
# enum.append(f'{pres.ljust(76)}{sufs}')
+
+ # Sink this row after every current entry only when its value
+ # is a genuine duplicate (``dup_values`` above) *and* its own
+ # notes mention "legacy", so the current name -- not the
+ # legacy one -- is the first member defined for that value
+ # and therefore wins ``LinkType(value).name``.
+ sink = legacy if (code_int in dup_values and 'legacy' in cmmt.lower()) else enum
sink.append(f'{sufs}\n {pres}')
except ValueError:
start, stop = map(int, temp.split('–'))
@@ -114,7 +164,34 @@ def process(self, data: 'list[Tag]') -> 'tuple[list[str], list[str]]':
# sufs = f"\n{' '*80}{sufs}"
# enum.append(f'{pres.ljust(76)}{sufs}')
- sink.append(f'{sufs}\n {pres}')
+
+ # Range rows are generic, user-definable placeholders,
+ # never a deprecated/current pair -- so every expanded
+ # member always lands in ``enum`` unconditionally,
+ # rather than through the per-row ``sink`` above.
+ # Otherwise one range row whose notes happened to mention
+ # "legacy" would sink all sixteen expanded members at
+ # once instead of just itself (item 2 in #852).
+ enum.append(f'{sufs}\n {pres}')
+
+ if legacy:
+ # A reader scanning the generated file by value will find these
+ # members out of numeric order: sinking appends them here, after
+ # every current entry, rather than leaving them at their
+ # original position in tcpdump's table (item 4 in #852). Say so
+ # once, right where they are emitted -- as a plain ``#`` comment,
+ # deliberately, not a ``#:`` one: a ``#:`` line immediately above
+ # an attribute becomes that attribute's rendered Sphinx docstring,
+ # and this note is about the generator's own source layout, not
+ # about what ``IPMB_LINUX`` (or whichever name is sunk here) means
+ # to a caller.
+ legacy[0] = (
+ "# The following legacy alias(es) are emitted here, after every\n"
+ " # current entry above, instead of at their original numeric\n"
+ " # position -- see pcapkit.vendor.reg.linktype.LinkType.process\n"
+ " # for why.\n"
+ " " + legacy[0]
+ )
return enum + legacy, miss
diff --git a/tests/const/test_const_linktype_209_unit.py b/tests/const/test_const_linktype_209_unit.py
index c3155a633a..9a06252f25 100644
--- a/tests/const/test_const_linktype_209_unit.py
+++ b/tests/const/test_const_linktype_209_unit.py
@@ -21,6 +21,25 @@
*after* every other row, so the current name is always the first one defined
for a shared value regardless of the source table's own row order.
+GitHub issue #852, a follow-up to the cross-review of #848, pointed out that
+the original predicate tested the notes column for the word "legacy" alone,
+with no check that the row's value was actually claimed by another row --
+meaning a future *current* row whose notes happened to mention "legacy" about
+some unrelated code would be wrongly sunk, and a single ``USER0``-style range
+row would sink all sixteen expanded members at once instead of none.
+:meth:`~pcapkit.vendor.reg.linktype.LinkType.process` now only sinks a
+single-value row when its value is a genuine duplicate elsewhere in the same
+table, and never routes a range row through the sink at all.
+
+The cross-review of #852 itself then caught a second-order regression: the
+first cut of the duplicate-value pre-scan counted only single-value rows, so a
+value duplicated *across a range boundary* -- a single-value row sharing a
+value with one member of a ``USER0``-style range -- was invisible to it, and
+that row's legacy alias was no longer sunk even when correctly worded. The
+pre-scan now expands ``a–b`` ranges the same way the main loop does
+before counting, so no duplicate is missed regardless of which branch the
+value comes from.
+
This module pins both the symptom and the root cause:
- :class:`LinkType209ConstResolutionTests` -- against the committed, generated
@@ -34,6 +53,16 @@
CI while still exercising the exact code path a real crawl runs. A
regeneration that drops the fix reintroduces the wrong order here before it
ever reaches the committed const file.
+- :class:`LinkTypeGeneratorValueAwareTests` -- issue #852 item 1: a row whose
+ value has no duplicate must never be sunk merely for mentioning "legacy",
+ and the canonical member of a genuine duplicate pair must be the current
+ one regardless of which table position it occupies.
+- :class:`LinkTypeGeneratorRangeSinkIsolationTests` -- issue #852 item 2: a
+ single range row's wording must never sink its whole expansion.
+- :class:`LinkTypeGeneratorCrossRangeDuplicateTests` -- the #852 cross-review
+ regression: a single-value row's legacy wording must still be honoured when
+ the value it duplicates belongs to a range row, not another single-value
+ one.
"""
from __future__ import annotations
@@ -153,8 +182,8 @@ def test_current_name_is_emitted_before_the_legacy_one(self) -> None:
self.assertEqual(len(enum), 2, f'expected exactly 2 rendered members, got: {enum!r}')
- first_name = enum[0].splitlines()[1].split(' = ', 1)[0].strip()
- second_name = enum[1].splitlines()[1].split(' = ', 1)[0].strip()
+ first_name = enum[0].splitlines()[-1].split(' = ', 1)[0].strip()
+ second_name = enum[1].splitlines()[-1].split(' = ', 1)[0].strip()
self.assertEqual(first_name, 'I2C_LINUX',
f'the current name must be emitted first so it is '
@@ -172,5 +201,433 @@ def test_both_rows_still_render_with_their_own_comment(self) -> None:
self.assertIn('Legacy names (do not use) for Linux I2C below.', rendered)
+#: Two rows for two *different*, otherwise-unrelated values -- neither shares
+#: its value with anything else in the (2-row) table. ``FOO_LEGACY_WORDED``'s
+#: notes mention "legacy" only incidentally, about a different, unnamed old
+#: code -- not because ``FOO_LEGACY_WORDED`` itself is a deprecated alias of
+#: ``BAR_PLAIN``. This is GitHub issue #852 item 1's own example: "a future
+#: current row's notes say something like 'supersedes the legacy DLT_FOO'".
+NON_DUPLICATE_LEGACY_WORDING_TABLE_HTML = """
+
+| header row, skipped by request() |
+
+| LINKTYPE_FOO_LEGACY_WORDED |
+500 |
+DLT_FOO_LEGACY_WORDED |
+
+Supersedes the legacy DLT_BAZ encoding used by very old capture tools.
+ |
+
+
+| LINKTYPE_BAR_PLAIN |
+501 |
+DLT_BAR_PLAIN |
+
+An ordinary, unrelated link type.
+ |
+
+
+"""
+
+#: A genuine duplicate pair for a third value, worded exactly the way 209 is
+#: (the legacy row's own notes say "legacy", the current row's do not).
+#: ``{0}`` and ``{1}`` are substituted with the two rows' HTML in each order,
+#: so the same pair can be fed through :meth:`process` current-first and
+#: legacy-first without duplicating the row markup itself.
+_QUX_CURRENT_ROW_HTML = """
+
+| LINKTYPE_QUX_CURRENT |
+600 |
+DLT_QUX_CURRENT |
+
+The current name for this link type.
+ |
+
+"""
+
+_QUX_OLD_ROW_HTML = """
+
+| LINKTYPE_QUX_OLD |
+600 |
+DLT_QUX_OLD |
+
+Legacy name (do not use) for the same link type.
+ |
+
+"""
+
+
+def _duplicate_pair_table_html(first: str, second: str) -> str:
+ """Wrap two already-built rows in the table markup :meth:`process` expects."""
+ return f"""
+
+| header row, skipped by request() |
+{first}
+{second}
+
+"""
+
+
+@unittest.skipUnless(HAS_VENDOR_DEPS, 'requests, bs4 and/or html5lib not installed')
+class LinkTypeGeneratorValueAwareTests(unittest.TestCase):
+ """Item 1 of GitHub issue #852: the sink predicate must be value-aware.
+
+ :meth:`~pcapkit.vendor.reg.linktype.LinkType.process`'s sink rule
+ (added by #848) tested a row's notes for the word "legacy" alone, with no
+ check that the row's value is actually claimed by another row. Issue
+ #852 pointed out this can invert in the *other* direction: a future
+ *current* row whose notes happen to mention "legacy" about some unrelated
+ code gets sunk even though nothing else shares its value -- silently
+ reordering the generated file's *emission order* for a value that was
+ never ambiguous in the first place. (It also cannot, by itself, save a
+ row from being wrongly sunk when tcpdump's wording is attached to the
+ wrong member of a *genuine* duplicate pair; that residual risk is
+ documented at the ``dup_values`` computation in
+ :meth:`~pcapkit.vendor.reg.linktype.LinkType.process` rather than
+ claimed as solved here.)
+ """
+
+ def setUp(self) -> None:
+ purge_modules(['pcapkit'])
+
+ @staticmethod
+ def _process(html: str) -> 'list[str]':
+ import bs4
+
+ from pcapkit.vendor.reg.linktype import LinkType
+
+ soup = bs4.BeautifulSoup(html, 'html5lib')
+ rows = soup.select('table.linktypedlt tr')[1:]
+
+ inst = object.__new__(LinkType)
+ enum, _miss = inst.process(rows)
+ return enum
+
+ def test_non_duplicated_legacy_wording_does_not_reorder(self) -> None:
+ """A row cannot be sunk unless another row actually shares its value.
+
+ Fails against the value-blind, word-only predicate on ``main``: that
+ predicate sinks ``FOO_LEGACY_WORDED`` purely because its notes
+ contain "legacy", swapping it after ``BAR_PLAIN`` in the returned
+ list even though value 500 has no duplicate anywhere in this table.
+ The value-aware predicate leaves both rows in table order, because
+ neither of their values is duplicated.
+ """
+ enum = self._process(NON_DUPLICATE_LEGACY_WORDING_TABLE_HTML)
+
+ self.assertEqual(len(enum), 2, f'expected exactly 2 rendered members, got: {enum!r}')
+
+ first_name = enum[0].splitlines()[-1].split(' = ', 1)[0].strip()
+ second_name = enum[1].splitlines()[-1].split(' = ', 1)[0].strip()
+
+ self.assertEqual(
+ [first_name, second_name], ['FOO_LEGACY_WORDED', 'BAR_PLAIN'],
+ f'a row whose value has no duplicate must never be sunk, regardless '
+ f'of its own wording; got order {[first_name, second_name]!r} -- see '
+ f'GitHub issue #852 item 1'
+ )
+
+ def test_duplicate_pair_canonical_member_is_order_independent(self) -> None:
+ """The *current* row must win regardless of which table position it is in.
+
+ Unlike an earlier version of this test, which fed only the
+ current-first ordering and left order-independence resting entirely
+ on :class:`LinkTypeGeneratorLegacyOrderingTests`'s separate,
+ legacy-first 209 fixture, this drives the *same* duplicate pair
+ through :meth:`process` in **both** orders in one test -- so the name
+ is earned here rather than borrowed from a sibling class. A predicate
+ that regressed to counting only a positional prefix (e.g. "has this
+ value already been appended to ``enum``") would still pass a
+ single-order test like the old one, because that predicate is only
+ wrong for the *first* occurrence of a value it has not seen yet --
+ exactly what running the reversed order below would catch.
+ """
+ current_first = self._process(_duplicate_pair_table_html(_QUX_CURRENT_ROW_HTML, _QUX_OLD_ROW_HTML))
+ legacy_first = self._process(_duplicate_pair_table_html(_QUX_OLD_ROW_HTML, _QUX_CURRENT_ROW_HTML))
+
+ for label, enum in (('current-first', current_first), ('legacy-first', legacy_first)):
+ self.assertEqual(len(enum), 2, f'[{label}] expected exactly 2 rendered members, got: {enum!r}')
+
+ first_name = enum[0].splitlines()[-1].split(' = ', 1)[0].strip()
+ second_name = enum[1].splitlines()[-1].split(' = ', 1)[0].strip()
+
+ self.assertEqual(
+ [first_name, second_name], ['QUX_CURRENT', 'QUX_OLD'],
+ f'[{label}] the current member of a duplicated value must be '
+ f'canonical regardless of table order; got {[first_name, second_name]!r}'
+ )
+
+
+#: A range row (``USER0``..``USER3``, values 900-903) whose notes mention
+#: "legacy", followed by an unrelated plain row. The plain row is what makes
+#: mis-sinking the range *observable* through ordering: if the range branch
+#: shared the per-row ``sink`` the single-value branch computes, this single
+#: row's wording would route all four expanded members through ``legacy``
+#: at once, and they would come out *after* ``PLAIN_ROW`` instead of before
+#: it (table order).
+LEGACY_WORDED_RANGE_TABLE_HTML = """
+
+| header row, skipped by request() |
+
+| LINKTYPE_USER0 |
+900–903 |
+DLT_USER0 |
+
+Legacy user-defined range, retained for compatibility.
+ |
+
+
+| LINKTYPE_PLAIN_ROW |
+999 |
+DLT_PLAIN_ROW |
+
+An ordinary, unrelated link type.
+ |
+
+
+"""
+
+
+@unittest.skipUnless(HAS_VENDOR_DEPS, 'requests, bs4 and/or html5lib not installed')
+class LinkTypeGeneratorRangeSinkIsolationTests(unittest.TestCase):
+ """Item 2 of GitHub issue #852: a range row must never sink its whole expansion.
+
+ The ``ValueError`` branch that expands a ``USER0``-style range used to
+ reuse the same per-row ``sink`` the single-value branch computed, so one
+ range row whose notes mentioned "legacy" would have sunk every expanded
+ member at once. :meth:`~pcapkit.vendor.reg.linktype.LinkType.process`
+ now always appends range-expanded members straight into ``enum``,
+ unconditionally, and this pins that deliberately.
+ """
+
+ def setUp(self) -> None:
+ purge_modules(['pcapkit'])
+
+ def test_legacy_worded_range_row_still_lands_entirely_in_enum(self) -> None:
+ import bs4
+
+ from pcapkit.vendor.reg.linktype import LinkType
+
+ soup = bs4.BeautifulSoup(LEGACY_WORDED_RANGE_TABLE_HTML, 'html5lib')
+ rows = soup.select('table.linktypedlt tr')[1:]
+
+ inst = object.__new__(LinkType)
+ enum, _miss = inst.process(rows)
+
+ self.assertEqual(len(enum), 5, f'expected 4 expanded USER0..USER3 members plus PLAIN_ROW, got: {enum!r}')
+
+ names = [element.splitlines()[-1].split(' = ', 1)[0].strip() for element in enum]
+ self.assertEqual(
+ names, ['USER0', 'USER1', 'USER2', 'USER3', 'PLAIN_ROW'],
+ f'a range row whose notes mention "legacy" must emit every expanded '
+ f'member into `enum`, in table order, never sunk into `legacy` as a '
+ f'block after a later row -- got {names!r}'
+ )
+
+
+#: A single-value legacy row for 900, followed by a range row that expands to
+#: include 900 among its members (``USER0``..``USER3`` == 900-903). This is
+#: the #852 cross-review's own repro: value 900 is a genuine duplicate, but
+#: only because one of its two occurrences comes from a *range* expansion --
+#: a duplicate-value pre-scan that counts only single-value rows never sees
+#: it, so ``OLDUSER``'s "legacy" wording is silently ignored and the
+#: deprecated name stays canonical for 900.
+CROSS_RANGE_DUPLICATE_TABLE_HTML = """
+
+| header row, skipped by request() |
+
+| LINKTYPE_OLDUSER |
+900 |
+DLT_OLDUSER |
+
+Legacy name (do not use) for the range below.
+ |
+
+
+| LINKTYPE_USER0 |
+900–903 |
+DLT_USER0 |
+
+An ordinary user-defined range.
+ |
+
+
+"""
+
+
+@unittest.skipUnless(HAS_VENDOR_DEPS, 'requests, bs4 and/or html5lib not installed')
+class LinkTypeGeneratorCrossRangeDuplicateTests(unittest.TestCase):
+ """The #852 cross-review's regression: a value duplicated *across* a range boundary.
+
+ The first cut of the duplicate-value pre-scan built ``dup_values`` from
+ only the single-value rows (filtering the notes column's number on
+ ``str.isdigit()``), so a range row's expanded values never entered the
+ count. A single-value row sharing one of *those* values -- ``900``,
+ shared here with the first member of a ``USER0``..``USER3`` range
+ covering 900-903 -- was therefore never recognised as a duplicate, and
+ its correctly-worded "legacy" notes were silently ignored: it landed in
+ ``enum`` instead of ``legacy``, ahead of the range it duplicates, making
+ the deprecated ``OLDUSER`` canonical for value 900 instead of ``USER0``.
+ That is exactly the failure mode #844 and #852 both exist to close,
+ reopened by narrowing the predicate. :meth:`process`'s pre-scan now
+ expands ``a–b`` ranges the same way the main loop does before counting,
+ so this duplicate is caught regardless of which branch either occurrence
+ comes from.
+ """
+
+ def setUp(self) -> None:
+ purge_modules(['pcapkit'])
+
+ def test_legacy_row_duplicating_a_range_member_is_sunk(self) -> None:
+ import bs4
+
+ from pcapkit.vendor.reg.linktype import LinkType
+
+ soup = bs4.BeautifulSoup(CROSS_RANGE_DUPLICATE_TABLE_HTML, 'html5lib')
+ rows = soup.select('table.linktypedlt tr')[1:]
+
+ inst = object.__new__(LinkType)
+ enum, _miss = inst.process(rows)
+
+ self.assertEqual(len(enum), 5, f'expected 4 expanded USER0..USER3 members plus OLDUSER, got: {enum!r}')
+
+ names = [element.splitlines()[-1].split(' = ', 1)[0].strip() for element in enum]
+ self.assertEqual(
+ names, ['USER0', 'USER1', 'USER2', 'USER3', 'OLDUSER'],
+ f'OLDUSER duplicates the range\'s own USER0 (both value 900) and is '
+ f'correctly worded "legacy", so it must be sunk after the range -- '
+ f'making USER0, not OLDUSER, canonical for value 900. Got {names!r}'
+ )
+
+ def test_malformed_en_dash_range_still_raises_loudly(self) -> None:
+ """A malformed registry row must still fail loudly -- via the pre-existing main loop.
+
+ A number column with more than one en dash (``900–903–905``) cannot
+ ``start, stop = map(int, ...)`` -- too many values to unpack. The
+ pre-scan's local ``_expand`` swallows exactly that ``ValueError`` and
+ contributes no counted values for the row, quietly -- but that is
+ *not* what makes this test pass. This is a regression guard on the
+ main loop's own, pre-existing ``start, stop = map(int,
+ temp.split('–'))``, which hits the identical unpack error a few
+ lines later and still raises, uncaught, out of :meth:`process`.
+ That statement predates #852 entirely, so this assertion passes
+ identically on ``main``, on this PR's first commit (before
+ ``_expand`` existed), and here: nothing about the pre-scan is what
+ is being pinned.
+ """
+ import bs4
+
+ from pcapkit.vendor.reg.linktype import LinkType
+
+ html = """
+
+| header row, skipped by request() |
+
+| LINKTYPE_MALFORMED |
+900–903–905 |
+DLT_MALFORMED |
+
+A malformed range with two en dashes instead of one.
+ |
+
+
+"""
+ soup = bs4.BeautifulSoup(html, 'html5lib')
+ rows = soup.select('table.linktypedlt tr')[1:]
+
+ inst = object.__new__(LinkType)
+ with self.assertRaises(ValueError):
+ inst.process(rows)
+
+
+def _signed_or_underscored_duplicate_table_html(legacy_number: str, current_number: str) -> str:
+ """A legacy/current duplicate pair whose shared value is spelled ``legacy_number``
+ and ``current_number`` -- both parse to the same :func:`int`, but neither need be
+ a plain unsigned decimal.
+ """
+ return f"""
+
+| header row, skipped by request() |
+
+| LINKTYPE_OLD |
+{legacy_number} |
+DLT_OLD |
+
+Legacy name (do not use) for this link type.
+ |
+
+
+| LINKTYPE_CUR |
+{current_number} |
+DLT_CUR |
+
+The current name for this link type.
+ |
+
+
+"""
+
+
+@unittest.skipUnless(HAS_VENDOR_DEPS, 'requests, bs4 and/or html5lib not installed')
+class LinkTypeGeneratorSignedOrUnderscoredValueTests(unittest.TestCase):
+ """The second #852 cross-review's regression: ``_expand``'s recognition set was too narrow.
+
+ The pre-scan's single-value branch tested ``str.isdigit()``, which is
+ narrower than the main loop's own acceptance test, ``int(temp)`` --
+ ``int()`` also accepts a leading sign (``+209``, ``-5``) and PEP 515
+ underscore grouping (``2_09``). A value the main loop accepts but
+ ``_expand`` does not is invisible to ``dup_values``, so a correctly
+ "legacy"-worded row sharing that value silently stops being sunk -- the
+ same class of under-count the range-expansion fix above closes, just for
+ a different reason ``_expand`` can fail to recognise a value.
+
+ Not reachable against tcpdump's live table today: every one of the 220
+ committed members matches a plain unsigned decimal, so this is a latent
+ hole rather than a live defect, exactly as #844 and #852's original
+ word-only rule was. :meth:`process`'s ``_expand`` now tries ``int(temp)``
+ directly, before the en-dash range check, so it recognises the same
+ values the main loop does.
+ """
+
+ def setUp(self) -> None:
+ purge_modules(['pcapkit'])
+
+ @staticmethod
+ def _process(html: str) -> 'list[str]':
+ import bs4
+
+ from pcapkit.vendor.reg.linktype import LinkType
+
+ soup = bs4.BeautifulSoup(html, 'html5lib')
+ rows = soup.select('table.linktypedlt tr')[1:]
+
+ inst = object.__new__(LinkType)
+ enum, _miss = inst.process(rows)
+ return enum
+
+ def test_signed_value_pair_resolves_to_the_current_member(self) -> None:
+ enum = self._process(_signed_or_underscored_duplicate_table_html('+750', '750'))
+
+ self.assertEqual(len(enum), 2, f'expected exactly 2 rendered members, got: {enum!r}')
+ names = [element.splitlines()[-1].split(' = ', 1)[0].strip() for element in enum]
+ self.assertEqual(
+ names, ['CUR', 'OLD'],
+ f'a signed value ("+750") must still be recognised as a duplicate of '
+ f'its unsigned twin ("750"), so OLD is sunk after CUR -- got {names!r}'
+ )
+
+ def test_underscored_value_pair_resolves_to_the_current_member(self) -> None:
+ enum = self._process(_signed_or_underscored_duplicate_table_html('1_000', '1000'))
+
+ self.assertEqual(len(enum), 2, f'expected exactly 2 rendered members, got: {enum!r}')
+ names = [element.splitlines()[-1].split(' = ', 1)[0].strip() for element in enum]
+ self.assertEqual(
+ names, ['CUR', 'OLD'],
+ f'an underscore-grouped value ("1_000") must still be recognised as a '
+ f'duplicate of its plain twin ("1000"), so OLD is sunk after CUR -- '
+ f'got {names!r}'
+ )
+
+
if __name__ == '__main__':
unittest.main()