Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion sigmf/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
# SPDX-License-Identifier: LGPL-3.0-or-later

# version of this python module
__version__ = "1.13.0"
__version__ = "1.14.0"
# matching version of the SigMF specification
__specification__ = "1.2.6"

Expand Down
20 changes: 16 additions & 4 deletions sigmf/archive.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@
import zipfile
from pathlib import Path

from . import keys
from .error import SigMFFileError, SigMFFileExistsError
from .keys import (
SIGMF_ARCHIVE_EXT,
Expand Down Expand Up @@ -139,7 +140,11 @@ def __init__(self, sigmffile, name=None, fileobj=None, compression=None, overwri
# prepare temp files with metadata and data
tmpdir = Path(tempfile.mkdtemp())
meta_path = tmpdir / (arcname + SIGMF_METADATA_EXT)
data_path = tmpdir / (arcname + SIGMF_DATASET_EXT)
# for non-conforming datasets, keep the original file name so the
# archive contents match the `core:dataset` reference in the metadata
dataset_fn = self.sigmffile.get_global_field(keys.DATASET_KEY)
is_ncd = dataset_fn is not None
data_path = tmpdir / (dataset_fn if is_ncd else arcname + SIGMF_DATASET_EXT)

with open(meta_path, "w") as handle:
self.sigmffile.dump(handle)
Expand Down Expand Up @@ -170,7 +175,7 @@ def _write_zip(self, fileobj, arcname, tmpdir, meta_path, data_path):
"""Write archive as zip."""
with zipfile.ZipFile(fileobj, mode="w", compression=zipfile.ZIP_DEFLATED) as zf:
# add data file first (matches tar convention for faster metadata updates)
zf.write(data_path, arcname=f"{arcname}/{arcname}{SIGMF_DATASET_EXT}")
zf.write(data_path, arcname=f"{arcname}/{data_path.name}")
zf.write(meta_path, arcname=f"{arcname}/{arcname}{SIGMF_METADATA_EXT}")

@staticmethod
Expand All @@ -183,8 +188,15 @@ def chmod(tarinfo: tarfile.TarInfo):
return tarinfo

def _ensure_data_file_set(self):
if not self.sigmffile.data_file and not isinstance(self.sigmffile.data_buffer, io.BytesIO):
raise SigMFFileError("No data file in SigMFFile; use `set_data_file` before archiving.")
"""Raise if the SigMFFile has no dataset to archive."""
if self.sigmffile.data_file or isinstance(self.sigmffile.data_buffer, io.BytesIO):
return
if self.sigmffile.get_global_field(keys.METADATA_ONLY_KEY, False):
raise SigMFFileError(
"Cannot archive a metadata-only SigMF file because it has no dataset; "
"write a `.sigmf-meta` file instead."
)
raise SigMFFileError("No data file in SigMFFile; use `set_data_file` before archiving.")

def _validate(self):
self.sigmffile.validate()
Expand Down
135 changes: 87 additions & 48 deletions sigmf/archivereader.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
"""Access SigMF archives without extracting them."""

import io
import json
import tarfile
import zipfile
from pathlib import Path
Expand Down Expand Up @@ -97,29 +98,55 @@ def __init__(self, name=None, skip_checksum=False, map_readonly=True, archive_bu
else:
raise ValueError("Either `name` or `archive_buffer` must be not None.")

@staticmethod
def _get_ncd_dataset_name(json_contents):
"""Return the Non-Conforming Dataset filename referenced by core:dataset, or None."""
if json_contents is None:
return None
try:
return json.loads(json_contents)[SigMFFile.GLOBAL_KEY].get(keys.DATASET_KEY)
except (ValueError, KeyError, TypeError):
return None

@classmethod
def _resolve_data_member(cls, member_names, json_contents):
"""
Return the name of the dataset member within an archive.

Prefers a conforming `.sigmf-data` member; otherwise falls back to the
Non-Conforming Dataset file named by `core:dataset`, which is stored in
the archive under its original filename.
"""
for name in member_names:
if name.endswith(SIGMF_DATASET_EXT):
return name
dataset_fn = cls._get_ncd_dataset_name(json_contents)
if dataset_fn:
for name in member_names:
if name.endswith(dataset_fn):
return name
return None

def _read_tar_obj(self, tar_obj):
"""Extract metadata and data from an open tar object."""
json_contents = None
data_buffer = None
data_size_bytes = None

for memb in tar_obj.getmembers():
if memb.isdir():
continue
elif memb.isfile():
if memb.name.endswith(SIGMF_METADATA_EXT):
with tar_obj.extractfile(memb) as fid:
json_contents = fid.read()
elif memb.name.endswith(SIGMF_DATASET_EXT):
data_size_bytes = memb.size
with tar_obj.extractfile(memb) as fid:
data_buffer = io.BytesIO(fid.read())
members = [memb for memb in tar_obj.getmembers() if memb.isfile()]

json_contents = None
for memb in members:
if memb.name.endswith(SIGMF_METADATA_EXT):
with tar_obj.extractfile(memb) as fid:
json_contents = fid.read()
if json_contents is None:
raise SigMFFileError("No .sigmf-meta file found in archive!")
if data_buffer is None:

data_name = self._resolve_data_member([m.name for m in members], json_contents)
if data_name is None:
raise SigMFFileError("No .sigmf-data file found in archive!")
return json_contents, data_buffer, data_size_bytes

data_member = next(m for m in members if m.name == data_name)
with tar_obj.extractfile(data_member) as fid:
data_buffer = io.BytesIO(fid.read())
return json_contents, data_buffer, data_member.size

def _read_tar(self, path):
"""Read a tar archive (possibly compressed) from disk."""
Expand All @@ -140,32 +167,44 @@ def _read_zip_fileobj(self, fileobj):

def _read_zip_obj(self, zf):
"""Extract metadata and data from an open ZipFile object."""
json_contents = None
data_buffer = None
data_size_bytes = None

for member_name in zf.namelist():
if member_name.endswith(SIGMF_METADATA_EXT):
json_contents = zf.read(member_name)
elif member_name.endswith(SIGMF_DATASET_EXT):
raw = zf.read(member_name)
data_size_bytes = len(raw)
data_buffer = io.BytesIO(raw)
names = zf.namelist()

json_contents = None
for name in names:
if name.endswith(SIGMF_METADATA_EXT):
json_contents = zf.read(name)
if json_contents is None:
raise SigMFFileError("No .sigmf-meta file found in archive!")
if data_buffer is None:

data_name = self._resolve_data_member(names, json_contents)
if data_name is None:
raise SigMFFileError("No .sigmf-data file found in archive!")
return json_contents, data_buffer, data_size_bytes

raw = zf.read(data_name)
return json_contents, io.BytesIO(raw), len(raw)

def _ncd_byte_bounds(self, data_size_bytes):
"""
Return the (offset, size) of the sample bytes within a stored dataset member.

For Non-Conforming Datasets the stored file includes its own container
header and trailer, described by `core:header_bytes` and
`core:trailing_bytes`. For conforming datasets both are zero.
"""
offset = self.sigmffile._get_ncd_offset()
trailing = self.sigmffile.get_global_field(keys.TRAILING_BYTES_KEY, 0)
return offset, data_size_bytes - offset - trailing

def _init_from_buffer(self, json_contents, data_buffer, data_size_bytes, skip_checksum, map_readonly, autoscale):
"""Initialize sigmffile from in-memory data."""
self.sigmffile = SigMFFile(metadata=json_contents, autoscale=autoscale)
self.sigmffile.validate()
offset, size_bytes = self._ncd_byte_bounds(data_size_bytes)
self.sigmffile.set_data_file(
data_buffer=data_buffer,
skip_checksum=skip_checksum,
size_bytes=data_size_bytes,
offset=offset,
size_bytes=size_bytes,
map_readonly=map_readonly,
)
self.ndim = self.sigmffile.ndim
Expand All @@ -174,27 +213,26 @@ def _init_from_buffer(self, json_contents, data_buffer, data_size_bytes, skip_ch
def _init_from_tar_memmap(self, path, skip_checksum, map_readonly, autoscale):
"""Initialize sigmffile with memmap into uncompressed tar."""
tar_obj = tarfile.open(path)
json_contents = None
data_offset = None
data_size_bytes = None
try:
members = [memb for memb in tar_obj.getmembers() if memb.isfile()]

for memb in tar_obj.getmembers():
if memb.isdir():
continue
elif memb.isfile():
json_contents = None
for memb in members:
if memb.name.endswith(SIGMF_METADATA_EXT):
with tar_obj.extractfile(memb) as fid:
json_contents = fid.read()
elif memb.name.endswith(SIGMF_DATASET_EXT):
data_offset = memb.offset_data
data_size_bytes = memb.size
if json_contents is None:
raise SigMFFileError("No .sigmf-meta file found in archive!")

tar_obj.close()
data_name = self._resolve_data_member([m.name for m in members], json_contents)
if data_name is None:
raise SigMFFileError("No .sigmf-data file found in archive!")

if json_contents is None:
raise SigMFFileError("No .sigmf-meta file found in archive!")
if data_offset is None:
raise SigMFFileError("No .sigmf-data file found in archive!")
data_member = next(m for m in members if m.name == data_name)
data_offset = data_member.offset_data
data_size_bytes = data_member.size
finally:
tar_obj.close()

self.sigmffile = SigMFFile(metadata=json_contents, autoscale=autoscale)
self.sigmffile.validate()
Expand All @@ -208,11 +246,12 @@ def _init_from_tar_memmap(self, path, skip_checksum, map_readonly, autoscale):
self.sigmffile.set_global_field(keys.SHA512_KEY, data_hash)

# memmap directly into the tar file at the data offset
offset, size_bytes = self._ncd_byte_bounds(data_size_bytes)
self.sigmffile.set_data_file(
data_file=path,
skip_checksum=True,
offset=data_offset,
size_bytes=data_size_bytes,
offset=data_offset + offset,
size_bytes=size_bytes,
map_readonly=map_readonly,
)
# set_data_file sets DATASET_KEY for non-.sigmf-data files (NCD),
Expand Down
3 changes: 0 additions & 3 deletions sigmf/convert/blue.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,6 @@

import base64
import getpass
import io
import logging
import struct
import tempfile
Expand Down Expand Up @@ -716,7 +715,6 @@ def construct_sigmf(
global_info=global_info,
skip_checksum=True,
)
meta.data_buffer = io.BytesIO()
else:
meta = SigMFFile(
data_file=filenames["data_fn"],
Expand Down Expand Up @@ -780,7 +778,6 @@ def construct_sigmf_ncd(
# create NCD metadata-only SigMF pointing to original file
meta = SigMFFile(global_info=global_info, skip_checksum=True)
meta.set_data_file(data_file=blue_path, offset=header_bytes, skip_checksum=True, size_bytes=data_bytes)
meta.data_buffer = io.BytesIO()
meta.add_capture(0, metadata=capture_info)
log.debug("created NCD SigMF: %r", meta)

Expand Down
2 changes: 0 additions & 2 deletions sigmf/convert/signalhound.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,6 @@
"""Signal Hound Converter"""

import getpass
import io
import logging
import tempfile
from datetime import datetime, timedelta, timezone
Expand Down Expand Up @@ -398,7 +397,6 @@ def signalhound_to_sigmf(
# create metadata-only SigMF for NCD pointing to original file
meta = SigMFFile(global_info=global_info)
meta.set_data_file(data_file=data_file_path, offset=0)
meta.data_buffer = io.BytesIO()
meta.add_capture(0, metadata=capture_info)
_add_annotations(meta, annotations)

Expand Down
2 changes: 0 additions & 2 deletions sigmf/convert/wav.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,6 @@

"""converter for wav containers"""

import io
import logging
import tempfile
import wave
Expand Down Expand Up @@ -165,7 +164,6 @@ def wav_to_sigmf(
# create metadata-only SigMF for NCD pointing to original file
meta = SigMFFile(global_info=global_info)
meta.set_data_file(data_file=wav_path, offset=header_bytes)
meta.data_buffer = io.BytesIO()
meta.add_capture(0, metadata=capture_info)

# write metadata file if output path specified
Expand Down
45 changes: 35 additions & 10 deletions sigmf/sigmffile.py
Original file line number Diff line number Diff line change
Expand Up @@ -553,6 +553,11 @@ def add_capture(self, start_index, metadata=None):
capture_list,
key=lambda item: item[keys.SAMPLE_START_KEY],
)
# capture `header_bytes` changes how many bytes of the dataset are sample
# data, so a cached sample count must be recomputed when a dataset is set
if self.data_file is not None or self.data_buffer is not None:
self._check_byte_budget(sum([c.get(keys.HEADER_BYTES_KEY, 0) for c in self.get_captures()]))
self._count_samples()

def get_captures(self):
"""
Expand Down Expand Up @@ -695,6 +700,26 @@ def get_sample_size(self):
"""
return dtype_info(self.datatype)["sample_size"]

def _get_total_byte_count(self) -> int:
"""Return the total size in bytes of the dataset source (file or buffer)."""
if self.data_file is not None:
return self.data_file.stat().st_size
if self.data_buffer is not None:
return len(self.data_buffer.getbuffer())
return 0

def _check_byte_budget(self, skipped_bytes: int) -> None:
"""Raise if the dataset is smaller than the header/trailing bytes its metadata says to skip."""
if self.data_file is None and self.data_buffer is None:
return
trailing_bytes = self.get_global_field(keys.TRAILING_BYTES_KEY, 0)
total_bytes = self._get_total_byte_count()
if skipped_bytes + trailing_bytes > total_bytes:
raise SigMFFileError(
f"Dataset is {total_bytes} bytes but its metadata skips {skipped_bytes} header "
f"and {trailing_bytes} trailing bytes."
)

def _count_samples(self):
"""
Count, set, and return the total number of samples in the data file.
Expand All @@ -713,13 +738,8 @@ def _count_samples(self):
else:
# calculate from file size, subtracting header and trailing bytes
header_bytes = sum([c.get(keys.HEADER_BYTES_KEY, 0) for c in self.get_captures()])
if self.data_file is not None:
file_bytes = self.data_file.stat().st_size
elif self.data_buffer is not None:
file_bytes = len(self.data_buffer.getbuffer())
else:
file_bytes = 0
sample_bytes = file_bytes - self.get_global_field(keys.TRAILING_BYTES_KEY, 0) - header_bytes
trailing_bytes = self.get_global_field(keys.TRAILING_BYTES_KEY, 0)
sample_bytes = self._get_total_byte_count() - trailing_bytes - header_bytes

total_sample_size = self.get_sample_size() * self.num_channels
sample_count, remainder = divmod(sample_bytes, total_sample_size)
Expand Down Expand Up @@ -783,6 +803,7 @@ def set_data_file(
self.data_buffer = data_buffer
self.data_offset = offset
self.data_size_bytes = size_bytes
self._check_byte_budget(offset)
self._count_samples()

dtype = dtype_info(self.get_global_field(keys.DATATYPE_KEY))
Expand All @@ -792,7 +813,12 @@ def set_data_file(

complex_int_separates = dtype["is_complex"] and dtype["is_fixedpoint"]
mapped_dtype_size = dtype["component_size"] if complex_int_separates else dtype["sample_size"]
mapped_length = None if size_bytes is None else size_bytes // mapped_dtype_size
# bound the map to the sample data so that Non-Conforming Dataset header/trailing bytes are not exposed as samples
mapped_bytes = size_bytes
if mapped_bytes is None and (self.data_file is not None or self.data_buffer is not None):
trailing_bytes = self.get_global_field(keys.TRAILING_BYTES_KEY, 0)
mapped_bytes = self._get_total_byte_count() - offset - trailing_bytes
mapped_length = None if mapped_bytes is None else mapped_bytes // mapped_dtype_size
mapped_reshape = (-1,) # we can't use -1 in mapped_length ...
if num_channels > 1:
mapped_reshape = mapped_reshape + (num_channels,)
Expand Down Expand Up @@ -1010,11 +1036,10 @@ def _read_datafile(self, first_byte, nitems):
# account for data_offset when seeking (important for NCDs)
seek_position = first_byte + getattr(self, "data_offset", 0)
fp.seek(seek_position, 0)

data = np.fromfile(fp, dtype=data_type_in, count=nitems)
elif self.data_buffer is not None:
# handle offset for data_buffer like we do for data_file
buffer_data = self.data_buffer.getbuffer()[first_byte:]
buffer_data = self.data_buffer.getbuffer()[first_byte + getattr(self, "data_offset", 0) :]
data = np.frombuffer(buffer_data, dtype=data_type_in, count=nitems)
else:
data = self._memmap
Expand Down
Loading