diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 7a6237c..8b5f329 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -59,10 +59,10 @@ jobs:
steps:
- name: Install optional tools macOS
if: runner.os == 'macOS' && matrix.optional-deps
- run: brew install pigz pbzip2 isa-l zstd
+ run: brew install pigz pbzip2 isa-l zstd lz4
- name: Install optional tools Linux
if: runner.os == 'Linux' && matrix.optional-deps
- run: sudo apt-get install pigz pbzip2 isal zstd
+ run: sudo apt-get install pigz pbzip2 isal zstd lz4
- name: Remove xz
if: runner.os == 'Linux' && !matrix.optional-deps
run: while which xz; do sudo rm $(which xz); done
@@ -84,6 +84,9 @@ jobs:
- name: Test with zstandard
if: matrix.with-zstandard
run: tox run -e zstd
+ - name: Test with LZ4 bindings
+ if: matrix.os == 'ubuntu-latest' && matrix.python-version == '3.10' && matrix.optional-deps && matrix.with-libs
+ run: tox run -e lz4
- name: Upload coverage report
uses: codecov/codecov-action@v3
diff --git a/README.rst b/README.rst
index 1440d31..3690fd3 100644
--- a/README.rst
+++ b/README.rst
@@ -27,6 +27,7 @@ Supported compression formats are:
- bzip2 (``.bz2``)
- xz (``.xz``)
- Zstandard (``.zst``)
+- LZ4 (``.lz4``) (optional)
Example usage
@@ -71,7 +72,7 @@ The function opens the file using a function suitable for the detected
file format and returns an open file-like object.
When writing, the file format is chosen based on the file name extension:
-``.gz``, ``.bz2``, ``.xz``, ``.zst``. This can be overriden with ``format``.
+``.gz``, ``.bz2``, ``.xz``, ``.zst``, ``.lz4``. This can be overriden with ``format``.
If the extension is not recognized, no compression is used.
When reading and a file name extension is available, the format is detected
@@ -97,15 +98,15 @@ preferred locale encoding.
``encoding``, ``errors`` and ``newline`` are only used when opening a file in text mode.
**compresslevel**:
-The compression level for writing to gzip, xz and Zstandard files.
+The compression level for writing to gzip, xz, Zstandard and LZ4 files.
If set to None, a default depending on the format is used:
-gzip: 1, xz: 6, Zstandard: 3.
+gzip: 1, xz: 6, Zstandard: 3, LZ4: 0.
This parameter is ignored for other compression formats.
**format**:
Override the autodetection of the input or output format.
-Possible values are: ``"gz"``, ``"xz"``, ``"bz2"``, ``"zst"``.
+Possible values are: ``"gz"``, ``"xz"``, ``"bz2"``, ``"zst"``, ``"lz4"``.
**threads**:
Set the number of additional threads spawned for compression or decompression.
@@ -138,6 +139,11 @@ built-in support for multithreaded compression.
For bz2 files, `pbzip2 (parallel bzip2) `_ is used.
+For LZ4 files, install the `python-lz4 `_
+package with ``pip install xopen[lz4]``. The ``lz4`` command-line program can
+also read and write LZ4 files when ``threads`` is not 0. The Python package is
+required for ``threads=0``. On PyPy, use the command-line program.
+
``xopen`` falls back to Python’s built-in functions
(``gzip.open``, ``lzma.open``, ``bz2.open``)
if none of the other methods can be used.
diff --git a/pyproject.toml b/pyproject.toml
index ec0f923..71a6825 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -28,11 +28,15 @@ homepage = "https://github.com/pycompression/xopen/"
[project.optional-dependencies]
dev = ["pytest"]
+lz4 = ['lz4>=4.4.5; platform_python_implementation != "PyPy"']
zstd = [] # Leave this in here for backwards compatibility (Zstandard support used to be optional)
[tool.setuptools_scm]
write_to = "src/xopen/_version.py"
+[tool.ruff.lint]
+select = ["E4", "E7", "E9", "F"]
+
[tool.pytest.ini_options]
addopts = "--strict-markers"
diff --git a/src/xopen/__init__.py b/src/xopen/__init__.py
index 2e26d8b..8c21f68 100644
--- a/src/xopen/__init__.py
+++ b/src/xopen/__init__.py
@@ -69,6 +69,21 @@
else:
from backports import zstd
+try:
+ import lz4.frame # type: ignore
+except ImportError:
+ lz4 = None
+
+# The lz4 binary accepts nonnegative integer compression levels and clips them to [1, 12].
+# python-lz4 accepts both positive (clipped at 16) and negative (unlimited; converted to --fast) compression levels.
+# => The shared nonnegative range of [0, 16] works with both.
+if lz4 is None:
+ _LZ4_LEVEL_MIN, _LZ4_LEVEL_MAX = 0, 16
+else:
+ _LZ4_LEVEL_MIN = lz4.frame.COMPRESSIONLEVEL_MIN
+ _LZ4_LEVEL_MAX = lz4.frame.COMPRESSIONLEVEL_MAX
+XOPEN_DEFAULT_LZ4_COMPRESSION = _LZ4_LEVEL_MIN
+
try:
with open("/proc/sys/fs/pipe-max-size", "rt", encoding="ascii") as f:
_MAX_PIPE_SIZE = int(f.read())
@@ -106,6 +121,7 @@ class _ProgramSettings:
"zstd": _ProgramSettings(("zstd",), tuple(range(1, 20)), "-T"),
"pigz": _ProgramSettings(("pigz", "--no-name"), tuple(range(0, 10)) + (11,), "-p"),
"gzip": _ProgramSettings(("gzip", "--no-name"), tuple(range(1, 10))),
+ "lz4": _ProgramSettings(("lz4",), tuple(range(_LZ4_LEVEL_MIN, _LZ4_LEVEL_MAX + 1))),
}
@@ -530,6 +546,43 @@ def _open_zst(
return io.BufferedWriter(f) # mode "ab" and "wb"
+def _open_lz4(
+ filename: FileOrPath,
+ mode: str,
+ compresslevel: Optional[int],
+ threads: Optional[int],
+):
+ assert mode in ("rb", "ab", "wb")
+ if compresslevel is None:
+ compresslevel = XOPEN_DEFAULT_LZ4_COMPRESSION
+ if compresslevel not in _PROGRAM_SETTINGS["lz4"].acceptable_compression_levels:
+ raise ValueError(
+ f"LZ4 compresslevel must be in range "
+ f"{_LZ4_LEVEL_MIN}-{_LZ4_LEVEL_MAX}, got {compresslevel}."
+ )
+
+ if lz4 is not None and (mode == "rb" or threads == 0):
+ return lz4.frame.LZ4FrameFile(
+ filename, mode, compression_level=compresslevel, content_checksum=True
+ )
+ if threads == 0:
+ raise ImportError("LZ4 support with threads=0 requires xopen[lz4]")
+
+ try:
+ # ponytail: Older lz4 binaries lack -T; add version-aware threading only if needed.
+ return _PipedCompressionProgram(
+ filename, mode, compresslevel, threads, _PROGRAM_SETTINGS["lz4"]
+ )
+ except OSError as error:
+ if lz4 is None:
+ raise ImportError(
+ "LZ4 support requires xopen[lz4] or the lz4 program"
+ ) from error
+ return lz4.frame.LZ4FrameFile(
+ filename, mode, compression_level=compresslevel, content_checksum=True
+ )
+
+
def _open_gz(
filename: FileOrPath,
mode: str,
@@ -662,6 +715,10 @@ def _detect_format_from_content(filename: FileOrPath) -> Optional[str]:
elif bs[:4] == b"\x28\xb5\x2f\xfd":
# https://datatracker.ietf.org/doc/html/rfc8478#section-3.1.1
return "zst"
+ elif bs[:4] == b"\x04\x22\x4d\x18":
+ # https://github.com/lz4/lz4/blob/dev/doc/lz4_Frame_format.md
+ return "lz4"
+
return None
finally:
if closefd:
@@ -673,7 +730,7 @@ def _detect_format_from_extension(filename: Union[str, bytes]) -> Optional[str]:
Attempt to detect file format from the filename extension.
Return None if no format could be detected.
"""
- for ext in ("bz2", "xz", "gz", "zst"):
+ for ext in ("bz2", "xz", "gz", "zst", "lz4"):
if isinstance(filename, bytes):
if filename.endswith(b"." + ext.encode()):
return ext
@@ -696,7 +753,7 @@ def _file_or_path_to_binary_stream(
# object is not binary, this will crash at a later point.
return file_or_path, False # type: ignore
raise TypeError(
- f"Unsupported type for {file_or_path}, " f"{file_or_path.__class__.__name__}."
+ f"Unsupported type for {file_or_path}, {file_or_path.__class__.__name__}."
)
@@ -767,7 +824,7 @@ def xopen( # noqa: C901
"""
A replacement for the "open" function that can also read and write
compressed files transparently. The supported compression formats are gzip,
- bzip2, xz and zstandard. If the filename is '-', standard output (mode 'w') or
+ bzip2, xz, zstandard and LZ4. If the filename is '-', standard output (mode 'w') or
standard input (mode 'r') is returned. Filename can be a string or a
file object. (See https://docs.python.org/3/glossary.html#term-file-object.)
@@ -776,6 +833,7 @@ def xopen( # noqa: C901
- .bz2 uses bzip2 compression
- .xz uses xz/lzma compression
- .zst uses zstandard compression
+ - .lz4 uses lz4 compression
- otherwise, no compression is used
When reading, if a file name extension is available, the format is detected
@@ -784,10 +842,10 @@ def xopen( # noqa: C901
mode can be: 'rt', 'rb', 'at', 'ab', 'wt', or 'wb'. Also, the 't' can be omitted,
so instead of 'rt', 'wt' and 'at', the abbreviations 'r', 'w' and 'a' can be used.
- compresslevel is the compression level for writing to gzip, xz and zst files.
+ compresslevel is the compression level for writing to gzip, xz, zst and lz4 files.
This parameter is ignored for the other compression formats.
If set to None, a default depending on the format is used:
- gzip: 6, xz: 6, zstd: 3.
+ gzip: 6, xz: 6, zstd: 3, lz4: 0.
When threads is None (the default), compressed file formats are read or written
using a pipe to a subprocess running an external tool such as,
@@ -807,7 +865,7 @@ def xopen( # noqa: C901
format overrides the autodetection of input and output formats. This can be
useful when compressed output needs to be written to a file without an
- extension. Possible values are "gz", "xz", "bz2", "zst".
+ extension. Possible values are "gz", "xz", "bz2", "zst", "lz4".
"""
if mode in ("r", "w", "a"):
mode += "t" # type: ignore
@@ -823,10 +881,10 @@ def xopen( # noqa: C901
elif _file_is_a_socket_or_pipe(filename):
filename = open(filename, binary_mode) # type: ignore
- if format not in (None, "gz", "xz", "bz2", "zst"):
+ if format not in (None, "gz", "xz", "bz2", "zst", "lz4"):
raise ValueError(
f"Format not supported: {format}. "
- f"Choose one of: 'gz', 'xz', 'bz2', 'zst'"
+ f"Choose one of: 'gz', 'xz', 'bz2', 'zst', 'lz4'."
)
detected_format = format or _detect_format_from_extension(filepath)
if detected_format is None and "r" in mode:
@@ -840,6 +898,8 @@ def xopen( # noqa: C901
opened_file = _open_bz2(filename, binary_mode, compresslevel, threads)
elif detected_format == "zst":
opened_file = _open_zst(filename, binary_mode, compresslevel, threads)
+ elif detected_format == "lz4":
+ opened_file = _open_lz4(filename, binary_mode, compresslevel, threads)
else:
opened_file, _ = _file_or_path_to_binary_stream(filename, binary_mode)
diff --git a/tests/file.txt.lz4 b/tests/file.txt.lz4
new file mode 100644
index 0000000..5b2ed80
Binary files /dev/null and b/tests/file.txt.lz4 differ
diff --git a/tests/test_piped.py b/tests/test_piped.py
index 3b32677..554676b 100644
--- a/tests/test_piped.py
+++ b/tests/test_piped.py
@@ -18,7 +18,7 @@
_ProgramSettings,
)
-extensions = ["", ".gz", ".bz2", ".xz", ".zst"]
+extensions = ["", ".gz", ".bz2", ".xz", ".zst", ".lz4"]
try:
import fcntl
@@ -57,16 +57,24 @@ def available_zstd_programs():
return []
+def available_lz4_programs():
+ if shutil.which("lz4"):
+ return [_PROGRAM_SETTINGS["lz4"]]
+ return []
+
+
PIPED_GZIP_PROGRAMS = available_gzip_programs()
PIPED_BZIP2_PROGRAMS = available_bzip2_programs()
PIPED_XZ_PROGRAMS = available_xz_programs()
PIPED_ZST_PROGRAMS = available_zstd_programs()
+PIPED_LZ4_PROGRAMS = available_lz4_programs()
ALL_PROGRAMS_WITH_EXTENSION = (
list(zip(PIPED_GZIP_PROGRAMS, cycle([".gz"])))
+ list(zip(PIPED_BZIP2_PROGRAMS, cycle([".bz2"])))
+ list(zip(PIPED_XZ_PROGRAMS, cycle([".xz"])))
+ list(zip(PIPED_ZST_PROGRAMS, cycle([".zst"])))
+ + list(zip(PIPED_LZ4_PROGRAMS, cycle([".lz4"])))
)
diff --git a/tests/test_xopen.py b/tests/test_xopen.py
index 8f96db7..a9fbb7a 100644
--- a/tests/test_xopen.py
+++ b/tests/test_xopen.py
@@ -2,27 +2,32 @@
Tests for the xopen.xopen function
"""
import bz2
-import subprocess
-import sys
-import tempfile
-from contextlib import contextmanager
import functools
import gzip
import io
import lzma
import os
-from pathlib import Path
import shutil
+import subprocess
+import sys
+import tempfile
+from contextlib import contextmanager
+from pathlib import Path
+
import pytest
-from xopen import xopen, _detect_format_from_content
+from xopen import _detect_format_from_content, xopen
if sys.version_info >= (3, 14):
from compression import zstd
else:
from backports import zstd
+try:
+ import lz4.frame
+except ImportError:
+ lz4 = None
# TODO this is duplicated in test_piped.py
TEST_DIR = Path(__file__).parent
@@ -31,6 +36,8 @@
extensions = ["", ".gz", ".bz2", ".xz"]
if shutil.which("zstd") or zstd:
extensions += [".zst"]
+if lz4:
+ extensions += [".lz4"]
base = os.path.join(os.path.dirname(__file__), "file.txt")
files = [base + ext for ext in extensions]
@@ -369,6 +376,8 @@ def test_read_no_threads(ext):
}
if ext == ".zst" and zstd is None:
return
+ if ext == ".lz4":
+ klasses[".lz4"] = lz4.frame.LZ4FrameFile
klass = klasses[ext]
with xopen(TEST_DIR / f"file.txt{ext}", "rb", threads=0) as f:
assert isinstance(f, klass), f
@@ -401,6 +410,8 @@ def test_write_no_threads(tmp_path, ext):
# Skip zst because if zstd is not available,
# we fall back to an external process even when threads=0
return
+ if ext == ".lz4":
+ klasses[".lz4"] = lz4.frame.LZ4FrameFile
klass = klasses[ext]
with xopen(tmp_path / f"out{ext}", "wb", threads=0) as f:
if isinstance(f, io.BufferedWriter):
@@ -592,6 +603,20 @@ def test_xopen_zst_fails_when_zstd_not_available(monkeypatch):
f.read()
+@pytest.mark.skipif(not shutil.which("lz4"), reason="lz4 program not installed")
+def test_lz4_program_without_bindings(tmp_path, monkeypatch):
+ import xopen as xopen_module
+
+ monkeypatch.setattr(xopen_module, "lz4", None)
+ path = tmp_path / "file.lz4"
+ with xopen_module.xopen(path, "wb", threads=1) as f:
+ f.write(b"hello")
+ with xopen_module.xopen(path, "rb", threads=1) as f:
+ assert f.read() == b"hello"
+ with pytest.raises(ImportError, match="xopen\\[lz4\\]"):
+ xopen_module.xopen(path, "rb", threads=0)
+
+
@pytest.mark.parametrize("threads", (0, 1))
def test_xopen_zst_long_window_size(threads):
if threads == 0 and zstd is None:
@@ -613,7 +638,6 @@ def test_xopen_zst_long_window_size(threads):
def test_pass_file_object_for_reading(ext, threads):
if ext == ".zst" and zstd is None:
return
-
with open(TEST_DIR / f"file.txt{ext}", "rb") as fh:
with xopen(fh, mode="rb", threads=threads) as f:
assert f.readline() == CONTENT_LINES[0].encode("utf-8")
diff --git a/tox.toml b/tox.toml
index 3ee996d..015ed52 100644
--- a/tox.toml
+++ b/tox.toml
@@ -21,6 +21,9 @@ deps = [
"zstandard",
]
+[env.lz4]
+extras = ["lz4"]
+
[env.no-libs]
deps = ["{[env_run_base]deps}", "pip"]
commands = [