Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .gitignore
Original file line number Diff line number Diff line change
@@ -1,2 +1,2 @@
Manifest*.toml
data/
/data/
2 changes: 1 addition & 1 deletion Project.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
name = "ASDF"
uuid = "686f71d1-807d-59a4-a860-28280ea06d7b"
version = "2.0.2"
version = "2.1.0"
authors = ["Erik Schnetter <schnetter@gmail.com>"]

[workspace]
Expand Down
2 changes: 1 addition & 1 deletion docs/src/index.md
Original file line number Diff line number Diff line change
Expand Up @@ -108,7 +108,7 @@ af = load("intro_compressed.asdf")
name: "ASDF.jl"
author: "Erik Schnetter <schnetter@gmail.com>"
homepage: "https://github.com/JuliaAstro/ASDF.jl"
version: "2.0.2"
version: "2.1.0"
...
�BLK0 f�0xj�sq���r#ASDF BLOCK INDEX
%YAML 1.1
Expand Down
8 changes: 8 additions & 0 deletions docs/src/interop.md
Original file line number Diff line number Diff line change
Expand Up @@ -135,3 +135,11 @@ af.write_to("example.asdf")
```

See the [PythonCall.jl documentation](https://juliapy.github.io/PythonCall.jl/stable/) for more.

## Compression

Which [compression schemes](@ref ASDF.Compression) the Python `asdf` reader understands depends on what is installed on the Python side:

- `C_Zlib`, `C_Bzip2`, and `C_Lz4` (the chunked LZ4 block format of Python's built-in `lz4` compressor) are readable by a stock `asdf` install (`lz4` additionally needs the [`lz4`](https://python-lz4.readthedocs.io/) Python package).
- `C_Lz4F` (`lz4f`, LZ4 frame), `C_Zstd`, and `C_Blosc`/`C_Blosc2` need the [asdf-compression](https://github.com/asdf-format/asdf-compression) extension.
- `C_Xz` currently has no Python counterpart.
135 changes: 45 additions & 90 deletions src/ASDF.jl
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module ASDF

using ChunkCodecLibBlosc: BloscCodec, BloscEncodeOptions
using ChunkCodecLibBzip2: BZ2Codec, BZ2EncodeOptions
using ChunkCodecLibLz4: LZ4BlockCodec, LZ4FrameCodec, LZ4BlockEncodeOptions, LZ4FrameEncodeOptions
using ChunkCodecLibLz4: LZ4NumcodecsCodec, LZ4FrameCodec, LZ4NumcodecsEncodeOptions, LZ4FrameEncodeOptions
using ChunkCodecLibLzma: XZCodec, XZEncodeOptions
using ChunkCodecLibZlib: ZlibCodec, ZlibEncodeOptions
using ChunkCodecLibZstd: ZstdCodec, ZstdEncodeOptions, decode, encode
Expand Down Expand Up @@ -34,20 +34,23 @@ Identifies the compression algorithm used for a data block. Available variants:
| `C_Blosc` | `blsc` | ChunkCodecLibBlosc.jl | Multi-threaded, shuffle-aware, best with typed numeric arrays |
| `C_Blosc2` | `bls2` | See [Issue #49](https://github.com/JuliaIO/ChunkCodecs.jl/issues/49) | Like Blosc but supports more than 2 GB of data |
| `C_Bzip2` | `bzp2` | ChunkCodecLibBzip2.jl | Good ratio, moderate speed (default) |
| `C_Lz4` (:block) | `lz4\\0` | ChunkCodecLibLz4.jl | Fastest decompression, Python-compatible |
| `C_Lz4` (:frame) | `lz4\\0` | ChunkCodecLibLz4.jl | LZ4 frame format for non-Python consumers |
| `C_Lz4` | `lz4\\0` | ChunkCodecLibLz4.jl | Fastest decompression, Python-compatible |
| `C_Lz4F` | `lz4f` | ChunkCodecLibLz4.jl | LZ4 frame format for non-Python consumers |
| `C_Xz` | `xz\\0\\0` | ChunkCodecLibLzma.jl | Highest compression ratio, slowest |
| `C_Zlib` | `zlib` | ChunkCodecLibZlib.jl | Broad compatibility |
| `C_Zstd` | `zstd` | ChunkCodecLibZstd.jl | Best ratio/speed trade-off |

Only `zlib` and `bzp2` are defined by the ASDF standard. `lz4` follows the Python `asdf` implementation's built-in compressor, `blsc`/`bls2`/`lz4f`/`zstd` follow its [asdf-compression](https://github.com/asdf-format/asdf-compression) extension, and `xz` has no Python counterpart.
"""
@enum Compression C_None C_Blosc C_Blosc2 C_Bzip2 C_Lz4 C_Xz C_Zlib C_Zstd
@enum Compression C_None C_Blosc C_Blosc2 C_Bzip2 C_Lz4 C_Lz4F C_Xz C_Zlib C_Zstd

const compression_keys = Dict{Compression, Vector{UInt8}}(
C_None => UInt8[0, 0, 0, 0],
C_Blosc => Vector{UInt8}("blsc"),
C_Blosc2 => Vector{UInt8}("bls2"),
C_Bzip2 => Vector{UInt8}("bzp2"),
C_Lz4 => Vector{UInt8}("lz4\0"),
C_Lz4F => Vector{UInt8}("lz4f"),
C_Xz => Vector{UInt8}("xz\0\0"),
C_Zlib => Vector{UInt8}("zlib"),
C_Zstd => Vector{UInt8}("zstd"),
Expand Down Expand Up @@ -173,8 +176,6 @@ big2native_U8(bytes::AbstractVector{UInt8}) = bytes[1]
big2native_U16(bytes::AbstractVector{UInt8}) = (UInt16(bytes[1]) << 8) | bytes[2]
big2native_U32(bytes::AbstractVector{UInt8}) = (UInt32(big2native_U16(@view bytes[1:2])) << 16) | big2native_U16(@view bytes[3:4])
big2native_U64(bytes::AbstractVector{UInt8}) = (UInt64(big2native_U32(@view bytes[1:4])) << 32) | big2native_U32(@view bytes[5:8])
# Read a 4-byte little-endian UInt32 (used for lz4.block store_size prefix)
little2native_U32(bytes::AbstractVector{UInt8}) = (UInt32(bytes[1])) | (UInt32(bytes[2]) << 8) | (UInt32(bytes[3]) << 16) | (UInt32(bytes[4]) << 24)

native2big_U8(val::UInt8) = UInt8[val]
native2big_U16(val::UInt16) = UInt8[(val >>> 0x08) & 0xff, (val >>> 0x00) & 0xff]
Expand Down Expand Up @@ -283,12 +284,17 @@ function read_block(header::BlockHeader)
if compression == C_None
# do nothing, the block is uncompressed
elseif compression == C_Lz4
data = decode_Lz4(data)
# TODO: Once an `LZ4ASDFCodec` is upstreamed to ChunkCodecs.jl,
# this could be added to the uniform codec branches below.
# https://github.com/JuliaAstro/ASDF.jl/issues/72
data = decode_Lz4(data, Int(header.data_size))
else
if compression == C_Blosc
codec = BloscCodec()
elseif compression == C_Bzip2
codec = BZ2Codec()
elseif compression == C_Lz4F
codec = LZ4FrameCodec()
elseif compression == C_Xz
codec = XZCodec()
elseif compression == C_Zlib
Expand All @@ -310,45 +316,33 @@ function read_block(header::BlockHeader)
return data
end

function decode_Lz4(data)
if (# LZ4 Frame magic bytes 04 22 4D 18
length(data) >= 4 &&
data[1] == 0x04 && data[2] == 0x22 &&
data[3] == 0x4D && data[4] == 0x18
)
return decode(LZ4FrameCodec(), data)
else
# If the data was originally created from Python's ASDF, then it will be in block instead of frame layout,
# where each chunk is:
#
# [4 bytes, big-endian] compressed chunk size (the ASDF envelope)
# [4 bytes, little-endian] uncompressed chunk size (lz4.block store_size=True prefix)
# [N bytes] raw LZ4 block payload
#
# lz4.block.compress() defaults to store_size=True, which prepends the
# uncompressed size as a little-endian uint32. LZ4BlockCodec expects only
# the raw block, so both the outer BE envelope and the inner LE prefix must
# be stripped, with the LE value used as the uncompressed_size hint.

out = UInt8[]
pos = 1

while pos <= length(data)
# Outer ASDF envelope: big-endian compressed chunk size
compressed_chunk_size = Int(big2native_U32(@view data[pos:(pos + 3)]))
pos += 4
# Inner lz4.block store_size=True prefix: little-endian uncompressed size
uncompressed_chunk_size = Int(little2native_U32(@view data[pos:(pos + 3)]))
pos += 4
# Raw LZ4 block payload (compressed_chunk_size includes the 4-byte LE prefix)
payload_len = compressed_chunk_size - 4
payload = @view data[pos:(pos + payload_len - 1)]
pos += payload_len
append!(out, decode(LZ4BlockCodec(), payload; max_size = uncompressed_chunk_size, size_hint = uncompressed_chunk_size))
end
# `lz4\0` blocks, as written by Python asdf's built-in `lz4` compressor: a concatenation of chunks
# `[UInt32 big-endian n][n bytes]`, where the n bytes are the numcodecs LZ4 format
# (Int32 little-endian decoded size + raw LZ4 block), i.e. `lz4.block.compress(store_size=True)`.
const LZ4_CHUNK_SIZE = 1 << 22 # Python asdf's default `compression_block_size`

function decode_Lz4(data::AbstractVector{UInt8}, data_size::Int)
out = sizehint!(UInt8[], data_size)
pos = 1
while pos <= length(data)
pos + 3 <= length(data) || error("Truncated LZ4 chunk header at byte $pos")
n = Int(big2native_U32(@view data[pos:(pos + 3)]))
pos + 3 + n <= length(data) || error("LZ4 chunk at byte $pos extends past the end of the block")
chunk = @view data[(pos + 4):(pos + 3 + n)]
append!(out, decode(LZ4NumcodecsCodec(), chunk; max_size = data_size - length(out)))
pos += 4 + n
end
return out
end

return out
function encode_Lz4(input::AbstractVector{UInt8}; chunk_size::Integer = LZ4_CHUNK_SIZE)
out = UInt8[]
for start in 1:chunk_size:length(input)
chunk = @view input[start:min(start + chunk_size - 1, end)]
encoded = encode(LZ4NumcodecsEncodeOptions(; compressionLevel = 12), chunk)
append!(out, native2big_U32(length(encoded)), encoded)
end
return out
end

################################################################################
Expand Down Expand Up @@ -1440,7 +1434,7 @@ long.asdf
├─ name::String | ASDF.jl
├─ author::String | Erik Schnetter <schnetter@gmail.com>
├─ homepage::String | https://github.com/JuliaAstro/ASDF.jl
└─ version::String | 2.0.2
└─ version::String | 2.1.0
```
"""
function info(io::IO, af::ASDFFile; max_rows = 20)
Expand Down Expand Up @@ -1558,7 +1552,7 @@ myfile.asdf
├─ name::String | ASDF.jl
├─ author::String | Erik Schnetter <schnetter@gmail.com>
├─ homepage::String | https://github.com/JuliaAstro/ASDF.jl
└─ version::String | 2.0.2
└─ version::String | 2.1.0
```
"""
function fileio_load(f::File{format"ASDF"}; kwargs...)
Expand Down Expand Up @@ -1597,7 +1591,6 @@ Parameter | Default | Description
| :---------- | :-------- | :------------------------------------------------------------------------ |
`compression` | `C_Bzip2` | Applied compression scheme |
`inline` | `false` | Embed data in YAML instead of a binary block |
`lz4_layout` | `:block` | `:block` for Python-compatible chunked LZ4, `:frame` for LZ4 frame format |

!!! note
If the compressed output is larger than the raw input, the block is stored uncompressed regardless of the chosen compression.
Expand All @@ -1606,10 +1599,9 @@ struct NDArrayWrapper
array::AbstractArray
compression::Compression
inline::Bool
lz4_layout::Symbol
end
function NDArrayWrapper(array::AbstractArray; compression::Compression = C_Bzip2, inline::Bool = false, lz4_layout::Symbol = :block)
return NDArrayWrapper(array, compression, inline, lz4_layout)
function NDArrayWrapper(array::AbstractArray; compression::Compression = C_Bzip2, inline::Bool = false)
return NDArrayWrapper(array, compression, inline)
end
Base.getindex(val::NDArrayWrapper) = val.array

Expand Down Expand Up @@ -1717,42 +1709,6 @@ function YAML._print(io::IO, val::NDArrayWrapper, level::Int = 0, ignore_level::
return YAML._print(io, ndarray, level, ignore_level)
end

function encode_Lz4_block(input::AbstractVector{UInt8}; chunk_size::Int = 1024 * 1024 * 8)
out = UInt8[]
offset = 1
while offset <= length(input)
chunk_end = min(offset + chunk_size - 1, length(input))
chunk = @view input[offset:chunk_end]

# Compress the raw chunk with LZ4 block codec
# LZ4BlockEncodeOptions does NOT prepend the uncompressed size,
# so we must prepend the LE uint32 ourselves to match Python's
# lz4.block.compress(store_size=True) behaviour.
compressed_payload = encode(LZ4BlockEncodeOptions(), chunk)
uncompressed_size = UInt32(length(chunk))
compressed_chunk_size = UInt32(4 + length(compressed_payload)) # LE prefix + raw payload

# Outer ASDF envelope: big-endian compressed chunk size (includes the 4-byte LE prefix)
append!(out, native2big_U32(compressed_chunk_size))

# Inner lz4.block store_size=True prefix: little-endian uncompressed size
append!(
out, [
(uncompressed_size >>> 0x00) & 0xff,
(uncompressed_size >>> 0x08) & 0xff,
(uncompressed_size >>> 0x10) & 0xff,
(uncompressed_size >>> 0x18) & 0xff,
]
)

# Raw LZ4 block payload
append!(out, compressed_payload)
offset = chunk_end + 1
end

return out
end

"""
write_file(filename::AbstractString, document::AbstractDict)

Expand Down Expand Up @@ -1836,15 +1792,14 @@ function write_file(filename::AbstractString, document::AbstractDict)
# TODO: Write directly to file
if array.compression == C_None
data = input
elseif array.compression == C_Lz4 && array.lz4_layout == :block
data = encode_Lz4_block(input)
#data = encode(LZ4BlockEncodeOptions(), input) # Not compatible with Python asdf
elseif array.compression == C_Lz4
data = encode_Lz4(input)
else
if array.compression == C_Blosc
encode_options = BloscEncodeOptions(; clevel = 9, doshuffle = 2, typesize = sizeof(eltype(array.array)), compressor = "zstd")
elseif array.compression == C_Bzip2
encode_options = BZ2EncodeOptions(; blockSize100k = 9)
elseif array.compression == C_Lz4 && array.lz4_layout == :frame
elseif array.compression == C_Lz4F
encode_options = LZ4FrameEncodeOptions(; compressionLevel = 12, blockSizeID = 7)
elseif array.compression == C_Xz
encode_options = XZEncodeOptions(; preset = UInt32(6))
Expand Down
Binary file added test/data/asdf-1.6.0/lz4.asdf
Binary file not shown.
Binary file added test/data/asdf-1.6.0/lz4f.asdf
Binary file not shown.
5 changes: 5 additions & 0 deletions test/test-blocks.jl
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,11 @@ end
"Invalid compression format found: C_Blosc2";
compression = ASDF.C_Blosc2,
)
test_read_block(
# The 4 data bytes parse as a big-endian chunk length of 0x05060708
"LZ4 chunk at byte 1 extends past the end of the block";
compression = ASDF.C_Lz4,
)
test_read_block(
"Actual data size different from declared data size in header.";
data_size = UInt64(9),
Expand Down
35 changes: 35 additions & 0 deletions test/test-compression.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,35 @@
@testset "LZ4 chunked codec" begin
x = rand(UInt8(0):UInt8(3), 10_000)
enc = ASDF.encode_Lz4(x; chunk_size = 1024) # 10 chunks
@test ASDF.decode_Lz4(enc, length(x)) == x
@test ASDF.decode_Lz4(ASDF.encode_Lz4(UInt8[]), 0) == UInt8[]
@test_throws "LZ4 chunk" ASDF.decode_Lz4(enc[1:(end - 1)], length(x))
@test_throws "LZ4 chunk" ASDF.decode_Lz4(enc[1:2], length(x))
end

#=
Regenerate lz4(f).asdf

Run the below to generate the test files used by this testset.
Needs: pip install asdf lz4 "asdf_compression[lz4f] @ git+https://github.com/asdf-format/asdf-compression"

```python
import asdf
import numpy as np

arr = np.arange(2048, dtype=np.int64) # 16 KiB
# lz4: stock asdf's chunked block format, 16 chunks of 128 elements; lz4f: asdf-compression's LZ4 frame
for label, kwargs in (("lz4", {"compression_block_size": 1024}), ("lz4f", {})):
af = asdf.AsdfFile({"arr": arr})
af.set_array_compression(af["arr"], label, **kwargs)
af.write_to(f"{label}.asdf")

```
=#
@testset "Python-generated LZ4 fixtures" begin
for (name, key) in ("lz4" => ASDF.C_Lz4, "lz4f" => ASDF.C_Lz4F)
af = ASDF.load_file(joinpath("data", "asdf-1.6.0", name * ".asdf"))
@test af.lazy_block_headers.block_headers[1].compression == ASDF.compression_keys[key]
@test af["arr"][] == 0:2047
end
end
10 changes: 8 additions & 2 deletions test/test-write.jl
Original file line number Diff line number Diff line change
Expand Up @@ -10,8 +10,8 @@
"element1" => ASDF.NDArrayWrapper(array; compression = ASDF.C_None),
"element2" => ASDF.NDArrayWrapper(array; compression = ASDF.C_Blosc),
"element3" => ASDF.NDArrayWrapper(array; compression = ASDF.C_Bzip2),
"element4" => ASDF.NDArrayWrapper(array; compression = ASDF.C_Lz4, lz4_layout = :block),
"element5" => ASDF.NDArrayWrapper(array; compression = ASDF.C_Lz4, lz4_layout = :frame),
"element4" => ASDF.NDArrayWrapper(array; compression = ASDF.C_Lz4),
"element5" => ASDF.NDArrayWrapper(array; compression = ASDF.C_Lz4F),
"element6" => ASDF.NDArrayWrapper(array; compression = ASDF.C_Xz),
"element7" => ASDF.NDArrayWrapper(array; compression = ASDF.C_Zlib),
"element8" => ASDF.NDArrayWrapper(array; compression = ASDF.C_Zstd),
Expand Down Expand Up @@ -42,6 +42,12 @@
@test element′ == element
end

# Blocks fall back to uncompressed storage when compression does not shrink them, so check the
# labels to make sure the two LZ4 code paths actually ran.
label(name) = doc′.lazy_block_headers.block_headers[doc′["group"][name].source + 1].compression
@test label("element4") == ASDF.compression_keys[ASDF.C_Lz4]
@test label("element5") == ASDF.compression_keys[ASDF.C_Lz4F]

@test_throws "`array` has invalid state: `compression` field has value not specified in `Compression` enum." begin
doc = Dict{Any, Any}("field1" => ASDF.NDArrayWrapper([5, 6, 7, 8]; compression = ASDF.C_Blosc2))
ASDF.write_file(filename, doc)
Expand Down
Loading