Repository navigation
Expand file tree
/
Copy pathbaseline.py
More file actions
89 lines (74 loc) · 3.25 KB
/
Copy pathbaseline.py
File metadata and controls
89 lines (74 loc) · 3.25 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
"""What do the existing tools already achieve on these three shapes?
Run this BEFORE writing any transform. If zstd -19 is already close to the
entropy of the data, there is nothing to win and the honest answer is to stop.
python baseline.py
"""
import bz2
import gzip
import io
import lzma
import os
import sys
import tarfile
import time
import zstandard
DATA = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'data')
def codecs():
return {
'gzip -6': lambda d: gzip.compress(d, 6),
'gzip -9': lambda d: gzip.compress(d, 9),
'zstd -19': lambda d: zstandard.ZstdCompressor(level=19).compress(d),
'bz2 -9': lambda d: bz2.compress(d, 9),
'xz -9': lambda d: lzma.compress(d, preset=9),
}
# Two flags that matter only on a big binary blob, so they are not in the
# default table: a 128 MB window (the decoder's own default limit, so the frame
# still decodes everywhere) and the x86 branch filter for ELF-heavy tars.
def oci_codecs():
p = zstandard.ZstdCompressionParameters.from_level(19, window_log=27, enable_ldm=True)
return {
'zstd -19 --long': lambda d: zstandard.ZstdCompressor(compression_params=p).compress(d),
'xz -9 +BCJ': lambda d: lzma.compress(d, filters=[
{'id': lzma.FILTER_X86}, {'id': lzma.FILTER_LZMA2, 'preset': 9}]),
}
def report(label, data, note='', extra=None):
print(f'\n{label} - {len(data):,} bytes {note}')
print(f" {'codec':<15} {'size':>12} {'ratio':>7} {'% of raw':>9} {'MB/s':>7}")
best = None
for name, fn in {**codecs(), **(extra or {})}.items():
t = time.perf_counter()
out = fn(data)
el = time.perf_counter() - t
mbs = len(data) / el / 1e6
print(f' {name:<15} {len(out):>12,} {len(data)/len(out):>6.2f}x '
f'{100*len(out)/len(data):>8.1f}% {mbs:>7.1f}')
if best is None or len(out) < best[1]:
best = (name, len(out))
print(f' -> best: {best[0]} at {best[1]:,} bytes')
return best
if __name__ == '__main__':
which = sys.argv[1] if len(sys.argv) > 1 else 'all'
if which in ('all', 'sql'):
for f in ('wiki.sql', 'chinook.sql'):
p = os.path.join(DATA, f)
if os.path.exists(p):
report(f, open(p, 'rb').read(), '(real SQL dump)')
if which in ('all', 'sqlite'):
for f in ('wiki.db', 'chinook.db'):
p = os.path.join(DATA, f)
if os.path.exists(p):
report(f, open(p, 'rb').read(), '(real SQLite file)')
if which in ('all', 'oci'):
p = os.path.join(DATA, 'layer.tar.gz')
if os.path.exists(p):
blob = open(p, 'rb').read()
print(f'\nlayer.tar.gz - {len(blob):,} bytes (real Docker layer, as shipped)')
tar = gzip.decompress(blob)
print(f' inner tar is {len(tar):,} bytes -> the gzip achieves '
f'{len(tar)/len(blob):.2f}x')
with tarfile.open(fileobj=io.BytesIO(tar)) as t:
members = t.getmembers()
print(f' {len(members):,} tar members')
# A registry re-encode: same tar content, better container.
report('layer, re-encoded from the inner tar', tar,
f'(vs {len(blob):,} B as shipped)', extra=oci_codecs())