Skip to content

Commit d625e96

Browse files
authored
Merge pull request #242 from rtibblesbot/issue-241-85627b
Add archive contents_sha256 and ARCHIVE_FORMATS
2 parents 2caaf5b + 28c428b commit d625e96

5 files changed

Lines changed: 140 additions & 0 deletions

File tree

‎README.md‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -202,6 +202,13 @@ for generating proquint identifiers for content channels. These are short string
202202
that are easy to enter on devices without a full keyboard, e.g. `sutul-hakuh`.
203203

204204

205+
Archive contents hash
206+
---------------------
207+
[le_utils/archive.py](./le_utils/archive.py) provides `contents_sha256`, a hash of a zip
208+
archive's member paths and bytes that ignores compression and metadata. It identifies files
209+
in `file_formats.ARCHIVE_FORMATS`. Its definition is frozen: ricecooker and Studio index archives by it.
210+
211+
205212
Roles
206213
-----
207214
The `role` constants are used for Role-based access control (RBAC) within the

‎le_utils/archive.py‎

Lines changed: 38 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,38 @@
1+
import hashlib
2+
import zipfile
3+
4+
_CHUNK_SIZE = 64 * 1024
5+
6+
7+
def _member_sha256(zf, info):
8+
digest = hashlib.sha256()
9+
with zf.open(info) as member:
10+
for chunk in iter(lambda: member.read(_CHUNK_SIZE), b""):
11+
digest.update(chunk)
12+
return digest.hexdigest()
13+
14+
15+
def contents_sha256(fileobj):
16+
"""
17+
Identity hash of a zip archive's member paths and bytes, independent of compression and metadata.
18+
FROZEN: ricecooker and Studio's upload Cloud Function index archives by this digest.
19+
:param fileobj: A seekable binary file object of a zip archive
20+
:return: A hex SHA-256 digest
21+
"""
22+
with zipfile.ZipFile(fileobj) as zf:
23+
members = {}
24+
for info in zf.infolist():
25+
# Not .filename: Windows rewrites "\\" in it and Python >= 3.12 overrides it from 0x7075 extra fields.
26+
path = info.orig_filename
27+
# Before the directory skip: zipfile reads "x\0/" as file "x".
28+
if "\0" in path:
29+
raise ValueError(f"NUL in member path {path!r}")
30+
if path.endswith("/"):
31+
continue
32+
if path in members:
33+
raise ValueError(f"Duplicate member path {path!r}")
34+
members[path] = info
35+
digest = hashlib.sha256()
36+
for path in sorted(members):
37+
digest.update(f"{path}\0{_member_sha256(zf, members[path])}\n".encode("utf-8"))
38+
return digest.hexdigest()

‎le_utils/constants/file_formats.py‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -85,6 +85,10 @@
8585
BLOOMD = "bloomd"
8686
BLOOMPUB_MIMETYPE = "application/bloompub+zip"
8787

88+
# Zip-based formats identified by le_utils.archive.contents_sha256.
89+
# FROZEN: ricecooker and Studio's upload Cloud Function pin this; removing a format orphans its index entries.
90+
ARCHIVE_FORMATS = (HTML5, H5P, EPUB, HTML5_ARTICLE, BLOOMPUB, BLOOMD)
91+
8892
choices = (
8993
(MP4, "MP4 Video"),
9094
(WEBM, "WEBM Video"),

‎tests/test_archive.py‎

Lines changed: 87 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,87 @@
1+
import io
2+
import struct
3+
import zipfile
4+
import zlib
5+
6+
import pytest
7+
8+
from le_utils.archive import contents_sha256
9+
10+
MEMBERS = [
11+
("README.txt", b"Kolibri\n"),
12+
("index.html", b"<html><body><p>Kolibri</p></body></html>\n"),
13+
("css/site/style.css", b"p { color: #333; }\n"),
14+
("assets/data.bin", bytes(range(256)) * 400),
15+
("données/été.txt", "café\n".encode("utf-8")),
16+
("empty.txt", b""),
17+
]
18+
EXPECTED_DIGEST = "9ccc8c88bedb2106e59879ba70e7c84f38f098bf8fa4a7a2476dad63eab0ae15"
19+
20+
21+
def _zip(members, compression=zipfile.ZIP_DEFLATED):
22+
archive = io.BytesIO()
23+
with zipfile.ZipFile(archive, "w", compression) as zf:
24+
for name, data in members:
25+
zf.writestr(name, data)
26+
archive.seek(0)
27+
return archive
28+
29+
30+
@pytest.mark.parametrize("compression", [zipfile.ZIP_STORED, zipfile.ZIP_DEFLATED], ids=["stored", "deflated"])
31+
def test_contents_sha256_is_pinned(compression):
32+
assert contents_sha256(_zip(MEMBERS, compression)) == EXPECTED_DIGEST
33+
34+
35+
def test_contents_sha256_ignores_member_order():
36+
assert contents_sha256(_zip(reversed(MEMBERS))) == EXPECTED_DIGEST
37+
38+
39+
def test_contents_sha256_ignores_directory_entries():
40+
directories = [("css/", b""), ("css/site/", b""), ("données/", b"")]
41+
assert contents_sha256(_zip(directories + MEMBERS)) == EXPECTED_DIGEST
42+
43+
44+
def test_contents_sha256_depends_on_member_paths():
45+
moved = [("moved/" + name if name == "index.html" else name, data) for name, data in MEMBERS]
46+
assert contents_sha256(_zip(moved)) != EXPECTED_DIGEST
47+
48+
49+
def _with_metadata(name):
50+
info = zipfile.ZipInfo(name, date_time=(2001, 2, 3, 4, 5, 6))
51+
info.external_attr = 0o100755 << 16
52+
info.comment = b"comment"
53+
# Python >= 3.12 replaces ZipInfo.filename with this Unicode Path field's name.
54+
renamed = ("renamed/" + name).encode("utf-8")
55+
info.extra = struct.pack("<HHBL", 0x7075, 5 + len(renamed), 1, zlib.crc32(name.encode("utf-8"))) + renamed
56+
return info
57+
58+
59+
def test_contents_sha256_ignores_metadata():
60+
archive = io.BytesIO()
61+
with zipfile.ZipFile(archive, "w") as zf:
62+
zf.comment = b"archive comment"
63+
for name, data in MEMBERS:
64+
zf.writestr(_with_metadata(name), data)
65+
assert contents_sha256(archive) == EXPECTED_DIGEST
66+
67+
68+
def test_contents_sha256_is_independent_of_file_position():
69+
archive = _zip(MEMBERS)
70+
archive.read()
71+
assert contents_sha256(archive) == EXPECTED_DIGEST
72+
assert not archive.closed
73+
74+
75+
def test_contents_sha256_rejects_duplicate_paths():
76+
with pytest.warns(UserWarning):
77+
archive = _zip(MEMBERS + [("index.html", b"other")])
78+
with pytest.raises(ValueError):
79+
contents_sha256(archive)
80+
81+
82+
def test_contents_sha256_rejects_nul_in_path():
83+
# zipfile truncates names at NUL when writing, so patch the stored bytes.
84+
# Trailing "/": NUL must be rejected before directory entries are skipped.
85+
archive = io.BytesIO(_zip([("nul#name/", b"x")]).getvalue().replace(b"nul#name", b"nul\0name"))
86+
with pytest.raises(ValueError):
87+
contents_sha256(archive)

‎tests/test_formats.py‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -18,3 +18,7 @@ def test_file_format_extensions_are_synced():
1818

1919
def test_FORMATLIST_exists():
2020
assert file_formats.FORMATLIST, "FORMATLIST did not genereate properly"
21+
22+
23+
def test_ARCHIVE_FORMATS_does_not_change():
24+
assert set(file_formats.ARCHIVE_FORMATS) == {"zip", "h5p", "epub", "kpub", "bloompub", "bloomd"}

0 commit comments

Comments
 (0)