Skip to content

Commit 5fd27c0

Browse files
authored
Create filetype.py
1 parent c721411 commit 5fd27c0

1 file changed

Lines changed: 74 additions & 0 deletions

File tree

‎app/infra/filetype.py‎

Lines changed: 74 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,74 @@
1+
"""Magic-byte content sniffing for uploads.
2+
3+
A filename extension is a claim, not a fact -- ``sales.xlsx`` can hold
4+
anything. This checks the leading bytes of an upload against known file
5+
signatures so the *content* must match one of an allow-listed set of
6+
types before it is accepted. A pure ``infra`` primitive: no session, no
7+
entities, stdlib only.
8+
9+
Signatures are deliberately small and well-known. Note that modern
10+
Office formats (xlsx/docx/pptx) are ZIP containers, so they share the
11+
ZIP signature -- ``xlsx`` is accepted as "a zip" here; distinguishing
12+
the Office subtype would require reading the archive, which is out of
13+
scope for a byte-signature gate.
14+
"""
15+
16+
from __future__ import annotations
17+
18+
# short name -> list of acceptable leading-byte signatures
19+
_SIGNATURES: dict[str, list[bytes]] = {
20+
"zip": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"],
21+
"xlsx": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"], # zip container
22+
"docx": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"],
23+
"pptx": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"],
24+
"pdf": [b"%PDF-"],
25+
"png": [b"\x89PNG\r\n\x1a\n"],
26+
"jpg": [b"\xff\xd8\xff"],
27+
"gif": [b"GIF87a", b"GIF89a"],
28+
"json": [], # text: no signature, validated as utf-8 text below
29+
"csv": [],
30+
"txt": [],
31+
"text": [],
32+
}
33+
34+
_TEXT_TYPES = {"json", "csv", "txt", "text"}
35+
36+
37+
def is_probably_text(head: bytes) -> bool:
38+
"""Heuristic: decodable as UTF-8 and free of NUL bytes."""
39+
if b"\x00" in head:
40+
return False
41+
try:
42+
head.decode("utf-8")
43+
return True
44+
except UnicodeDecodeError:
45+
# a multibyte char may be split at the chunk boundary; tolerate a
46+
# short tail by retrying without the last few bytes
47+
try:
48+
head[:-3].decode("utf-8")
49+
return True
50+
except UnicodeDecodeError:
51+
return False
52+
53+
54+
def sniff_matches(head: bytes, allowed: list[str]) -> bool:
55+
"""Does ``head`` match at least one of the ``allowed`` type names?
56+
57+
Empty ``allowed`` means validation is disabled -> always True.
58+
Unknown type names in ``allowed`` are ignored (they can never match),
59+
so a typo fails closed rather than silently allowing everything.
60+
"""
61+
if not allowed:
62+
return True
63+
for name in allowed:
64+
key = name.lower().lstrip(".")
65+
sigs = _SIGNATURES.get(key)
66+
if sigs is None:
67+
continue # unknown type name: cannot match
68+
if key in _TEXT_TYPES:
69+
if is_probably_text(head):
70+
return True
71+
continue
72+
if any(head.startswith(sig) for sig in sigs):
73+
return True
74+
return False

0 commit comments

Comments
 (0)