|
| 1 | +"""Magic-byte content sniffing for uploads. |
| 2 | +
|
| 3 | +A filename extension is a claim, not a fact -- ``sales.xlsx`` can hold |
| 4 | +anything. This checks the leading bytes of an upload against known file |
| 5 | +signatures so the *content* must match one of an allow-listed set of |
| 6 | +types before it is accepted. A pure ``infra`` primitive: no session, no |
| 7 | +entities, stdlib only. |
| 8 | +
|
| 9 | +Signatures are deliberately small and well-known. Note that modern |
| 10 | +Office formats (xlsx/docx/pptx) are ZIP containers, so they share the |
| 11 | +ZIP signature -- ``xlsx`` is accepted as "a zip" here; distinguishing |
| 12 | +the Office subtype would require reading the archive, which is out of |
| 13 | +scope for a byte-signature gate. |
| 14 | +""" |
| 15 | + |
| 16 | +from __future__ import annotations |
| 17 | + |
| 18 | +# short name -> list of acceptable leading-byte signatures |
| 19 | +_SIGNATURES: dict[str, list[bytes]] = { |
| 20 | + "zip": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"], |
| 21 | + "xlsx": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"], # zip container |
| 22 | + "docx": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"], |
| 23 | + "pptx": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"], |
| 24 | + "pdf": [b"%PDF-"], |
| 25 | + "png": [b"\x89PNG\r\n\x1a\n"], |
| 26 | + "jpg": [b"\xff\xd8\xff"], |
| 27 | + "gif": [b"GIF87a", b"GIF89a"], |
| 28 | + "json": [], # text: no signature, validated as utf-8 text below |
| 29 | + "csv": [], |
| 30 | + "txt": [], |
| 31 | + "text": [], |
| 32 | +} |
| 33 | + |
| 34 | +_TEXT_TYPES = {"json", "csv", "txt", "text"} |
| 35 | + |
| 36 | + |
| 37 | +def is_probably_text(head: bytes) -> bool: |
| 38 | + """Heuristic: decodable as UTF-8 and free of NUL bytes.""" |
| 39 | + if b"\x00" in head: |
| 40 | + return False |
| 41 | + try: |
| 42 | + head.decode("utf-8") |
| 43 | + return True |
| 44 | + except UnicodeDecodeError: |
| 45 | + # a multibyte char may be split at the chunk boundary; tolerate a |
| 46 | + # short tail by retrying without the last few bytes |
| 47 | + try: |
| 48 | + head[:-3].decode("utf-8") |
| 49 | + return True |
| 50 | + except UnicodeDecodeError: |
| 51 | + return False |
| 52 | + |
| 53 | + |
| 54 | +def sniff_matches(head: bytes, allowed: list[str]) -> bool: |
| 55 | + """Does ``head`` match at least one of the ``allowed`` type names? |
| 56 | +
|
| 57 | + Empty ``allowed`` means validation is disabled -> always True. |
| 58 | + Unknown type names in ``allowed`` are ignored (they can never match), |
| 59 | + so a typo fails closed rather than silently allowing everything. |
| 60 | + """ |
| 61 | + if not allowed: |
| 62 | + return True |
| 63 | + for name in allowed: |
| 64 | + key = name.lower().lstrip(".") |
| 65 | + sigs = _SIGNATURES.get(key) |
| 66 | + if sigs is None: |
| 67 | + continue # unknown type name: cannot match |
| 68 | + if key in _TEXT_TYPES: |
| 69 | + if is_probably_text(head): |
| 70 | + return True |
| 71 | + continue |
| 72 | + if any(head.startswith(sig) for sig in sigs): |
| 73 | + return True |
| 74 | + return False |
0 commit comments