From 5efa37baa88c506966f57c2f547ed5ecefb7a513 Mon Sep 17 00:00:00 2001 From: Hrishikesh Yadav Date: Tue, 6 Oct 2026 08:36:52 +0530 Subject: [PATCH 1/2] feat(tokenizer): implement deterministic dataset tokenization with Merkle verification - Add abstract load, encode, decode methods to BaseTokenizer - Implement load, encode, and decode for BPETokenizer and SentencePieceTokenizer - Add load_tokenizer factory function with automatic artifact detection - Support SentencePiece and BPE in hash_tokenizer_config - Implement memory-efficient streaming tokenize_dataset with little-endian binary output, SHA256 integrity, Merkle root computation, and cryptographic manifest chain linkage - Implement verify_tokenized_dataset producing detailed VerificationReport across file existence, binary SHA256, Merkle root, input dataset hash, tokenizer config, and parent manifest chain - Add CLI scripts scripts/tokenize_dataset.py and scripts/verify_tokenized_dataset.py - Add comprehensive 13-test suite in tests/test_tokenize_dataset.py Closes #61 Addresses #51, #52, #55 --- .../openverifiablellm/tokenizer/__init__.py | 12 + Legacy/openverifiablellm/tokenizer/base.py | 18 +- .../tokenizer/bpe_tokenizer.py | 40 +- Legacy/openverifiablellm/tokenizer/factory.py | 48 ++- .../tokenizer/sentencepiece_tokenizer.py | 44 +- .../tokenizer/tokenize_dataset.py | 388 ++++++++++++++++++ Legacy/openverifiablellm/tokenizer/train.py | 82 +++- Legacy/scripts/tokenize_dataset.py | 93 +++++ Legacy/scripts/verify_tokenized_dataset.py | 63 +++ Legacy/tests/test_tokenize_dataset.py | 336 +++++++++++++++ 10 files changed, 1098 insertions(+), 26 deletions(-) create mode 100644 Legacy/openverifiablellm/tokenizer/tokenize_dataset.py create mode 100644 Legacy/scripts/tokenize_dataset.py create mode 100644 Legacy/scripts/verify_tokenized_dataset.py create mode 100644 Legacy/tests/test_tokenize_dataset.py diff --git a/Legacy/openverifiablellm/tokenizer/__init__.py b/Legacy/openverifiablellm/tokenizer/__init__.py index 10ea2ae2..b04fda0c 100644 --- a/Legacy/openverifiablellm/tokenizer/__init__.py +++ b/Legacy/openverifiablellm/tokenizer/__init__.py @@ -1,6 +1,18 @@ +from .base import BaseTokenizer +from .bpe_tokenizer import BPETokenizer +from .factory import create_tokenizer, load_tokenizer +from .sentencepiece_tokenizer import SentencePieceTokenizer +from .tokenize_dataset import tokenize_dataset, verify_tokenized_dataset from .train import hash_tokenizer_config, train_tokenizer __all__ = [ + "BaseTokenizer", + "BPETokenizer", + "SentencePieceTokenizer", + "create_tokenizer", + "load_tokenizer", + "tokenize_dataset", + "verify_tokenized_dataset", "train_tokenizer", "hash_tokenizer_config", ] diff --git a/Legacy/openverifiablellm/tokenizer/base.py b/Legacy/openverifiablellm/tokenizer/base.py index 5a8d3fd0..f5d7f8ac 100644 --- a/Legacy/openverifiablellm/tokenizer/base.py +++ b/Legacy/openverifiablellm/tokenizer/base.py @@ -1,5 +1,6 @@ from abc import ABC, abstractmethod from pathlib import Path +from typing import List, Optional, Union class BaseTokenizer(ABC): @@ -22,10 +23,25 @@ def train(self, text_file: Path, save_path: Path): """Train tokenizer and save model.""" pass + @abstractmethod + def load(self, tokenizer_dir: Path): + """Load trained tokenizer model from directory.""" + pass + + @abstractmethod + def encode(self, text: str) -> List[int]: + """Encode text into a list of token IDs.""" + pass + + @abstractmethod + def decode(self, token_ids: List[int]) -> str: + """Decode a list of token IDs back into text.""" + pass + @abstractmethod def get_vocab_path(self, tokenizer_dir: Path) -> Path: pass @abstractmethod - def get_merges_path(self, tokenizer_dir: Path): + def get_merges_path(self, tokenizer_dir: Path) -> Optional[Path]: pass diff --git a/Legacy/openverifiablellm/tokenizer/bpe_tokenizer.py b/Legacy/openverifiablellm/tokenizer/bpe_tokenizer.py index 9a18d912..0545880f 100644 --- a/Legacy/openverifiablellm/tokenizer/bpe_tokenizer.py +++ b/Legacy/openverifiablellm/tokenizer/bpe_tokenizer.py @@ -1,4 +1,5 @@ from pathlib import Path +from typing import List, Optional from tokenizers import ByteLevelBPETokenizer @@ -8,7 +9,17 @@ class BPETokenizer(BaseTokenizer): + def __init__(self, vocab_size: int = 32000, min_frequency: int = 2): + super().__init__(vocab_size, min_frequency) + self._tokenizer: Optional[ByteLevelBPETokenizer] = None + def train(self, text_file: Path, save_path: Path): + text_file = Path(text_file) + save_path = Path(save_path) + if not text_file.is_file(): + raise FileNotFoundError(f"Text file not found: {text_file}") + + save_path.mkdir(parents=True, exist_ok=True) tokenizer = ByteLevelBPETokenizer() tokenizer.train( @@ -19,9 +30,34 @@ def train(self, text_file: Path, save_path: Path): ) tokenizer.save_model(str(save_path)) + self._tokenizer = tokenizer + + def load(self, tokenizer_dir: Path): + tokenizer_dir = Path(tokenizer_dir) + vocab_path = self.get_vocab_path(tokenizer_dir) + merges_path = self.get_merges_path(tokenizer_dir) + + if not vocab_path.is_file(): + raise FileNotFoundError(f"vocab.json not found at {vocab_path}") + if not merges_path.is_file(): + raise FileNotFoundError(f"merges.txt not found at {merges_path}") + + self._tokenizer = ByteLevelBPETokenizer.from_file(str(vocab_path), str(merges_path)) + + def encode(self, text: str) -> List[int]: + if not isinstance(text, str): + raise TypeError(f"text must be str, got {type(text).__name__}") + if self._tokenizer is None: + raise RuntimeError("Tokenizer is not trained or loaded. Call train() or load() first.") + return self._tokenizer.encode(text).ids + + def decode(self, token_ids: List[int]) -> str: + if self._tokenizer is None: + raise RuntimeError("Tokenizer is not trained or loaded. Call train() or load() first.") + return self._tokenizer.decode(token_ids) def get_vocab_path(self, tokenizer_dir: Path) -> Path: - return tokenizer_dir / "vocab.json" + return Path(tokenizer_dir) / "vocab.json" def get_merges_path(self, tokenizer_dir: Path) -> Path: - return tokenizer_dir / "merges.txt" + return Path(tokenizer_dir) / "merges.txt" diff --git a/Legacy/openverifiablellm/tokenizer/factory.py b/Legacy/openverifiablellm/tokenizer/factory.py index de004f64..19502595 100644 --- a/Legacy/openverifiablellm/tokenizer/factory.py +++ b/Legacy/openverifiablellm/tokenizer/factory.py @@ -1,8 +1,16 @@ +from pathlib import Path +from typing import Optional, Union + +from .base import BaseTokenizer from .bpe_tokenizer import BPETokenizer from .sentencepiece_tokenizer import SentencePieceTokenizer -def create_tokenizer(tokenizer_type, vocab_size, min_frequency): +def create_tokenizer( + tokenizer_type: str, + vocab_size: int = 32000, + min_frequency: int = 2, +) -> BaseTokenizer: tokenizer_type = tokenizer_type.lower() if tokenizer_type == "bpe": @@ -12,3 +20,41 @@ def create_tokenizer(tokenizer_type, vocab_size, min_frequency): return SentencePieceTokenizer(vocab_size, min_frequency) raise ValueError(f"Unsupported tokenizer: {tokenizer_type}") + + +def load_tokenizer( + tokenizer_dir: Union[str, Path], + tokenizer_type: Optional[str] = None, +) -> BaseTokenizer: + """ + Load a trained tokenizer from directory. + + If tokenizer_type is not provided, automatically detects whether + SentencePiece (spm.model) or BPE (vocab.json, merges.txt) artifacts are present. + """ + tokenizer_dir = Path(tokenizer_dir) + if not tokenizer_dir.is_dir(): + raise NotADirectoryError(f"Tokenizer directory not found: {tokenizer_dir}") + + t_type = tokenizer_type.lower() if tokenizer_type else None + + has_spm = (tokenizer_dir / "spm.model").is_file() + has_bpe = (tokenizer_dir / "vocab.json").is_file() + + if t_type == "sentencepiece" or (t_type is None and has_spm and not has_bpe): + tok = SentencePieceTokenizer() + tok.load(tokenizer_dir) + return tok + + if t_type == "bpe" or (t_type is None and has_bpe): + tok = BPETokenizer() + tok.load(tokenizer_dir) + return tok + + if t_type is not None: + raise ValueError(f"Unsupported tokenizer type: {tokenizer_type}") + + raise FileNotFoundError( + f"Could not identify tokenizer artifacts in {tokenizer_dir}. " + "Expected spm.model for SentencePiece or vocab.json/merges.txt for BPE." + ) diff --git a/Legacy/openverifiablellm/tokenizer/sentencepiece_tokenizer.py b/Legacy/openverifiablellm/tokenizer/sentencepiece_tokenizer.py index 91fb9d08..59335fc7 100644 --- a/Legacy/openverifiablellm/tokenizer/sentencepiece_tokenizer.py +++ b/Legacy/openverifiablellm/tokenizer/sentencepiece_tokenizer.py @@ -1,4 +1,5 @@ from pathlib import Path +from typing import List, Optional import sentencepiece as spm @@ -10,7 +11,17 @@ class SentencePieceTokenizer(BaseTokenizer): SentencePiece tokenizer implementation. """ + def __init__(self, vocab_size: int = 32000, min_frequency: int = 2): + super().__init__(vocab_size, min_frequency) + self._sp: Optional[spm.SentencePieceProcessor] = None + def train(self, text_file: Path, save_path: Path): + text_file = Path(text_file) + save_path = Path(save_path) + if not text_file.is_file(): + raise FileNotFoundError(f"Text file not found: {text_file}") + + save_path.mkdir(parents=True, exist_ok=True) model_prefix = save_path / "spm" spm.SentencePieceTrainer.train( @@ -19,9 +30,36 @@ def train(self, text_file: Path, save_path: Path): vocab_size=self.vocab_size, ) - def get_vocab_path(self, tokenizer_dir: Path): - return tokenizer_dir / "spm.vocab" + model_file = save_path / "spm.model" + if model_file.is_file(): + self._sp = spm.SentencePieceProcessor(model_file=str(model_file)) + + def load(self, tokenizer_dir: Path): + tokenizer_dir = Path(tokenizer_dir) + model_file = self.get_model_path(tokenizer_dir) + if not model_file.is_file(): + raise FileNotFoundError(f"spm.model not found at {model_file}") + + self._sp = spm.SentencePieceProcessor(model_file=str(model_file)) + + def encode(self, text: str) -> List[int]: + if not isinstance(text, str): + raise TypeError(f"text must be str, got {type(text).__name__}") + if self._sp is None: + raise RuntimeError("Tokenizer is not trained or loaded. Call train() or load() first.") + return [int(tok) for tok in self._sp.encode(text, out_type=int)] + + def decode(self, token_ids: List[int]) -> str: + if self._sp is None: + raise RuntimeError("Tokenizer is not trained or loaded. Call train() or load() first.") + return self._sp.decode([int(t) for t in token_ids]) + + def get_model_path(self, tokenizer_dir: Path) -> Path: + return Path(tokenizer_dir) / "spm.model" + + def get_vocab_path(self, tokenizer_dir: Path) -> Path: + return Path(tokenizer_dir) / "spm.vocab" - def get_merges_path(self, tokenizer_dir: Path): + def get_merges_path(self, tokenizer_dir: Path) -> Optional[Path]: # SentencePiece does not use merges return None diff --git a/Legacy/openverifiablellm/tokenizer/tokenize_dataset.py b/Legacy/openverifiablellm/tokenizer/tokenize_dataset.py new file mode 100644 index 00000000..67ac42e4 --- /dev/null +++ b/Legacy/openverifiablellm/tokenizer/tokenize_dataset.py @@ -0,0 +1,388 @@ +from datetime import datetime, timezone +import json +import logging +from pathlib import Path +from typing import Any, Dict, List, Optional, Union + +import numpy as np + +from openverifiablellm.manifest_chain import ( + get_parent_manifest_hash, + verify_manifest_chain_link, +) +from openverifiablellm.utils import ( + compute_merkle_root, + compute_sha256, +) +from openverifiablellm.verify import ( + CheckResult, + CheckStatus, + VerificationReport, +) + +from .base import BaseTokenizer +from .factory import load_tokenizer +from .train import hash_tokenizer_config + +logger = logging.getLogger(__name__) + +DEFAULT_CHUNK_SIZE_BYTES = 1048576 # 1 MB +SUPPORTED_DTYPES = {"uint16", "uint32"} +DTYPE_MAP = { + "uint16": np.dtype(" str: + return json.dumps(obj, indent=2, sort_keys=True) + + +def tokenize_dataset( + input_file: Union[str, Path], + tokenizer: Union[BaseTokenizer, str, Path], + output_file: Union[str, Path], + manifest_path: Optional[Union[str, Path]] = None, + previous_manifest_path: Optional[Union[str, Path]] = None, + chunk_size_bytes: int = DEFAULT_CHUNK_SIZE_BYTES, + dtype: str = "uint32", + write_manifest: bool = True, +) -> Dict[str, Any]: + """ + Tokenize a text dataset in a deterministic, memory-efficient streaming fashion. + + Parameters + ---------- + input_file : str or Path + Path to the preprocessed dataset text file. + tokenizer : BaseTokenizer or str or Path + Tokenizer instance with encode() method, or path to trained tokenizer directory. + output_file : str or Path + Path where tokenized binary dataset (.bin) will be written. + manifest_path : str or Path, optional + Path where tokenized dataset manifest will be written. Defaults to output_file.parent / "tokenized_manifest.json". + previous_manifest_path : str or Path, optional + Path to the previous manifest (e.g. preprocessing dataset_manifest.json) to establish a cryptographic chain. + chunk_size_bytes : int, default 1MB + Byte size of each chunk for Merkle tree construction. + dtype : str, default "uint32" + Data type for stored token IDs ("uint16" or "uint32"). + write_manifest : bool, default True + Whether to write the generated manifest JSON to disk. + + Returns + ------- + dict + Tokenization manifest dictionary with hashes, Merkle root, and metadata. + """ + input_path = Path(input_file) + output_path = Path(output_file) + + if not input_path.is_file(): + raise FileNotFoundError(f"Input dataset file not found: {input_path}") + + if dtype not in SUPPORTED_DTYPES: + raise ValueError( + f"Unsupported dtype: '{dtype}'. Supported dtypes are: {sorted(SUPPORTED_DTYPES)}" + ) + + if chunk_size_bytes <= 0: + raise ValueError("chunk_size_bytes must be > 0") + + tokenizer_dir: Optional[Path] = None + if isinstance(tokenizer, (str, Path)): + tokenizer_dir = Path(tokenizer) + tok_instance = load_tokenizer(tokenizer_dir) + elif isinstance(tokenizer, BaseTokenizer): + tok_instance = tokenizer + else: + tok_instance = tokenizer + + if not hasattr(tok_instance, "encode") or not callable(getattr(tok_instance, "encode")): + raise TypeError( + f"Tokenizer instance must implement a callable encode() method, got {type(tok_instance).__name__}" + ) + + output_path.parent.mkdir(parents=True, exist_ok=True) + + # Use explicit little-endian byte ordering for guaranteed cross-platform reproducibility + np_dtype = DTYPE_MAP[dtype] + + total_tokens = 0 + total_bytes = 0 + + logger.info("Starting deterministic streaming tokenization: %s -> %s", input_path, output_path) + + with input_path.open("r", encoding="utf-8") as fin, output_path.open("wb") as fout: + for line in fin: + text = line.strip() + if not text: + continue + + encoded = tok_instance.encode(text) + if isinstance(encoded, list): + token_ids = encoded + elif hasattr(encoded, "ids"): + token_ids = encoded.ids + else: + raise TypeError( + f"Tokenizer.encode() returned unsupported type: {type(encoded).__name__}. " + "Expected list of ints or object with 'ids' attribute." + ) + + if not token_ids: + continue + + arr = np.array(token_ids, dtype=np_dtype) + raw_bytes = arr.tobytes() + fout.write(raw_bytes) + + total_tokens += len(token_ids) + total_bytes += len(raw_bytes) + + fout.flush() + + input_sha256 = compute_sha256(file_path=input_path) + tokenized_sha256 = compute_sha256(file_path=output_path) + merkle_root = compute_merkle_root(output_path, chunk_size=chunk_size_bytes) + + parent_manifest_hash: Optional[str] = None + if previous_manifest_path is not None: + prev_path = Path(previous_manifest_path) + if prev_path.is_file(): + parent_manifest_hash = get_parent_manifest_hash(prev_path) + else: + logger.warning("Previous manifest path specified but not found: %s", prev_path) + + tok_config: Optional[Dict[str, Any]] = None + if tokenizer_dir is not None and tokenizer_dir.is_dir(): + try: + tok_config = hash_tokenizer_config(tokenizer_dir) + except Exception as e: + logger.warning("Could not compute tokenizer config hash: %s", e) + + manifest_data: Dict[str, Any] = { + "version": "1.0.0", + "created_at": datetime.now(timezone.utc).isoformat(), + "input_dataset_path": str(input_path), + "input_dataset_sha256": input_sha256, + "tokenized_dataset_path": str(output_path), + "tokenized_dataset_sha256": tokenized_sha256, + "merkle_root": merkle_root, + "chunk_size_bytes": chunk_size_bytes, + "total_tokens": total_tokens, + "total_bytes": total_bytes, + "dtype": dtype, + "tokenizer_config": tok_config, + "parent_manifest_hash": parent_manifest_hash, + } + + if write_manifest: + if manifest_path is None: + manifest_path = output_path.parent / "tokenized_manifest.json" + target_manifest = Path(manifest_path) + target_manifest.parent.mkdir(parents=True, exist_ok=True) + target_manifest.write_text(_canonical_json(manifest_data), encoding="utf-8") + logger.info("Saved tokenized dataset manifest to %s", target_manifest) + + return manifest_data + + +def verify_tokenized_dataset( + tokenized_file: Union[str, Path], + manifest_path: Union[str, Path], + tokenizer_path: Optional[Union[str, Path]] = None, + input_file: Optional[Union[str, Path]] = None, + previous_manifest_path: Optional[Union[str, Path]] = None, +) -> VerificationReport: + """ + Verify tokenized dataset binary artifact against its cryptographic manifest. + + Validates: + - Tokenized binary file existence + - Byte-level SHA256 integrity + - Merkle root verification over chunk hashes + - Tokenizer config hashes (if tokenizer_path provided) + - Input dataset SHA256 integrity (if input_file provided) + - Parent manifest chain linkage (if previous_manifest_path provided) + + Returns + ------- + VerificationReport + Detailed verification report with individual check results. + """ + tok_file = Path(tokenized_file) + mf_path = Path(manifest_path) + + report = VerificationReport( + input_dump=str(tok_file), + manifest_path=str(mf_path), + previous_manifest_path=str(previous_manifest_path) if previous_manifest_path else None, + ) + + if not mf_path.is_file(): + report.add( + CheckResult( + name="manifest_exists", + status=CheckStatus.FAIL, + expected=str(mf_path), + actual="FileNotFound", + detail=f"Tokenized manifest not found at {mf_path}", + ) + ) + return report + + try: + manifest_data = json.loads(mf_path.read_text(encoding="utf-8")) + except json.JSONDecodeError as e: + report.add( + CheckResult( + name="manifest_json_valid", + status=CheckStatus.FAIL, + detail=f"Malformed manifest JSON: {e}", + ) + ) + return report + + # 1. Check tokenized file existence + if not tok_file.is_file(): + report.add( + CheckResult( + name="tokenized_file_exists", + status=CheckStatus.FAIL, + expected=str(tok_file), + actual="FileNotFound", + detail=f"Tokenized dataset binary not found at {tok_file}", + ) + ) + return report + else: + report.add(CheckResult(name="tokenized_file_exists", status=CheckStatus.PASS)) + + # 2. Check tokenized dataset SHA256 + expected_sha = manifest_data.get("tokenized_dataset_sha256") + actual_sha = compute_sha256(file_path=tok_file) + if expected_sha and actual_sha == expected_sha: + report.add(CheckResult(name="tokenized_sha256", status=CheckStatus.PASS)) + else: + report.add( + CheckResult( + name="tokenized_sha256", + status=CheckStatus.FAIL, + expected=expected_sha, + actual=actual_sha, + detail="Tokenized binary SHA256 does not match recorded manifest hash", + ) + ) + + # 3. Check Merkle root + expected_merkle = manifest_data.get("merkle_root") + chunk_size = manifest_data.get("chunk_size_bytes", DEFAULT_CHUNK_SIZE_BYTES) + actual_merkle = compute_merkle_root(tok_file, chunk_size=chunk_size) + if expected_merkle and actual_merkle == expected_merkle: + report.add(CheckResult(name="merkle_root", status=CheckStatus.PASS)) + else: + report.add( + CheckResult( + name="merkle_root", + status=CheckStatus.FAIL, + expected=expected_merkle, + actual=actual_merkle, + detail=f"Merkle root mismatch using chunk size {chunk_size} bytes", + ) + ) + + # 4. Optional: check input dataset SHA256 + if input_file is not None: + in_path = Path(input_file) + expected_in_sha = manifest_data.get("input_dataset_sha256") + if not in_path.is_file(): + report.add( + CheckResult( + name="input_dataset_sha256", + status=CheckStatus.FAIL, + expected=expected_in_sha, + actual="FileNotFound", + detail=f"Input file not found at {in_path}", + ) + ) + else: + actual_in_sha = compute_sha256(file_path=in_path) + if expected_in_sha and actual_in_sha == expected_in_sha: + report.add(CheckResult(name="input_dataset_sha256", status=CheckStatus.PASS)) + else: + report.add( + CheckResult( + name="input_dataset_sha256", + status=CheckStatus.FAIL, + expected=expected_in_sha, + actual=actual_in_sha, + detail="Input dataset text SHA256 mismatch", + ) + ) + + # 5. Optional: check tokenizer configuration hashes + if tokenizer_path is not None: + tok_dir = Path(tokenizer_path) + expected_tok_config = manifest_data.get("tokenizer_config") + if not tok_dir.is_dir(): + report.add( + CheckResult( + name="tokenizer_config", + status=CheckStatus.FAIL, + detail=f"Tokenizer directory not found: {tok_dir}", + ) + ) + else: + try: + actual_tok_config = hash_tokenizer_config(tok_dir) + if expected_tok_config and actual_tok_config == expected_tok_config: + report.add(CheckResult(name="tokenizer_config", status=CheckStatus.PASS)) + else: + report.add( + CheckResult( + name="tokenizer_config", + status=CheckStatus.FAIL, + expected=str(expected_tok_config), + actual=str(actual_tok_config), + detail="Tokenizer configuration hashes do not match recorded config", + ) + ) + except Exception as e: + report.add( + CheckResult( + name="tokenizer_config", + status=CheckStatus.FAIL, + detail=f"Failed to hash tokenizer directory: {e}", + ) + ) + + # 6. Optional: check cryptographic manifest chain link + if previous_manifest_path is not None: + prev_path = Path(previous_manifest_path) + if not prev_path.is_file(): + report.add( + CheckResult( + name="parent_manifest_chain", + status=CheckStatus.FAIL, + detail=f"Previous manifest file not found: {prev_path}", + ) + ) + else: + is_valid = verify_manifest_chain_link(prev_path, manifest_data) + if is_valid: + report.add(CheckResult(name="parent_manifest_chain", status=CheckStatus.PASS)) + else: + expected_parent_hash = manifest_data.get("parent_manifest_hash") + actual_parent_hash = get_parent_manifest_hash(prev_path) + report.add( + CheckResult( + name="parent_manifest_chain", + status=CheckStatus.FAIL, + expected=expected_parent_hash, + actual=actual_parent_hash, + detail="Parent manifest cryptographic link mismatch", + ) + ) + + return report diff --git a/Legacy/openverifiablellm/tokenizer/train.py b/Legacy/openverifiablellm/tokenizer/train.py index bc8fcc34..31c42729 100644 --- a/Legacy/openverifiablellm/tokenizer/train.py +++ b/Legacy/openverifiablellm/tokenizer/train.py @@ -1,7 +1,7 @@ import json import logging from pathlib import Path -from typing import Union +from typing import Optional, Union from openverifiablellm.utils import compute_sha256 @@ -63,32 +63,76 @@ def train_tokenizer( return save_path -def hash_tokenizer_config(tokenizer_path: Union[str, Path]) -> dict: +def hash_tokenizer_config( + tokenizer_path: Union[str, Path], + tokenizer_type: Optional[str] = None, +) -> dict: """ Compute SHA256 hashes of tokenizer configuration files. - """ + Supports both BPE (vocab.json, merges.txt) and SentencePiece (spm.model, spm.vocab). + Automatically detects tokenizer type from artifacts present on disk if not explicitly specified. + """ tokenizer_path = Path(tokenizer_path) - vocab_path = tokenizer_path / "vocab.json" - merges_path = tokenizer_path / "merges.txt" + t_type = tokenizer_type.lower() if tokenizer_type else None + + has_spm = (tokenizer_path / "spm.model").is_file() or (tokenizer_path / "spm.vocab").is_file() + has_bpe = (tokenizer_path / "vocab.json").is_file() or (tokenizer_path / "merges.txt").is_file() + + if t_type == "sentencepiece" or (t_type is None and has_spm and not has_bpe): + model_path = tokenizer_path / "spm.model" + vocab_path = tokenizer_path / "spm.vocab" + + if not model_path.is_file(): + raise FileNotFoundError(f"spm.model not found at {model_path}") + if not vocab_path.is_file(): + raise FileNotFoundError(f"spm.vocab not found at {vocab_path}") + + vocab_bytes = vocab_path.read_bytes() + vocab_hash = compute_sha256(data=vocab_bytes) + vocab_lines = [ + line + for line in vocab_bytes.decode("utf-8", errors="replace").splitlines() + if line.strip() + ] + actual_vocab_size = len(vocab_lines) + + model_hash = compute_sha256(file_path=model_path) + + logger.info("SentencePiece tokenizer config hashed successfully") + + return { + "tokenizer_type": "sentencepiece", + "tokenizer_vocab_hash": vocab_hash, + "tokenizer_model_hash": model_hash, + "tokenizer_merges_hash": None, + "tokenizer_vocab_size": actual_vocab_size, + } + + elif t_type == "bpe" or (t_type is None and (has_bpe or not has_spm)): + vocab_path = tokenizer_path / "vocab.json" + merges_path = tokenizer_path / "merges.txt" - if not vocab_path.is_file(): - raise FileNotFoundError(f"vocab.json not found at {vocab_path}") + if not vocab_path.is_file(): + raise FileNotFoundError(f"vocab.json not found at {vocab_path}") + if not merges_path.is_file(): + raise FileNotFoundError(f"merges.txt not found at {merges_path}") - if not merges_path.is_file(): - raise FileNotFoundError(f"merges.txt not found at {merges_path}") + vocab_bytes = vocab_path.read_bytes() + vocab_hash = compute_sha256(data=vocab_bytes) + actual_vocab_size = len(json.loads(vocab_bytes.decode("utf-8"))) - vocab_bytes = vocab_path.read_bytes() - vocab_hash = compute_sha256(data=vocab_bytes) - actual_vocab_size = len(json.loads(vocab_bytes.decode("utf-8"))) + merges_hash = compute_sha256(file_path=merges_path) - merges_hash = compute_sha256(file_path=merges_path) + logger.info("BPE tokenizer config hashed successfully") - logger.info("Tokenizer config hashed successfully") + return { + "tokenizer_type": "bpe", + "tokenizer_vocab_hash": vocab_hash, + "tokenizer_merges_hash": merges_hash, + "tokenizer_vocab_size": actual_vocab_size, + } - return { - "tokenizer_vocab_hash": vocab_hash, - "tokenizer_merges_hash": merges_hash, - "tokenizer_vocab_size": actual_vocab_size, - } + else: + raise ValueError(f"Unsupported or ambiguous tokenizer type: {tokenizer_type}") diff --git a/Legacy/scripts/tokenize_dataset.py b/Legacy/scripts/tokenize_dataset.py new file mode 100644 index 00000000..5da73e2b --- /dev/null +++ b/Legacy/scripts/tokenize_dataset.py @@ -0,0 +1,93 @@ +import argparse +import json +import logging +from pathlib import Path +import sys + +from openverifiablellm.tokenizer import tokenize_dataset + +logging.basicConfig(level=logging.INFO, format="%(levelname)s - %(message)s") +logger = logging.getLogger(__name__) + + +def main(): + parser = argparse.ArgumentParser( + description="Deterministically tokenize preprocessed dataset text into binary format with Merkle verification." + ) + parser.add_argument( + "--input", + required=True, + help="Path to preprocessed input text file (e.g. data/processed/wiki_clean.txt)", + ) + parser.add_argument( + "--tokenizer", + required=True, + help="Path to trained tokenizer directory (containing spm.model or vocab.json)", + ) + parser.add_argument( + "--output", + required=True, + help="Path where tokenized binary output will be written (.bin)", + ) + parser.add_argument( + "--manifest", + default=None, + help="Optional path to output tokenized manifest JSON", + ) + parser.add_argument( + "--previous-manifest", + default=None, + help="Optional path to previous preprocessing manifest for cryptographic chain linkage", + ) + parser.add_argument( + "--chunk-size", + type=int, + default=1048576, + help="Merkle chunk size in bytes (default: 1048576 = 1MB)", + ) + parser.add_argument( + "--dtype", + choices=["uint16", "uint32"], + default="uint32", + help="Data type for token IDs (default: uint32)", + ) + parser.add_argument( + "--no-manifest", + action="store_true", + help="Skip writing manifest file to disk", + ) + + args = parser.parse_args() + + try: + manifest = tokenize_dataset( + input_file=Path(args.input), + tokenizer=Path(args.tokenizer), + output_file=Path(args.output), + manifest_path=Path(args.manifest) if args.manifest else None, + previous_manifest_path=Path(args.previous_manifest) if args.previous_manifest else None, + chunk_size_bytes=args.chunk_size, + dtype=args.dtype, + write_manifest=not args.no_manifest, + ) + + print("\n" + "=" * 60) + print("TOKENIZATION COMPLETE & VERIFIED") + print("=" * 60) + print(f"Input file : {manifest['input_dataset_path']}") + print(f"Output binary file : {manifest['tokenized_dataset_path']}") + print(f"Total tokens : {manifest['total_tokens']:,}") + print(f"Total bytes : {manifest['total_bytes']:,}") + print(f"Token file SHA-256 : {manifest['tokenized_dataset_sha256']}") + print(f"Merkle root : {manifest['merkle_root']}") + if manifest.get("parent_manifest_hash"): + print(f"Parent manifest hash : {manifest['parent_manifest_hash']}") + print("=" * 60 + "\n") + + except Exception as e: + logger.error("Tokenization failed: %s", e) + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/Legacy/scripts/verify_tokenized_dataset.py b/Legacy/scripts/verify_tokenized_dataset.py new file mode 100644 index 00000000..03e3f1bd --- /dev/null +++ b/Legacy/scripts/verify_tokenized_dataset.py @@ -0,0 +1,63 @@ +import argparse +import logging +from pathlib import Path +import sys + +from openverifiablellm.tokenizer import verify_tokenized_dataset + +logging.basicConfig(level=logging.INFO, format="%(levelname)s - %(message)s") +logger = logging.getLogger(__name__) + + +def main(): + parser = argparse.ArgumentParser( + description="Verify tokenized dataset binary artifact against its cryptographic manifest." + ) + parser.add_argument( + "tokenized_file", + help="Path to tokenized binary dataset (.bin)", + ) + parser.add_argument( + "--manifest", + required=True, + help="Path to tokenized dataset manifest JSON", + ) + parser.add_argument( + "--tokenizer", + default=None, + help="Optional path to tokenizer directory to verify configuration hashes", + ) + parser.add_argument( + "--input-file", + default=None, + help="Optional path to original preprocessed text file to verify input hash", + ) + parser.add_argument( + "--previous-manifest", + default=None, + help="Optional path to previous manifest to verify cryptographic chain link", + ) + + args = parser.parse_args() + + try: + report = verify_tokenized_dataset( + tokenized_file=Path(args.tokenized_file), + manifest_path=Path(args.manifest), + tokenizer_path=Path(args.tokenizer) if args.tokenizer else None, + input_file=Path(args.input_file) if args.input_file else None, + previous_manifest_path=Path(args.previous_manifest) if args.previous_manifest else None, + ) + + print("\n" + report.summary() + "\n") + + if not report.all_passed: + sys.exit(1) + + except Exception as e: + logger.error("Verification encountered an unhandled error: %s", e) + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/Legacy/tests/test_tokenize_dataset.py b/Legacy/tests/test_tokenize_dataset.py new file mode 100644 index 00000000..45da163c --- /dev/null +++ b/Legacy/tests/test_tokenize_dataset.py @@ -0,0 +1,336 @@ +import json +from pathlib import Path +import shutil +import tempfile +import unittest + +import numpy as np + +from openverifiablellm.manifest_chain import compute_manifest_hash +from openverifiablellm.tokenizer import ( + BPETokenizer, + SentencePieceTokenizer, + create_tokenizer, + hash_tokenizer_config, + load_tokenizer, + tokenize_dataset, + train_tokenizer, + verify_tokenized_dataset, +) +from openverifiablellm.utils import ( + compute_merkle_root, + compute_sha256, + generate_merkle_proof, + verify_merkle_proof, +) +from openverifiablellm.verify import CheckStatus + +SAMPLE_CORPUS = """Artificial intelligence and verifiable computing are foundational to modern AI systems. +Reproducibility guarantees that computation produces identical outputs across different execution environments. +Deterministic dataset tokenization bridges preprocessed text and model training. +Cryptographic Merkle trees ensure that no chunk of training data can be silently altered or tampered with. +""" + + +class TestTokenizerContract(unittest.TestCase): + def setUp(self): + self.tmp = Path(tempfile.mkdtemp()) + self.text_file = self.tmp / "sample.txt" + self.text_file.write_text(SAMPLE_CORPUS, encoding="utf-8") + + def tearDown(self): + shutil.rmtree(self.tmp, ignore_errors=True) + + def test_bpe_train_load_encode_decode(self): + tok_dir = self.tmp / "bpe_model" + tok = BPETokenizer(vocab_size=100, min_frequency=1) + + # Before training or loading, encode/decode must fail + with self.assertRaises(RuntimeError): + tok.encode("hello") + with self.assertRaises(RuntimeError): + tok.decode([1, 2]) + + tok.train(self.text_file, tok_dir) + self.assertTrue((tok_dir / "vocab.json").is_file()) + self.assertTrue((tok_dir / "merges.txt").is_file()) + + ids = tok.encode("verifiable computing") + self.assertIsInstance(ids, list) + self.assertGreater(len(ids), 0) + decoded = tok.decode(ids) + self.assertIn("verifiable", decoded) + + # Test loading from directory into a fresh instance + new_tok = BPETokenizer() + new_tok.load(tok_dir) + new_ids = new_tok.encode("verifiable computing") + self.assertEqual(ids, new_ids) + self.assertEqual(new_tok.decode(new_ids), decoded) + + def test_sentencepiece_train_load_encode_decode(self): + tok_dir = self.tmp / "spm_model" + tok = SentencePieceTokenizer(vocab_size=100) + + with self.assertRaises(RuntimeError): + tok.encode("hello") + with self.assertRaises(RuntimeError): + tok.decode([1, 2]) + + tok.train(self.text_file, tok_dir) + self.assertTrue((tok_dir / "spm.model").is_file()) + self.assertTrue((tok_dir / "spm.vocab").is_file()) + + ids = tok.encode("verifiable computing") + self.assertIsInstance(ids, list) + self.assertGreater(len(ids), 0) + decoded = tok.decode(ids) + self.assertIn("verifiable", decoded) + + # Test loading from directory into fresh instance + new_tok = SentencePieceTokenizer() + new_tok.load(tok_dir) + new_ids = new_tok.encode("verifiable computing") + self.assertEqual(ids, new_ids) + self.assertEqual(new_tok.decode(new_ids), decoded) + + def test_load_tokenizer_auto_detection(self): + bpe_dir = self.tmp / "bpe_tok" + spm_dir = self.tmp / "spm_tok" + + train_tokenizer(self.text_file, save_path=bpe_dir, tokenizer_type="bpe", vocab_size=100, min_frequency=1) + train_tokenizer(self.text_file, save_path=spm_dir, tokenizer_type="sentencepiece", vocab_size=100) + + loaded_bpe = load_tokenizer(bpe_dir) + self.assertIsInstance(loaded_bpe, BPETokenizer) + + loaded_spm = load_tokenizer(spm_dir) + self.assertIsInstance(loaded_spm, SentencePieceTokenizer) + + with self.assertRaises(NotADirectoryError): + load_tokenizer(self.tmp / "nonexistent") + + empty_dir = self.tmp / "empty_dir" + empty_dir.mkdir() + with self.assertRaises(FileNotFoundError): + load_tokenizer(empty_dir) + + +class TestTokenizeDatasetPipeline(unittest.TestCase): + def setUp(self): + self.tmp = Path(tempfile.mkdtemp()) + self.text_file = self.tmp / "input.txt" + self.text_file.write_text(SAMPLE_CORPUS, encoding="utf-8") + self.tok_dir = self.tmp / "tokenizer" + train_tokenizer(self.text_file, save_path=self.tok_dir, tokenizer_type="bpe", vocab_size=120, min_frequency=1) + self.output_bin = self.tmp / "output.bin" + self.manifest_file = self.tmp / "tokenized_manifest.json" + + def tearDown(self): + shutil.rmtree(self.tmp, ignore_errors=True) + + def test_streaming_tokenization_and_manifest(self): + manifest = tokenize_dataset( + input_file=self.text_file, + tokenizer=self.tok_dir, + output_file=self.output_bin, + manifest_path=self.manifest_file, + dtype="uint32", + ) + + self.assertTrue(self.output_bin.is_file()) + self.assertTrue(self.manifest_file.is_file()) + + self.assertGreater(manifest["total_tokens"], 0) + self.assertEqual(manifest["total_bytes"], manifest["total_tokens"] * 4) + self.assertEqual(self.output_bin.stat().st_size, manifest["total_bytes"]) + + # Validate token array content + tokens = np.fromfile(self.output_bin, dtype=" Date: Tue, 6 Oct 2026 09:51:31 +0530 Subject: [PATCH 2/2] fix(tokenize): reject samefile aliases, validate token ID bounds, and require tokenizer provenance for manifests --- .../tokenizer/tokenize_dataset.py | 37 +++++++++++++++++-- Legacy/tests/test_tokenize_dataset.py | 32 ++++++++++++++++ 2 files changed, 65 insertions(+), 4 deletions(-) diff --git a/Legacy/openverifiablellm/tokenizer/tokenize_dataset.py b/Legacy/openverifiablellm/tokenizer/tokenize_dataset.py index 67ac42e4..963d89bf 100644 --- a/Legacy/openverifiablellm/tokenizer/tokenize_dataset.py +++ b/Legacy/openverifiablellm/tokenizer/tokenize_dataset.py @@ -32,6 +32,10 @@ "uint16": np.dtype(" str: @@ -81,6 +85,16 @@ def tokenize_dataset( if not input_path.is_file(): raise FileNotFoundError(f"Input dataset file not found: {input_path}") + if output_path.exists(): + try: + if input_path.samefile(output_path): + raise ValueError(f"Input file and output file cannot be the same file: {input_path}") + except OSError: + if input_path.resolve() == output_path.resolve(): + raise ValueError(f"Input file and output file cannot be the same file: {input_path}") + elif input_path.resolve() == output_path.resolve(): + raise ValueError(f"Input file and output file cannot be the same file: {input_path}") + if dtype not in SUPPORTED_DTYPES: raise ValueError( f"Unsupported dtype: '{dtype}'. Supported dtypes are: {sorted(SUPPORTED_DTYPES)}" @@ -95,6 +109,8 @@ def tokenize_dataset( tok_instance = load_tokenizer(tokenizer_dir) elif isinstance(tokenizer, BaseTokenizer): tok_instance = tokenizer + if hasattr(tokenizer, "tokenizer_dir") and tokenizer.tokenizer_dir: + tokenizer_dir = Path(tokenizer.tokenizer_dir) else: tok_instance = tokenizer @@ -103,10 +119,18 @@ def tokenize_dataset( f"Tokenizer instance must implement a callable encode() method, got {type(tok_instance).__name__}" ) + if write_manifest: + if tokenizer_dir is None or not tokenizer_dir.is_dir(): + raise ValueError( + "A valid tokenizer artifact directory must be provided when write_manifest=True " + "to compute tokenizer provenance hashes. Pass write_manifest=False if skipping manifest." + ) + output_path.parent.mkdir(parents=True, exist_ok=True) # Use explicit little-endian byte ordering for guaranteed cross-platform reproducibility np_dtype = DTYPE_MAP[dtype] + min_val, max_val = DTYPE_RANGES[dtype] total_tokens = 0 total_bytes = 0 @@ -133,6 +157,14 @@ def tokenize_dataset( if not token_ids: continue + for tid in token_ids: + if not isinstance(tid, (int, np.integer)) or isinstance(tid, bool): + raise TypeError(f"Token ID must be an integer, got {type(tid).__name__}: {tid}") + if tid < min_val or tid > max_val: + raise ValueError( + f"Token ID {tid} is outside allowable range [{min_val}, {max_val}] for {dtype}" + ) + arr = np.array(token_ids, dtype=np_dtype) raw_bytes = arr.tobytes() fout.write(raw_bytes) @@ -156,10 +188,7 @@ def tokenize_dataset( tok_config: Optional[Dict[str, Any]] = None if tokenizer_dir is not None and tokenizer_dir.is_dir(): - try: - tok_config = hash_tokenizer_config(tokenizer_dir) - except Exception as e: - logger.warning("Could not compute tokenizer config hash: %s", e) + tok_config = hash_tokenizer_config(tokenizer_dir) manifest_data: Dict[str, Any] = { "version": "1.0.0", diff --git a/Legacy/tests/test_tokenize_dataset.py b/Legacy/tests/test_tokenize_dataset.py index 45da163c..2fbf1f7d 100644 --- a/Legacy/tests/test_tokenize_dataset.py +++ b/Legacy/tests/test_tokenize_dataset.py @@ -180,6 +180,38 @@ def test_invalid_dtype_and_arguments(self): with self.assertRaises(TypeError): tokenize_dataset(self.text_file, object(), self.output_bin) + def test_samefile_rejection(self): + with self.assertRaises(ValueError): + tokenize_dataset(self.text_file, self.tok_dir, self.text_file) + + def test_token_id_range_validation(self): + class DummyNegativeTokenizer: + def encode(self, text): + return [-1, 5] + + class DummyOverflowTokenizer: + def encode(self, text): + return [70000] + + with self.assertRaises(ValueError): + tokenize_dataset(self.text_file, DummyNegativeTokenizer(), self.output_bin, write_manifest=False) + + with self.assertRaises(ValueError): + tokenize_dataset(self.text_file, DummyOverflowTokenizer(), self.output_bin, dtype="uint16", write_manifest=False) + + def test_manifest_provenance_requirement(self): + class DummyValidTokenizer: + def encode(self, text): + return [1, 2, 3] + + # Fails when write_manifest=True without valid tokenizer dir + with self.assertRaises(ValueError): + tokenize_dataset(self.text_file, DummyValidTokenizer(), self.output_bin, write_manifest=True) + + # Passes cleanly when write_manifest=False + m = tokenize_dataset(self.text_file, DummyValidTokenizer(), self.output_bin, write_manifest=False) + self.assertGreater(m["total_tokens"], 0) + def test_determinism_across_runs(self): out1 = self.tmp / "run1.bin" out2 = self.tmp / "run2.bin"