diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml new file mode 100644 index 0000000..18d2e02 --- /dev/null +++ b/.github/workflows/release.yml @@ -0,0 +1,89 @@ +name: Release + +# Push a tag like v1.1.0 to build the package, create the GitHub release (notes +# taken from that version's CHANGELOG.md section) and publish to PyPI. +on: + push: + tags: ["v*"] + +permissions: + contents: read + +jobs: + build: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + + - name: Check the tag matches the package version + run: | + version=$(python -c "import tomllib; print(tomllib.load(open('pyproject.toml', 'rb'))['project']['version'])") + if [ "v$version" != "$GITHUB_REF_NAME" ]; then + echo "::error::Tag $GITHUB_REF_NAME does not match pyproject.toml version $version" + exit 1 + fi + + - name: Build and check the distributions + run: | + python -m pip install build twine + python -m build + twine check --strict dist/* + + - uses: actions/upload-artifact@v4 + with: + name: dist + path: dist/ + + github-release: + needs: build + runs-on: ubuntu-latest + permissions: + contents: write + steps: + - uses: actions/checkout@v4 + + - uses: actions/download-artifact@v4 + with: + name: dist + path: dist/ + + - name: Extract this version's notes from CHANGELOG.md + run: | + version="${GITHUB_REF_NAME#v}" + awk -v heading="## [$version]" ' + /^## \[/ { if (found) exit; if (index($0, heading) == 1) { found = 1; next } } + found { print } + ' CHANGELOG.md > notes.md + if [ ! -s notes.md ]; then + echo "::error::CHANGELOG.md has no section for $version" + exit 1 + fi + + - name: Create the GitHub release + env: + GH_TOKEN: ${{ github.token }} + run: | + gh release create "$GITHUB_REF_NAME" dist/* \ + --title "$GITHUB_REF_NAME" --notes-file notes.md --verify-tag + + pypi: + needs: build + runs-on: ubuntu-latest + # PyPI trusted publishing: the project must list this repository, + # workflow (release.yml) and environment (pypi) as a trusted publisher. + environment: + name: pypi + url: https://pypi.org/project/vayu-whisper/ + permissions: + id-token: write + steps: + - uses: actions/download-artifact@v4 + with: + name: dist + path: dist/ + + - uses: pypa/gh-action-pypi-publish@release/v1 diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..311d372 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,63 @@ +# Changelog + +All notable changes to this project are documented here. The release workflow +publishes the section for each version as its GitHub release notes. + +## [1.1.0] - 2026-10-01 + +### Fixed + +- Batched decoding (`batch_size > 1`) no longer drops text at the end of each + 30-second window, and every segment records its own window's `seek` (#3). +- `word_timestamps=True` now works with batched decoding, the default for + `LightningWhisperMLX` (#4). +- Batched temperature fallback follows the `temperature` schedule and + re-checks each step; a single temperature means no fallback (#5). +- The rule that stops timestamps going backwards during decoding now applies + (#6). +- CLI: each input file gets its own output name instead of every file + overwriting the first one's output (#7). +- CLI: missing files, a missing ffmpeg and undecodable audio are reported + clearly and listed in the failure summary (#7). +- Speculative decoding runs, uses a key/value cache and per-model + spectrograms, and gives exactly the target model's greedy output (#8). + +### Added + +- `AudioLoadError`, raised when ffmpeg is missing or cannot decode the audio. +- `scripts/benchmark.py` to measure the batched speed-up on your own Mac (#11). +- README sections on loading local models and on batched vs sequential + decoding (#11). +- CI that runs the test suite on Linux and Apple Silicon (#2). + +### Changed + +These may need a change in code that calls Vayu: + +- `load_audio()` raises `FileNotFoundError` for a missing file (previously + `ValueError`), and `transcribe()` checks the path before loading the model. +- `quant` raises `ValueError` when it can't be applied (an unknown level, a + model without a quantised build, or a repo path) instead of silently loading + full precision (#9). +- `beam_size` and `patience` raise `NotImplementedError` up front; beam search + is not implemented. The CLI's `--patience` option is removed (#10). +- `hallucination_silence_threshold` with `batch_size > 1` now warns that it is + ignored. +- Speculative decoding's default pair is distil-large-v3 → large-v3; draft and + target models must share a vocabulary (#8). +- Decoded timestamps can differ slightly from 1.0.0 because the timestamp rule + now applies (#6). +- Development dependencies live in the `dev` extra only: `pip install -e + ".[dev]"` or `uv sync --extra dev` (#2). + +### Deprecated + +- `parallel_chunk_transcribe`: it never ran chunks in parallel. Use + `transcribe(audio, batch_size=N)` (#8). + +## [1.0.0] - 2026-01-19 + +First release on PyPI as `vayu-whisper`. + +[1.1.0]: https://github.com/CodeWithBehnam/vayu/compare/v1.0.0...v1.1.0 +[1.0.0]: https://github.com/CodeWithBehnam/vayu/releases/tag/v1.0.0 diff --git a/pyproject.toml b/pyproject.toml index aae6cc3..0ac4691 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "vayu-whisper" -version = "1.0.0" +version = "1.1.0" description = "The fastest Whisper speech-to-text on Apple Silicon. 3-5x faster transcription via MLX batched decoding on M1/M2/M3/M4 Macs." readme = "README.md" license = {text = "MIT"} diff --git a/tests/test_version.py b/tests/test_version.py new file mode 100644 index 0000000..0375bc9 --- /dev/null +++ b/tests/test_version.py @@ -0,0 +1,9 @@ +"""The package's __version__ must match the version in pyproject.toml.""" + +from importlib.metadata import version + +import whisper_mlx + + +def test_version_matches_installed_metadata(): + assert whisper_mlx.__version__ == version("vayu-whisper") diff --git a/whisper_mlx/__init__.py b/whisper_mlx/__init__.py index 84d7403..93df02f 100644 --- a/whisper_mlx/__init__.py +++ b/whisper_mlx/__init__.py @@ -74,7 +74,7 @@ parallel_chunk_transcribe, ) -__version__ = "1.0.0" +__version__ = "1.1.0" __all__ = [ # Main transcription function