diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 94f5531..6cf2d62 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -2,11 +2,12 @@ name: Publish to PyPI # Releases https://pypi.org/p/telegram-channel-scraper with PyPI Trusted Publishing (no API token stored). # To release: bump `version` in pyproject.toml and merge to main (or push a v* tag, or run manually). -# An already-published version is skipped, so unrelated edits to pyproject.toml are harmless. +# An already-published version is skipped, so unrelated edits are harmless. +# A GitHub release v is created with the matching CHANGELOG.md section as notes. on: push: branches: [main] - paths: [pyproject.toml, .github/workflows/publish.yml] + paths: [pyproject.toml, CHANGELOG.md, .github/workflows/publish.yml] tags: ["v*"] workflow_dispatch: @@ -42,3 +43,27 @@ jobs: - uses: pypa/gh-action-pypi-publish@release/v1 with: skip-existing: true + + release: + needs: publish + runs-on: ubuntu-latest + permissions: + contents: write + steps: + - uses: actions/checkout@v4 + - uses: actions/download-artifact@v4 + with: + name: dist + path: dist/ + - name: Create GitHub release + env: + GH_TOKEN: ${{ github.token }} + run: | + VERSION=$(python3 -c "import tomllib;print(tomllib.load(open('pyproject.toml','rb'))['project']['version'])") + TAG="v$VERSION" + if gh release view "$TAG" >/dev/null 2>&1; then + echo "Release $TAG already exists"; exit 0 + fi + awk -v v="$VERSION" '$0 ~ "^## "v"$"{f=1;next} /^## /{f=0} f' CHANGELOG.md > notes.md + printf '\n---\n**PyPI:** https://pypi.org/project/telegram-channel-scraper/%s/\n' "$VERSION" >> notes.md + gh release create "$TAG" dist/* --target "$GITHUB_SHA" --title "$TAG" --notes-file notes.md diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..dffb871 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,29 @@ +# Changelog + +## 2.0.0 + +First release on PyPI: **`pip install "telegram-channel-scraper[all]"`** + +A complete rewrite of the old single-class script into the `tgscraper` package: a Python library, a CLI, an MCP server for AI assistants, and a web dashboard. It reads public Telegram channels through `t.me/s/`, with no API key, no login and no phone number. + +### Highlights +- **Rich data per post:** id, date, text and HTML, views, reactions per emoji, author, edited flag, forwards, replies, media (photos, videos, voice notes, documents, stickers, link previews), hashtags, mentions and links. Also channel info: title, description, subscribers and counters. +- **One-line API and CLI:** `tg.scrape("durov")` in Python, `tgscraper durov` in the terminal. Every command supports `--json`. +- **Search and filters:** Telegram's server-side search, date ranges, keywords, regex, hashtag, media type, minimum views, and full-history scraping. +- **Export:** JSON, JSONL, CSV (Excel-friendly), XLSX, SQLite (upsert archive) and Markdown. +- **Robust networking:** sync and async clients, concurrent multi-channel scraping, retries with backoff, `Retry-After` handling, rate limiting, and rotating HTTP/SOCKS proxies. +- **Incremental scraping, monitoring and media:** fetch only new posts since the last run; watch channels and push new posts to a webhook or a Telegram bot; download photos, videos and voice notes. +- **Analytics:** top posts, activity by day, hour and weekday, hashtags, top words, reactions, and English/Persian sentiment. +- **For AI assistants:** + - MCP server `tgscraper-mcp` with 8 tools, 2 prompts and 1 resource, for Claude, Cursor, VS Code, Windsurf and others; + - a Claude Skill; + - `llms.txt` and `AGENTS.md`. +- **Dashboard and Docker:** `tgscraper dashboard` opens a Streamlit UI; `docker compose up dashboard` runs it in a container. +- **Quality:** + - offline test suite on Python 3.9–3.13; + - a weekly live test against real `t.me`; + - releases published to PyPI with Trusted Publishing. + +### Compatibility +- `from telegram_scraper import TelegramScraper` still works. +- The PyPI package name is `telegram-channel-scraper`, because `tgscraper` was already taken. The import name and the commands are still `tgscraper`.