diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7eac58a..23c82ea 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -18,20 +18,24 @@ jobs: with: python-version: "3.11" - - name: Install ruff + - name: Install dependencies run: | python -m pip install --upgrade pip - pip install ruff + pip install -e ".[dev]" - name: Run ruff run: | ruff check . + - name: Run mypy + run: | + mypy csv2vcard + test: runs-on: ubuntu-latest strategy: matrix: - python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"] + python-version: ["3.10", "3.11", "3.12", "3.13", "3.14"] steps: - uses: actions/checkout@v4 diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index ce4d87e..e744081 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -42,6 +42,14 @@ jobs: exit 1 fi + - name: Verify __version__ matches tag + run: | + MODULE_VERSION=$(python -c "import re; print(re.search(r'__version__ = \"(.+?)\"', open('csv2vcard/_version.py').read()).group(1))") + if [ "$MODULE_VERSION" != "${{ steps.get_version.outputs.VERSION }}" ]; then + echo "Error: csv2vcard/_version.py ($MODULE_VERSION) doesn't match tag version (${{ steps.get_version.outputs.VERSION }})" + exit 1 + fi + - name: Install build dependencies run: | python -m pip install --upgrade pip @@ -57,15 +65,13 @@ jobs: run: | twine check dist/* + # Uses PyPI Trusted Publishing (OIDC via id-token: write) - no API token secret. + # Requires a trusted publisher for this repo/workflow configured on pypi.org. - name: Publish to PyPI - env: - TWINE_USERNAME: __token__ - TWINE_PASSWORD: ${{ secrets.PYPI_API_TOKEN }} - run: | - twine upload dist/* + uses: pypa/gh-action-pypi-publish@release/v1 - name: Upload release assets - uses: softprops/action-gh-release@v1 + uses: softprops/action-gh-release@v2 with: files: | dist/*.whl diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..bb95266 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,48 @@ +# Changelog + +## 0.6.0 + +### Fixed + +- **Contacts with the same name no longer overwrite each other.** Files are suffixed (`smith_john_2.vcf`, ...) instead of silently replaced. +- **Excel "CSV UTF-8" files work.** The byte order mark no longer corrupts the first column, which previously dropped every contact's last name. +- **Line breaks in a cell can no longer inject properties.** All values are escaped or sanitized, including `\r`. +- **Output follows the vCard specs:** CRLF line endings (also on Windows) and line folding at 75 octets. +- **vCard 4.0:** + - Inline `PHOTO`, `LOGO` and `KEY` use `data:` URIs (`ENCODING=b` is not valid in 4.0). + - Phone numbers are valid `tel:` URIs; local numbers without a country code are written as text. + - `CATEGORIES` and `NICKNAME` keep their list separators instead of collapsing into one value. +- **vCard 3.0:** + - The vCard 2.1 `CHARSET=UTF-8` parameter is no longer emitted. + - `data:` URI photos are converted properly. + - Time zone names use `VALUE=text`. +- **Invalid values are skipped with a warning** instead of being written: geo coordinates, and dates in 3.0. +- **`iter_contacts` streams rows** instead of loading the whole file first. +- **`test_csv2vcard` is no longer collected by pytest** as a test. + +### Added + +- **vCard 2.1 output** (`-V 2.1`) for legacy Outlook, car kits and feature phones, with quoted-printable encoding for non-ASCII text. +- **RFC 9554 properties in vCard 4.0:** `PRONOUNS` and `SOCIALPROFILE`, plus `LANG`. Columns: `pronouns`, `social_profile`, `language`. +- **Stable UIDs:** UIDs come from the name, organization and email, or from a `uid` / `contact_id` / `external_id` column. Re-importing a re-converted CSV updates contacts instead of duplicating them. +- **Multiple values per field:** numbered columns such as `email_2` or `Phone 3` for phones, emails, websites and social profiles. +- **`--keep-unmapped` / `keep_unmapped=True`:** writes columns that match no field as `X-` properties. +- **Organization-only rows** become company cards (`KIND:org` in 4.0, `X-ABSHOWAS:COMPANY` in 3.0). +- **`PRODID`** is written in 3.0 and 4.0. +- **More date formats:** `DD.MM.YYYY`, unambiguous slashed dates, and dates without a year (`--MM-DD`). +- **`csv2vcard --version`** works without a subcommand. +- **`--strict`** now also fails on malformed rows and undecodable bytes. + +### Changed + +- **Column headers match loosely:** case-insensitive, and spaces, hyphens and underscores are treated alike (`First Name` = `first_name`). +- **`mobile`, `cell` and `cellphone` columns now map to `phone_cell`** (`TEL;TYPE=CELL`) instead of the default work phone. +- **A `location` column is no longer treated as geo coordinates.** +- **Multi-value fields keep every matching column**, not just the first. +- **Python 3.10+ is required**; Python 3.9 is end-of-life. Python 3.14 is supported. + +### Packaging and CI + +- **Removed the stale `setup.py`**; `pyproject.toml` is the only build configuration. setuptools >= 77 is required. +- **CI runs mypy** and tests Python 3.10 to 3.14. +- **Releases publish through PyPI Trusted Publishing** (no API token); `action-gh-release` is updated to v2. diff --git a/DESCRIPTION.md b/DESCRIPTION.md index 47178c7..cbb02d0 100644 --- a/DESCRIPTION.md +++ b/DESCRIPTION.md @@ -1,20 +1,23 @@ # csv2vcard -A Python library for converting CSV files to vCard format (3.0 and 4.0). +A Python library for converting CSV files to vCard format (2.1, 3.0 and 4.0). Create vCards from a spreadsheet of contacts - useful for business cards, QR codes, CRM imports, or transferring contacts between systems. ## Features -- **vCard 3.0 and 4.0 support** - Generate either format +- **vCard 2.1, 3.0 and 4.0** - Standards-compliant output (CRLF line endings, line folding, escaping); 4.0 includes RFC 9554 properties such as pronouns and social profiles, 2.1 targets legacy Outlook, car kits and feature phones +- **Stable UIDs** - Converting the same CSV again produces the same UIDs, so re-imports update contacts instead of duplicating them - **Custom CSV mapping** - Map any CSV column names to vCard fields - **Batch processing** - Convert entire directories of CSV files - **Single-file output** - Combine all contacts into one .vcf file - **File splitting** - Split output by size (`--max-vcard-file-size`) or contact count (`--max-vcards-per-file`) - **Multi-type fields** - Multiple phones (`phone_cell`, `phone_home`, `phone_work`, `phone_fax`), emails (`email_home`, `email_work`), and addresses (work + home) +- **Multiple values per field** - Extra emails, phones, websites and social profiles via numbered columns (`email_2`, `Phone 3`, ...) +- **Keep extra columns** - `--keep-unmapped` writes columns that match no field as `X-` properties instead of dropping them - **Media embedding** - Embed photos, logos, and keys (base64 or URL) - **Accent stripping** - Remove diacritics for compatibility (`--strip-accents`) -- **Auto-detect encoding** - Handles various file encodings +- **Auto-detect encoding** - Handles various file encodings, including Excel's UTF-8 with BOM - **Command-line interface** - Convert files directly from terminal - **Library API** - Use programmatically in your Python code - **Type hints** - Full typing support for IDE autocomplete @@ -48,6 +51,12 @@ csv2vcard convert contacts.csv # Specify output directory and vCard version csv2vcard convert contacts.csv -o ./vcards -V 4.0 +# vCard 2.1 for legacy Outlook, car kits and feature phones +csv2vcard convert contacts.csv -V 2.1 + +# Keep columns that match no vCard field as X- properties +csv2vcard convert contacts.csv --keep-unmapped + # Convert all CSVs in a directory csv2vcard convert ./csv_folder/ @@ -93,6 +102,7 @@ csv2vcard( mapping_file="mapping.json", # Custom column names strip_accents=True, # Remove diacritics max_vcards_per_file=100, # Split into multiple files + keep_unmapped=True, # Keep unknown columns as X- properties ) # Convert entire directory @@ -106,13 +116,15 @@ test_csv2vcard() Your CSV file should have column headers that match vCard fields. Use the default names or create a custom mapping. +Headers are matched case-insensitively and treat spaces, hyphens and underscores alike, so `First Name`, `first-name` and `first_name` are equivalent. Exports from Excel (including "CSV UTF-8" with a byte order mark) and Outlook-style headers such as `Business Street` or `Mobile Phone` work out of the box. + ### Default Column Names -**Required:** `last_name`, `first_name` +**Required:** `last_name`, `first_name` (rows with only `org` become organization cards) **Basic fields:** ``` -last_name, first_name, middle_name, name_prefix, name_suffix, nickname, gender, birthday, anniversary, org, title, role, note +last_name, first_name, middle_name, name_prefix, name_suffix, nickname, gender, birthday, anniversary, pronouns, language, org, title, role, note, uid ``` **Contact fields (single):** @@ -147,9 +159,15 @@ photo, logo, key **Additional fields:** ``` -categories, geo, tz +categories, geo, tz, social_profile ``` +**Multiple values:** `phone*`, `email*`, `website` and `social_profile` accept numbered columns for extra values, e.g. `email`, `email_2`, `email_3` or `Phone 1`, `Phone 2`. + +**Dates:** `birthday` and `anniversary` accept `YYYY-MM-DD`, `YYYYMMDD`, `DD.MM.YYYY`, `--MM-DD` (no year) and slashed dates when day and month can be told apart. Ambiguous dates such as `06/07/1990` are reported and kept as text in vCard 4.0. + +**UIDs:** each vCard gets a UID derived from the name, organization and email, or from the `uid` column (`uid`, `contact_id`, `external_id`) when present. + ### Example CSV ```csv @@ -187,16 +205,18 @@ Arguments: Options: -d, --delimiter TEXT CSV field delimiter (default: ",") -o, --output PATH Output directory (default: ./export/) - -V, --vcard-version TEXT vCard version: 3.0 or 4.0 (default: 3.0) + -V, --vcard-version TEXT vCard version: 2.1, 3.0 or 4.0 (default: 3.0) -1, --single-vcard Export all contacts to a single .vcf file -m, --mapping PATH Path to JSON mapping file -e, --encoding TEXT CSV file encoding (auto-detected if not set) -a, --strip-accents Remove accents/diacritics from contact fields --max-vcard-file-size INT Split output by file size (bytes) --max-vcards-per-file INT Split output by contact count - --strict Exit on validation errors + --keep-unmapped Keep unmapped columns as X- properties + --strict Fail on validation errors, malformed rows + and undecodable bytes -v, --verbose Enable verbose output - --version Show version and exit + --version Show version and exit (also: csv2vcard --version) --help Show help message ``` @@ -221,6 +241,7 @@ files = csv2vcard( strip_accents=False, # Remove diacritics max_file_size=None, # Split by file size (bytes) max_vcards_per_file=None, # Split by contact count + keep_unmapped=False, # Keep unknown columns as X- properties ) # Returns: List[Path] of created vCard files @@ -251,13 +272,14 @@ contact = Contact( contact = Contact.from_dict({"last_name": "Doe", "first_name": "John"}) # vCard versions +VCardVersion.V2_1 # vCard 2.1 (legacy) VCardVersion.V3_0 # vCard 3.0 (RFC 2426) -VCardVersion.V4_0 # vCard 4.0 (RFC 6350) +VCardVersion.V4_0 # vCard 4.0 (RFC 6350 + RFC 9554) ``` ## Requirements -- Python 3.9 or higher +- Python 3.10 or higher - For CLI: `typer` (installed with `csv2vcard[cli]`) - For encoding detection: `charset-normalizer` (installed with `csv2vcard[encoding]`) diff --git a/README.md b/README.md index 50a36be..837b3fd 100644 --- a/README.md +++ b/README.md @@ -13,21 +13,24 @@ -A Python library for converting CSV files to vCard format (3.0 and 4.0). +A Python library for converting CSV files to vCard format (2.1, 3.0 and 4.0). Create vCards from a spreadsheet of contacts - useful for business cards, QR codes, CRM imports, or transferring contacts between systems. ## Features -- **vCard 3.0 and 4.0 support** - Generate either format +- **vCard 2.1, 3.0 and 4.0** - Standards-compliant output (CRLF line endings, line folding, escaping); 4.0 includes RFC 9554 properties such as pronouns and social profiles, 2.1 targets legacy Outlook, car kits and feature phones +- **Stable UIDs** - Converting the same CSV again produces the same UIDs, so re-imports update contacts instead of duplicating them - **Custom CSV mapping** - Map any CSV column names to vCard fields - **Batch processing** - Convert entire directories of CSV files - **Single-file output** - Combine all contacts into one .vcf file - **File splitting** - Split output by size or contact count - **Multi-type fields** - Multiple phone numbers, emails, and addresses per contact +- **Multiple values per field** - Extra emails, phones, websites and social profiles via numbered columns (`email_2`, `Phone 3`, ...) +- **Keep extra columns** - `--keep-unmapped` writes columns that match no field as `X-` properties instead of dropping them - **Media embedding** - Embed photos, logos, and keys (base64 or URL) - **Accent stripping** - Remove diacritics for compatibility -- **Auto-detect encoding** - Handles various file encodings +- **Auto-detect encoding** - Handles various file encodings, including Excel's UTF-8 with BOM - **Command-line interface** - Convert files directly from terminal - **Library API** - Use programmatically in your Python code - **Type hints** - Full typing support for IDE autocomplete @@ -61,6 +64,12 @@ csv2vcard convert contacts.csv # Specify output directory and vCard version csv2vcard convert contacts.csv -o ./vcards -V 4.0 +# vCard 2.1 for legacy Outlook, car kits and feature phones +csv2vcard convert contacts.csv -V 2.1 + +# Keep columns that match no vCard field as X- properties +csv2vcard convert contacts.csv --keep-unmapped + # Convert all CSVs in a directory csv2vcard convert ./csv_folder/ @@ -106,6 +115,7 @@ csv2vcard( mapping_file="mapping.json", # Custom column names strip_accents=True, # Remove diacritics max_vcards_per_file=100, # Split into multiple files + keep_unmapped=True, # Keep unknown columns as X- properties ) # Convert entire directory @@ -119,13 +129,15 @@ test_csv2vcard() Your CSV file should have column headers that match vCard fields. Use the default names or create a custom mapping. +Headers are matched case-insensitively and treat spaces, hyphens and underscores alike, so `First Name`, `first-name` and `first_name` are equivalent. Exports from Excel (including "CSV UTF-8" with a byte order mark) and Outlook-style headers such as `Business Street` or `Mobile Phone` work out of the box. + ### Default Column Names -**Required:** `last_name`, `first_name` +**Required:** `last_name`, `first_name` (rows with only `org` become organization cards) **Basic fields:** ``` -last_name, first_name, middle_name, name_prefix, name_suffix, nickname, gender, birthday, anniversary, org, title, role, note +last_name, first_name, middle_name, name_prefix, name_suffix, nickname, gender, birthday, anniversary, pronouns, language, org, title, role, note, uid ``` **Contact fields (single):** @@ -160,9 +172,15 @@ photo, logo, key **Additional fields:** ``` -categories, geo, tz +categories, geo, tz, social_profile ``` +**Multiple values:** `phone*`, `email*`, `website` and `social_profile` accept numbered columns for extra values, e.g. `email`, `email_2`, `email_3` or `Phone 1`, `Phone 2`. + +**Dates:** `birthday` and `anniversary` accept `YYYY-MM-DD`, `YYYYMMDD`, `DD.MM.YYYY`, `--MM-DD` (no year) and slashed dates when day and month can be told apart. Ambiguous dates such as `06/07/1990` are reported and kept as text in vCard 4.0. + +**UIDs:** each vCard gets a UID derived from the name, organization and email, or from the `uid` column (`uid`, `contact_id`, `external_id`) when present. + ### Example CSV ```csv @@ -200,16 +218,18 @@ Arguments: Options: -d, --delimiter TEXT CSV field delimiter (default: ",") -o, --output PATH Output directory (default: ./export/) - -V, --vcard-version TEXT vCard version: 3.0 or 4.0 (default: 3.0) + -V, --vcard-version TEXT vCard version: 2.1, 3.0 or 4.0 (default: 3.0) -1, --single-vcard Export all contacts to a single .vcf file -m, --mapping PATH Path to JSON mapping file -e, --encoding TEXT CSV file encoding (auto-detected if not set) -a, --strip-accents Remove accents/diacritics from contact fields --max-vcard-file-size INT Split output by file size (bytes) --max-vcards-per-file INT Split output by contact count - --strict Exit on validation errors + --keep-unmapped Keep unmapped columns as X- properties + --strict Fail on validation errors, malformed rows + and undecodable bytes -v, --verbose Enable verbose output - --version Show version and exit + --version Show version and exit (also: csv2vcard --version) --help Show help message ``` @@ -234,6 +254,7 @@ files = csv2vcard( strip_accents=False, # Remove diacritics (é→e, ü→u) max_file_size=None, # Split by file size (bytes) max_vcards_per_file=None, # Split by contact count + keep_unmapped=False, # Keep unknown columns as X- properties ) # Returns: List[Path] of created vCard files @@ -264,8 +285,9 @@ contact = Contact( contact = Contact.from_dict({"last_name": "Doe", "first_name": "John"}) # vCard versions +VCardVersion.V2_1 # vCard 2.1 (legacy) VCardVersion.V3_0 # vCard 3.0 (RFC 2426) -VCardVersion.V4_0 # vCard 4.0 (RFC 6350) +VCardVersion.V4_0 # vCard 4.0 (RFC 6350 + RFC 9554) ``` ## Supported vCard Fields @@ -283,17 +305,20 @@ VCardVersion.V4_0 # vCard 4.0 (RFC 6350) | `gender` | Gender (M/F/O/N/U) | M | | `birthday` | Birth date (YYYY-MM-DD) | 1990-01-15 | | `anniversary` | Anniversary date | 2015-06-20 | +| `pronouns` | Pronouns (vCard 4.0) | they/them | +| `language` | Preferred language (vCard 4.0) | en | | `org` | Organization | Acme Corp | | `title` | Job title | Developer | | `role` | Role/function | Team Lead | | `note` | Notes | Any additional info | +| `uid` | Stable ID from the source system | crm-42 | ### Contact Fields | Field | Description | vCard Type | |-------|-------------|------------| | `phone` | Default phone | TEL;TYPE=WORK | -| `phone_cell` | Mobile phone | TEL;TYPE=CELL | +| `phone_cell` | Mobile phone (also `mobile`, `cell`) | TEL;TYPE=CELL | | `phone_home` | Home phone | TEL;TYPE=HOME | | `phone_work` | Work phone | TEL;TYPE=WORK | | `phone_fax` | Fax number | TEL;TYPE=FAX | @@ -301,6 +326,7 @@ VCardVersion.V4_0 # vCard 4.0 (RFC 6350) | `email_home` | Personal email | EMAIL;TYPE=HOME | | `email_work` | Work email | EMAIL;TYPE=WORK | | `website` | Website URL | URL | +| `social_profile` | Social profile URL | SOCIALPROFILE (4.0), X-SOCIALPROFILE (2.1/3.0) | ### Address Fields @@ -321,9 +347,9 @@ VCardVersion.V4_0 # vCard 4.0 (RFC 6350) | Field | Description | Format | |-------|-------------|--------| -| `photo` | Contact photo | URL or base64 | -| `logo` | Company logo | URL or base64 | -| `key` | Public key | URL or base64 | +| `photo` | Contact photo | URL, data: URI or base64 | +| `logo` | Company logo | URL, data: URI or base64 | +| `key` | Public key | URL, data: URI, base64 or ASCII-armored PGP | ### Additional Fields @@ -335,7 +361,7 @@ VCardVersion.V4_0 # vCard 4.0 (RFC 6350) ## Requirements -- Python 3.9 or higher +- Python 3.10 or higher - For CLI: `typer` (installed with `csv2vcard[cli]`) - For encoding detection: `charset-normalizer` (installed with `csv2vcard[encoding]`) @@ -359,7 +385,7 @@ pytest --cov=csv2vcard --cov-report=term-missing mypy csv2vcard # Linting -ruff check csv2vcard +ruff check . ``` ## License diff --git a/csv2vcard/__init__.py b/csv2vcard/__init__.py index b06898a..5cb4162 100644 --- a/csv2vcard/__init__.py +++ b/csv2vcard/__init__.py @@ -1,13 +1,12 @@ -"""csv2vcard - Convert CSV files to vCard format (3.0 and 4.0).""" - -__version__ = "0.5.1" +"""csv2vcard - Convert CSV files to vCard format (2.1, 3.0 and 4.0).""" # For backwards compatibility, users can still do: # from csv2vcard import csv2vcard # csv2vcard.csv2vcard("file.csv", ",") # This works because Python allows importing submodules directly. - +# # For cleaner new API, also expose the main functions: +from csv2vcard._version import __version__ from csv2vcard.csv2vcard import csv2vcard, test_csv2vcard __all__ = [ diff --git a/csv2vcard/_version.py b/csv2vcard/_version.py new file mode 100644 index 0000000..19a1c27 --- /dev/null +++ b/csv2vcard/_version.py @@ -0,0 +1,3 @@ +"""Package version (kept separate so internal modules can import it without cycles).""" + +__version__ = "0.6.0" diff --git a/csv2vcard/cli.py b/csv2vcard/cli.py index 64b1ff2..ec7105b 100644 --- a/csv2vcard/cli.py +++ b/csv2vcard/cli.py @@ -3,7 +3,6 @@ import logging import sys from pathlib import Path -from typing import Optional # Check if typer is available try: @@ -33,8 +32,9 @@ def _check_typer() -> None: app = typer.Typer( name="csv2vcard", - help="Convert CSV files to vCard format (3.0 and 4.0).", + help="Convert CSV files to vCard format (2.1, 3.0 and 4.0).", add_completion=False, + no_args_is_help=True, ) def version_callback(value: bool) -> None: @@ -43,6 +43,27 @@ def version_callback(value: bool) -> None: print(f"csv2vcard version {__version__}") raise typer.Exit() + @app.callback() + def main( + version: Annotated[ + bool | None, + typer.Option( + "--version", + callback=version_callback, + is_eager=True, + help="Show version and exit", + ), + ] = None, + ) -> None: + """Convert CSV files to vCard format (2.1, 3.0 and 4.0).""" + + def _parse_version(vcard_version: str) -> VCardVersion: + try: + return VCardVersion(vcard_version) + except ValueError: + typer.echo(f"Error: Invalid vCard version '{vcard_version}'. Use 2.1, 3.0 or 4.0.") + raise typer.Exit(code=1) from None + @app.command() def convert( source: Annotated[ @@ -61,7 +82,7 @@ def convert( ), ] = ",", output_dir: Annotated[ - Optional[Path], + Path | None, typer.Option( "--output", "-o", @@ -73,7 +94,7 @@ def convert( typer.Option( "--vcard-version", "-V", - help="vCard version to generate: 3.0 or 4.0", + help="vCard version to generate: 2.1, 3.0 or 4.0", ), ] = "3.0", single_file: Annotated[ @@ -85,7 +106,7 @@ def convert( ), ] = False, mapping_file: Annotated[ - Optional[Path], + Path | None, typer.Option( "--mapping", "-m", @@ -93,7 +114,7 @@ def convert( ), ] = None, encoding: Annotated[ - Optional[str], + str | None, typer.Option( "--encoding", "-e", @@ -109,19 +130,26 @@ def convert( ), ] = False, max_file_size: Annotated[ - Optional[int], + int | None, typer.Option( "--max-vcard-file-size", help="Maximum file size in bytes for split output files", ), ] = None, max_vcards_per_file: Annotated[ - Optional[int], + int | None, typer.Option( "--max-vcards-per-file", help="Maximum number of vCards per output file", ), ] = None, + keep_unmapped: Annotated[ + bool, + typer.Option( + "--keep-unmapped", + help="Keep CSV columns that match no field as X- properties", + ), + ] = False, strict: Annotated[ bool, typer.Option( @@ -138,7 +166,7 @@ def convert( ), ] = False, version: Annotated[ - Optional[bool], + bool | None, typer.Option( "--version", callback=version_callback, @@ -163,6 +191,8 @@ def convert( csv2vcard convert data.csv --strip-accents csv2vcard convert data.csv --max-vcards-per-file 100 + + csv2vcard convert data.csv -V 2.1 --keep-unmapped """ # Configure logging log_level = logging.DEBUG if verbose else logging.INFO @@ -171,12 +201,7 @@ def convert( format="%(levelname)s: %(message)s", ) - # Parse vCard version - try: - vc_version = VCardVersion(vcard_version) - except ValueError: - typer.echo(f"Error: Invalid vCard version '{vcard_version}'. Use 3.0 or 4.0.") - raise typer.Exit(code=1) from None + vc_version = _parse_version(vcard_version) try: files = csv2vcard_func( @@ -191,6 +216,7 @@ def convert( strip_accents=strip_accents_opt, max_file_size=max_file_size, max_vcards_per_file=max_vcards_per_file, + keep_unmapped=keep_unmapped, ) if files: typer.echo(f"Successfully created {len(files)} vCard file(s).") @@ -205,7 +231,7 @@ def convert( @app.command() def test( output_dir: Annotated[ - Optional[Path], + Path | None, typer.Option( "--output", "-o", @@ -217,7 +243,7 @@ def test( typer.Option( "--vcard-version", "-V", - help="vCard version to generate: 3.0 or 4.0", + help="vCard version to generate: 2.1, 3.0 or 4.0", ), ] = "3.0", ) -> None: @@ -226,11 +252,7 @@ def test( This is useful for verifying the installation works correctly. """ - try: - vc_version = VCardVersion(vcard_version) - except ValueError: - typer.echo(f"Error: Invalid vCard version '{vcard_version}'. Use 3.0 or 4.0.") - raise typer.Exit(code=1) from None + vc_version = _parse_version(vcard_version) test_csv2vcard_func(output_dir=output_dir, version=vc_version) typer.echo("Test vCard created successfully.") @@ -248,7 +270,7 @@ def show_mapping() -> None: else: # Fallback app when Typer is not installed - def app() -> None: + def app() -> None: # type: ignore[misc] """Fallback CLI without Typer.""" _check_typer() diff --git a/csv2vcard/create_vcard.py b/csv2vcard/create_vcard.py index ba30a70..8c83271 100644 --- a/csv2vcard/create_vcard.py +++ b/csv2vcard/create_vcard.py @@ -3,11 +3,53 @@ from __future__ import annotations import logging +import re +from collections.abc import Iterator +from typing import NamedTuple +from csv2vcard._version import __version__ from csv2vcard.models import Contact, VCardOutput, VCardVersion +from csv2vcard.utils import normalize_date, parse_geo, parse_utc_offset, phone_to_tel_uri logger = logging.getLogger(__name__) +PRODID = f"-//tech4242//csv2vcard {__version__}//EN" + +# RFC 2425/2426/6350: content lines end with CRLF and are folded at 75 octets +CRLF = "\r\n" +MAX_LINE_OCTETS = 75 + +# (field, vCard 3.0 TYPE) - 4.0 uses the lower-case form, 2.1 separates with ";" +PHONE_TYPES = ( + ("phone", "WORK,VOICE"), + ("phone_cell", "CELL"), + ("phone_home", "HOME,VOICE"), + ("phone_work", "WORK,VOICE"), + ("phone_fax", "FAX"), +) +EMAIL_TYPES = ( + ("email", "WORK"), + ("email_home", "HOME"), + ("email_work", "WORK"), +) + +GENDER_WORDS = {"MALE": "M", "FEMALE": "F", "OTHER": "O", "NONE": "N", "UNKNOWN": "U"} + +_CONTROL_CHARS = re.compile(r"[\x00-\x08\x0b-\x1f\x7f]") # all except TAB and LF +_ANY_CONTROL_CHARS = re.compile(r"[\x00-\x1f\x7f]+") +_URI = re.compile(r"^[a-zA-Z][a-zA-Z0-9+.\-]+:\S") +_BARE_URL = re.compile(r"^(www\.)?[\w-]+(\.[\w-]+)*\.[a-z]{2,}(/\S*)?$", re.IGNORECASE) +_DATA_URI = re.compile( + r"^data:(?P[^;,]*)(?P(?:;[^;,]*)*);base64,(?P.*)$", + re.IGNORECASE | re.DOTALL, +) +_BASE64_SIGNATURES = ( + ("/9j/", "image/jpeg"), + ("iVBOR", "image/png"), + ("R0lGOD", "image/gif"), + ("UklGR", "image/webp"), +) + def create_vcard( contact: dict[str, str] | Contact, @@ -18,7 +60,7 @@ def create_vcard( Args: contact: Contact data (dict or Contact object) - version: vCard version to generate (3.0 or 4.0) + version: vCard version to generate (2.1, 3.0 or 4.0) Returns: Dictionary with 'filename', 'output', and 'name' keys @@ -29,6 +71,8 @@ def create_vcard( if version == VCardVersion.V4_0: output = _create_vcard_4(contact_obj) + elif version == VCardVersion.V2_1: + output = _create_vcard_21(contact_obj) else: output = _create_vcard_3(contact_obj) @@ -55,7 +99,7 @@ def create_vcard_typed( Args: contact: Contact data (dict or Contact object) - version: vCard version to generate (3.0 or 4.0) + version: vCard version to generate (2.1, 3.0 or 4.0) Returns: VCardOutput dataclass @@ -69,16 +113,28 @@ def create_vcard_typed( ) +# --------------------------------------------------------------------------- +# Value encoding helpers +# --------------------------------------------------------------------------- + + +def _normalize_text(value: str) -> str: + """Normalize line breaks to LF and drop other control characters.""" + value = value.replace("\r\n", "\n").replace("\r", "\n") + return _CONTROL_CHARS.sub("", value) + + def _escape_vcard_value(value: str) -> str: """ - Escape special characters in vCard values. + Escape special characters in vCard 3.0/4.0 text values. Args: value: Raw string value Returns: - Escaped string safe for vCard + Escaped string safe for vCard (never contains a raw line break) """ + value = _normalize_text(value) # Escape backslashes first, then other special chars value = value.replace("\\", "\\\\") value = value.replace(",", "\\,") @@ -87,6 +143,169 @@ def _escape_vcard_value(value: str) -> str: return value +def _escape_list(value: str) -> str: + """Escape a comma-separated list value (CATEGORIES, NICKNAME), keeping the separators.""" + return ",".join(_escape_vcard_value(item.strip()) for item in value.split(",") if item.strip()) + + +def _structured(*components: str) -> str: + """Build a structured value (N, ADR) from escaped components.""" + return ";".join(_escape_vcard_value(c) for c in components) + + +def _clean(value: str) -> str: + """Make a URI-like or single-token value safe: no control characters or line breaks.""" + return _ANY_CONTROL_CHARS.sub(" ", value).strip() + + +def _is_uri(value: str) -> bool: + return bool(_URI.match(value)) + + +def _fold_line(line: str) -> str: + """Fold a content line at 75 octets without splitting UTF-8 sequences (RFC 6350 3.2).""" + if len(line.encode("utf-8")) <= MAX_LINE_OCTETS: + return line + parts: list[str] = [] + current = "" + size = 0 + for char in line: + char_size = len(char.encode("utf-8")) + if size + char_size > MAX_LINE_OCTETS: + parts.append(current) + current = char + size = 1 + char_size # continuation lines start with a space + else: + current += char + size += char_size + parts.append(current) + return (CRLF + " ").join(parts) + + +def _serialize(lines: list[str], *, fold: bool = True) -> str: + """Join content lines with CRLF, folding long lines (pre-formatted lines are kept).""" + if fold: + lines = [line if CRLF in line else _fold_line(line) for line in lines] + return CRLF.join(lines) + CRLF + + +def _typed_values( + contact: Contact, specs: tuple[tuple[str, str], ...] +) -> Iterator[tuple[str, str]]: + """Yield (value, types) for multi-type fields, skipping duplicate values.""" + seen: set[str] = set() + for field_name, types in specs: + for value in contact.values(field_name): + value = _clean(value) + if value and value.casefold() not in seen: + seen.add(value.casefold()) + yield value, types + + +def _addresses(contact: Contact) -> Iterator[tuple[str, tuple[str, ...]]]: + """Yield (type, ADR components) for work and home addresses that have data.""" + work = (contact.street, contact.city, contact.region, contact.p_code, contact.country) + home = ( + contact.home_street, + contact.home_city, + contact.home_region, + contact.home_p_code, + contact.home_country, + ) + for adr_type, (street, city, region, p_code, country) in (("WORK", work), ("HOME", home)): + if any((street, city, region, p_code, country)): + # ADR format: PO Box;Extended;Street;City;Region;PostalCode;Country + yield adr_type, ("", "", street, city, region, p_code, country) + + +def _n_components(contact: Contact) -> tuple[str, ...]: + # N field: LastName;FirstName;MiddleName;Prefix;Suffix + return ( + contact.last_name, + contact.first_name, + contact.middle_name, + contact.name_prefix, + contact.name_suffix, + ) + + +def _date(contact: Contact, field_name: str) -> str | None: + """Normalize a date field, logging a warning if it cannot be interpreted.""" + raw = getattr(contact, field_name) + if not raw: + return None + normalized = normalize_date(raw) + if normalized is None: + logger.warning( + f"{contact.get_formatted_name()}: unrecognized {field_name} '{raw}' " + "(use YYYY-MM-DD)" + ) + return normalized + + +def _social_profiles(contact: Contact) -> Iterator[str]: + for value in contact.values("social_profile"): + value = _clean(value) + if _BARE_URL.match(value): + value = f"https://{value}" + if _is_uri(value): + yield value + else: + logger.warning( + f"{contact.get_formatted_name()}: social_profile '{value}' is not a URL, skipped" + ) + + +def _geo(contact: Contact) -> tuple[str, str] | None: + if not contact.geo: + return None + coords = parse_geo(contact.geo) + if coords is None: + logger.warning(f"{contact.get_formatted_name()}: invalid geo '{contact.geo}', skipped") + return coords + + +class _Media(NamedTuple): + """A PHOTO, LOGO or KEY value: either a URI or inline base64 data.""" + + uri: str | None + media_type: str + data: str + + +def _parse_media(value: str, default_type: str) -> _Media: + value = value.strip() + match = _DATA_URI.match(value) + if match: + media_type = match["type"] or default_type + return _Media(None, media_type.lower(), re.sub(r"\s+", "", match["data"])) + if _is_uri(value): + return _Media(_clean(value), "", "") + data = re.sub(r"\s+", "", value) + for signature, media_type in _BASE64_SIGNATURES: + if data.startswith(signature): + return _Media(None, media_type, data) + return _Media(None, default_type, data) + + +def _v3_type(media_type: str) -> str: + """Map a media type to the vCard 2.1/3.0 TYPE parameter (image/png -> PNG).""" + if "pgp" in media_type: + return "PGP" + if "pkix" in media_type or "x509" in media_type: + return "X509" + return media_type.rsplit("/", 1)[-1].upper() + + +def _is_armored_key(value: str) -> bool: + return value.lstrip().startswith("-----BEGIN") + + +# --------------------------------------------------------------------------- +# vCard 3.0 +# --------------------------------------------------------------------------- + + def _create_vcard_3(contact: Contact) -> str: """ Generate vCard 3.0 format (RFC 2426). @@ -100,168 +319,118 @@ def _create_vcard_3(contact: Contact) -> str: lines = [ "BEGIN:VCARD", "VERSION:3.0", + f"PRODID:{PRODID}", ] - # N field: LastName;FirstName;MiddleName;Prefix;Suffix - n_parts = [ - _escape_vcard_value(contact.last_name), - _escape_vcard_value(contact.first_name), - _escape_vcard_value(contact.middle_name), - _escape_vcard_value(contact.name_prefix), - _escape_vcard_value(contact.name_suffix), - ] - lines.append(f"N;CHARSET=UTF-8:{';'.join(n_parts)}") + lines.append(f"N:{_structured(*_n_components(contact))}") + lines.append(f"FN:{_escape_vcard_value(contact.get_formatted_name())}") - # FN field: Formatted name - fn = contact.get_formatted_name() - lines.append(f"FN;CHARSET=UTF-8:{_escape_vcard_value(fn)}") + if contact.is_organization: + # Apple Contacts extension: display the card as a company + lines.append("X-ABSHOWAS:COMPANY") # Optional fields - only include if non-empty if contact.nickname: - lines.append(f"NICKNAME;CHARSET=UTF-8:{_escape_vcard_value(contact.nickname)}") + lines.append(f"NICKNAME:{_escape_list(contact.nickname)}") if contact.gender: # vCard 3.0 doesn't have GENDER, use X-GENDER extension lines.append(f"X-GENDER:{_escape_vcard_value(contact.gender)}") - if contact.birthday: - # Format: YYYYMMDD or YYYY-MM-DD - bday = contact.birthday.replace("-", "") - lines.append(f"BDAY:{bday}") - - if contact.anniversary: - # vCard 3.0 uses X-ANNIVERSARY extension - anniv = contact.anniversary.replace("-", "") - lines.append(f"X-ANNIVERSARY:{anniv}") + for field_name, prop in (("birthday", "BDAY"), ("anniversary", "X-ANNIVERSARY")): + date = _date(contact, field_name) + if date and date.startswith("--"): + # vCard 3.0 has no year-less dates; Apple's convention is year 1604 + lines.append(f"{prop};X-APPLE-OMIT-YEAR=1604:1604-{date[2:4]}-{date[4:6]}") + elif date: + lines.append(f"{prop}:{date[:4]}-{date[4:6]}-{date[6:8]}") if contact.title: - lines.append(f"TITLE;CHARSET=UTF-8:{_escape_vcard_value(contact.title)}") + lines.append(f"TITLE:{_escape_vcard_value(contact.title)}") if contact.role: - lines.append(f"ROLE;CHARSET=UTF-8:{_escape_vcard_value(contact.role)}") + lines.append(f"ROLE:{_escape_vcard_value(contact.role)}") if contact.org: - lines.append(f"ORG;CHARSET=UTF-8:{_escape_vcard_value(contact.org)}") - - # Phone numbers - backwards compatible single phone - if contact.phone: - lines.append(f"TEL;TYPE=WORK,VOICE:{contact.phone}") - - # Multi-type phone numbers (v0.5.0) - if contact.phone_cell: - lines.append(f"TEL;TYPE=CELL:{contact.phone_cell}") - if contact.phone_home: - lines.append(f"TEL;TYPE=HOME,VOICE:{contact.phone_home}") - if contact.phone_work: - lines.append(f"TEL;TYPE=WORK,VOICE:{contact.phone_work}") - if contact.phone_fax: - lines.append(f"TEL;TYPE=FAX:{contact.phone_fax}") - - # Email - backwards compatible single email - if contact.email: - lines.append(f"EMAIL;TYPE=WORK:{contact.email}") - - # Multi-type email (v0.5.0) - if contact.email_home: - lines.append(f"EMAIL;TYPE=HOME:{contact.email_home}") - if contact.email_work: - lines.append(f"EMAIL;TYPE=WORK:{contact.email_work}") - - if contact.website: - lines.append(f"URL;TYPE=WORK:{contact.website}") - - # Address (default/work) - only if at least one component is present - if any([contact.street, contact.city, contact.region, contact.p_code, contact.country]): - # ADR format: PO Box;Extended;Street;City;Region;PostalCode;Country - adr_parts = [ - "", # PO Box - "", # Extended address - _escape_vcard_value(contact.street), - _escape_vcard_value(contact.city), - _escape_vcard_value(contact.region), - _escape_vcard_value(contact.p_code), - _escape_vcard_value(contact.country), - ] - lines.append(f"ADR;TYPE=WORK;CHARSET=UTF-8:{';'.join(adr_parts)}") - - # Home address (v0.5.0) - if any([contact.home_street, contact.home_city, contact.home_region, - contact.home_p_code, contact.home_country]): - adr_parts = [ - "", # PO Box - "", # Extended address - _escape_vcard_value(contact.home_street), - _escape_vcard_value(contact.home_city), - _escape_vcard_value(contact.home_region), - _escape_vcard_value(contact.home_p_code), - _escape_vcard_value(contact.home_country), - ] - lines.append(f"ADR;TYPE=HOME;CHARSET=UTF-8:{';'.join(adr_parts)}") - - # Media fields (v0.5.0) + lines.append(f"ORG:{_escape_vcard_value(contact.org)}") + + for value, types in _typed_values(contact, PHONE_TYPES): + lines.append(f"TEL;TYPE={types}:{_escape_vcard_value(value)}") + + for value, types in _typed_values(contact, EMAIL_TYPES): + lines.append(f"EMAIL;TYPE={types}:{_escape_vcard_value(value)}") + + for value in contact.values("website"): + lines.append(f"URL;TYPE=WORK:{_clean(value)}") + + for adr_type, components in _addresses(contact): + lines.append(f"ADR;TYPE={adr_type}:{_structured(*components)}") + if contact.photo: lines.append(_format_media_field_v3("PHOTO", contact.photo)) if contact.logo: lines.append(_format_media_field_v3("LOGO", contact.logo)) - # New vCard fields (v0.5.0) if contact.categories: - lines.append(f"CATEGORIES;CHARSET=UTF-8:{_escape_vcard_value(contact.categories)}") + lines.append(f"CATEGORIES:{_escape_list(contact.categories)}") - if contact.geo: + geo = _geo(contact) + if geo: # vCard 3.0 GEO format: lat;lon - geo = contact.geo.replace(",", ";") - lines.append(f"GEO:{geo}") + lines.append(f"GEO:{geo[0]};{geo[1]}") if contact.tz: - lines.append(f"TZ:{_escape_vcard_value(contact.tz)}") + offset = parse_utc_offset(contact.tz) + if offset: + lines.append(f"TZ:{offset[0]}{offset[1]}:{offset[2]}") + else: + lines.append(f"TZ;VALUE=text:{_escape_vcard_value(contact.tz)}") if contact.key: lines.append(_format_key_field_v3(contact.key)) + for profile in _social_profiles(contact): + # Apple Contacts extension (SOCIALPROFILE is only standard in vCard 4.0) + lines.append(f"X-SOCIALPROFILE:{profile}") + if contact.note: - lines.append(f"NOTE;CHARSET=UTF-8:{_escape_vcard_value(contact.note)}") + lines.append(f"NOTE:{_escape_vcard_value(contact.note)}") - # Add REV timestamp - lines.append(f"REV:{Contact.generate_rev()}") + for name, value in contact.extensions.items(): + lines.append(f"{name}:{_escape_vcard_value(value)}") - # Add UID + lines.append(f"REV:{Contact.generate_rev()}") lines.append(f"UID:{contact.generate_uid()}") - lines.append("END:VCARD") - return "\n".join(lines) + "\n" + return _serialize(lines) def _format_media_field_v3(field_name: str, value: str) -> str: """Format PHOTO or LOGO field for vCard 3.0.""" - if value.startswith(("http://", "https://")): - return f"{field_name};VALUE=URI:{value}" - else: - # Assume base64 encoded data - # Try to detect image type from data or default to JPEG - if value.startswith("/9j/"): - media_type = "JPEG" - elif value.startswith("iVBOR"): - media_type = "PNG" - elif value.startswith("R0lGOD"): - media_type = "GIF" - else: - media_type = "JPEG" - return f"{field_name};ENCODING=b;TYPE={media_type}:{value}" + media = _parse_media(value, "image/jpeg") + if media.uri: + return f"{field_name};VALUE=URI:{media.uri}" + return f"{field_name};ENCODING=b;TYPE={_v3_type(media.media_type)}:{media.data}" def _format_key_field_v3(value: str) -> str: """Format KEY field for vCard 3.0.""" - if value.startswith(("http://", "https://")): - return f"KEY;VALUE=URI:{value}" - else: - # Assume base64 encoded key data - return f"KEY;ENCODING=b:{value}" + if _is_armored_key(value): + return f"KEY;TYPE=PGP;VALUE=text:{_escape_vcard_value(value.strip())}" + media = _parse_media(value, "application/pgp-keys") + if media.uri: + return f"KEY;VALUE=URI:{media.uri}" + return f"KEY;ENCODING=b;TYPE={_v3_type(media.media_type)}:{media.data}" + + +# --------------------------------------------------------------------------- +# vCard 4.0 +# --------------------------------------------------------------------------- def _create_vcard_4(contact: Contact) -> str: """ - Generate vCard 4.0 format (RFC 6350). + Generate vCard 4.0 format (RFC 6350, plus RFC 9554 properties). Args: contact: Contact object @@ -272,44 +441,43 @@ def _create_vcard_4(contact: Contact) -> str: lines = [ "BEGIN:VCARD", "VERSION:4.0", + f"PRODID:{PRODID}", ] - # N field: LastName;FirstName;MiddleName;Prefix;Suffix - n_parts = [ - _escape_vcard_value(contact.last_name), - _escape_vcard_value(contact.first_name), - _escape_vcard_value(contact.middle_name), - _escape_vcard_value(contact.name_prefix), - _escape_vcard_value(contact.name_suffix), - ] - lines.append(f"N:{';'.join(n_parts)}") + if contact.is_organization: + lines.append("KIND:org") - # FN field: Formatted name - fn = contact.get_formatted_name() - lines.append(f"FN:{_escape_vcard_value(fn)}") - - # Optional fields - only include if non-empty # Note: CHARSET is not used in vCard 4.0 (UTF-8 is mandatory) + lines.append(f"N:{_structured(*_n_components(contact))}") + lines.append(f"FN:{_escape_vcard_value(contact.get_formatted_name())}") + if contact.nickname: - lines.append(f"NICKNAME:{_escape_vcard_value(contact.nickname)}") + lines.append(f"NICKNAME:{_escape_list(contact.nickname)}") if contact.gender: # vCard 4.0 GENDER format: single letter (M/F/O/N/U) or ;text gender = contact.gender.upper() + gender = GENDER_WORDS.get(gender, gender) if gender in ("M", "F", "O", "N", "U"): lines.append(f"GENDER:{gender}") else: - # Use full text after semicolon lines.append(f"GENDER:;{_escape_vcard_value(contact.gender)}") - if contact.birthday: - # vCard 4.0 format: YYYYMMDD or --MMDD - bday = contact.birthday.replace("-", "") - lines.append(f"BDAY:{bday}") + for field_name, prop in (("birthday", "BDAY"), ("anniversary", "ANNIVERSARY")): + raw = getattr(contact, field_name) + date = _date(contact, field_name) + if date: + # vCard 4.0 format: YYYYMMDD or --MMDD + lines.append(f"{prop}:{date}") + elif raw: + # vCard 4.0 allows free-form text dates, so nothing is lost + lines.append(f"{prop};VALUE=text:{_escape_vcard_value(raw)}") + + if contact.pronouns: + lines.append(f"PRONOUNS:{_escape_vcard_value(contact.pronouns)}") - if contact.anniversary: - anniv = contact.anniversary.replace("-", "") - lines.append(f"ANNIVERSARY:{anniv}") + if contact.language: + lines.append(f"LANG:{_clean(contact.language)}") if contact.title: lines.append(f"TITLE:{_escape_vcard_value(contact.title)}") @@ -320,115 +488,230 @@ def _create_vcard_4(contact: Contact) -> str: if contact.org: lines.append(f"ORG:{_escape_vcard_value(contact.org)}") - # Phone numbers - backwards compatible single phone - if contact.phone: - lines.append(f"TEL;TYPE=work,voice;VALUE=uri:tel:{contact.phone}") - - # Multi-type phone numbers (v0.5.0) - if contact.phone_cell: - lines.append(f"TEL;TYPE=cell;VALUE=uri:tel:{contact.phone_cell}") - if contact.phone_home: - lines.append(f"TEL;TYPE=home,voice;VALUE=uri:tel:{contact.phone_home}") - if contact.phone_work: - lines.append(f"TEL;TYPE=work,voice;VALUE=uri:tel:{contact.phone_work}") - if contact.phone_fax: - lines.append(f"TEL;TYPE=fax;VALUE=uri:tel:{contact.phone_fax}") - - # Email - backwards compatible single email - if contact.email: - lines.append(f"EMAIL;TYPE=work:{contact.email}") - - # Multi-type email (v0.5.0) - if contact.email_home: - lines.append(f"EMAIL;TYPE=home:{contact.email_home}") - if contact.email_work: - lines.append(f"EMAIL;TYPE=work:{contact.email_work}") - - if contact.website: - lines.append(f"URL;TYPE=work:{contact.website}") - - # Address (default/work) - if any([contact.street, contact.city, contact.region, contact.p_code, contact.country]): - adr_parts = [ - "", # PO Box - "", # Extended address - _escape_vcard_value(contact.street), - _escape_vcard_value(contact.city), - _escape_vcard_value(contact.region), - _escape_vcard_value(contact.p_code), - _escape_vcard_value(contact.country), - ] - lines.append(f"ADR;TYPE=work:{';'.join(adr_parts)}") - - # Home address (v0.5.0) - if any([contact.home_street, contact.home_city, contact.home_region, - contact.home_p_code, contact.home_country]): - adr_parts = [ - "", # PO Box - "", # Extended address - _escape_vcard_value(contact.home_street), - _escape_vcard_value(contact.home_city), - _escape_vcard_value(contact.home_region), - _escape_vcard_value(contact.home_p_code), - _escape_vcard_value(contact.home_country), - ] - lines.append(f"ADR;TYPE=home:{';'.join(adr_parts)}") - - # Media fields (v0.5.0) + for value, types in _typed_values(contact, PHONE_TYPES): + tel_uri = phone_to_tel_uri(value) + if tel_uri: + lines.append(f"TEL;TYPE={types.lower()};VALUE=uri:{tel_uri}") + else: + # Local numbers can't be tel: URIs without a phone-context; use text + lines.append(f"TEL;TYPE={types.lower()}:{_escape_vcard_value(value)}") + + for value, types in _typed_values(contact, EMAIL_TYPES): + lines.append(f"EMAIL;TYPE={types.lower()}:{_escape_vcard_value(value)}") + + for value in contact.values("website"): + lines.append(f"URL;TYPE=work:{_clean(value)}") + + for adr_type, components in _addresses(contact): + lines.append(f"ADR;TYPE={adr_type.lower()}:{_structured(*components)}") + if contact.photo: lines.append(_format_media_field_v4("PHOTO", contact.photo)) if contact.logo: lines.append(_format_media_field_v4("LOGO", contact.logo)) - # New vCard fields (v0.5.0) if contact.categories: - lines.append(f"CATEGORIES:{_escape_vcard_value(contact.categories)}") + lines.append(f"CATEGORIES:{_escape_list(contact.categories)}") - if contact.geo: + geo = _geo(contact) + if geo: # vCard 4.0 GEO format: geo:lat,lon - lines.append(f"GEO:geo:{contact.geo}") + lines.append(f"GEO:geo:{geo[0]},{geo[1]}") if contact.tz: - lines.append(f"TZ:{_escape_vcard_value(contact.tz)}") + offset = parse_utc_offset(contact.tz) + if offset: + lines.append(f"TZ;VALUE=utc-offset:{offset[0]}{offset[1]}{offset[2]}") + else: + lines.append(f"TZ:{_escape_vcard_value(contact.tz)}") if contact.key: lines.append(_format_key_field_v4(contact.key)) + for profile in _social_profiles(contact): + lines.append(f"SOCIALPROFILE:{profile}") + if contact.note: lines.append(f"NOTE:{_escape_vcard_value(contact.note)}") - # Add REV timestamp - lines.append(f"REV:{Contact.generate_rev()}") + for name, value in contact.extensions.items(): + lines.append(f"{name}:{_escape_vcard_value(value)}") - # Add UID + lines.append(f"REV:{Contact.generate_rev()}") lines.append(f"UID:urn:uuid:{contact.generate_uid()}") - lines.append("END:VCARD") - return "\n".join(lines) + "\n" + return _serialize(lines) def _format_media_field_v4(field_name: str, value: str) -> str: - """Format PHOTO or LOGO field for vCard 4.0.""" - if value.startswith(("http://", "https://")): - return f"{field_name}:{value}" - else: - # Assume base64 encoded data - # Try to detect media type from data or default to JPEG - if value.startswith("/9j/"): - media_type = "image/jpeg" - elif value.startswith("iVBOR"): - media_type = "image/png" - elif value.startswith("R0lGOD"): - media_type = "image/gif" - else: - media_type = "image/jpeg" - return f"{field_name};ENCODING=b;MEDIATYPE={media_type}:{value}" + """Format PHOTO or LOGO field for vCard 4.0 (inline data as a data: URI).""" + media = _parse_media(value, "image/jpeg") + if media.uri: + return f"{field_name}:{media.uri}" + return f"{field_name}:data:{media.media_type};base64,{media.data}" def _format_key_field_v4(value: str) -> str: """Format KEY field for vCard 4.0.""" - if value.startswith(("http://", "https://")): - return f"KEY:{value}" - else: - # Assume base64 encoded key data (PGP or similar) - return f"KEY;MEDIATYPE=application/pgp-keys:{value}" + if _is_armored_key(value): + return f"KEY;VALUE=text:{_escape_vcard_value(value.strip())}" + media = _parse_media(value, "application/pgp-keys") + if media.uri: + return f"KEY:{media.uri}" + return f"KEY:data:{media.media_type};base64,{media.data}" + + +# --------------------------------------------------------------------------- +# vCard 2.1 +# --------------------------------------------------------------------------- + + +def _escape_21(value: str) -> str: + """Escape a vCard 2.1 component: only semicolons are escaped.""" + return _normalize_text(value).replace(";", "\\;") + + +def _qp_encode(text: str) -> list[str]: + """ + Quoted-printable encode text into atoms, one per character. + + Keeping a character's =XX sequences in one atom means soft line breaks + never split a multi-byte UTF-8 character, which line-based parsers choke on. + """ + text = text.replace("\n", "\r\n") + atoms: list[str] = [] + for index, char in enumerate(text): + is_last = index == len(text) - 1 + if ("!" <= char <= "~" and char != "=") or (char in "\t " and not is_last): + atoms.append(char) + else: + atoms.append("".join(f"={byte:02X}" for byte in char.encode("utf-8"))) + return atoms + + +def _prop_21(name: str, value: str, *, force_qp: bool = False) -> str: + """ + Build a vCard 2.1 text property, switching to quoted-printable if needed. + + vCard 2.1 has no RFC 2425 line folding, so non-ASCII, multi-line or long + values are quoted-printable encoded with soft line breaks instead. + """ + line = f"{name}:{value}" + if not force_qp and value.isascii() and "\n" not in value and len(line) <= MAX_LINE_OCTETS: + return line + + lines: list[str] = [] + current = f"{name};CHARSET=UTF-8;ENCODING=QUOTED-PRINTABLE:" + for atom in _qp_encode(value): + if len(current) + len(atom) > MAX_LINE_OCTETS: # leave room for the soft break + lines.append(current + "=") + # A leading space would read as RFC 2425 line folding, so encode it + current = {" ": "=20", "\t": "=09"}.get(atom, atom) + else: + current += atom + lines.append(current) + return CRLF.join(lines) + + +def _base64_21(prefix: str, data: str) -> str: + """Format inline base64 data the vCard 2.1 way: indented lines, then a blank line.""" + chunks = [data[i:i + 72] for i in range(0, len(data), 72)] or [""] + return CRLF.join([prefix + chunks[0], *(" " + chunk for chunk in chunks[1:])]) + CRLF + + +def _create_vcard_21(contact: Contact) -> str: + """ + Generate vCard 2.1 format (for legacy Outlook, car kits and feature phones). + + Args: + contact: Contact object + + Returns: + vCard 2.1 formatted string + """ + lines = [ + "BEGIN:VCARD", + "VERSION:2.1", + ] + + lines.append(_prop_21("N", ";".join(_escape_21(c) for c in _n_components(contact)))) + lines.append(_prop_21("FN", _normalize_text(contact.get_formatted_name()))) + + if contact.nickname: + lines.append(_prop_21("NICKNAME", _normalize_text(contact.nickname))) + + if contact.gender: + lines.append(_prop_21("X-GENDER", _normalize_text(contact.gender))) + + for field_name, prop in (("birthday", "BDAY"), ("anniversary", "X-ANNIVERSARY")): + date = _date(contact, field_name) + if date and not date.startswith("--"): + lines.append(f"{prop}:{date}") + + for field_name, prop in (("title", "TITLE"), ("role", "ROLE")): + value = getattr(contact, field_name) + if value: + lines.append(_prop_21(prop, _normalize_text(value))) + + if contact.org: + lines.append(_prop_21("ORG", _escape_21(contact.org))) + + for value, types in _typed_values(contact, PHONE_TYPES): + lines.append(f"TEL;{types.replace(',', ';')}:{value}") + + for value, types in _typed_values(contact, EMAIL_TYPES): + lines.append(f"EMAIL;INTERNET;{types}:{value}") + + for value in contact.values("website"): + lines.append(f"URL;WORK:{_clean(value)}") + + for adr_type, components in _addresses(contact): + lines.append(_prop_21(f"ADR;{adr_type}", ";".join(_escape_21(c) for c in components))) + + for field_name, prop in (("photo", "PHOTO"), ("logo", "LOGO")): + value = getattr(contact, field_name) + if value: + media = _parse_media(value, "image/jpeg") + if media.uri: + lines.append(f"{prop};VALUE=URL:{media.uri}") + else: + prefix = f"{prop};ENCODING=BASE64;{_v3_type(media.media_type)}:" + lines.append(_base64_21(prefix, media.data)) + + if contact.categories: + lines.append(_prop_21("CATEGORIES", _normalize_text(contact.categories))) + + geo = _geo(contact) + if geo: + lines.append(f"GEO:{geo[0]},{geo[1]}") + + if contact.tz: + # vCard 2.1 TZ only supports UTC offsets + offset = parse_utc_offset(contact.tz) + if offset: + lines.append(f"TZ:{offset[0]}{offset[1]}:{offset[2]}") + + if contact.key: + if _is_armored_key(contact.key): + lines.append(_prop_21("KEY;PGP", _normalize_text(contact.key.strip()), force_qp=True)) + else: + media = _parse_media(contact.key, "application/pgp-keys") + if media.uri: + lines.append(f"KEY;VALUE=URL:{media.uri}") + else: + prefix = f"KEY;ENCODING=BASE64;{_v3_type(media.media_type)}:" + lines.append(_base64_21(prefix, media.data)) + + for profile in _social_profiles(contact): + lines.append(f"X-SOCIALPROFILE:{profile}") + + if contact.note: + lines.append(_prop_21("NOTE", _normalize_text(contact.note))) + + for name, value in contact.extensions.items(): + lines.append(_prop_21(name, _normalize_text(value))) + + lines.append(f"REV:{Contact.generate_rev()}") + lines.append(f"UID:{contact.generate_uid()}") + lines.append("END:VCARD") + return _serialize(lines, fold=False) diff --git a/csv2vcard/csv2vcard.py b/csv2vcard/csv2vcard.py index e26f052..9459b43 100644 --- a/csv2vcard/csv2vcard.py +++ b/csv2vcard/csv2vcard.py @@ -14,7 +14,7 @@ export_vcards_split, ) from csv2vcard.mapping import load_mapping -from csv2vcard.models import VCardVersion +from csv2vcard.models import Contact, VCardVersion from csv2vcard.parse_csv import find_csv_files, parse_csv from csv2vcard.utils import strip_accents_from_contact @@ -34,6 +34,7 @@ def csv2vcard( strip_accents: bool = False, max_file_size: int | None = None, max_vcards_per_file: int | None = None, + keep_unmapped: bool = False, csv_delimeter: str | None = None, # Legacy parameter name (deprecated) ) -> list[Path]: """ @@ -43,7 +44,7 @@ def csv2vcard( csv_filename: Path to the CSV file or directory containing CSV files csv_delimiter: Field delimiter character (default: ",") output_dir: Output directory (default: ./export/) - version: vCard version to generate (default: 3.0) + version: vCard version to generate: 2.1, 3.0 or 4.0 (default: 3.0) strict: Raise errors on validation issues (default: False) single_file: Export all contacts to a single .vcf file (default: False) encoding: File encoding (auto-detected if None) @@ -51,6 +52,7 @@ def csv2vcard( strip_accents: Remove accents from contact fields (default: False) max_file_size: Maximum file size in bytes for split files (v0.5.0) max_vcards_per_file: Maximum vCards per file for split files (v0.5.0) + keep_unmapped: Keep CSV columns that match no field as X- properties (v0.6.0) csv_delimeter: DEPRECATED - use csv_delimiter instead Returns: @@ -90,6 +92,8 @@ def csv2vcard( # Parse all CSV files and generate vCards all_vcards: list[dict[str, str]] = [] + used_filenames: set[str] = set() + uid_counts: dict[str, int] = {} for csv_file in csv_files: contacts = parse_csv( csv_file, @@ -97,13 +101,24 @@ def csv2vcard( strict=strict, encoding=encoding, mapping=mapping, + keep_unmapped=keep_unmapped, ) for contact in contacts: # Apply accent stripping if requested (v0.5.0) if strip_accents: contact = strip_accents_from_contact(contact) - vcard = create_vcard(contact, version=version) + contact_obj = Contact.from_dict(contact) + + # Identical contacts would share a UID; number repeats so each stays + # distinct (and still stable across runs) + uid = contact_obj.generate_uid() + uid_counts[uid] = uid_counts.get(uid, 0) + 1 + if uid_counts[uid] > 1: + contact_obj.uid = f"{uid}/{uid_counts[uid]}" + + vcard = create_vcard(contact_obj, version=version) + vcard["filename"] = _unique_filename(vcard["filename"], used_filenames) all_vcards.append(vcard) if not all_vcards: @@ -137,6 +152,19 @@ def csv2vcard( return created_files +def _unique_filename(filename: str, used: set[str]) -> str: + """Suffix a filename (_2, _3, ...) so contacts with the same name don't overwrite each other.""" + stem, suffix = filename.rsplit(".", 1) + candidate = filename + counter = 1 + # Compare case-insensitively: macOS and Windows filesystems are case-insensitive + while candidate.lower() in used: + counter += 1 + candidate = f"{stem}_{counter}.{suffix}" + used.add(candidate.lower()) + return candidate + + def test_csv2vcard( output_dir: str | Path | None = None, version: VCardVersion = VCardVersion.V3_0, @@ -146,7 +174,7 @@ def test_csv2vcard( Args: output_dir: Output directory (default: ./export/) - version: vCard version to generate (default: 3.0) + version: vCard version to generate: 2.1, 3.0 or 4.0 (default: 3.0) """ mock_contact = { "last_name": "Gump", @@ -173,3 +201,7 @@ def test_csv2vcard( print(vcard["output"]) export_vcard(vcard, output_dir) + + +# Not a test: keep pytest from collecting this public helper +test_csv2vcard.__test__ = False # type: ignore[attr-defined] diff --git a/csv2vcard/export_vcard.py b/csv2vcard/export_vcard.py index cc20a0a..63831d0 100644 --- a/csv2vcard/export_vcard.py +++ b/csv2vcard/export_vcard.py @@ -4,6 +4,7 @@ import logging import warnings +from collections.abc import Sequence from pathlib import Path from csv2vcard.exceptions import ExportError @@ -61,7 +62,7 @@ def export_vcard( ) from None try: - output_file.write_text(output, encoding="utf-8") + output_file.write_text(output, encoding="utf-8", newline="") logger.info(f"Created vCard for {name}: {output_file}") return output_file except OSError as e: @@ -100,7 +101,7 @@ def ensure_export_dir(output_dir: str | Path | None = None) -> Path: def export_vcards_combined( - vcards: list[dict[str, str] | VCardOutput], + vcards: Sequence[dict[str, str] | VCardOutput], output_path: str | Path, ) -> Path: """ @@ -133,7 +134,7 @@ def export_vcards_combined( combined = "".join(outputs) try: - output_file.write_text(combined, encoding="utf-8") + output_file.write_text(combined, encoding="utf-8", newline="") logger.info(f"Created combined vCard with {len(vcards)} contacts: {output_file}") return output_file except OSError as e: @@ -142,7 +143,7 @@ def export_vcards_combined( def export_vcards_split( - vcards: list[dict[str, str] | VCardOutput], + vcards: Sequence[dict[str, str] | VCardOutput], output_dir: str | Path, base_filename: str = "contacts", max_file_size: int | None = None, @@ -194,7 +195,7 @@ def write_chunk() -> None: combined = "".join(current_chunk) try: - output_file.write_text(combined, encoding="utf-8") + output_file.write_text(combined, encoding="utf-8", newline="") created_files.append(output_file) logger.info(f"Created split vCard with {len(current_chunk)} contacts: {output_file}") except OSError as e: diff --git a/csv2vcard/mapping.py b/csv2vcard/mapping.py index 87040bb..f4a82ec 100644 --- a/csv2vcard/mapping.py +++ b/csv2vcard/mapping.py @@ -4,13 +4,16 @@ import json import logging +import re from pathlib import Path -from csv2vcard.models import ALL_FIELDS +from csv2vcard.models import ALL_FIELDS, EXTENSION_PREFIX, MULTI_VALUE_FIELDS logger = logging.getLogger(__name__) -# Default mapping: vCard field -> list of possible CSV column names +# Default mapping: vCard field -> list of possible CSV column names. +# Column names are matched case-insensitively, treating spaces, hyphens and +# underscores alike ("First Name" matches "first_name"). DEFAULT_MAPPING: dict[str, list[str]] = { # Name components "last_name": ["last_name", "lastname", "last", "surname", "family_name", "familyname"], @@ -23,28 +26,46 @@ "gender": ["gender", "sex"], "birthday": ["birthday", "birthdate", "birth_date", "dob", "date_of_birth", "bday"], "anniversary": ["anniversary", "wedding_anniversary", "wedding_date"], + "pronouns": ["pronouns", "pronoun"], + "language": ["language", "lang", "preferred_language"], # Contact - single (backwards compatible) - "phone": ["phone", "telephone", "tel", "mobile", "cell", "cellphone", "phone_number"], - "email": ["email", "e-mail", "email_address", "mail"], - "website": ["website", "url", "web", "homepage", "webpage", "site"], + "phone": ["phone", "telephone", "tel", "phone_number"], + "email": ["email", "e-mail", "email_address", "e-mail_address", "mail"], + "website": ["website", "url", "web", "homepage", "webpage", "web_page", "site"], # Contact - multi-type phone (v0.5.0) - "phone_cell": ["phone_cell", "cell_phone", "mobile_phone", "mobile"], + "phone_cell": [ + "phone_cell", "cell_phone", "mobile_phone", "mobile", "cell", "cellphone", + ], "phone_home": ["phone_home", "home_phone", "personal_phone"], "phone_work": ["phone_work", "work_phone", "business_phone", "office_phone"], - "phone_fax": ["phone_fax", "fax", "fax_number"], + "phone_fax": ["phone_fax", "fax", "fax_number", "business_fax"], # Contact - multi-type email (v0.5.0) "email_home": ["email_home", "home_email", "personal_email"], "email_work": ["email_work", "work_email", "business_email", "office_email"], + # Social profile URL - RFC 9554 (v0.6.0) + "social_profile": ["social_profile", "social_profiles", "social", "social_url", "profile_url"], # Organization "org": ["org", "organization", "organisation", "company", "employer", "business"], "title": ["title", "job_title", "jobtitle", "position"], "role": ["role", "job_role", "function", "occupation"], # Address (default/work) - "street": ["street", "street_address", "address", "address1", "street1", "work_street"], - "city": ["city", "locality", "town", "work_city"], - "region": ["region", "state", "province", "county", "state_province", "work_state"], - "p_code": ["p_code", "postal_code", "postalcode", "zip", "zipcode", "zip_code", "postcode"], - "country": ["country", "country_name", "nation", "work_country"], + "street": [ + "street", "street_address", "address", "address1", "street1", "work_street", + "business_street", + ], + "city": ["city", "locality", "town", "work_city", "business_city"], + "region": [ + "region", "state", "province", "county", "state_province", "work_state", + "business_state", + ], + "p_code": [ + "p_code", "postal_code", "postalcode", "zip", "zipcode", "zip_code", "postcode", + "business_postal_code", + ], + "country": [ + "country", "country_name", "nation", "work_country", "business_country", + "business_country_region", + ], # Address - home (v0.5.0) # Support both "home_street" and "street_home" naming conventions "home_street": [ @@ -57,17 +78,20 @@ "home_p_code": [ "home_p_code", "p_code_home", "home_postal_code", "home_zip", "zip_home", "personal_zip", ], - "home_country": ["home_country", "country_home", "personal_country"], + "home_country": [ + "home_country", "country_home", "personal_country", "home_country_region", + ], # Media (v0.5.0) "photo": ["photo", "picture", "image", "avatar", "photo_url"], "logo": ["logo", "company_logo", "org_logo", "logo_url"], # New vCard fields (v0.5.0) "categories": ["categories", "category", "tags", "groups", "labels"], - "geo": ["geo", "coordinates", "location", "lat_lon", "gps"], + "geo": ["geo", "coordinates", "lat_lon", "latlng", "gps"], "tz": ["tz", "timezone", "time_zone"], "key": ["key", "public_key", "pgp_key", "gpg_key"], # Other "note": ["note", "notes", "comment", "comments", "remarks", "description"], + "uid": ["uid", "contact_id", "external_id"], } @@ -123,37 +147,101 @@ def load_mapping(mapping_path: str | Path | None = None) -> dict[str, list[str]] return merged +def normalize_column_name(name: str) -> str: + """ + Normalize a CSV column name for matching. + + Lower-cases the name and collapses runs of spaces, hyphens and other + punctuation into single underscores. + + Examples: + >>> normalize_column_name(" First Name ") + 'first_name' + >>> normalize_column_name("E-mail 2") + 'e_mail_2' + """ + return re.sub(r"[^0-9a-z]+", "_", name.strip().lower()).strip("_") + + def apply_mapping( row: dict[str, str], mapping: dict[str, list[str]], + *, + keep_unmapped: bool = False, ) -> dict[str, str]: """ Apply field mapping to a CSV row, converting column names to vCard field names. + Single-value fields take the first matching non-empty column. Multi-value + fields (phones, emails, websites, social profiles) collect every matching + column, including numbered variants such as "email_2" or "Phone 3"; the + extra values are returned as "_2", "_3", ... + Args: row: Dictionary with CSV column names as keys mapping: Field mapping (vCard field -> list of CSV column names) + keep_unmapped: Also return non-empty columns that match no field, + as extension properties ("Department" -> "X-DEPARTMENT") Returns: Dictionary with vCard field names as keys """ result: dict[str, str] = {} - # Normalize row keys for case-insensitive matching - normalized_row = {k.lower().strip(): v for k, v in row.items()} + # Normalize row keys for case- and separator-insensitive matching + normalized_row: dict[str, str] = {} + for key, value in row.items(): + normalized_row.setdefault(normalize_column_name(key), value) + consumed: set[str] = set() for vcard_field, csv_columns in mapping.items(): - for csv_col in csv_columns: - csv_col_lower = csv_col.lower().strip() - if csv_col_lower in normalized_row: - value = normalized_row[csv_col_lower] - if value: # Only set if non-empty - result[vcard_field] = value - break # First match wins + aliases = [normalize_column_name(c) for c in csv_columns] + consumed.update(a for a in aliases if a in normalized_row) + + if vcard_field in MULTI_VALUE_FIELDS: + values = _collect_values(normalized_row, aliases, consumed) + if values: + result[vcard_field] = values[0] + for index, value in enumerate(values[1:], start=2): + result[f"{vcard_field}_{index}"] = value + continue + + for alias in aliases: + value = normalized_row.get(alias, "").strip() + if value: # Only set if non-empty + result[vcard_field] = value + break # First match wins + + if keep_unmapped: + for column, value in normalized_row.items(): + prop = EXTENSION_PREFIX + column.upper().replace("_", "-") + if column not in consumed and value.strip() and column: + result.setdefault(prop, value.strip()) return result +def _collect_values( + normalized_row: dict[str, str], + aliases: list[str], + consumed: set[str], +) -> list[str]: + """Collect all values of a multi-value field: exact aliases first, then numbered columns.""" + values = [normalized_row[a].strip() for a in aliases if a in normalized_row] + + numbered: list[tuple[int, str]] = [] + for alias in aliases: + pattern = re.compile(re.escape(alias) + r"_?(\d+)") + for column, value in normalized_row.items(): + match = pattern.fullmatch(column) + if match and column not in consumed: + consumed.add(column) + numbered.append((int(match[1]), value.strip())) + values.extend(value for _, value in sorted(numbered, key=lambda item: item[0])) + + return list(dict.fromkeys(v for v in values if v)) + + def create_example_mapping() -> str: """ Create an example mapping JSON for documentation purposes. diff --git a/csv2vcard/models.py b/csv2vcard/models.py index 842e895..f5d0cdf 100644 --- a/csv2vcard/models.py +++ b/csv2vcard/models.py @@ -4,7 +4,7 @@ import re import uuid -from dataclasses import dataclass, field +from dataclasses import dataclass, field, fields from datetime import datetime, timezone from enum import Enum @@ -12,6 +12,7 @@ class VCardVersion(Enum): """Supported vCard versions.""" + V2_1 = "2.1" V3_0 = "3.0" V4_0 = "4.0" @@ -32,6 +33,8 @@ class VCardVersion(Enum): "gender", "birthday", "anniversary", + "pronouns", # RFC 9554 (v0.6.0) + "language", # Preferred language, e.g. "en" (v0.6.0) # Contact - single (backwards compatible) "phone", "email", @@ -44,6 +47,8 @@ class VCardVersion(Enum): # Contact - multi-type email (v0.5.0) "email_home", "email_work", + # Social profile URL - RFC 9554 (v0.6.0) + "social_profile", # Organization "org", "title", @@ -61,8 +66,8 @@ class VCardVersion(Enum): "home_p_code", "home_country", # Media (v0.5.0) - "photo", # URL or base64-encoded image - "logo", # URL or base64-encoded image + "photo", # URL, data: URI or base64-encoded image + "logo", # URL, data: URI or base64-encoded image # New vCard fields (v0.5.0) "categories", # Comma-separated list "geo", # latitude,longitude @@ -70,8 +75,32 @@ class VCardVersion(Enum): "key", # Public key URL or base64 # Other "note", + "uid", # Stable identifier from the source system (v0.6.0) }) +# Fields that may hold several values (v0.6.0). Extra values are passed in +# contact dicts as "_2", "_3", ... (e.g. "email_2"). +MULTI_VALUE_FIELDS: frozenset[str] = frozenset({ + "phone", + "phone_cell", + "phone_home", + "phone_work", + "phone_fax", + "email", + "email_home", + "email_work", + "website", + "social_profile", +}) + +# Prefix marking unmapped CSV columns passed through as vCard extension properties +EXTENSION_PREFIX = "X-" + +_NUMBERED_KEY = re.compile(r"^(?P[a-z_]+)_(?P\d+)$") + +# Namespace for deterministic UIDs, so re-converting the same CSV yields the same UIDs +UID_NAMESPACE = uuid.uuid5(uuid.NAMESPACE_URL, "https://github.com/tech4242/csv2vcard") + @dataclass class Contact: @@ -87,8 +116,10 @@ class Contact: # Basic info nickname: str = "" gender: str = "" # M, F, O, N, U or full words - birthday: str = "" # YYYY-MM-DD or YYYYMMDD - anniversary: str = "" # YYYY-MM-DD or YYYYMMDD + birthday: str = "" # YYYY-MM-DD, YYYYMMDD, --MM-DD, DD.MM.YYYY, ... + anniversary: str = "" # same formats as birthday + pronouns: str = "" # e.g., "they/them" (v0.6.0) + language: str = "" # e.g., "en", "de-AT" (v0.6.0) # Contact - single (backwards compatible) phone: str = "" @@ -105,6 +136,9 @@ class Contact: email_home: str = "" email_work: str = "" + # Social profile URL (v0.6.0) + social_profile: str = "" + # Organization org: str = "" title: str = "" @@ -125,8 +159,8 @@ class Contact: home_country: str = "" # Media (v0.5.0) - photo: str = "" # URL or base64-encoded image - logo: str = "" # URL or base64-encoded image + photo: str = "" # URL, data: URI or base64-encoded image + logo: str = "" # URL, data: URI or base64-encoded image # New vCard fields (v0.5.0) categories: str = "" # Comma-separated list @@ -136,6 +170,12 @@ class Contact: # Other note: str = "" + uid: str = "" # Source-system identifier; hashed into a stable UUID (v0.6.0) + + # Additional values for MULTI_VALUE_FIELDS, e.g. {"email": ["second@example.com"]} + extra_values: dict[str, list[str]] = field(default_factory=dict) + # Extension properties, e.g. {"X-DEPARTMENT": "Sales"} + extensions: dict[str, str] = field(default_factory=dict) def __post_init__(self) -> None: """Validate and sanitize contact data after initialization.""" @@ -144,128 +184,77 @@ def __post_init__(self) -> None: value = getattr(self, field_name, "") if isinstance(value, str): setattr(self, field_name, value.strip()) + self.extra_values = { + name: [v.strip() for v in values if v.strip()] + for name, values in self.extra_values.items() + } + self.extensions = { + name: value.strip() for name, value in self.extensions.items() if value.strip() + } @classmethod def from_dict(cls, data: dict[str, str]) -> Contact: """ Create Contact from dictionary, providing defaults for missing fields. + Besides the standard field names, the dictionary may contain numbered + keys for multi-value fields ("email_2", "phone_cell_3", ...) and + extension properties ("X-DEPARTMENT"). + Args: data: Dictionary with contact field values Returns: Contact instance """ - return cls( - # Name components - last_name=data.get("last_name", ""), - first_name=data.get("first_name", ""), - middle_name=data.get("middle_name", ""), - name_prefix=data.get("name_prefix", ""), - name_suffix=data.get("name_suffix", ""), - # Basic info - nickname=data.get("nickname", ""), - gender=data.get("gender", ""), - birthday=data.get("birthday", ""), - anniversary=data.get("anniversary", ""), - # Contact - single - phone=data.get("phone", ""), - email=data.get("email", ""), - website=data.get("website", ""), - # Contact - multi-type phone (v0.5.0) - phone_cell=data.get("phone_cell", ""), - phone_home=data.get("phone_home", ""), - phone_work=data.get("phone_work", ""), - phone_fax=data.get("phone_fax", ""), - # Contact - multi-type email (v0.5.0) - email_home=data.get("email_home", ""), - email_work=data.get("email_work", ""), - # Organization - org=data.get("org", ""), - title=data.get("title", ""), - role=data.get("role", ""), - # Address (default/work) - street=data.get("street", ""), - city=data.get("city", ""), - region=data.get("region", ""), - p_code=data.get("p_code", ""), - country=data.get("country", ""), - # Address - home (v0.5.0) - home_street=data.get("home_street", ""), - home_city=data.get("home_city", ""), - home_region=data.get("home_region", ""), - home_p_code=data.get("home_p_code", ""), - home_country=data.get("home_country", ""), - # Media (v0.5.0) - photo=data.get("photo", ""), - logo=data.get("logo", ""), - # New vCard fields (v0.5.0) - categories=data.get("categories", ""), - geo=data.get("geo", ""), - tz=data.get("tz", ""), - key=data.get("key", ""), - # Other - note=data.get("note", ""), - ) + values = {name: data.get(name, "") for name in ALL_FIELDS} + + numbered: dict[str, list[tuple[int, str]]] = {} + extensions: dict[str, str] = {} + for key, value in data.items(): + if key.startswith(EXTENSION_PREFIX): + extensions[key] = value + continue + match = _NUMBERED_KEY.match(key) + if match and match["field"] in MULTI_VALUE_FIELDS: + numbered.setdefault(match["field"], []).append((int(match["index"]), value)) + + extra_values = { + name: [value for _, value in sorted(items)] for name, items in numbered.items() + } + return cls(**values, extra_values=extra_values, extensions=extensions) def to_dict(self) -> dict[str, str]: """ Convert to dictionary for backwards compatibility. Returns: - Dictionary with all contact fields + Dictionary with all contact fields (plus numbered and extension keys) """ - return { - # Name components - "last_name": self.last_name, - "first_name": self.first_name, - "middle_name": self.middle_name, - "name_prefix": self.name_prefix, - "name_suffix": self.name_suffix, - # Basic info - "nickname": self.nickname, - "gender": self.gender, - "birthday": self.birthday, - "anniversary": self.anniversary, - # Contact - single - "phone": self.phone, - "email": self.email, - "website": self.website, - # Contact - multi-type phone (v0.5.0) - "phone_cell": self.phone_cell, - "phone_home": self.phone_home, - "phone_work": self.phone_work, - "phone_fax": self.phone_fax, - # Contact - multi-type email (v0.5.0) - "email_home": self.email_home, - "email_work": self.email_work, - # Organization - "org": self.org, - "title": self.title, - "role": self.role, - # Address (default/work) - "street": self.street, - "city": self.city, - "region": self.region, - "p_code": self.p_code, - "country": self.country, - # Address - home (v0.5.0) - "home_street": self.home_street, - "home_city": self.home_city, - "home_region": self.home_region, - "home_p_code": self.home_p_code, - "home_country": self.home_country, - # Media (v0.5.0) - "photo": self.photo, - "logo": self.logo, - # New vCard fields (v0.5.0) - "categories": self.categories, - "geo": self.geo, - "tz": self.tz, - "key": self.key, - # Other - "note": self.note, - } + result = {f.name: getattr(self, f.name) for f in fields(self) if f.name in ALL_FIELDS} + for name, extra in self.extra_values.items(): + for index, value in enumerate(extra, start=2): + result[f"{name}_{index}"] = value + result.update(self.extensions) + return result + + def values(self, field_name: str) -> list[str]: + """ + Get all non-empty values of a field (primary value first). + + Args: + field_name: Contact field name + + Returns: + List of values, deduplicated in order + """ + candidates = [getattr(self, field_name), *self.extra_values.get(field_name, [])] + return list(dict.fromkeys(v for v in candidates if v)) + + @property + def is_organization(self) -> bool: + """True if the contact represents an organization rather than a person.""" + return bool(self.org) and not (self.first_name or self.middle_name or self.last_name) def get_safe_filename(self) -> str: """ @@ -276,17 +265,12 @@ def get_safe_filename(self) -> str: Returns: Safe filename ending in .vcf """ - # Remove or replace unsafe characters (keep only alphanumeric, underscore, hyphen) - safe_last = re.sub(r"[^\w\-]", "_", self.last_name.lower()) - safe_first = re.sub(r"[^\w\-]", "_", self.first_name.lower()) - - # Prevent path traversal - safe_last = safe_last.replace("..", "_").strip("_.") - safe_first = safe_first.replace("..", "_").strip("_.") + if self.is_organization: + return f"{_sanitize_filename_part(self.org) or 'unknown'}.vcf" # Ensure we have something valid - safe_last = safe_last or "unknown" - safe_first = safe_first or "contact" + safe_last = _sanitize_filename_part(self.last_name) or "unknown" + safe_first = _sanitize_filename_part(self.first_name) or "contact" return f"{safe_last}_{safe_first}.vcf" @@ -297,27 +281,45 @@ def get_formatted_name(self) -> str: Returns: Formatted name string """ - parts = [] - if self.name_prefix: - parts.append(self.name_prefix) - if self.first_name: - parts.append(self.first_name) - if self.middle_name: - parts.append(self.middle_name) - if self.last_name: - parts.append(self.last_name) - if self.name_suffix: - parts.append(self.name_suffix) - return " ".join(parts) or "Unknown" + parts = [ + self.name_prefix, + self.first_name, + self.middle_name, + self.last_name, + self.name_suffix, + ] + return " ".join(p for p in parts if p) or self.org or "Unknown" def generate_uid(self) -> str: """ - Generate a unique identifier for this contact. + Generate a stable unique identifier for this contact. + + Uses the ``uid`` field when set (kept as-is if it is a UUID, otherwise + hashed into one). Without it, the UID is derived from the name, + organization and primary email, so converting the same CSV again + yields the same UIDs and re-imports update instead of duplicating. Returns: UUID string """ - return str(uuid.uuid4()) + if self.uid: + try: + return str(uuid.UUID(self.uid)) + except ValueError: + return str(uuid.uuid5(UID_NAMESPACE, f"uid:{self.uid}")) + + emails = self.values("email") + self.values("email_work") + self.values("email_home") + identity = "|".join( + part.casefold() + for part in ( + self.first_name, + self.middle_name, + self.last_name, + self.org, + emails[0] if emails else "", + ) + ) + return str(uuid.uuid5(UID_NAMESPACE, f"contact:{identity}")) @staticmethod def generate_rev() -> str: @@ -330,6 +332,12 @@ def generate_rev() -> str: return datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") +def _sanitize_filename_part(value: str) -> str: + """Keep only alphanumerics, underscores and hyphens; prevent path traversal.""" + safe = re.sub(r"[^\w\-]", "_", value.lower()) + return safe.replace("..", "_").strip("_.") + + @dataclass class VCardOutput: """Output from vCard generation.""" diff --git a/csv2vcard/parse_csv.py b/csv2vcard/parse_csv.py index db45905..29dde60 100644 --- a/csv2vcard/parse_csv.py +++ b/csv2vcard/parse_csv.py @@ -77,44 +77,35 @@ def find_csv_files(source: str | Path) -> list[Path]: raise ValueError(f"Invalid source path: {source_path}") -def parse_csv( - csv_filename: str | Path, - csv_delimiter: str = ",", - *, - strict: bool = False, - encoding: str | None = None, - mapping: dict[str, list[str]] | None = None, -) -> list[dict[str, str]]: - """ - Parse a CSV file and return a list of contact dictionaries. +def _resolve_encoding(filepath: Path, encoding: str | None) -> str: + """Pick the encoding to read with, so a UTF-8 byte order mark is always removed.""" + if encoding is None: + encoding = detect_encoding(filepath) + # Excel's "CSV UTF-8" export starts with a BOM; "utf-8-sig" strips it + if encoding.lower().replace("_", "-") in ("utf-8", "utf8", "ascii"): + return "utf-8-sig" + return encoding - Args: - csv_filename: Path to the CSV file - csv_delimiter: Field delimiter character (default: ",") - strict: If True, raise errors on validation issues - encoding: File encoding (auto-detected if None) - mapping: Field mapping (uses default if None) - Returns: - List of contact dictionaries with vCard field names +def _iter_contact_dicts( + filepath: Path, + csv_delimiter: str, + *, + strict: bool, + encoding: str | None, + mapping: dict[str, list[str]] | None, + keep_unmapped: bool, +) -> Iterator[dict[str, str]]: + """ + Stream contact dictionaries from a CSV file. Raises: - ParseError: If file cannot be parsed (only in strict mode) - ValidationError: If strict=True and validation fails + ValidationError: If the file is invalid (non-strict callers handle this) + ParseError: On empty files, malformed rows or read errors (strict mode) """ - filepath = Path(csv_filename) + validate_csv_file(filepath, strict=strict) - try: - validate_csv_file(filepath, strict=strict) - except ValidationError: - if strict: - raise - logger.error(f"CSV validation failed: {filepath}") - return [] - - # Detect encoding if not specified - if encoding is None: - encoding = detect_encoding(filepath) + encoding = _resolve_encoding(filepath, encoding) # Use default mapping if not provided if mapping is None: @@ -123,7 +114,9 @@ def parse_csv( logger.info(f"Parsing CSV file: {filepath} (encoding: {encoding})") try: - with open(filepath, encoding=encoding, newline="", errors="replace") as f: + # Strict mode fails on undecodable bytes instead of silently replacing them + errors = "strict" if strict else "replace" + with open(filepath, encoding=encoding, newline="", errors=errors) as f: reader = csv.reader(f, delimiter=csv_delimiter) try: @@ -132,45 +125,124 @@ def parse_csv( logger.error(f"CSV file is empty: {filepath}") if strict: raise ParseError(f"CSV file is empty: {filepath}") from None - return [] + return - # Normalize header names (strip whitespace) - header = [col.strip() for col in header] + # Normalize header names (strip whitespace and any leftover BOM) + header = [col.strip().lstrip("\ufeff").strip() for col in header] logger.debug(f"CSV headers: {header}") - contacts: list[dict[str, str]] = [] + count = 0 for row_num, row in enumerate(reader, start=2): + if not any(cell.strip() for cell in row): + continue # Skip blank lines + if len(row) != len(header): + msg = f"Row {row_num} has {len(row)} columns, expected {len(header)}" + if strict: + raise ParseError(f"{filepath}: {msg}") + logger.warning(f"{msg}, skipped") + continue + + if any("\ufffd" in cell for cell in row): logger.warning( - f"Row {row_num} has {len(row)} columns, expected {len(header)}" + f"Row {row_num} contains bytes that are invalid in {encoding} " + "(replaced with U+FFFD); try --encoding" ) - continue # Create raw contact dict from CSV - raw_contact = dict(zip(header, row)) + raw_contact = dict(zip(header, row, strict=True)) # Apply field mapping - contact = apply_mapping(raw_contact, mapping) + contact = apply_mapping(raw_contact, mapping, keep_unmapped=keep_unmapped) validation_warnings = validate_contact(contact, strict=strict) for warning in validation_warnings: logger.warning(f"Row {row_num}: {warning}") - contacts.append(contact) + count += 1 + yield contact - logger.info(f"Parsed {len(contacts)} contacts from CSV") - return contacts + logger.info(f"Parsed {count} contacts from {filepath}") - except csv.Error as e: - logger.error(f"CSV parsing error: {e}") + except (csv.Error, UnicodeDecodeError) as e: + logger.error(f"CSV parsing error in {filepath}: {e}") if strict: raise ParseError(f"Failed to parse CSV: {e}") from e - return [] except OSError as e: logger.error(f"I/O error reading {filepath}: {e}") if strict: raise ParseError(f"Failed to read CSV file: {e}") from e - return [] + + +def parse_csv( + csv_filename: str | Path, + csv_delimiter: str = ",", + *, + strict: bool = False, + encoding: str | None = None, + mapping: dict[str, list[str]] | None = None, + keep_unmapped: bool = False, +) -> list[dict[str, str]]: + """ + Parse a CSV file and return a list of contact dictionaries. + + Args: + csv_filename: Path to the CSV file + csv_delimiter: Field delimiter character (default: ",") + strict: If True, raise errors on validation issues, malformed rows + and undecodable bytes + encoding: File encoding (auto-detected if None) + mapping: Field mapping (uses default if None) + keep_unmapped: Keep unmapped columns as X- extension properties + + Returns: + List of contact dictionaries with vCard field names + + Raises: + ParseError: If file cannot be parsed (only in strict mode) + ValidationError: If strict=True and validation fails + """ + return list(iter_contact_dicts( + csv_filename, + csv_delimiter, + strict=strict, + encoding=encoding, + mapping=mapping, + keep_unmapped=keep_unmapped, + )) + + +def iter_contact_dicts( + csv_filename: str | Path, + csv_delimiter: str = ",", + *, + strict: bool = False, + encoding: str | None = None, + mapping: dict[str, list[str]] | None = None, + keep_unmapped: bool = False, +) -> Iterator[dict[str, str]]: + """ + Iterate over contact dictionaries in a CSV file without loading it all. + + Takes the same arguments as :func:`parse_csv`. + + Yields: + Contact dictionaries with vCard field names + """ + filepath = Path(csv_filename) + try: + yield from _iter_contact_dicts( + filepath, + csv_delimiter, + strict=strict, + encoding=encoding, + mapping=mapping, + keep_unmapped=keep_unmapped, + ) + except ValidationError: + if strict: + raise + logger.error(f"CSV validation failed: {filepath}") def parse_csv_files( @@ -180,6 +252,7 @@ def parse_csv_files( strict: bool = False, encoding: str | None = None, mapping_file: str | Path | None = None, + keep_unmapped: bool = False, ) -> list[dict[str, str]]: """ Parse one or more CSV files from a file or directory path. @@ -190,6 +263,7 @@ def parse_csv_files( strict: If True, raise errors on validation issues encoding: File encoding (auto-detected if None) mapping_file: Path to JSON mapping file (uses default if None) + keep_unmapped: Keep unmapped columns as X- extension properties Returns: List of all contact dictionaries from all CSV files @@ -209,6 +283,7 @@ def parse_csv_files( strict=strict, encoding=encoding, mapping=mapping, + keep_unmapped=keep_unmapped, ) all_contacts.extend(contacts) @@ -222,21 +297,27 @@ def iter_contacts( *, encoding: str | None = None, mapping: dict[str, list[str]] | None = None, + keep_unmapped: bool = False, ) -> Iterator[Contact]: """ - Iterate over contacts in a CSV file (memory efficient). + Iterate over contacts in a CSV file (memory efficient: rows are streamed). Args: csv_filename: Path to the CSV file csv_delimiter: Field delimiter character encoding: File encoding (auto-detected if None) mapping: Field mapping (uses default if None) + keep_unmapped: Keep unmapped columns as X- extension properties Yields: Contact objects """ - for contact_dict in parse_csv( - csv_filename, csv_delimiter, encoding=encoding, mapping=mapping + for contact_dict in iter_contact_dicts( + csv_filename, + csv_delimiter, + encoding=encoding, + mapping=mapping, + keep_unmapped=keep_unmapped, ): yield Contact.from_dict(contact_dict) diff --git a/csv2vcard/utils.py b/csv2vcard/utils.py index e51def8..f2c7fd8 100644 --- a/csv2vcard/utils.py +++ b/csv2vcard/utils.py @@ -2,7 +2,9 @@ from __future__ import annotations +import re import unicodedata +from datetime import date def strip_accents(text: str) -> str: @@ -58,3 +60,142 @@ def strip_accents_from_contact(contact: dict[str, str]) -> dict[str, str]: key: strip_accents(value) if isinstance(value, str) else value for key, value in contact.items() } + + +_ISO_DATE = re.compile(r"^(\d{4})[-/.]?(\d{2})[-/.]?(\d{2})$") +_NO_YEAR_DATE = re.compile(r"^--(\d{2})-?(\d{2})$") +_DOTTED_DATE = re.compile(r"^(\d{1,2})\.(\d{1,2})\.(\d{4})$") +_SLASHED_DATE = re.compile(r"^(\d{1,2})/(\d{1,2})/(\d{4})$") +_UTC_OFFSET = re.compile(r"^(?:UTC|GMT)?\s*([+-])(\d{1,2}):?(\d{2})?$", re.IGNORECASE) +_GLOBAL_PHONE = re.compile(r"^\+[\d\s().\-]+$") + + +def normalize_date(value: str) -> str | None: + """ + Normalize a date to vCard basic format. + + Accepts ISO dates (1944-06-06, 19440606, 1944/06/06), dates without a + year (--06-06), European dotted dates (06.06.1944) and slashed dates + when the day/month order is unambiguous (13/06/1944, 06/13/1944). + A time part after "T" or a space is ignored. + + Args: + value: Raw date string + + Returns: + "YYYYMMDD", "--MMDD" for dates without a year, or None if the value + cannot be interpreted unambiguously + + Examples: + >>> normalize_date("1944-06-06") + '19440606' + >>> normalize_date("--12-24") + '--1224' + >>> normalize_date("06/06/1944") is None + True + """ + value = re.split(r"[T ]", value.strip(), maxsplit=1)[0] + + if match := _NO_YEAR_DATE.match(value): + month, day = int(match[1]), int(match[2]) + # 2000 is a leap year, so --0229 is accepted + return f"--{month:02d}{day:02d}" if _is_valid_date(2000, month, day) else None + + if match := _ISO_DATE.match(value): + year, month, day = int(match[1]), int(match[2]), int(match[3]) + elif match := _DOTTED_DATE.match(value): + day, month, year = int(match[1]), int(match[2]), int(match[3]) + elif match := _SLASHED_DATE.match(value): + first, second, year = int(match[1]), int(match[2]), int(match[3]) + if first > 12 >= second: + day, month = first, second + elif second > 12 >= first: + month, day = first, second + elif first == second: + day = month = first + else: + return None # Ambiguous: could be MM/DD or DD/MM + else: + return None + + return f"{year:04d}{month:02d}{day:02d}" if _is_valid_date(year, month, day) else None + + +def _is_valid_date(year: int, month: int, day: int) -> bool: + try: + date(year, month, day) + except ValueError: + return False + return True + + +def parse_geo(value: str) -> tuple[str, str] | None: + """ + Parse geographic coordinates. + + Accepts "lat,lon", "lat;lon" and "geo:lat,lon". + + Args: + value: Coordinate string + + Returns: + (latitude, longitude) as strings, or None if invalid or out of range + """ + value = value.strip() + if value.lower().startswith("geo:"): + value = value[4:].split(";", 1)[0] + parts = [p.strip() for p in value.replace(";", ",").split(",")] + if len(parts) != 2: + return None + try: + lat, lon = float(parts[0]), float(parts[1]) + except ValueError: + return None + if not (-90.0 <= lat <= 90.0 and -180.0 <= lon <= 180.0): + return None + return parts[0], parts[1] + + +def parse_utc_offset(value: str) -> tuple[str, str, str] | None: + """ + Parse a UTC offset such as "-05:00", "+0530" or "UTC+1". + + Args: + value: Timezone string + + Returns: + (sign, hours, minutes) zero-padded, or None if not a UTC offset + """ + match = _UTC_OFFSET.match(value.strip()) + if not match: + return None + hours, minutes = int(match[2]), int(match[3] or 0) + if hours > 14 or minutes > 59: + return None + return match[1], f"{hours:02d}", f"{minutes:02d}" + + +def phone_to_tel_uri(value: str) -> str | None: + """ + Convert an international phone number to an RFC 3966 tel: URI. + + Only global numbers (starting with "+") can be expressed as a tel: URI + without a phone-context, so local numbers return None. + + Args: + value: Phone number, e.g. "+49 170 5 25 25 25" + + Returns: + URI such as "tel:+49-170-5-25-25-25", or None + + Examples: + >>> phone_to_tel_uri("+1 (555) 123-4567") + 'tel:+1-555-123-4567' + >>> phone_to_tel_uri("0170 123") is None + True + """ + value = value.strip() + if not _GLOBAL_PHONE.match(value) or not any(c.isdigit() for c in value): + return None + digits = re.sub(r"[\s().\-]+", "-", value[1:]).strip("-") + return f"tel:+{digits}" diff --git a/csv2vcard/validators.py b/csv2vcard/validators.py index 82e23ca..2a17f17 100644 --- a/csv2vcard/validators.py +++ b/csv2vcard/validators.py @@ -8,6 +8,7 @@ from csv2vcard.exceptions import ValidationError from csv2vcard.models import REQUIRED_FIELDS +from csv2vcard.utils import normalize_date, parse_geo logger = logging.getLogger(__name__) @@ -17,6 +18,8 @@ # Email validation regex (RFC 5322 simplified) EMAIL_REGEX = re.compile(r"^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$") +_EMAIL_KEY = re.compile(r"^email(_home|_work)?(_\d+)?$") + def validate_contact(contact: dict[str, str], strict: bool = False) -> list[str]: """ @@ -50,15 +53,16 @@ def validate_contact(contact: dict[str, str], strict: bool = False) -> list[str] raise ValidationError(msg) warnings.append(msg) - # Validate email format if provided - email = contact.get("email", "").strip() - if email and not validate_email(email): - warnings.append(f"Invalid email format: {email}") - - # Validate multi-type emails (v0.5.0) - for email_field in ("email_home", "email_work"): - email_val = contact.get(email_field, "").strip() - if email_val and not validate_email(email_val): + # Validate email format, including multi-type and numbered emails ("email_2") + for email_field, email_val in contact.items(): + if not _EMAIL_KEY.match(email_field): + continue + email_val = email_val.strip() + if not email_val or validate_email(email_val): + continue + if email_field == "email": + warnings.append(f"Invalid email format: {email_val}") + else: warnings.append(f"Invalid email format in {email_field}: {email_val}") # Validate gender (v0.5.0) @@ -71,6 +75,15 @@ def validate_contact(contact: dict[str, str], strict: bool = False) -> list[str] if geo and not validate_geo(geo): warnings.append(f"Invalid geo coordinates: {geo}") + # Validate dates (v0.6.0) + for date_field in ("birthday", "anniversary"): + date_val = contact.get(date_field, "").strip() + if date_val and normalize_date(date_val) is None: + warnings.append( + f"Unrecognized {date_field} '{date_val}' (use YYYY-MM-DD; " + "DD/MM vs MM/DD can't be told apart)" + ) + return warnings @@ -164,25 +177,7 @@ def validate_geo(geo: str) -> bool: >>> validate_geo("invalid") False """ - if not geo: - return False - - # Allow semicolon separator (vCard 3.0 format) or comma (common format) - parts = geo.replace(";", ",").split(",") - - if len(parts) != 2: - return False - - try: - lat = float(parts[0].strip()) - lon = float(parts[1].strip()) - except ValueError: - return False - - # Check valid ranges - if not (-90.0 <= lat <= 90.0): - return False - return -180.0 <= lon <= 180.0 + return bool(geo) and parse_geo(geo) is not None def validate_csv_file(filepath: Path, strict: bool = False) -> None: diff --git a/pyproject.toml b/pyproject.toml index e342742..bab9873 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,11 +1,11 @@ [build-system] -requires = ["setuptools>=61.0", "wheel"] +requires = ["setuptools>=77.0"] build-backend = "setuptools.build_meta" [project] name = "csv2vcard" -version = "0.5.1" -description = "A library for converting CSVs to vCards (vCard 3.0 and 4.0)" +version = "0.6.0" +description = "A library for converting CSVs to vCards (vCard 2.1, 3.0 and 4.0)" readme = "DESCRIPTION.md" license = "MIT" license-files = ["LICENSE.txt"] @@ -16,16 +16,16 @@ keywords = ["csv", "vcard", "contacts", "export", "vcf"] classifiers = [ "Development Status :: 4 - Beta", "Programming Language :: Python :: 3", - "Programming Language :: Python :: 3.9", "Programming Language :: Python :: 3.10", "Programming Language :: Python :: 3.11", "Programming Language :: Python :: 3.12", "Programming Language :: Python :: 3.13", + "Programming Language :: Python :: 3.14", "Typing :: Typed", "Operating System :: OS Independent", "Topic :: Utilities", ] -requires-python = ">=3.9" +requires-python = ">=3.10" dependencies = [] [project.optional-dependencies] @@ -47,6 +47,7 @@ csv2vcard = "csv2vcard.cli:app" [project.urls] "Homepage" = "https://github.com/tech4242/csv2vcard" "Bug Tracker" = "https://github.com/tech4242/csv2vcard/issues" +"Changelog" = "https://github.com/tech4242/csv2vcard/blob/master/CHANGELOG.md" [tool.setuptools.packages.find] where = ["."] @@ -56,22 +57,18 @@ include = ["csv2vcard*"] csv2vcard = ["py.typed"] [tool.mypy] -python_version = "3.9" +python_version = "3.10" warn_return_any = true warn_unused_configs = true strict = true [tool.ruff] -target-version = "py39" +target-version = "py310" line-length = 100 [tool.ruff.lint] select = ["E", "F", "I", "N", "W", "UP", "B", "C4", "SIM"] -[tool.ruff.lint.per-file-ignores] -# cli.py uses Optional[X] for Python 3.9 compatibility with Typer runtime type evaluation -"csv2vcard/cli.py" = ["UP045"] - [tool.pytest.ini_options] testpaths = ["tests"] addopts = "-v --cov=csv2vcard --cov-report=term-missing" diff --git a/setup.py b/setup.py deleted file mode 100644 index 0b8f6c8..0000000 --- a/setup.py +++ /dev/null @@ -1,33 +0,0 @@ -# Legacy setup.py for older pip versions -# Configuration is now in pyproject.toml - -import os - -from setuptools import setup - -with open(os.path.join(os.path.dirname(__file__), 'README.md')) as readme: - README = readme.read() - -setup( - name='csv2vcard', - packages=['csv2vcard'], - version='0.3.0', - description='A library for converting CSVs to vCards (vCard 3.0 and 4.0)', - long_description=README, - long_description_content_type='text/markdown', - author='tech4242', - url='https://github.com/tech4242/csv2vcard', - keywords=['csv', 'vcard', 'contacts', 'export', 'vcf'], - python_requires='>=3.9', - classifiers=[ - 'Development Status :: 4 - Beta', - 'License :: OSI Approved :: MIT License', - 'Programming Language :: Python :: 3', - 'Programming Language :: Python :: 3.9', - 'Programming Language :: Python :: 3.10', - 'Programming Language :: Python :: 3.11', - 'Programming Language :: Python :: 3.12', - 'Programming Language :: Python :: 3.13', - 'Typing :: Typed', - ], -) diff --git a/tests/test_cli.py b/tests/test_cli.py index 1451a88..82ed46e 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -183,3 +183,43 @@ def test_convert_help(self, runner: CliRunner) -> None: assert "CSV file" in result.stdout # Check for "delimiter" without dashes due to ANSI escape codes in rich output assert "delimiter" in result.stdout + + +class TestCLIV060: + """Test CLI options added in v0.6.0.""" + + def test_top_level_version(self, runner: CliRunner) -> None: + """Test `csv2vcard --version` without a subcommand.""" + from csv2vcard import __version__ + + result = runner.invoke(app, ["--version"]) + + assert result.exit_code == 0 + assert f"csv2vcard version {__version__}" in result.stdout + + def test_convert_vcard_21( + self, runner: CliRunner, sample_csv: Path, temp_dir: Path + ) -> None: + """Test creating vCard 2.1 via CLI.""" + output_dir = temp_dir / "output" + + result = runner.invoke( + app, ["convert", str(sample_csv), "-V", "2.1", "-o", str(output_dir)] + ) + + assert result.exit_code == 0 + content = next(output_dir.glob("*.vcf")).read_text(encoding="utf-8") + assert "VERSION:2.1" in content + + def test_keep_unmapped(self, runner: CliRunner, temp_dir: Path) -> None: + """Test the --keep-unmapped option.""" + csv_path = temp_dir / "extra.csv" + csv_path.write_text("last_name,first_name,Team\nDoe,John,Core\n", encoding="utf-8") + output_dir = temp_dir / "output" + + result = runner.invoke( + app, ["convert", str(csv_path), "--keep-unmapped", "-o", str(output_dir)] + ) + + assert result.exit_code == 0 + assert "X-TEAM:Core" in (output_dir / "doe_john.vcf").read_text(encoding="utf-8") diff --git a/tests/test_create_vcard.py b/tests/test_create_vcard.py index 8089210..526c074 100644 --- a/tests/test_create_vcard.py +++ b/tests/test_create_vcard.py @@ -50,8 +50,8 @@ def test_create_vcard_minimal_contact(self, minimal_contact: dict[str, str]) -> result = create_vcard(minimal_contact) assert result["filename"] == "doe_john.vcf" - assert "N;" in result["output"] - assert "FN;" in result["output"] + assert "N:Doe;John;;;" in result["output"] + assert "FN:John Doe" in result["output"] # Optional fields should not be present assert "TITLE:" not in result["output"] or "TITLE:;" in result["output"] @@ -93,27 +93,27 @@ def test_with_v4(self, sample_contact: dict[str, str]) -> None: class TestVCard3Format: """Test vCard 3.0 format specifics.""" - def test_v3_has_charset(self, sample_contact: dict[str, str]) -> None: - """Test that vCard 3.0 includes CHARSET.""" + def test_v3_no_charset(self, sample_contact: dict[str, str]) -> None: + """Test that vCard 3.0 doesn't use the vCard 2.1 CHARSET parameter.""" contact = Contact.from_dict(sample_contact) output = _create_vcard_3(contact) - assert "CHARSET=UTF-8" in output + assert "CHARSET" not in output def test_v3_name_format(self, sample_contact: dict[str, str]) -> None: """Test vCard 3.0 name format.""" contact = Contact.from_dict(sample_contact) output = _create_vcard_3(contact) - assert "N;CHARSET=UTF-8:Gump;Forrest;;;" in output - assert "FN;CHARSET=UTF-8:Forrest Gump" in output + assert "N:Gump;Forrest;;;" in output + assert "FN:Forrest Gump" in output def test_v3_address_format(self, sample_contact: dict[str, str]) -> None: """Test vCard 3.0 address format.""" contact = Contact.from_dict(sample_contact) output = _create_vcard_3(contact) - assert "ADR;TYPE=WORK;CHARSET=UTF-8:" in output + assert "ADR;TYPE=WORK:" in output def test_v3_phone_format(self, sample_contact: dict[str, str]) -> None: """Test vCard 3.0 phone format.""" @@ -461,9 +461,8 @@ def test_categories_v3(self) -> None: } result = create_vcard(contact, version=VCardVersion.V3_0) - assert "CATEGORIES;CHARSET=UTF-8:" in result["output"] - # Commas are escaped in vCard format - assert "Work" in result["output"] + # Commas separate list values and must not be escaped + assert "CATEGORIES:Work,Friends,VIP" in result["output"] def test_categories_v4(self) -> None: """Test categories in vCard 4.0.""" @@ -507,8 +506,8 @@ def test_timezone_v3(self) -> None: } result = create_vcard(contact, version=VCardVersion.V3_0) - assert "TZ:" in result["output"] - assert "America/New_York" in result["output"] + # vCard 3.0 TZ defaults to a UTC offset, so names need VALUE=text + assert "TZ;VALUE=text:America/New_York" in result["output"] def test_timezone_v4(self) -> None: """Test timezone in vCard 4.0.""" @@ -519,4 +518,312 @@ def test_timezone_v4(self) -> None: } result = create_vcard(contact, version=VCardVersion.V4_0) - assert "TZ:" in result["output"] + assert "TZ;VALUE=utc-offset:-0500" in result["output"] + + +def _unfold(output: str) -> str: + """Undo RFC 6350 line folding.""" + return output.replace("\r\n ", "") + + +class TestSerialization: + """Test RFC-compliant line endings, folding and escaping (v0.6.0).""" + + def test_crlf_line_endings(self, sample_contact: dict[str, str]) -> None: + """Test that every line ends with CRLF.""" + for version in VCardVersion: + output = create_vcard(sample_contact, version=version)["output"] + assert output.endswith("END:VCARD\r\n") + assert "\n" not in output.replace("\r\n", "") + + def test_long_lines_folded_at_75_octets(self) -> None: + """Test that long lines are folded and unfold back to the original value.""" + note = "Ünïcödé " * 40 + output = create_vcard({"last_name": "Doe", "first_name": "John", "note": note})["output"] + + for line in output.split("\r\n"): + assert len(line.encode("utf-8")) <= 75 + assert f"NOTE:{note.strip()}" in _unfold(output) + + def test_newline_in_value_cannot_inject_properties(self) -> None: + """Test that line breaks in any field can't create extra properties.""" + contact = { + "last_name": "Doe", + "first_name": "John", + "phone": "1\r\nEMAIL:evil@example.com", + "website": "https://example.com\nNOTE:injected", + "note": "line1\r\nline2", + } + for version in VCardVersion: + output = create_vcard(contact, version=version)["output"] + lines = _unfold(output).split("\r\n") + assert not any(line.startswith("EMAIL") for line in lines) + assert not any(line.startswith("NOTE:injected") for line in lines) + + def test_note_newlines_escaped(self) -> None: + """Test that CRLF and LF in text become an escaped \\n.""" + contact = {"last_name": "Doe", "first_name": "John", "note": "a\r\nb\rc\nd"} + output = create_vcard(contact)["output"] + assert "NOTE:a\\nb\\nc\\nd" in output + + def test_prodid(self, minimal_contact: dict[str, str]) -> None: + """Test that PRODID identifies the generator.""" + output = create_vcard(minimal_contact)["output"] + assert "PRODID:-//tech4242//csv2vcard " in output + + +class TestListValues: + """Test list-valued properties (v0.6.0).""" + + def test_categories_v4_not_escaped(self) -> None: + """Test that category separators are kept and values escaped.""" + contact = {"last_name": "Doe", "first_name": "John", "categories": "Work, VIP;Gold"} + output = create_vcard(contact, version=VCardVersion.V4_0)["output"] + assert "CATEGORIES:Work,VIP\\;Gold" in output + + def test_nickname_list(self) -> None: + """Test that multiple nicknames stay separate.""" + contact = {"last_name": "Doe", "first_name": "Robert", "nickname": "Bob,Rob"} + assert "NICKNAME:Bob,Rob" in create_vcard(contact)["output"] + + +class TestVCard4Compliance: + """Test vCard 4.0 value formats (v0.6.0).""" + + def test_tel_uri_has_no_spaces(self, sample_contact: dict[str, str]) -> None: + """Test that international numbers become valid tel: URIs.""" + output = create_vcard(sample_contact, version=VCardVersion.V4_0)["output"] + assert "TEL;TYPE=work,voice;VALUE=uri:tel:+49-170-5-25-25-25" in output + + def test_local_number_is_text(self) -> None: + """Test that numbers without a country code are written as text.""" + contact = {"last_name": "Doe", "first_name": "John", "phone_cell": "0170 123456"} + output = create_vcard(contact, version=VCardVersion.V4_0)["output"] + assert "TEL;TYPE=cell:0170 123456" in output + assert "tel:0170" not in output + + def test_photo_base64_is_data_uri(self) -> None: + """Test that inline photos use a data: URI instead of ENCODING=b.""" + contact = {"last_name": "Doe", "first_name": "John", "photo": "iVBORw0KGgo="} + output = create_vcard(contact, version=VCardVersion.V4_0)["output"] + assert "PHOTO:data:image/png;base64,iVBORw0KGgo=" in _unfold(output) + assert "ENCODING" not in output + + def test_key_base64_is_data_uri(self) -> None: + """Test that inline keys use a data: URI.""" + contact = {"last_name": "Doe", "first_name": "John", "key": "mQENBFabc="} + output = create_vcard(contact, version=VCardVersion.V4_0)["output"] + assert "KEY:data:application/pgp-keys;base64,mQENBFabc=" in output + + def test_armored_key_is_text(self) -> None: + """Test that ASCII-armored keys are kept as escaped text.""" + key = "-----BEGIN PGP PUBLIC KEY BLOCK-----\nmQENBF\n-----END PGP PUBLIC KEY BLOCK-----" + contact = {"last_name": "Doe", "first_name": "John", "key": key} + output = _unfold(create_vcard(contact, version=VCardVersion.V4_0)["output"]) + assert "KEY;VALUE=text:-----BEGIN PGP PUBLIC KEY BLOCK-----\\nmQENBF\\n" in output + + def test_gender_words_normalized(self) -> None: + """Test that full gender words map to RFC 6350 codes.""" + contact = {"last_name": "Doe", "first_name": "Jane", "gender": "Female"} + output = create_vcard(contact, version=VCardVersion.V4_0)["output"] + assert "GENDER:F\r\n" in output + + def test_unparseable_date_kept_as_text(self) -> None: + """Test that ambiguous dates are preserved with VALUE=text.""" + contact = {"last_name": "Doe", "first_name": "John", "birthday": "06/07/1990"} + output = create_vcard(contact, version=VCardVersion.V4_0)["output"] + assert "BDAY;VALUE=text:06/07/1990" in output + + def test_dates_basic_format(self) -> None: + """Test that dates are written in basic format, year-less as --MMDD.""" + contact = { + "last_name": "Doe", + "first_name": "John", + "birthday": "24.12.1990", + "anniversary": "--06-15", + } + output = create_vcard(contact, version=VCardVersion.V4_0)["output"] + assert "BDAY:19901224" in output + assert "ANNIVERSARY:--0615" in output + + def test_invalid_geo_skipped(self) -> None: + """Test that invalid coordinates are not written.""" + contact = {"last_name": "Doe", "first_name": "John", "geo": "Berlin"} + output = create_vcard(contact, version=VCardVersion.V4_0)["output"] + assert "GEO" not in output + + def test_kind_org(self) -> None: + """Test that organization-only rows become KIND:org cards.""" + contact = {"org": "Acme Inc."} + result = create_vcard(contact, version=VCardVersion.V4_0) + assert "KIND:org" in result["output"] + assert "FN:Acme Inc." in result["output"] + assert result["filename"] == "acme_inc.vcf" + + def test_rfc9554_properties(self) -> None: + """Test PRONOUNS, SOCIALPROFILE and LANG.""" + contact = { + "last_name": "Doe", + "first_name": "Alex", + "pronouns": "they/them", + "language": "en", + "social_profile": "https://mastodon.social/@alex", + "social_profile_2": "github.com/alex", + } + output = create_vcard(contact, version=VCardVersion.V4_0)["output"] + assert "PRONOUNS:they/them" in output + assert "LANG:en" in output + assert "SOCIALPROFILE:https://mastodon.social/@alex" in output + assert "SOCIALPROFILE:https://github.com/alex" in output + + +class TestVCard3Compliance: + """Test vCard 3.0 value formats (v0.6.0).""" + + def test_data_uri_photo(self) -> None: + """Test that a data: URI photo is split into TYPE and raw base64.""" + contact = { + "last_name": "Doe", + "first_name": "John", + "photo": "data:image/png;base64,iVBORw0KGgo=", + } + output = create_vcard(contact)["output"] + assert "PHOTO;ENCODING=b;TYPE=PNG:iVBORw0KGgo=" in output + + def test_dates(self) -> None: + """Test ISO dates, and Apple's convention for dates without a year.""" + contact = { + "last_name": "Doe", + "first_name": "John", + "birthday": "19901224", + "anniversary": "--06-15", + } + output = create_vcard(contact)["output"] + assert "BDAY:1990-12-24" in output + assert "X-ANNIVERSARY;X-APPLE-OMIT-YEAR=1604:1604-06-15" in output + + def test_invalid_date_skipped(self) -> None: + """Test that unrecognized dates are not written as invalid BDAY values.""" + contact = {"last_name": "Doe", "first_name": "John", "birthday": "sometime"} + assert "BDAY" not in create_vcard(contact)["output"] + + def test_utc_offset_tz(self) -> None: + """Test that UTC offsets are written without VALUE=text.""" + contact = {"last_name": "Doe", "first_name": "John", "tz": "+0530"} + assert "TZ:+05:30" in create_vcard(contact)["output"] + + def test_org_only_shows_as_company(self) -> None: + """Test Apple's company display hint for organization-only rows.""" + output = create_vcard({"org": "Acme"})["output"] + assert "X-ABSHOWAS:COMPANY" in output + assert "FN:Acme" in output + + def test_social_profile_extension(self) -> None: + """Test that social profiles use Apple's X-SOCIALPROFILE in 3.0.""" + contact = {"last_name": "Doe", "first_name": "John", "social_profile": "https://x.com/jd"} + assert "X-SOCIALPROFILE:https://x.com/jd" in create_vcard(contact)["output"] + + +class TestMultiValueAndExtensions: + """Test numbered multi-value fields and X- extensions (v0.6.0).""" + + def test_numbered_values(self) -> None: + """Test that email_2 / phone_cell_2 produce extra properties.""" + contact = { + "last_name": "Doe", + "first_name": "John", + "email": "a@example.com", + "email_2": "b@example.com", + "phone_cell": "+111", + "phone_cell_2": "+222", + } + output = create_vcard(contact)["output"] + assert "EMAIL;TYPE=WORK:a@example.com" in output + assert "EMAIL;TYPE=WORK:b@example.com" in output + assert output.count("TEL;TYPE=CELL:") == 2 + + def test_duplicate_values_written_once(self) -> None: + """Test that the same email in two columns is written once.""" + contact = { + "last_name": "Doe", + "first_name": "John", + "email": "a@example.com", + "email_work": "A@example.com", + } + assert create_vcard(contact)["output"].count("EMAIL") == 1 + + def test_extensions(self) -> None: + """Test that X- keys are written as escaped extension properties.""" + contact = {"last_name": "Doe", "first_name": "John", "X-DEPARTMENT": "R&D, Berlin"} + output = create_vcard(contact)["output"] + assert "X-DEPARTMENT:R&D\\, Berlin" in output + + +class TestUID: + """Test deterministic UIDs (v0.6.0).""" + + def test_uid_is_stable(self, sample_contact: dict[str, str]) -> None: + """Test that the same contact always gets the same UID.""" + first = Contact.from_dict(sample_contact).generate_uid() + second = Contact.from_dict(sample_contact).generate_uid() + assert first == second + + def test_uid_differs_between_contacts(self) -> None: + """Test that different contacts get different UIDs.""" + a = Contact.from_dict({"last_name": "Doe", "first_name": "John"}).generate_uid() + b = Contact.from_dict({"last_name": "Doe", "first_name": "Jane"}).generate_uid() + assert a != b + + def test_uid_field_used(self) -> None: + """Test that a UUID in the uid column is kept as-is.""" + uid = "6ba7b810-9dad-11d1-80b4-00c04fd430c8" + output = create_vcard({"last_name": "Doe", "uid": uid}, version=VCardVersion.V4_0) + assert f"UID:urn:uuid:{uid}" in output["output"] + + def test_non_uuid_uid_hashed(self) -> None: + """Test that non-UUID source IDs become stable UUIDs.""" + a = Contact.from_dict({"uid": "crm-42"}).generate_uid() + b = Contact.from_dict({"uid": "crm-42", "last_name": "Changed"}).generate_uid() + assert a == b + + +class TestVCard21: + """Test vCard 2.1 output (v0.6.0).""" + + def test_basic_structure(self, sample_contact: dict[str, str]) -> None: + """Test 2.1 version line and parameter style.""" + output = create_vcard(sample_contact, version=VCardVersion.V2_1)["output"] + assert "VERSION:2.1" in output + assert "TEL;WORK;VOICE:+49 170 5 25 25 25" in output + assert "EMAIL;INTERNET;WORK:forrestgump@example.com" in output + assert "N:Gump;Forrest;;;" in output + + def test_non_ascii_quoted_printable(self) -> None: + """Test that non-ASCII text is quoted-printable encoded.""" + contact = {"last_name": "Müller", "first_name": "Jürgen"} + output = create_vcard(contact, version=VCardVersion.V2_1)["output"] + assert "N;CHARSET=UTF-8;ENCODING=QUOTED-PRINTABLE:M=C3=BCller;J=C3=BCrgen;;;" in output + + def test_quoted_printable_soft_breaks(self) -> None: + """Test that long values use QP soft line breaks of at most 76 chars.""" + contact = {"last_name": "Doe", "first_name": "John", "note": "é" * 100} + output = create_vcard(contact, version=VCardVersion.V2_1)["output"] + note_lines = output.split("NOTE;", 1)[1].split("\r\nREV")[0].split("\r\n") + assert len(note_lines) > 1 + assert all(len(line) <= 76 for line in note_lines) + assert all(line.endswith("=") for line in note_lines[:-1]) + + def test_multiline_note(self) -> None: + """Test that line breaks are encoded, not written raw.""" + contact = {"last_name": "Doe", "first_name": "John", "note": "a\nb"} + output = create_vcard(contact, version=VCardVersion.V2_1)["output"] + assert "NOTE;CHARSET=UTF-8;ENCODING=QUOTED-PRINTABLE:a=0D=0Ab" in output + + def test_base64_photo(self) -> None: + """Test 2.1 inline photo layout: indented continuation and a blank line.""" + data = "iVBOR" + "A" * 200 + contact = {"last_name": "Doe", "first_name": "John", "photo": data} + output = create_vcard(contact, version=VCardVersion.V2_1)["output"] + assert "PHOTO;ENCODING=BASE64;PNG:iVBOR" in output + block = output.split("PHOTO;", 1)[1].split("\r\n\r\n")[0] + assert "".join(line.strip() for line in block.split(":", 1)[1].split("\r\n")) == data diff --git a/tests/test_csv2vcard.py b/tests/test_csv2vcard.py index 1cd1e7a..d88f6e7 100644 --- a/tests/test_csv2vcard.py +++ b/tests/test_csv2vcard.py @@ -207,3 +207,63 @@ def test_old_positional_args(self, sample_csv: Path, temp_dir: Path) -> None: files = csv2vcard(sample_csv, ",", output_dir=temp_dir) assert len(files) == 2 + + +class TestCSV2VCardV060: + """Test batch-level behavior added in v0.6.0.""" + + def test_same_name_contacts_not_overwritten(self, temp_dir: Path) -> None: + """Test that two contacts with the same name produce two files.""" + csv_path = temp_dir / "dupes.csv" + csv_path.write_text( + "last_name,first_name,phone\nSmith,John,+1\nSmith,John,+2\nsmith,john,+3\n", + encoding="utf-8", + ) + output_dir = temp_dir / "out" + + files = csv2vcard(csv_path, output_dir=output_dir) + + assert [f.name for f in files] == ["smith_john.vcf", "smith_john_2.vcf", "smith_john_3.vcf"] + assert len(list(output_dir.glob("*.vcf"))) == 3 + + def test_uids_stable_across_runs_and_unique(self, temp_dir: Path) -> None: + """Test that re-running yields the same UIDs, and duplicates stay distinct.""" + csv_path = temp_dir / "dupes.csv" + csv_path.write_text( + "last_name,first_name\nSmith,John\nSmith,John\nDoe,Jane\n", encoding="utf-8" + ) + + def uids(run: str) -> list[str]: + files = csv2vcard(csv_path, output_dir=temp_dir / run) + return [ + line + for f in files + for line in f.read_text(encoding="utf-8").splitlines() + if line.startswith("UID:") + ] + + first, second = uids("a"), uids("b") + assert first == second + assert len(set(first)) == 3 + + def test_files_use_crlf(self, sample_csv: Path, temp_dir: Path) -> None: + """Test that written files keep CRLF line endings on every platform.""" + files = csv2vcard(sample_csv, output_dir=temp_dir / "out", single_file=True) + data = files[0].read_bytes() + assert b"\r\n" in data + assert b"\r\r\n" not in data + assert data.count(b"\n") == data.count(b"\r\n") + + def test_keep_unmapped(self, temp_dir: Path) -> None: + """Test that unmapped columns are written as X- properties on request.""" + csv_path = temp_dir / "extra.csv" + csv_path.write_text("last_name,first_name,Department\nDoe,John,Sales\n", encoding="utf-8") + + files = csv2vcard(csv_path, output_dir=temp_dir / "out", keep_unmapped=True) + + assert "X-DEPARTMENT:Sales" in files[0].read_text(encoding="utf-8") + + def test_vcard_21(self, sample_csv: Path, temp_dir: Path) -> None: + """Test generating vCard 2.1 files.""" + files = csv2vcard(sample_csv, output_dir=temp_dir / "out", version=VCardVersion.V2_1) + assert "VERSION:2.1" in files[0].read_text(encoding="utf-8") diff --git a/tests/test_mapping.py b/tests/test_mapping.py index fc0fe06..875af6b 100644 --- a/tests/test_mapping.py +++ b/tests/test_mapping.py @@ -180,3 +180,51 @@ def test_all_values_are_lists(self) -> None: for field, columns in DEFAULT_MAPPING.items(): assert isinstance(columns, list), f"{field} mapping is not a list" assert len(columns) > 0, f"{field} mapping is empty" + + +class TestMappingV060: + """Test column normalization, multi-value columns and passthrough (v0.6.0).""" + + def test_spaces_and_hyphens_match(self) -> None: + """Test that 'First Name' / 'E-Mail Address' match default aliases.""" + row = {"First Name": "John", "Last Name": "Doe", "E-mail Address": "j@example.com"} + result = apply_mapping(row, DEFAULT_MAPPING) + assert result == {"first_name": "John", "last_name": "Doe", "email": "j@example.com"} + + def test_mobile_is_cell_phone(self) -> None: + """Test that 'mobile' maps to phone_cell only (not also to phone).""" + result = apply_mapping({"mobile": "+111"}, DEFAULT_MAPPING) + assert result == {"phone_cell": "+111"} + + def test_location_not_geo(self) -> None: + """Test that a 'location' column (usually a place name) isn't used as GEO.""" + assert "geo" not in apply_mapping({"location": "Berlin"}, DEFAULT_MAPPING) + + def test_numbered_columns(self) -> None: + """Test that numbered columns become extra values in order.""" + row = {"Email 3": "c@x.com", "email": "a@x.com", "Email 2": "b@x.com", "Phone2": "+2"} + result = apply_mapping(row, DEFAULT_MAPPING) + assert result["email"] == "a@x.com" + assert result["email_2"] == "b@x.com" + assert result["email_3"] == "c@x.com" + assert result["phone"] == "+2" + + def test_multi_value_collects_all_aliases(self) -> None: + """Test that all matching alias columns are kept for multi-value fields.""" + row = {"website": "https://a.com", "homepage": "https://b.com"} + result = apply_mapping(row, DEFAULT_MAPPING) + assert result["website"] == "https://a.com" + assert result["website_2"] == "https://b.com" + + def test_keep_unmapped(self) -> None: + """Test that unmapped non-empty columns become X- properties.""" + row = {"first_name": "John", "Cost Center": "42", "Empty": "", "last_name": "Doe"} + result = apply_mapping(row, DEFAULT_MAPPING, keep_unmapped=True) + assert result["X-COST-CENTER"] == "42" + assert "X-EMPTY" not in result + assert "X-FIRST-NAME" not in result + + def test_unmapped_dropped_by_default(self) -> None: + """Test that unmapped columns are dropped unless requested.""" + result = apply_mapping({"first_name": "John", "Cost Center": "42"}, DEFAULT_MAPPING) + assert result == {"first_name": "John"} diff --git a/tests/test_models.py b/tests/test_models.py index 7595b9b..d43fb71 100644 --- a/tests/test_models.py +++ b/tests/test_models.py @@ -147,12 +147,16 @@ def test_all_fields(self) -> None: "last_name", "first_name", "middle_name", "name_prefix", "name_suffix", # Basic info "nickname", "gender", "birthday", "anniversary", + # RFC 9554 / preferences (v0.6.0) + "pronouns", "language", # Contact - single (backwards compatible) "phone", "email", "website", # Contact - multi-type phone (v0.5.0) "phone_cell", "phone_home", "phone_work", "phone_fax", # Contact - multi-type email (v0.5.0) "email_home", "email_work", + # Social profile (v0.6.0) + "social_profile", # Organization "org", "title", "role", # Address (default/work) @@ -164,6 +168,6 @@ def test_all_fields(self) -> None: # New vCard fields (v0.5.0) "categories", "geo", "tz", "key", # Other - "note", + "note", "uid", } assert expected == ALL_FIELDS diff --git a/tests/test_parse_csv.py b/tests/test_parse_csv.py index 2e9331b..20b09fc 100644 --- a/tests/test_parse_csv.py +++ b/tests/test_parse_csv.py @@ -140,3 +140,63 @@ def test_iter_contacts_with_delimiter(self, semicolon_csv: Path) -> None: assert len(contacts) == 1 assert contacts[0].last_name == "Smith" + + +class TestParseCSVRobustness: + """Test encoding and malformed-input handling (v0.6.0).""" + + def test_utf8_bom(self, temp_dir: Path) -> None: + """Test that Excel's UTF-8 BOM doesn't break the first column.""" + csv_path = temp_dir / "bom.csv" + csv_path.write_bytes("last_name,first_name\r\nMüller,Jürgen\r\n".encode()) + + contacts = parse_csv(csv_path) + + assert contacts == [{"last_name": "Müller", "first_name": "Jürgen"}] + + def test_utf8_bom_with_explicit_encoding(self, temp_dir: Path) -> None: + """Test that the BOM is stripped even when encoding='utf-8' is passed.""" + csv_path = temp_dir / "bom.csv" + csv_path.write_bytes("last_name,first_name\nDoe,John\n".encode()) + + contacts = parse_csv(csv_path, encoding="utf-8") + + assert contacts[0]["last_name"] == "Doe" + + def test_column_mismatch_strict_raises(self, malformed_csv: Path) -> None: + """Test that malformed rows fail in strict mode instead of being dropped.""" + with pytest.raises(ParseError, match="columns"): + parse_csv(malformed_csv, strict=True) + + def test_undecodable_bytes_strict_raises(self, temp_dir: Path) -> None: + """Test that invalid bytes fail in strict mode.""" + csv_path = temp_dir / "latin1.csv" + csv_path.write_bytes(b"last_name,first_name\nM\xfcller,J\xfcrgen\n") + + with pytest.raises(ParseError): + parse_csv(csv_path, encoding="utf-8", strict=True) + + def test_undecodable_bytes_warn( + self, temp_dir: Path, caplog: pytest.LogCaptureFixture + ) -> None: + """Test that replaced bytes are reported in non-strict mode.""" + csv_path = temp_dir / "latin1.csv" + csv_path.write_bytes(b"last_name,first_name\nM\xfcller,J\xfcrgen\n") + + with caplog.at_level(logging.WARNING): + contacts = parse_csv(csv_path, encoding="utf-8") + + assert len(contacts) == 1 + assert any("U+FFFD" in record.message for record in caplog.records) + + def test_blank_lines_skipped(self, temp_dir: Path) -> None: + """Test that blank lines are ignored.""" + csv_path = temp_dir / "blank.csv" + csv_path.write_text("last_name,first_name\n\nDoe,John\n\n", encoding="utf-8") + + assert len(parse_csv(csv_path)) == 1 + + def test_iter_contacts_is_lazy(self, sample_csv: Path) -> None: + """Test that iter_contacts streams rows instead of parsing everything first.""" + iterator = iter_contacts(sample_csv) + assert next(iterator).last_name == "Gump" diff --git a/tests/test_utils.py b/tests/test_utils.py index 9d8e7a9..cea0bc2 100644 --- a/tests/test_utils.py +++ b/tests/test_utils.py @@ -2,7 +2,16 @@ from __future__ import annotations -from csv2vcard.utils import strip_accents, strip_accents_from_contact +import pytest + +from csv2vcard.utils import ( + normalize_date, + parse_geo, + parse_utc_offset, + phone_to_tel_uri, + strip_accents, + strip_accents_from_contact, +) class TestStripAccents: @@ -90,3 +99,59 @@ def test_preserves_all_keys(self) -> None: result = strip_accents_from_contact(contact) assert set(result.keys()) == set(contact.keys()) + + +class TestNormalizeDate: + """Test date normalization (v0.6.0).""" + + @pytest.mark.parametrize( + ("value", "expected"), + [ + ("1944-06-06", "19440606"), + ("19440606", "19440606"), + ("1944/06/06", "19440606"), + ("1944-06-06T10:00:00Z", "19440606"), + ("06.06.1944", "19440606"), + ("24/12/1990", "19901224"), + ("12/24/1990", "19901224"), + ("05/05/1990", "19900505"), + ("--12-24", "--1224"), + ("--0229", "--0229"), + ], + ) + def test_valid(self, value: str, expected: str) -> None: + assert normalize_date(value) == expected + + @pytest.mark.parametrize("value", ["06/07/1990", "1990-02-30", "tomorrow", "", "--13-01"]) + def test_invalid_or_ambiguous(self, value: str) -> None: + assert normalize_date(value) is None + + +class TestPhoneToTelUri: + """Test tel: URI conversion (v0.6.0).""" + + def test_global_number(self) -> None: + assert phone_to_tel_uri("+1 (555) 123-4567") == "tel:+1-555-123-4567" + + def test_local_number(self) -> None: + assert phone_to_tel_uri("0170 1234") is None + + def test_extension_not_supported(self) -> None: + assert phone_to_tel_uri("+1 555 1234 ext. 5") is None + + +class TestParseHelpers: + """Test geo and UTC offset parsing (v0.6.0).""" + + def test_geo_formats(self) -> None: + assert parse_geo("37.38,-122.08") == ("37.38", "-122.08") + assert parse_geo("37.38;-122.08") == ("37.38", "-122.08") + assert parse_geo("geo:37.38,-122.08") == ("37.38", "-122.08") + assert parse_geo("Berlin") is None + assert parse_geo("91,0") is None + + def test_utc_offset(self) -> None: + assert parse_utc_offset("-05:00") == ("-", "05", "00") + assert parse_utc_offset("+0530") == ("+", "05", "30") + assert parse_utc_offset("UTC+1") == ("+", "01", "00") + assert parse_utc_offset("America/New_York") is None diff --git a/tests/test_version.py b/tests/test_version.py new file mode 100644 index 0000000..a0768d3 --- /dev/null +++ b/tests/test_version.py @@ -0,0 +1,17 @@ +"""Tests for package version consistency.""" + +from __future__ import annotations + +import re +from pathlib import Path + +from csv2vcard import __version__ + + +def test_version_matches_pyproject() -> None: + """Test that __version__ and pyproject.toml agree (the release workflow checks both).""" + pyproject = (Path(__file__).parent.parent / "pyproject.toml").read_text(encoding="utf-8") + match = re.search(r'^version = "(.+?)"', pyproject, re.MULTILINE) + + assert match is not None + assert match[1] == __version__