diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 000000000..4e00cd8bf --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,10 @@ +version: 2 +updates: + - package-ecosystem: "julia" + directory: "/" + schedule: + interval: "weekly" + - package-ecosystem: "github-actions" + directory: "/" + schedule: + interval: "weekly" \ No newline at end of file diff --git a/.github/workflows/jlpkgbutler-butler-dev-workflow.yml b/.github/workflows/jlpkgbutler-butler-dev-workflow.yml deleted file mode 100644 index 4c6813360..000000000 --- a/.github/workflows/jlpkgbutler-butler-dev-workflow.yml +++ /dev/null @@ -1,22 +0,0 @@ -name: Run the Julia Package Butler - -on: - push: - branches: - - main - - master - schedule: - - cron: '*/5 * * * *' - workflow_dispatch: - -jobs: - butler: - name: "Run Package Butler" - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - uses: davidanthoff/julia-pkgbutler@releases/v1 - with: - github-token: ${{ secrets.GITHUB_TOKEN }} - ssh-private-key: ${{ secrets.JLPKGBUTLER_TOKEN }} - channel: dev diff --git a/.github/workflows/jlpkgbutler-ci-master-workflow.yml b/.github/workflows/jlpkgbutler-ci-master-workflow.yml deleted file mode 100644 index 1f8ab264a..000000000 --- a/.github/workflows/jlpkgbutler-ci-master-workflow.yml +++ /dev/null @@ -1,43 +0,0 @@ -name: Run CI on main - -on: - push: - branches: - - main - - master - workflow_dispatch: - -jobs: - test: - runs-on: ${{ matrix.os }} - strategy: - matrix: - julia-version: ['1.6', '1.7', '1.8', '1.9', '1.10', '1.11'] - julia-arch: [x64, x86] - os: [ubuntu-latest, windows-latest, macos-13] - exclude: - - os: macos-13 - julia-arch: x86 - - os: macos-13 - julia-version: "1.4" - - steps: - - uses: actions/checkout@v4 - - uses: julia-actions/setup-julia@v2 - with: - version: ${{ matrix.julia-version }} - arch: ${{ matrix.julia-arch }} - - uses: julia-actions/cache@v2 - - uses: julia-actions/julia-buildpkg@v1 - env: - PYTHON: "" - - uses: julia-actions/julia-runtest@v1 - env: - PYTHON: "" - - uses: julia-actions/julia-processcoverage@v1 - - uses: codecov/codecov-action@v4 - with: - files: ./lcov.info - flags: unittests - token: ${{ secrets.CODECOV_TOKEN }} - \ No newline at end of file diff --git a/.github/workflows/jlpkgbutler-ci-pr-workflow.yml b/.github/workflows/jlpkgbutler-ci-pr-workflow.yml deleted file mode 100644 index 76683999a..000000000 --- a/.github/workflows/jlpkgbutler-ci-pr-workflow.yml +++ /dev/null @@ -1,39 +0,0 @@ -name: Run CI on PR - -on: - pull_request: - types: [opened, synchronize, reopened] - -jobs: - test: - runs-on: ${{ matrix.os }} - strategy: - matrix: - julia-version: ['1.6', '1.7', '1.8', '1.9', '1.10', '1.11'] - julia-arch: [x64, x86] - os: [ubuntu-latest, windows-latest, macos-13] - exclude: - - os: macos-13 - julia-arch: x86 - - os: macos-13 - julia-version: "1.4" - - steps: - - uses: actions/checkout@v4 - - uses: julia-actions/setup-julia@v2 - with: - version: ${{ matrix.julia-version }} - arch: ${{ matrix.julia-arch }} - - uses: julia-actions/cache@v2 - - uses: julia-actions/julia-buildpkg@v1 - env: - PYTHON: "" - - uses: julia-actions/julia-runtest@v1 - env: - PYTHON: "" - - uses: julia-actions/julia-processcoverage@v1 - - uses: codecov/codecov-action@v4 - with: - files: ./lcov.info - flags: unittests - token: ${{ secrets.CODECOV_TOKEN }} diff --git a/.github/workflows/jlpkgbutler-codeformat-pr-workflow.yml b/.github/workflows/jlpkgbutler-codeformat-pr-workflow.yml deleted file mode 100644 index d99a8e060..000000000 --- a/.github/workflows/jlpkgbutler-codeformat-pr-workflow.yml +++ /dev/null @@ -1,23 +0,0 @@ -name: Code Formatting - -on: - push: - branches: - - main - - master - workflow_dispatch: - -jobs: - format: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - uses: julia-actions/julia-codeformat@releases/v1 - - name: Create Pull Request - uses: peter-evans/create-pull-request@v6 - with: - token: ${{ secrets.GITHUB_TOKEN }} - commit-message: Format files using DocumentFormat - title: '[AUTO] Format files using DocumentFormat' - body: '[DocumentFormat.jl](https://github.com/julia-vscode/DocumentFormat.jl) would suggest these formatting changes' - labels: no changelog diff --git a/.github/workflows/jlpkgbutler-compathelper-workflow.yml b/.github/workflows/jlpkgbutler-compathelper-workflow.yml deleted file mode 100644 index b31583129..000000000 --- a/.github/workflows/jlpkgbutler-compathelper-workflow.yml +++ /dev/null @@ -1,20 +0,0 @@ -name: Run CompatHelper - -on: - schedule: - - cron: '00 * * * *' - issues: - types: [opened, reopened] - workflow_dispatch: - -jobs: - CompatHelper: - name: "Run CompatHelper.jl" - runs-on: ubuntu-latest - steps: - - name: Pkg.add("CompatHelper") - run: julia -e 'using Pkg; Pkg.add("CompatHelper")' - - name: CompatHelper.main() - env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} - run: julia -e 'using CompatHelper; CompatHelper.main()' diff --git a/.github/workflows/jlpkgbutler-docdeploy-workflow.yml b/.github/workflows/jlpkgbutler-docdeploy-workflow.yml deleted file mode 100644 index 6656d97ff..000000000 --- a/.github/workflows/jlpkgbutler-docdeploy-workflow.yml +++ /dev/null @@ -1,23 +0,0 @@ -name: Deploy documentation - -on: - push: - branches: - - main - - master - tags: - - v* - workflow_dispatch: - -jobs: - docdeploy: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - uses: julia-actions/julia-buildpkg@v1 - env: - PYTHON: "" - - uses: julia-actions/julia-docdeploy@latest - env: - DOCUMENTER_KEY: ${{ secrets.JLPKGBUTLER_TOKEN }} - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/.github/workflows/jlpkgbutler-tagbot-workflow.yml b/.github/workflows/jlpkgbutler-tagbot-workflow.yml deleted file mode 100644 index d3ca956ea..000000000 --- a/.github/workflows/jlpkgbutler-tagbot-workflow.yml +++ /dev/null @@ -1,17 +0,0 @@ -name: TagBot -on: - issue_comment: - types: - - created - workflow_dispatch: - -jobs: - TagBot: - if: github.event_name == 'workflow_dispatch' || github.actor == 'JuliaTagBot' - runs-on: ubuntu-latest - steps: - - uses: JuliaRegistries/TagBot@v1 - with: - token: ${{ secrets.GITHUB_TOKEN }} - ssh: ${{ secrets.JLPKGBUTLER_TOKEN }} - branches: true diff --git a/.github/workflows/juliaci.yml b/.github/workflows/juliaci.yml new file mode 100644 index 000000000..6aa24b5de --- /dev/null +++ b/.github/workflows/juliaci.yml @@ -0,0 +1,17 @@ +name: Julia CI + +on: + push: {branches: [main,master]} + pull_request: {types: [opened,synchronize,reopened,ready_for_review,converted_to_draft]} + issue_comment: {types: [created]} + workflow_dispatch: {inputs: {feature: {type: choice, description: What to run, options: [DocDeploy,LintAndTest,TagBot]}}} + +jobs: + julia-ci: + uses: julia-testitems/testitem-workflow/.github/workflows/juliaci.yml@v2 + with: + include-all-compatible-minor-versions: true + include-rc-versions: true + permissions: write-all + secrets: + codecov_token: ${{ secrets.CODECOV_TOKEN }} diff --git a/.gitignore b/.gitignore index 28fa6a049..080f53dac 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,8 @@ *.jl.cov *.jl.mem .vscode +*.jl.*.cov +docs/build/ +docs/Manifest.toml +test-output*.* +Manifest.toml diff --git a/.jlpkgbutler.toml b/.jlpkgbutler.toml deleted file mode 100644 index b72304ff0..000000000 --- a/.jlpkgbutler.toml +++ /dev/null @@ -1 +0,0 @@ -template = "bach" diff --git a/LICENSE.md b/LICENSE.md index e21939a8e..9fecfc8ba 100644 --- a/LICENSE.md +++ b/LICENSE.md @@ -1,6 +1,6 @@ The ExcelReaders.jl package is licensed under the MIT "Expat" License: -> Copyright (c) 2016-2019: David Anthoff. +> Copyright (c) 2016-2026: David Anthoff. > > Permission is hereby granted, free of charge, to any person obtaining > a copy of this software and associated documentation files (the diff --git a/Project.toml b/Project.toml index 7f6b7327d..667eaf495 100644 --- a/Project.toml +++ b/Project.toml @@ -1,22 +1,23 @@ name = "ExcelReaders" uuid = "c04bee98-12a5-510c-87df-2a230cb6e075" -version = "0.12.1-DEV" +version = "1.0.0-DEV" [deps] -Dates = "ade2ca70-3891-5945-98fb-dc099432e06a" -Conda = "8f4d0f93-b110-5947-807f-2305c1781a2d" DataValues = "e7dc6d0d-1eca-5fa6-8ad6-5aecde8b7ea5" -PyCall = "438e738f-606a-5dbb-bf0a-cddfbfd45ab0" - -[extras] -TestItemRunner = "f8b46487-2199-4994-9208-9a1283c18c0a" -Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" +Dates = "ade2ca70-3891-5945-98fb-dc099432e06a" +LibXLS = "221edcab-ec84-5148-84a2-7385855c517f" +XLSX = "fdbf4ff8-1666-58a4-91e7-1b58723a45e0" [compat] +DataValues = "0.4.4, 0.5, 1" +Dates = "1" +LibXLS = "0.1, 0.2, 1" +XLSX = "0.12" julia = "1.6" -Conda = "1.9" -DataValues = "0.4.4" -PyCall = "1.96" + +[extras] +Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" +TestItemRunner = "f8b46487-2199-4994-9208-9a1283c18c0a" [targets] test = ["Test", "TestItemRunner"] diff --git a/README.md b/README.md index 61f8a89b5..3d2c04007 100644 --- a/README.md +++ b/README.md @@ -1,28 +1,21 @@ # ExcelReaders -[![Build Status](https://travis-ci.org/queryverse/ExcelReaders.jl.svg?branch=master)](https://travis-ci.org/queryverse/ExcelReaders.jl) -[![Build status](https://ci.appveyor.com/api/projects/status/v7b60gfrg65qkqt5/branch/master?svg=true)](https://ci.appveyor.com/project/queryverse/excelreaders-jl/branch/master) -[![Coverage Status](https://coveralls.io/repos/queryverse/ExcelReaders.jl/badge.svg)](https://coveralls.io/r/queryverse/ExcelReaders.jl) -[![codecov](https://codecov.io/gh/queryverse/ExcelReaders.jl/branch/master/graph/badge.svg)](https://codecov.io/gh/queryverse/ExcelReaders.jl) +[![Build Status](https://github.com/queryverse/ExcelReaders.jl/actions/workflows/juliaci.yml/badge.svg?branch=main)](https://github.com/queryverse/ExcelReaders.jl/actions/workflows/juliaci.yml) ExcelReaders is a package that provides functionality to read Excel files. -**WARNING**: Version v0.12 removed support for modern Excel files. This package is now _only_ supporting legacy xls files. The reason for this is that the underlying Python package made that move a couple of years ago as well. - -The [XLSX.jl](https://github.com/felipenoris/XLSX.jl) provides excellent support for modern Excel files. +Both legacy xls files (Excel 97-2003) and modern xlsx files are supported. +The file format is detected from the content of the file, not its extension. +Under the hood legacy files are read via +[LibXLS.jl](https://github.com/queryverse/LibXLS.jl) (a wrapper of the +[libxls](https://github.com/libxls/libxls) C library) and modern files via +[XLSX.jl](https://github.com/JuliaData/XLSX.jl), so the package has no Python +or Java dependency. ## Installation Use ``Pkg.add("ExcelReaders")`` in Julia to install ExcelReaders and its dependencies. -The package uses the Python xlrd library. If either Python or the xlrd package are not installed on your Mac or Windows system, the package will use the [Conda.jl](https://github.com/Luthaf/Conda.jl) package to install all necessary dependencies automatically. If you are on another system you can either install Python and xlrd yourself or instruct PyCall to use Conda.jl to manage its own python install (`ENV["PYTHON"]=""; Pkg.build("PyCall")` and restart Julia). - -## Alternatives - -The [XLSX.jl](https://github.com/felipenoris/XLSX.jl) provides excellent support for modern Excel files. - -The [Taro](https://github.com/aviks/Taro.jl) package also provides Excel file reading functionality. The main difference between the two packages (in terms of Excel functionality) is that ExcelReaders uses the Python package [xlrd](https://github.com/python-excel/xlrd) for its processing, whereas Taro uses the Java packages Apache [Tika](http://tika.apache.org/) and Apache [POI](http://poi.apache.org/). - ## Basic usage The most basic usage is this: @@ -33,9 +26,11 @@ using ExcelReaders data = readxl("Filename.xls", "Sheet1!A1:C4") ```` -This will return an array with all the data in the cell range A1 to C4 on Sheet1 in the Excel file Filename.xls. +This will return an array with all the data in the cell range A1 to C4 on +Sheet1 in the Excel file Filename.xls. -If you expect to read multiple ranges from the same Excel file you can get much better performance by opening the Excel file only once: +If you expect to read multiple ranges from the same Excel file you can get much +better performance by opening the Excel file only once: ````julia using ExcelReaders @@ -64,3 +59,9 @@ This will read all content on Sheet1 in the file Filename.xls. Eventual blank ro - ``ncols`` accepts either ``:all`` (default) or a postiive integer. With ``:all``, all columns (except skipped ones) are read. An integer specifies the exact number of columns to be read. ``readxlsheet`` also accepts an ExcelFile (as obtained from ``openxl``) as its first argument. + +## Alternatives + +[XLSX.jl](https://github.com/JuliaData/XLSX.jl) provides excellent, more +fully-featured support for modern Excel files (including write support), and +is used by this package as its xlsx backend. diff --git a/docs/Project.toml b/docs/Project.toml index f2a273e56..1814eb330 100644 --- a/docs/Project.toml +++ b/docs/Project.toml @@ -2,4 +2,4 @@ Documenter = "e30172f5-a6a5-5a46-863b-614d45cd2de4" [compat] -Documenter = "~0.24" +Documenter = "1" diff --git a/docs/make.jl b/docs/make.jl index 29bef7baa..a75e3ba4f 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -2,7 +2,8 @@ using Documenter, ExcelReaders makedocs(modules = [ExcelReaders], sitename = "ExcelReaders.jl", - analytics = "UA-132838790-1", + format = Documenter.HTML(analytics = "UA-132838790-1"), + warnonly = [:missing_docs], pages = [ "Introduction" => "index.md" ]) diff --git a/src/ExcelReaders.jl b/src/ExcelReaders.jl index 4d9f7a54d..22845c77a 100644 --- a/src/ExcelReaders.jl +++ b/src/ExcelReaders.jl @@ -1,17 +1,13 @@ module ExcelReaders -using PyCall, DataValues, Dates +using DataValues, Dates -export openxl, readxl, readxlsheet, ExcelErrorCell, ExcelFile, readxlnames, readxlrange +import LibXLS, XLSX -const xlrd = PyNULL() +export openxl, readxl, readxlsheet, ExcelErrorCell, ExcelFile, readxlnames, readxlrange include("package_documentation.jl") -function __init__() - copy!(xlrd, pyimport_conda("xlrd", "xlrd")) -end - """ ExcelFile @@ -20,8 +16,8 @@ A handle to an open Excel file. You can create an instance of an ``ExcelFile`` by calling ``openxl``. """ mutable struct ExcelFile - workbook::PyObject - filename::AbstractString + workbook::Union{LibXLS.Workbook,XLSX.XLSXFile} + filename::String end """ @@ -30,20 +26,36 @@ end An Excel cell that has an Excel error. You cannot create ``ExcelErrorCell`` objects, they are returned if a cell in an -Excel file has an Excel error. +Excel file has an Excel error. ``errorcode`` is the BIFF error code. """ mutable struct ExcelErrorCell errorcode::Int end +const ERROR_CODE_TEXTS = Dict{Int,String}( + 0 => "#NULL!", + 7 => "#DIV/0!", + 15 => "#VALUE!", + 23 => "#REF!", + 29 => "#NAME?", + 36 => "#NUM!", + 42 => "#N/A", + 43 => "#GETTING_DATA", +) + +const ERROR_TEXT_CODES = Dict{String,Int}(v => k for (k, v) in ERROR_CODE_TEXTS) + function Base.show(io::IO, o::ExcelFile) print(io, "ExcelFile <$(o.filename)>") end function Base.show(io::IO, o::ExcelErrorCell) - print(io, xlrd.error_text_from_code[o.errorcode]) + print(io, get(ERROR_CODE_TEXTS, o.errorcode, "#ERROR($(o.errorcode))")) end +const OLE2_FILE_HEADER = [0xd0, 0xcf, 0x11, 0xe0] # legacy xls (BIFF) +const ZIP_FILE_HEADER = [0x50, 0x4b, 0x03, 0x04] # xlsx (OOXML) + """ openxl(filename) @@ -54,6 +66,9 @@ The returned ``ExcelFile`` handle can later be passed as the first argument to of those functions more than once, performance will be better if you open the file only once with ``openxl``. +Both legacy xls files and modern xlsx files are supported; the file format is +detected from the content of the file, not its extension. + # Example ````julia f = openxl("filename.xls") @@ -61,18 +76,85 @@ data = readxl(f, "Sheet1!A1:C4") ```` """ function openxl(filename::AbstractString) - wb = xlrd.open_workbook(filename) - return ExcelFile(wb, basename(filename)) + isfile(filename) || error("File $filename not found.") + + header = open(io -> Base.read(io, 4), filename) + + if header == OLE2_FILE_HEADER + return ExcelFile(LibXLS.openxls(filename), basename(filename)) + elseif header == ZIP_FILE_HEADER + return ExcelFile(XLSX.readxlsx(filename), basename(filename)) + else + error("$filename is not a valid Excel file.") + end +end + +function Base.close(file::ExcelFile) + file.workbook isa LibXLS.Workbook && close(file.workbook) + return nothing +end + +sheetnames(file::ExcelFile) = sheetnames(file.workbook) +sheetnames(wb::LibXLS.Workbook) = LibXLS.sheetnames(wb) +sheetnames(wb::XLSX.XLSXFile) = XLSX.sheetnames(wb) + +sheet_handle(file::ExcelFile, sheetname::AbstractString) = sheet_handle(file.workbook, sheetname) + +function sheet_handle(wb::LibXLS.Workbook, sheetname::AbstractString) + LibXLS.is_valid_sheetname(wb, sheetname) || error("Sheet $sheetname not found.") + return LibXLS.getworksheet(wb, sheetname) +end + +function sheet_handle(wb::XLSX.XLSXFile, sheetname::AbstractString) + XLSX.hassheet(wb, sheetname) || error("Sheet $sheetname not found.") + return wb[sheetname] +end + +sheet_dims(ws::LibXLS.Worksheet) = size(ws) + +function sheet_dims(ws::XLSX.Worksheet) + dim = XLSX.get_dimension(ws) + dim === nothing && return (0, 0) + return (dim.stop.row_number, dim.stop.column_number) +end + +# Map a backend cell value into the ExcelReaders vocabulary: NA for blank +# cells (and cells holding an empty string, as in previous versions), Float64 +# for all numbers, String, Bool, DateTime, Time and ExcelErrorCell. +normalize_value(::Missing) = NA +normalize_value(v::AbstractString) = isempty(v) ? NA : String(v) +normalize_value(v::Bool) = v +normalize_value(v::Real) = Float64(v) +normalize_value(v::Date) = DateTime(v) +normalize_value(v::LibXLS.CellError) = ExcelErrorCell(Int(v.code)) +normalize_value(v) = v + +function cell_value(ws::LibXLS.Worksheet, row::Integer, col::Integer) + nrows, ncols = size(ws) + (1 <= row <= nrows && 1 <= col <= ncols) || return NA + return normalize_value(ws[row, col]) end +function cell_value(ws::XLSX.Worksheet, row::Integer, col::Integer) + (1 <= row && 1 <= col) || return NA + cell = XLSX.getcell(ws, XLSX.CellRef(row, col)) + cell isa XLSX.EmptyCell && return NA + if XLSX.iserror(cell) + errortext = XLSX.get_error_string(XLSX.getval(cell)) + return ExcelErrorCell(get(ERROR_TEXT_CODES, errortext, -1)) + end + return normalize_value(XLSX.getdata(ws, cell)) +end + +isblank(v) = v isa DataValue && DataValues.isna(v) + function readxlsheet(filename::AbstractString, sheetindex::Int; args...) file = openxl(filename) return readxlsheet(file, sheetindex; args...) end function readxlsheet(file::ExcelFile, sheetindex::Int; args...) - sheetnames = file.workbook.sheet_names() - return readxlsheet(file, sheetnames[sheetindex]; args...) + return readxlsheet(file, sheetnames(file)[sheetindex]; args...) end function readxlsheet(filename::AbstractString, sheetname::AbstractString; args...) @@ -81,16 +163,14 @@ function readxlsheet(filename::AbstractString, sheetname::AbstractString; args.. end function readxlsheet(file::ExcelFile, sheetname::AbstractString; args...) - sheet = file.workbook.sheet_by_name(sheetname) - startrow, startcol, endrow, endcol = convert_args_to_row_col(sheet; args...) - - data = readxl_internal(file, sheetname, startrow, startcol, endrow, endcol) + ws = sheet_handle(file, sheetname) + startrow, startcol, endrow, endcol = convert_args_to_row_col(ws; args...) - return data + return readxl_internal(ws, startrow, startcol, endrow, endcol) end # Function converts "relative" range like skip rows/cols and size of range to "absolute" from row/col to row/col -function convert_args_to_row_col(sheet;skipstartrows::Union{Int,Symbol} = :blanks, skipstartcols::Union{Int,Symbol} = :blanks, nrows::Union{Int,Symbol} = :all, ncols::Union{Int,Symbol} = :all) +function convert_args_to_row_col(ws; skipstartrows::Union{Int,Symbol}=:blanks, skipstartcols::Union{Int,Symbol}=:blanks, nrows::Union{Int,Symbol}=:all, ncols::Union{Int,Symbol}=:all) isa(skipstartrows, Symbol) && skipstartrows != :blanks && error("Only :blank or an integer is a valid argument for skipstartrows") isa(skipstartrows, Int) && skipstartrows < 0 && error("Can't skip a negative number of rows") isa(skipstartcols, Symbol) && skipstartcols != :blanks && error("Only :blank or an integer is a valid argument for skipstartcols") @@ -99,16 +179,13 @@ function convert_args_to_row_col(sheet;skipstartrows::Union{Int,Symbol} = :blank isa(nrows, Int) && nrows < 0 && error("nrows should be :all or positive") isa(ncols, Symbol) && ncols != :all && error("Only :all or an integer is a valid argument for ncols") isa(ncols, Int) && ncols < 0 && error("ncols should be :all or positive") - sheet_rows = sheet.nrows - sheet_cols = sheet.ncols - cell_value = sheet.cell_value + sheet_rows, sheet_cols = sheet_dims(ws) if skipstartrows == :blanks startrow = -1 for cur_row in 1:sheet_rows, cur_col in 1:sheet_cols - cellval = cell_value(cur_row - 1, cur_col - 1) - if cellval != "" + if !isblank(cell_value(ws, cur_row, cur_col)) startrow = cur_row break end @@ -125,8 +202,7 @@ function convert_args_to_row_col(sheet;skipstartrows::Union{Int,Symbol} = :blank if skipstartcols == :blanks startcol = -1 for cur_col in 1:sheet_cols, cur_row in 1:sheet_rows - cellval = cell_value(cur_row - 1, cur_col - 1) - if cellval != "" + if !isblank(cell_value(ws, cur_row, cur_col)) startcol = cur_col break end @@ -167,18 +243,18 @@ end function convert_ref_to_sheet_row_col(range::AbstractString) r = r"('?[^']+'?|[^!]+)!([A-Za-z]*)(\d*)(:([A-Za-z]*)(\d*))?" m = match(r, range) - m == nothing && error("Invalid Excel range specified.") + m === nothing && error("Invalid Excel range specified.") sheetname = String(m.captures[1]) startrow = parse(Int, m.captures[3]) startcol = colnum(m.captures[2]) - if m.captures[4] == nothing + if m.captures[4] === nothing endrow = startrow endcol = startcol else endrow = parse(Int, m.captures[6]) endcol = colnum(m.captures[5]) end - if (startrow > endrow ) || (startcol > endcol) + if (startrow > endrow) || (startcol > endcol) error("Please provide rectangular region from top left to bottom right corner") end return sheetname, startrow, startcol, endrow, endcol @@ -192,49 +268,23 @@ end function readxl(file::ExcelFile, range::AbstractString) sheetname, startrow, startcol, endrow, endcol = convert_ref_to_sheet_row_col(range) - readxl_internal(file, sheetname, startrow, startcol, endrow, endcol) -end - -function get_cell_value(ws, row, col, wb) - cellval = ws.cell_value(row - 1, col - 1) - if cellval == "" - return NA - else - celltype = ws.cell_type(row - 1, col - 1) - if celltype == xlrd.XL_CELL_TEXT - return convert(String, cellval) - elseif celltype == xlrd.XL_CELL_NUMBER - return convert(Float64, cellval) - elseif celltype == xlrd.XL_CELL_DATE - date_year, date_month, date_day, date_hour, date_minute, date_sec = xlrd.xldate_as_tuple(cellval, wb.datemode) - if date_month == 0 - return Time(date_hour, date_minute, date_sec) - else - return DateTime(date_year, date_month, date_day, date_hour, date_minute, date_sec) - end - elseif celltype == xlrd.XL_CELL_BOOLEAN - return convert(Bool, cellval) - elseif celltype == xlrd.XL_CELL_ERROR - return ExcelErrorCell(cellval) - else - error("Unknown cell type") - end - end + ws = sheet_handle(file, sheetname) + readxl_internal(ws, startrow, startcol, endrow, endcol) end function readxl_internal(file::ExcelFile, sheetname::AbstractString, startrow::Integer, startcol::Integer, endrow::Integer, endcol::Integer) - wb = file.workbook - ws = wb.sheet_by_name(sheetname) + return readxl_internal(sheet_handle(file, sheetname), startrow, startcol, endrow, endcol) +end +function readxl_internal(ws, startrow::Integer, startcol::Integer, endrow::Integer, endcol::Integer) if startrow == endrow && startcol == endcol - return get_cell_value(ws, startrow, startcol, wb) + return cell_value(ws, startrow, startcol) else - data = Array{Any}(undef, endrow - startrow + 1, endcol - startcol + 1) for row in startrow:endrow for col in startcol:endcol - data[row - startrow + 1, col - startcol + 1] = get_cell_value(ws, row, col, wb) + data[row - startrow + 1, col - startcol + 1] = cell_value(ws, row, col) end end @@ -243,20 +293,14 @@ function readxl_internal(file::ExcelFile, sheetname::AbstractString, startrow::I end function readxlnames(f::ExcelFile) - return [lowercase(i.name) for i in f.workbook.name_obj_list if i.hidden == 0] + f.workbook isa XLSX.XLSXFile || error("Defined names are not supported for legacy xls files.") + return sort!(collect(keys(f.workbook.workbook.workbook_names))) end function readxlrange(f::ExcelFile, range::AbstractString) - name = f.workbook.name_map[lowercase(range)] - if length(name) != 1 - error("More than one reference per name, this case is not yet handled by ExcelReaders.") - end - - formula_text = name[1].formula_text - formula_text = replace(formula_text, "\$" => "") - formula_text = replace(formula_text, "'" => "") - - return readxl(f, formula_text) + f.workbook isa XLSX.XLSXFile || error("Defined names are not supported for legacy xls files.") + data = XLSX.getdata(f.workbook, range) + return data isa AbstractArray ? map(normalize_value, data) : normalize_value(data) end end # module diff --git a/test/TestData.xlsx b/test/TestData.xlsx new file mode 100644 index 000000000..817ae1391 Binary files /dev/null and b/test/TestData.xlsx differ diff --git a/test/test_excelreaders.jl b/test/test_excelreaders.jl index f28ed9739..115d38851 100644 --- a/test/test_excelreaders.jl +++ b/test/test_excelreaders.jl @@ -1,16 +1,13 @@ @testitem "ExcelReaders" begin - using Dates, PyCall, DataValues + using Dates, DataValues -# TODO Throw julia specific exceptions for these errors - @test_throws PyCall.PyError openxl("FileThatDoesNotExist.xls") - @test_throws PyCall.PyError openxl("runtests.jl") + @test_throws ErrorException openxl("FileThatDoesNotExist.xls") + @test_throws ErrorException openxl(normpath(@__DIR__, "runtests.jl")) filename = normpath(@__DIR__, "TestData.xls") file = openxl(filename) @test file.filename == "TestData.xls" - buffer = IOBuffer() - @test sprint(show, file) == "ExcelFile " for (k, v) in Dict(0 => "#NULL!", 7 => "#DIV/0!", 23 => "#REF!", 42 => "#N/A", 29 => "#NAME?", 36 => "#NUM!", 15 => "#VALUE!") @@ -112,3 +109,85 @@ end end + +@testitem "ExcelReaders xlsx backend" begin + using Dates, DataValues + + filename = normpath(@__DIR__, "TestData.xlsx") + + file = openxl(filename) + @test file.filename == "TestData.xlsx" + @test sprint(show, file) == "ExcelFile " + + # TestData.xlsx holds the same content as TestData.xls, so the very same + # assertions must hold for both backends. + for f in [file, filename] + @test_throws ErrorException readxl(f, "Sheet1!C4:G3") + @test_throws ErrorException readxl(f, "Sheet1!G2:B5") + @test_throws ErrorException readxl(f, "Sheet1!G5:B2") + + data = readxl(f, "Sheet1!C3:N7") + @test size(data) == (5, 12) + @test data[4,1] == 2.0 + @test data[2,2] == "A" + @test data[2,3] == true + @test DataValues.isna(data[4,5]) + @test data[2,9] == Date(2015, 3, 3) + @test data[2,9] isa DateTime # backend-independent type vocabulary + @test data[3,9] == DateTime(2015, 2, 4, 10, 14) + @test data[4,9] == DateTime(1988, 4, 9, 0, 0) + @test data[5,9] == Time(15, 2, 0) + @test data[3,10] == DateTime(1950, 8, 9, 18, 40) + @test DataValues.isna(data[5,10]) + @test isa(data[2,11], ExcelErrorCell) + @test isa(data[3,11], ExcelErrorCell) + @test isa(data[4,12], ExcelErrorCell) + @test DataValues.isna(data[5,12]) + + # single cell read + @test readxl(f, "Sheet1!C4") == 1.0 + + @test_throws ErrorException readxlsheet(f, "Empty Sheet") + + data = readxlsheet(f, "Second Sheet") + @test size(data) == (6, 6) + @test data[2,1] == 1. + @test data[5,2] == "CCC" + @test data[3,3] == false + @test data[6,6] == Time(15, 2, 00) + @test DataValues.isna(data[4,3]) + @test DataValues.isna(data[4,6]) + end +end + +@testitem "Cross-backend parity" begin + using Dates, DataValues + + # The two test files hold the same content in the two file formats, so + # every cell in the shared range must come back identical from both + # backends. + xls = openxl(normpath(@__DIR__, "TestData.xls")) + xlsx = openxl(normpath(@__DIR__, "TestData.xlsx")) + + function compare_cells(data_xls, data_xlsx) + @test size(data_xls) == size(data_xlsx) + for i in eachindex(data_xls) + a, b = data_xls[i], data_xlsx[i] + if a isa ExcelErrorCell + @test b isa ExcelErrorCell + @test a.errorcode == b.errorcode + elseif a isa DataValue + @test b isa DataValue && DataValues.isna(b) + else + @test typeof(a) == typeof(b) + @test a == b + end + end + end + + compare_cells(readxl(xls, "Sheet1!C3:N7"), readxl(xlsx, "Sheet1!C3:N7")) + + for sheet in ["Second Sheet", 2] + compare_cells(readxlsheet(xls, sheet), readxlsheet(xlsx, sheet)) + end +end