Skip to content

Commit 2e7bc67

Browse files
fix: round integer values during Excel and CSV import (#1365)
1 parent 35e7e00 commit 2e7bc67

2 files changed

Lines changed: 43 additions & 9 deletions

File tree

‎backend/apps/datasource/api/datasource.py‎

Lines changed: 2 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -31,7 +31,7 @@
3131
from ..crud.table import get_tables_by_ds_id
3232
from ..models.datasource import CoreDatasource, CreateDatasource, TableObj, CoreTable, CoreField, FieldObj, \
3333
TableSchemaResponse, ColumnSchemaResponse, PreviewResponse, ImportRequest
34-
from ..utils.excel import parse_excel_preview, USER_TYPE_TO_PANDAS
34+
from ..utils.excel import parse_excel_preview, read_import_dataframe
3535

3636
router = APIRouter(tags=["Datasource"], prefix="/datasource")
3737
path = settings.EXCEL_PATH
@@ -576,17 +576,10 @@ def inner():
576576
fields = sheet_info.fields
577577

578578
field_mapping = {f.fieldName: f.fieldType for f in fields}
579-
dtype_dict = {
580-
col: USER_TYPE_TO_PANDAS.get(field_mapping.get(col, 'string'), 'string')
581-
for col in field_mapping.keys()
582-
}
583-
584579
try:
580+
df = read_import_dataframe(save_path, sheet_name, field_mapping)
585581
if save_path.endswith(".csv"):
586-
df = pd.read_csv(save_path, engine='c', dtype=dtype_dict)
587582
sheet_name = "Sheet1"
588-
else:
589-
df = pd.read_excel(save_path, sheet_name=sheet_name, engine='calamine', dtype=dtype_dict)
590583
except Exception as e:
591584
raise HTTPException(500, f"{trans('i18n_ds_upload_error')}: {str(e)}")
592585

‎backend/apps/datasource/utils/excel.py‎

Lines changed: 41 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,5 @@
1+
from decimal import Decimal, InvalidOperation, ROUND_HALF_UP
2+
13
import pandas as pd
24

35
FIELD_TYPE_MAP = {
@@ -25,6 +27,45 @@ def infer_field_type(dtype) -> str:
2527
return FIELD_TYPE_MAP.get(dtype_str, 'string')
2628

2729

30+
def _round_import_integer(value):
31+
"""Round explicitly selected integer values without a float intermediate."""
32+
if pd.isna(value) or (isinstance(value, str) and not value.strip()):
33+
return pd.NA
34+
if isinstance(value, bool):
35+
return int(value)
36+
try:
37+
number = Decimal(str(value))
38+
except InvalidOperation as exc:
39+
raise ValueError('Invalid numeric value for integer field') from exc
40+
if not number.is_finite():
41+
raise ValueError('Integer fields require finite numeric values')
42+
rounded = number.to_integral_value(rounding=ROUND_HALF_UP)
43+
if rounded < -(2 ** 63) or rounded > 2 ** 63 - 1:
44+
raise ValueError('Rounded value is outside the signed 64-bit integer range')
45+
return int(rounded)
46+
47+
48+
def read_import_dataframe(save_path: str, sheet_name: str, field_mapping: dict):
49+
"""Read stored cell values; round only columns explicitly selected as integers."""
50+
dtype_dict = {
51+
column: 'object' if field_type == 'int' else USER_TYPE_TO_PANDAS.get(field_type, 'string')
52+
for column, field_type in field_mapping.items()
53+
}
54+
if save_path.endswith('.csv'):
55+
df = pd.read_csv(save_path, engine='c', dtype=dtype_dict)
56+
else:
57+
df = pd.read_excel(save_path, sheet_name=sheet_name, engine='calamine', dtype=dtype_dict)
58+
for column, field_type in field_mapping.items():
59+
if field_type == 'int' and column in df.columns:
60+
try:
61+
# Build a nullable integer array directly: Series.map can coerce
62+
# large integers plus missing values to lossy float64 values.
63+
df[column] = pd.array([_round_import_integer(value) for value in df[column]], dtype='Int64')
64+
except ValueError as exc:
65+
raise ValueError(f"Column '{column}': {exc}") from exc
66+
return df
67+
68+
2869
def parse_excel_preview(save_path: str, max_rows: int = 10):
2970
sheets_data = []
3071
if save_path.endswith(".csv"):

0 commit comments

Comments
 (0)