From 2e7bc672fe69f1118e04e64e29c07b7f32eef47e Mon Sep 17 00:00:00 2001 From: xuwei-fit2cloud Date: Thu, 17 Sep 2026 15:45:03 +0800 Subject: [PATCH] fix: round integer values during Excel and CSV import (#1365) --- backend/apps/datasource/api/datasource.py | 11 ++---- backend/apps/datasource/utils/excel.py | 41 +++++++++++++++++++++++ 2 files changed, 43 insertions(+), 9 deletions(-) diff --git a/backend/apps/datasource/api/datasource.py b/backend/apps/datasource/api/datasource.py index f1ca6a295..6f91cc22a 100644 --- a/backend/apps/datasource/api/datasource.py +++ b/backend/apps/datasource/api/datasource.py @@ -31,7 +31,7 @@ from ..crud.table import get_tables_by_ds_id from ..models.datasource import CoreDatasource, CreateDatasource, TableObj, CoreTable, CoreField, FieldObj, \ TableSchemaResponse, ColumnSchemaResponse, PreviewResponse, ImportRequest -from ..utils.excel import parse_excel_preview, USER_TYPE_TO_PANDAS +from ..utils.excel import parse_excel_preview, read_import_dataframe router = APIRouter(tags=["Datasource"], prefix="/datasource") path = settings.EXCEL_PATH @@ -576,17 +576,10 @@ def inner(): fields = sheet_info.fields field_mapping = {f.fieldName: f.fieldType for f in fields} - dtype_dict = { - col: USER_TYPE_TO_PANDAS.get(field_mapping.get(col, 'string'), 'string') - for col in field_mapping.keys() - } - try: + df = read_import_dataframe(save_path, sheet_name, field_mapping) if save_path.endswith(".csv"): - df = pd.read_csv(save_path, engine='c', dtype=dtype_dict) sheet_name = "Sheet1" - else: - df = pd.read_excel(save_path, sheet_name=sheet_name, engine='calamine', dtype=dtype_dict) except Exception as e: raise HTTPException(500, f"{trans('i18n_ds_upload_error')}: {str(e)}") diff --git a/backend/apps/datasource/utils/excel.py b/backend/apps/datasource/utils/excel.py index dd4044358..2d982097c 100644 --- a/backend/apps/datasource/utils/excel.py +++ b/backend/apps/datasource/utils/excel.py @@ -1,3 +1,5 @@ +from decimal import Decimal, InvalidOperation, ROUND_HALF_UP + import pandas as pd FIELD_TYPE_MAP = { @@ -25,6 +27,45 @@ def infer_field_type(dtype) -> str: return FIELD_TYPE_MAP.get(dtype_str, 'string') +def _round_import_integer(value): + """Round explicitly selected integer values without a float intermediate.""" + if pd.isna(value) or (isinstance(value, str) and not value.strip()): + return pd.NA + if isinstance(value, bool): + return int(value) + try: + number = Decimal(str(value)) + except InvalidOperation as exc: + raise ValueError('Invalid numeric value for integer field') from exc + if not number.is_finite(): + raise ValueError('Integer fields require finite numeric values') + rounded = number.to_integral_value(rounding=ROUND_HALF_UP) + if rounded < -(2 ** 63) or rounded > 2 ** 63 - 1: + raise ValueError('Rounded value is outside the signed 64-bit integer range') + return int(rounded) + + +def read_import_dataframe(save_path: str, sheet_name: str, field_mapping: dict): + """Read stored cell values; round only columns explicitly selected as integers.""" + dtype_dict = { + column: 'object' if field_type == 'int' else USER_TYPE_TO_PANDAS.get(field_type, 'string') + for column, field_type in field_mapping.items() + } + if save_path.endswith('.csv'): + df = pd.read_csv(save_path, engine='c', dtype=dtype_dict) + else: + df = pd.read_excel(save_path, sheet_name=sheet_name, engine='calamine', dtype=dtype_dict) + for column, field_type in field_mapping.items(): + if field_type == 'int' and column in df.columns: + try: + # Build a nullable integer array directly: Series.map can coerce + # large integers plus missing values to lossy float64 values. + df[column] = pd.array([_round_import_integer(value) for value in df[column]], dtype='Int64') + except ValueError as exc: + raise ValueError(f"Column '{column}': {exc}") from exc + return df + + def parse_excel_preview(save_path: str, max_rows: int = 10): sheets_data = [] if save_path.endswith(".csv"):