|
| 1 | +from decimal import Decimal, InvalidOperation, ROUND_HALF_UP |
| 2 | + |
1 | 3 | import pandas as pd |
2 | 4 |
|
3 | 5 | FIELD_TYPE_MAP = { |
@@ -25,6 +27,45 @@ def infer_field_type(dtype) -> str: |
25 | 27 | return FIELD_TYPE_MAP.get(dtype_str, 'string') |
26 | 28 |
|
27 | 29 |
|
| 30 | +def _round_import_integer(value): |
| 31 | + """Round explicitly selected integer values without a float intermediate.""" |
| 32 | + if pd.isna(value) or (isinstance(value, str) and not value.strip()): |
| 33 | + return pd.NA |
| 34 | + if isinstance(value, bool): |
| 35 | + return int(value) |
| 36 | + try: |
| 37 | + number = Decimal(str(value)) |
| 38 | + except InvalidOperation as exc: |
| 39 | + raise ValueError('Invalid numeric value for integer field') from exc |
| 40 | + if not number.is_finite(): |
| 41 | + raise ValueError('Integer fields require finite numeric values') |
| 42 | + rounded = number.to_integral_value(rounding=ROUND_HALF_UP) |
| 43 | + if rounded < -(2 ** 63) or rounded > 2 ** 63 - 1: |
| 44 | + raise ValueError('Rounded value is outside the signed 64-bit integer range') |
| 45 | + return int(rounded) |
| 46 | + |
| 47 | + |
| 48 | +def read_import_dataframe(save_path: str, sheet_name: str, field_mapping: dict): |
| 49 | + """Read stored cell values; round only columns explicitly selected as integers.""" |
| 50 | + dtype_dict = { |
| 51 | + column: 'object' if field_type == 'int' else USER_TYPE_TO_PANDAS.get(field_type, 'string') |
| 52 | + for column, field_type in field_mapping.items() |
| 53 | + } |
| 54 | + if save_path.endswith('.csv'): |
| 55 | + df = pd.read_csv(save_path, engine='c', dtype=dtype_dict) |
| 56 | + else: |
| 57 | + df = pd.read_excel(save_path, sheet_name=sheet_name, engine='calamine', dtype=dtype_dict) |
| 58 | + for column, field_type in field_mapping.items(): |
| 59 | + if field_type == 'int' and column in df.columns: |
| 60 | + try: |
| 61 | + # Build a nullable integer array directly: Series.map can coerce |
| 62 | + # large integers plus missing values to lossy float64 values. |
| 63 | + df[column] = pd.array([_round_import_integer(value) for value in df[column]], dtype='Int64') |
| 64 | + except ValueError as exc: |
| 65 | + raise ValueError(f"Column '{column}': {exc}") from exc |
| 66 | + return df |
| 67 | + |
| 68 | + |
28 | 69 | def parse_excel_preview(save_path: str, max_rows: int = 10): |
29 | 70 | sheets_data = [] |
30 | 71 | if save_path.endswith(".csv"): |
|
0 commit comments