Skip to content

Commit 178f81d

Browse files
authored
Merge pull request #209 from lukepayyapilli/fix/pandas-3.0-string-dtype-compatibility
fix: pandas 3.0 compatibility for strict string dtype enforcement
2 parents a0c5784 + 865e926 commit 178f81d

3 files changed

Lines changed: 58 additions & 4 deletions

File tree

‎tableone/preprocessors.py‎

Lines changed: 4 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -118,6 +118,8 @@ def handle_categorical_nulls(df: pd.DataFrame, categorical: list, null_value: st
118118
df = df.copy()
119119
for column in categorical:
120120
if df[column].isnull().any():
121-
df[column] = df[column].astype(object).astype(str)
122-
df[column] = df[column].replace('nan', null_value)
121+
# Convert to object dtype to allow mixed types, then to string.
122+
# Use fillna() instead of replace() for pandas 3.0 compatibility,
123+
# as NA values are not the string 'nan' in pandas 3.0+.
124+
df[column] = df[column].astype(object).fillna(null_value).astype(str)
123125
return df

‎tableone/tableone.py‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -629,6 +629,13 @@ def _insert_n_row(self, table, data):
629629
except TypeError:
630630
table = pd.concat([n_row, table])
631631

632+
# Ensure columns are object dtype to allow mixed type assignment.
633+
# This is required for pandas 3.0+ which enforces strict string dtype.
634+
# See: https://pandas.pydata.org/docs/user_guide/migration-3-strings.html
635+
for col in table.columns:
636+
if pd.api.types.is_string_dtype(table[col]):
637+
table[col] = table[col].astype(object)
638+
632639
if self._groupbylvls == ['Overall']:
633640
table.loc['n', 'Overall'] = len(data.index)
634641
else:

‎tests/unit/test_tableone.py‎

Lines changed: 47 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -121,8 +121,9 @@ def data_mixed(n=20):
121121
mu, sigma = 50, 5
122122
data_mixed['mixed numeric data'] = np.random.normal(mu, sigma, n)
123123

124-
# FutureWarning: Setting an item of incompatible dtype is deprecated
125-
# and will raise an error in a future version of pandas.
124+
# Convert to object dtype to allow mixed types (string in numeric column).
125+
# This is required for pandas 3.0+ which enforces strict dtype.
126+
data_mixed['mixed numeric data'] = data_mixed['mixed numeric data'].astype(object)
126127
data_mixed.loc[1, 'mixed numeric data'] = 'could not measure'
127128

128129
return data_mixed
@@ -1430,3 +1431,47 @@ def test_ttest_equal_var_flag():
14301431
t2 = TableOne(df, columns=['x'], groupby='group', pval=True, ttest_equal_var=True, pval_digits=5)
14311432
pval = t2.tableone[('Grouped by group', 'P-Value')].iloc[1]
14321433
assert pval == "0.00010"
1434+
1435+
1436+
def test_pandas_string_dtype_compatibility():
1437+
"""
1438+
Test that TableOne works with pandas string dtype columns.
1439+
1440+
Pandas 3.0+ enforces strict string dtype, which raises TypeError when
1441+
assigning non-string values (like integers) to string-typed columns.
1442+
This test verifies the fix for GitHub issue #207.
1443+
1444+
See: https://pandas.pydata.org/docs/user_guide/migration-3-strings.html
1445+
"""
1446+
# Create DataFrame with explicit string dtype (simulates pandas 3.0 behavior)
1447+
df = pd.DataFrame({
1448+
'group': pd.array(['A', 'A', 'B', 'B', 'B'], dtype='string'),
1449+
'category': pd.array(['x', 'y', 'x', 'y', 'y'], dtype='string'),
1450+
})
1451+
1452+
# This should not raise TypeError when inserting 'n' row counts
1453+
table = TableOne(df, columns=['category'], categorical=['category'], groupby='group')
1454+
1455+
# Verify the table was created successfully
1456+
assert table.tableone is not None
1457+
1458+
# Verify the 'n' row exists and has correct counts
1459+
n_row = table.tableone.loc['n', :]
1460+
assert n_row is not None
1461+
1462+
1463+
def test_pandas_string_dtype_with_overall():
1464+
"""
1465+
Test that TableOne works with string dtype when overall=True.
1466+
"""
1467+
df = pd.DataFrame({
1468+
'group': pd.array(['A', 'A', 'B', 'B', 'B'], dtype='string'),
1469+
'category': pd.array(['x', 'y', 'x', 'y', 'y'], dtype='string'),
1470+
})
1471+
1472+
# This should not raise TypeError
1473+
table = TableOne(df, columns=['category'], categorical=['category'],
1474+
groupby='group', overall=True)
1475+
1476+
assert table.tableone is not None
1477+
assert 'Overall' in str(table.tableone.columns)

0 commit comments

Comments
 (0)