Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 20 additions & 5 deletions patsy/state.py
Original file line number Diff line number Diff line change
Expand Up @@ -105,7 +105,14 @@ def __init__(self):

def memorize_chunk(self, x):
x = atleast_2d_column_default(x)
self._count += x.shape[0]
if safe_issubdtype(x.dtype, np.inexact):
# Missing values are left out of the mean, column by column; rows
# containing them are dropped later by the NA action.
missing = np.isnan(x)
self._count += np.sum(~missing, 0)
x = np.where(missing, 0, x)
else:
self._count += x.shape[0]
this_total = np.sum(x, 0, dtype=wide_dtype_for(x))
# This is to handle potentially multi-column x's:
if self._sum is None:
Expand Down Expand Up @@ -168,12 +175,20 @@ def memorize_chunk(self, x, center=True, rescale=True, ddof=0):
if self.current_mean is None:
self.current_mean = np.zeros(x.shape[1], dtype=wide_dtype_for(x))
self.current_M2 = np.zeros(x.shape[1], dtype=wide_dtype_for(x))
# Missing values are left out of the mean and variance, column by
# column; rows containing them are dropped later by the NA action.
if safe_issubdtype(x.dtype, np.inexact):
present = ~np.isnan(x)
else:
present = np.ones(x.shape, dtype=bool)
# XX this can surely be vectorized but I am feeling lazy:
for i in range(x.shape[0]):
self.current_n += 1
delta = x[i, :] - self.current_mean
self.current_mean += delta / self.current_n
self.current_M2 += delta * (x[i, :] - self.current_mean)
self.current_n = self.current_n + present[i, :]
delta = np.where(present[i, :], x[i, :] - self.current_mean, 0)
self.current_mean += delta / np.maximum(self.current_n, 1)
self.current_M2 += delta * np.where(
present[i, :], x[i, :] - self.current_mean, 0
)

def memorize_finish(self):
pass
Expand Down
30 changes: 30 additions & 0 deletions patsy/test_state.py
Original file line number Diff line number Diff line change
Expand Up @@ -135,6 +135,20 @@ def test_Center():
check_stateful(Center, True, [1.3, -10.1, 7.0, 12.0], [-1.25, -12.65, 4.45, 9.45])


def test_Center_missing_values():
# NaNs are skipped column by column when computing the mean, and stay NaN
data = np.array([[1.0, 10.0], [np.nan, 20.0], [3.0, np.nan], [4.0, 30.0]])
expected = np.array(
[[-1.666666, -10.0], [np.nan, 0.0], [0.333333, np.nan], [1.333333, 10.0]]
)
for chunks in ([data], [data[:2], data[2:]]):
t = Center()
for chunk in chunks:
t.memorize_chunk(chunk)
t.memorize_finish()
assert np.allclose(t.transform(data), expected, equal_nan=True, atol=1e-6)


def test_stateful_transform_wrapper():
assert np.allclose(center([1, 2, 3]), [-1, 0, 1])
assert np.allclose(center([1, 2, 1, 2]), [-0.5, 0.5, -0.5, 0.5])
Expand Down Expand Up @@ -205,3 +219,19 @@ def test_Standardize():
ddof=1,
)
check_stateful(Standardize, True, r20, r20, center=False, rescale=False, ddof=1)


def test_Standardize_missing_values():
# NaNs are skipped column by column when computing the mean and variance
data = np.array([[1.0, 10.0], [np.nan, 20.0], [3.0, np.nan], [4.0, 30.0]])
expected = np.empty_like(data)
for j in range(data.shape[1]):
col = data[:, j]
observed = col[~np.isnan(col)]
expected[:, j] = (col - observed.mean()) / observed.std(ddof=1)
for chunks in ([data], [data[:2], data[2:]]):
t = Standardize()
for chunk in chunks:
t.memorize_chunk(chunk, ddof=1)
t.memorize_finish()
assert np.allclose(t.transform(data, ddof=1), expected, equal_nan=True)