diff --git a/hed/models/column_source.py b/hed/models/column_source.py index 9f3453cb..0a2f7312 100644 --- a/hed/models/column_source.py +++ b/hed/models/column_source.py @@ -11,7 +11,7 @@ from __future__ import annotations -import math +import numbers from typing import Protocol, runtime_checkable # Cell texts that mean "no value here". None and float NaN are also missing. @@ -45,11 +45,18 @@ def distinct_values(values) -> dict[str, list[int]]: def is_missing(value) -> bool: - """Return True if a cell value stands for "no value": None, a float NaN, ``""`` or ``"n/a"``.""" + """Return True if a cell value stands for "no value": None, a NaN of any float width, ``""`` or ``"n/a"``. + + The NaN test accepts any real number rather than only ``float``: numpy's float32 and float16 do not + subclass Python's float (only float64 does), and a numpy-backed table (an NWB column, a pandas + frame read with a narrow dtype) hands those NaNs over as values. NaN is the one real value that is + not equal to itself, and that test needs no conversion to float, so an integer too large for a + float or a Fraction is simply a value. + """ if value is None: return True - if isinstance(value, float) and math.isnan(value): - return True + if isinstance(value, numbers.Real): + return bool(value != value) return isinstance(value, str) and value in MISSING_VALUES diff --git a/tests/models/test_column_source.py b/tests/models/test_column_source.py index 10b8d38c..a99f8e1b 100644 --- a/tests/models/test_column_source.py +++ b/tests/models/test_column_source.py @@ -1,9 +1,12 @@ import math import unittest +from fractions import Fraction +import numpy as np import pandas as pd from hed.models import ColumnSource, ListColumnSource, TabularInput, distinct_values +from hed.models.column_source import is_missing class TestDistinctValues(unittest.TestCase): @@ -15,6 +18,27 @@ def test_missing_values_are_dropped(self): values = [None, float("nan"), "", "n/a", "Red", math.nan, "n/a"] self.assertEqual(distinct_values(values), {"Red": [4]}) + def test_numpy_nan_of_any_width_is_missing(self): + """numpy float32 and float16 are not instances of float; their NaN is still a missing value.""" + for dtype in (np.float16, np.float32, np.float64): + values = np.array([1.5, np.nan, 1.5], dtype=dtype) + self.assertEqual(distinct_values(values), {"1.5": [0, 2]}, dtype) + self.assertTrue(is_missing(values[1]), dtype) + self.assertFalse(is_missing(values[0]), dtype) + self.assertFalse(is_missing(np.int32(0))) + self.assertFalse(is_missing(np.bool_(False))) + self.assertFalse(is_missing(0)) + self.assertTrue(is_missing(None)) + self.assertTrue(is_missing("n/a")) + self.assertFalse(is_missing("nan")) # the text nan is a value + + def test_real_values_that_do_not_fit_a_float_are_values(self): + """An integer too large for a float, or a Fraction, is a value: the NaN test converts nothing.""" + huge = 10**1000 + self.assertFalse(is_missing(huge)) + self.assertFalse(is_missing(Fraction(1, 3))) + self.assertEqual(distinct_values([huge, Fraction(1, 3), huge]), {str(huge): [0, 2], "1/3": [1]}) + def test_non_strings_become_text_and_bytes_are_decoded(self): values = [3, 3.5, b"Blue", 3] self.assertEqual(distinct_values(values), {"3": [0, 3], "3.5": [1], "Blue": [2]})