Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
test_histograms.py856 linesDownload Raw Back to tests
1import warnings
2
3import pytest
4
5import numpy as np
6from numpy import histogram, histogram_bin_edges, histogramdd
7from numpy.testing import (
8    assert_,
9    assert_allclose,
10    assert_almost_equal,
11    assert_array_almost_equal,
12    assert_array_equal,
13    assert_array_max_ulp,
14    assert_equal,
15    assert_raises,
16    assert_raises_regex,
17)
18
19
20class TestHistogram:
21
22    def setup_method(self):
23        pass
24
25    def teardown_method(self):
26        pass
27
28    def test_simple(self):
29        n = 100
30        v = np.random.rand(n)
31        (a, b) = histogram(v)
32        # check if the sum of the bins equals the number of samples
33        assert_equal(np.sum(a, axis=0), n)
34        # check that the bin counts are evenly spaced when the data is from
35        # a linear function
36        (a, b) = histogram(np.linspace(0, 10, 100))
37        assert_array_equal(a, 10)
38
39    def test_one_bin(self):
40        # Ticket 632
41        hist, edges = histogram([1, 2, 3, 4], [1, 2])
42        assert_array_equal(hist, [2, ])
43        assert_array_equal(edges, [1, 2])
44        assert_raises(ValueError, histogram, [1, 2], bins=0)
45        h, e = histogram([1, 2], bins=1)
46        assert_equal(h, np.array([2]))
47        assert_allclose(e, np.array([1., 2.]))
48
49    def test_density(self):
50        # Check that the integral of the density equals 1.
51        n = 100
52        v = np.random.rand(n)
53        a, b = histogram(v, density=True)
54        area = np.sum(a * np.diff(b))
55        assert_almost_equal(area, 1)
56
57        # Check with non-constant bin widths
58        v = np.arange(10)
59        bins = [0, 1, 3, 6, 10]
60        a, b = histogram(v, bins, density=True)
61        assert_array_equal(a, .1)
62        assert_equal(np.sum(a * np.diff(b)), 1)
63
64        # Test that passing False works too
65        a, b = histogram(v, bins, density=False)
66        assert_array_equal(a, [1, 2, 3, 4])
67
68        # Variable bin widths are especially useful to deal with
69        # infinities.
70        v = np.arange(10)
71        bins = [0, 1, 3, 6, np.inf]
72        a, b = histogram(v, bins, density=True)
73        assert_array_equal(a, [.1, .1, .1, 0.])
74
75        # Taken from a bug report from N. Becker on the numpy-discussion
76        # mailing list Aug. 6, 2010.
77        counts, dmy = np.histogram(
78            [1, 2, 3, 4], [0.5, 1.5, np.inf], density=True)
79        assert_equal(counts, [.25, 0])
80
81    def test_outliers(self):
82        # Check that outliers are not tallied
83        a = np.arange(10) + .5
84
85        # Lower outliers
86        h, b = histogram(a, range=[0, 9])
87        assert_equal(h.sum(), 9)
88
89        # Upper outliers
90        h, b = histogram(a, range=[1, 10])
91        assert_equal(h.sum(), 9)
92
93        # Normalization
94        h, b = histogram(a, range=[1, 9], density=True)
95        assert_almost_equal((h * np.diff(b)).sum(), 1, decimal=15)
96
97        # Weights
98        w = np.arange(10) + .5
99        h, b = histogram(a, range=[1, 9], weights=w, density=True)
100        assert_equal((h * np.diff(b)).sum(), 1)
101
102        h, b = histogram(a, bins=8, range=[1, 9], weights=w)
103        assert_equal(h, w[1:-1])
104
105    def test_arr_weights_mismatch(self):
106        a = np.arange(10) + .5
107        w = np.arange(11) + .5
108        with assert_raises_regex(ValueError, "same shape as"):
109            h, b = histogram(a, range=[1, 9], weights=w, density=True)
110
111    def test_type(self):
112        # Check the type of the returned histogram
113        a = np.arange(10) + .5
114        h, b = histogram(a)
115        assert_(np.issubdtype(h.dtype, np.integer))
116
117        h, b = histogram(a, density=True)
118        assert_(np.issubdtype(h.dtype, np.floating))
119
120        h, b = histogram(a, weights=np.ones(10, int))
121        assert_(np.issubdtype(h.dtype, np.integer))
122
123        h, b = histogram(a, weights=np.ones(10, float))
124        assert_(np.issubdtype(h.dtype, np.floating))
125
126    def test_f32_rounding(self):
127        # gh-4799, check that the rounding of the edges works with float32
128        x = np.array([276.318359, -69.593948, 21.329449], dtype=np.float32)
129        y = np.array([5005.689453, 4481.327637, 6010.369629], dtype=np.float32)
130        counts_hist, xedges, yedges = np.histogram2d(x, y, bins=100)
131        assert_equal(counts_hist.sum(), 3.)
132
133    def test_bool_conversion(self):
134        # gh-12107
135        # Reference integer histogram
136        a = np.array([1, 1, 0], dtype=np.uint8)
137        int_hist, int_edges = np.histogram(a)
138
139        # Should raise a warning on booleans
140        # Ensure that the histograms are equivalent, need to suppress
141        # the warnings to get the actual outputs
142        with pytest.warns(RuntimeWarning, match='Converting input from .*'):
143            hist, edges = np.histogram([True, True, False])
144            # A warning should be issued
145            assert_array_equal(hist, int_hist)
146            assert_array_equal(edges, int_edges)
147
148    def test_weights(self):
149        v = np.random.rand(100)
150        w = np.ones(100) * 5
151        a, b = histogram(v)
152        na, nb = histogram(v, density=True)
153        wa, wb = histogram(v, weights=w)
154        nwa, nwb = histogram(v, weights=w, density=True)
155        assert_array_almost_equal(a * 5, wa)
156        assert_array_almost_equal(na, nwa)
157
158        # Check weights are properly applied.
159        v = np.linspace(0, 10, 10)
160        w = np.concatenate((np.zeros(5), np.ones(5)))
161        wa, wb = histogram(v, bins=np.arange(11), weights=w)
162        assert_array_almost_equal(wa, w)
163
164        # Check with integer weights
165        wa, wb = histogram([1, 2, 2, 4], bins=4, weights=[4, 3, 2, 1])
166        assert_array_equal(wa, [4, 5, 0, 1])
167        wa, wb = histogram(
168            [1, 2, 2, 4], bins=4, weights=[4, 3, 2, 1], density=True)
169        assert_array_almost_equal(wa, np.array([4, 5, 0, 1]) / 10. / 3. * 4)
170
171        # Check weights with non-uniform bin widths
172        a, b = histogram(
173            np.arange(9), [0, 1, 3, 6, 10],
174            weights=[2, 1, 1, 1, 1, 1, 1, 1, 1], density=True)
175        assert_almost_equal(a, [.2, .1, .1, .075])
176
177    def test_exotic_weights(self):
178
179        # Test the use of weights that are not integer or floats, but e.g.
180        # complex numbers or object types.
181
182        # Complex weights
183        values = np.array([1.3, 2.5, 2.3])
184        weights = np.array([1, -1, 2]) + 1j * np.array([2, 1, 2])
185
186        # Check with custom bins
187        wa, wb = histogram(values, bins=[0, 2, 3], weights=weights)
188        assert_array_almost_equal(wa, np.array([1, 1]) + 1j * np.array([2, 3]))
189
190        # Check with even bins
191        wa, wb = histogram(values, bins=2, range=[1, 3], weights=weights)
192        assert_array_almost_equal(wa, np.array([1, 1]) + 1j * np.array([2, 3]))
193
194        # Decimal weights
195        from decimal import Decimal
196        values = np.array([1.3, 2.5, 2.3])
197        weights = np.array([Decimal(1), Decimal(2), Decimal(3)])
198
199        # Check with custom bins
200        wa, wb = histogram(values, bins=[0, 2, 3], weights=weights)
201        assert_array_almost_equal(wa, [Decimal(1), Decimal(5)])
202
203        # Check with even bins
204        wa, wb = histogram(values, bins=2, range=[1, 3], weights=weights)
205        assert_array_almost_equal(wa, [Decimal(1), Decimal(5)])
206
207    def test_no_side_effects(self):
208        # This is a regression test that ensures that values passed to
209        # ``histogram`` are unchanged.
210        values = np.array([1.3, 2.5, 2.3])
211        np.histogram(values, range=[-10, 10], bins=100)
212        assert_array_almost_equal(values, [1.3, 2.5, 2.3])
213
214    def test_empty(self):
215        a, b = histogram([], bins=([0, 1]))
216        assert_array_equal(a, np.array([0]))
217        assert_array_equal(b, np.array([0, 1]))
218
219    def test_error_binnum_type(self):
220        # Tests if right Error is raised if bins argument is float
221        vals = np.linspace(0.0, 1.0, num=100)
222        histogram(vals, 5)
223        assert_raises(TypeError, histogram, vals, 2.4)
224
225    def test_finite_range(self):
226        # Normal ranges should be fine
227        vals = np.linspace(0.0, 1.0, num=100)
228        histogram(vals, range=[0.25, 0.75])
229        assert_raises(ValueError, histogram, vals, range=[np.nan, 0.75])
230        assert_raises(ValueError, histogram, vals, range=[0.25, np.inf])
231
232    def test_invalid_range(self):
233        # start of range must be < end of range
234        vals = np.linspace(0.0, 1.0, num=100)
235        with assert_raises_regex(ValueError, "max must be larger than"):
236            np.histogram(vals, range=[0.1, 0.01])
237
238    def test_bin_edge_cases(self):
239        # Ensure that floating-point computations correctly place edge cases.
240        arr = np.array([337, 404, 739, 806, 1007, 1811, 2012])
241        hist, edges = np.histogram(arr, bins=8296, range=(2, 2280))
242        mask = hist > 0
243        left_edges = edges[:-1][mask]
244        right_edges = edges[1:][mask]
245        for x, left, right in zip(arr, left_edges, right_edges):
246            assert_(x >= left)
247            assert_(x < right)
248
249    def test_last_bin_inclusive_range(self):
250        arr = np.array([0.,  0.,  0.,  1.,  2.,  3.,  3.,  4.,  5.])
251        hist, edges = np.histogram(arr, bins=30, range=(-0.5, 5))
252        assert_equal(hist[-1], 1)
253
254    def test_bin_array_dims(self):
255        # gracefully handle bins object > 1 dimension
256        vals = np.linspace(0.0, 1.0, num=100)
257        bins = np.array([[0, 0.5], [0.6, 1.0]])
258        with assert_raises_regex(ValueError, "must be 1d"):
259            np.histogram(vals, bins=bins)
260
261    def test_unsigned_monotonicity_check(self):
262        # Ensures ValueError is raised if bins not increasing monotonically
263        # when bins contain unsigned values (see #9222)
264        arr = np.array([2])
265        bins = np.array([1, 3, 1], dtype='uint64')
266        with assert_raises(ValueError):
267            hist, edges = np.histogram(arr, bins=bins)
268
269    def test_object_array_of_0d(self):
270        # gh-7864
271        assert_raises(ValueError,
272            histogram, [np.array(0.4) for i in range(10)] + [-np.inf])
273        assert_raises(ValueError,
274            histogram, [np.array(0.4) for i in range(10)] + [np.inf])
275
276        # these should not crash
277        np.histogram([np.array(0.5) for i in range(10)] + [.500000000000002])
278        np.histogram([np.array(0.5) for i in range(10)] + [.5])
279
280    def test_some_nan_values(self):
281        # gh-7503
282        one_nan = np.array([0, 1, np.nan])
283        all_nan = np.array([np.nan, np.nan])
284
285        # the internal comparisons with NaN give warnings
286        with warnings.catch_warnings():
287            warnings.simplefilter('ignore', RuntimeWarning)
288            # can't infer range with nan
289            assert_raises(ValueError, histogram, one_nan, bins='auto')
290            assert_raises(ValueError, histogram, all_nan, bins='auto')
291
292            # explicit range solves the problem
293            h, b = histogram(one_nan, bins='auto', range=(0, 1))
294            assert_equal(h.sum(), 2)  # nan is not counted
295            h, b = histogram(all_nan, bins='auto', range=(0, 1))
296            assert_equal(h.sum(), 0)  # nan is not counted
297
298            # as does an explicit set of bins
299            h, b = histogram(one_nan, bins=[0, 1])
300            assert_equal(h.sum(), 2)  # nan is not counted
301            h, b = histogram(all_nan, bins=[0, 1])
302            assert_equal(h.sum(), 0)  # nan is not counted
303
304    def test_datetime(self):
305        begin = np.datetime64('2000-01-01', 'D')
306        offsets = np.array([0, 0, 1, 1, 2, 3, 5, 10, 20])
307        bins = np.array([0, 2, 7, 20])
308        dates = begin + offsets
309        date_bins = begin + bins
310
311        td = np.dtype('timedelta64[D]')
312
313        # Results should be the same for integer offsets or datetime values.
314        # For now, only explicit bins are supported, since linspace does not
315        # work on datetimes or timedeltas
316        d_count, d_edge = histogram(dates, bins=date_bins)
317        t_count, t_edge = histogram(offsets.astype(td), bins=bins.astype(td))
318        i_count, i_edge = histogram(offsets, bins=bins)
319
320        assert_equal(d_count, i_count)
321        assert_equal(t_count, i_count)
322
323        assert_equal((d_edge - begin).astype(int), i_edge)
324        assert_equal(t_edge.astype(int), i_edge)
325
326        assert_equal(d_edge.dtype, dates.dtype)
327        assert_equal(t_edge.dtype, td)
328
329    def do_signed_overflow_bounds(self, dtype):
330        exponent = 8 * np.dtype(dtype).itemsize - 1
331        arr = np.array([-2**exponent + 4, 2**exponent - 4], dtype=dtype)
332        hist, e = histogram(arr, bins=2)
333        assert_equal(e, [-2**exponent + 4, 0, 2**exponent - 4])
334        assert_equal(hist, [1, 1])
335
336    def test_signed_overflow_bounds(self):
337        self.do_signed_overflow_bounds(np.byte)
338        self.do_signed_overflow_bounds(np.short)
339        self.do_signed_overflow_bounds(np.intc)
340        self.do_signed_overflow_bounds(np.int_)
341        self.do_signed_overflow_bounds(np.longlong)
342
343    def do_precision_lower_bound(self, float_small, float_large):
344        eps = np.finfo(float_large).eps
345
346        arr = np.array([1.0], float_small)
347        range = np.array([1.0 + eps, 2.0], float_large)
348
349        # test is looking for behavior when the bounds change between dtypes
350        if range.astype(float_small)[0] != 1:
351            return
352
353        # previously crashed
354        count, x_loc = np.histogram(arr, bins=1, range=range)
355        assert_equal(count, [0])
356        assert_equal(x_loc.dtype, float_large)
357
358    def do_precision_upper_bound(self, float_small, float_large):
359        eps = np.finfo(float_large).eps
360
361        arr = np.array([1.0], float_small)
362        range = np.array([0.0, 1.0 - eps], float_large)
363
364        # test is looking for behavior when the bounds change between dtypes
365        if range.astype(float_small)[-1] != 1:
366            return
367
368        # previously crashed
369        count, x_loc = np.histogram(arr, bins=1, range=range)
370        assert_equal(count, [0])
371
372        assert_equal(x_loc.dtype, float_large)
373
374    def do_precision(self, float_small, float_large):
375        self.do_precision_lower_bound(float_small, float_large)
376        self.do_precision_upper_bound(float_small, float_large)
377
378    def test_precision(self):
379        # not looping results in a useful stack trace upon failure
380        self.do_precision(np.half, np.single)
381        self.do_precision(np.half, np.double)
382        self.do_precision(np.half, np.longdouble)
383        self.do_precision(np.single, np.double)
384        self.do_precision(np.single, np.longdouble)
385        self.do_precision(np.double, np.longdouble)
386
387    def test_histogram_bin_edges(self):
388        hist, e = histogram([1, 2, 3, 4], [1, 2])
389        edges = histogram_bin_edges([1, 2, 3, 4], [1, 2])
390        assert_array_equal(edges, e)
391
392        arr = np.array([0.,  0.,  0.,  1.,  2.,  3.,  3.,  4.,  5.])
393        hist, e = histogram(arr, bins=30, range=(-0.5, 5))
394        edges = histogram_bin_edges(arr, bins=30, range=(-0.5, 5))
395        assert_array_equal(edges, e)
396
397        hist, e = histogram(arr, bins='auto', range=(0, 1))
398        edges = histogram_bin_edges(arr, bins='auto', range=(0, 1))
399        assert_array_equal(edges, e)
400
401    def test_small_value_range(self):
402        arr = np.array([1, 1 + 2e-16] * 10)
403        with pytest.raises(ValueError, match="Too many bins for data range"):
404            histogram(arr, bins=10)
405
406    # @requires_memory(free_bytes=1e10)
407    # @pytest.mark.slow
408    @pytest.mark.skip(reason="Bad memory reports lead to OOM in ci testing")
409    def test_big_arrays(self):
410        sample = np.zeros([100000000, 3])
411        xbins = 400
412        ybins = 400
413        zbins = np.arange(16000)
414        hist = np.histogramdd(sample=sample, bins=(xbins, ybins, zbins))
415        assert_equal(type(hist), type((1, 2)))
416
417    def test_gh_23110(self):
418        hist, e = np.histogram(np.array([-0.9e-308], dtype='>f8'),
419                               bins=2,
420                               range=(-1e-308, -2e-313))
421        expected_hist = np.array([1, 0])
422        assert_array_equal(hist, expected_hist)
423
424    def test_gh_28400(self):
425        e = 1 + 1e-12
426        Z = [0, 1, 1, 1, 1, 1, e, e, e, e, e, e, 2]
427        counts, edges = np.histogram(Z, bins="auto")
428        assert len(counts) < 10
429        assert edges[0] == Z[0]
430        assert edges[-1] == Z[-1]
431
432class TestHistogramOptimBinNums:
433    """
434    Provide test coverage when using provided estimators for optimal number of
435    bins
436    """
437
438    def test_empty(self):
439        estimator_list = ['fd', 'scott', 'rice', 'sturges',
440                          'doane', 'sqrt', 'auto', 'stone']
441        # check it can deal with empty data
442        for estimator in estimator_list:
443            a, b = histogram([], bins=estimator)
444            assert_array_equal(a, np.array([0]))
445            assert_array_equal(b, np.array([0, 1]))
446
447    def test_simple(self):
448        """
449        Straightforward testing with a mixture of linspace data (for
450        consistency). All test values have been precomputed and the values
451        shouldn't change
452        """
453        # Some basic sanity checking, with some fixed data.
454        # Checking for the correct number of bins
455        basic_test = {50:   {'fd': 4,  'scott': 4,  'rice': 8,  'sturges': 7,
456                             'doane': 8, 'sqrt': 8, 'auto': 7, 'stone': 2},
457                      500:  {'fd': 8,  'scott': 8,  'rice': 16, 'sturges': 10,
458                             'doane': 12, 'sqrt': 23, 'auto': 10, 'stone': 9},
459                      5000: {'fd': 17, 'scott': 17, 'rice': 35, 'sturges': 14,
460                             'doane': 17, 'sqrt': 71, 'auto': 17, 'stone': 20}}
461
462        for testlen, expectedResults in basic_test.items():
463            # Create some sort of non uniform data to test with
464            # (2 peak uniform mixture)
465            x1 = np.linspace(-10, -1, testlen // 5 * 2)
466            x2 = np.linspace(1, 10, testlen // 5 * 3)
467            x = np.concatenate((x1, x2))
468            for estimator, numbins in expectedResults.items():
469                a, b = np.histogram(x, estimator)
470                assert_equal(len(a), numbins, err_msg=f"For the {estimator} estimator "
471                             f"with datasize of {testlen}")
472
473    def test_small(self):
474        """
475        Smaller datasets have the potential to cause issues with the data
476        adaptive methods, especially the FD method. All bin numbers have been
477        precalculated.
478        """
479        small_dat = {1: {'fd': 1, 'scott': 1, 'rice': 1, 'sturges': 1,
480                         'doane': 1, 'sqrt': 1, 'stone': 1},
481                     2: {'fd': 2, 'scott': 1, 'rice': 3, 'sturges': 2,
482                         'doane': 1, 'sqrt': 2, 'stone': 1},
483                     3: {'fd': 2, 'scott': 2, 'rice': 3, 'sturges': 3,
484                         'doane': 3, 'sqrt': 2, 'stone': 1}}
485
486        for testlen, expectedResults in small_dat.items():
487            testdat = np.arange(testlen).astype(float)
488            for estimator, expbins in expectedResults.items():
489                a, b = np.histogram(testdat, estimator)
490                assert_equal(len(a), expbins, err_msg=f"For the {estimator} estimator "
491                             f"with datasize of {testlen}")
492
493    def test_incorrect_methods(self):
494        """
495        Check a Value Error is thrown when an unknown string is passed in
496        """
497        check_list = ['mad', 'freeman', 'histograms', 'IQR']
498        for estimator in check_list:
499            assert_raises(ValueError, histogram, [1, 2, 3], estimator)
500
501    def test_novariance(self):
502        """
503        Check that methods handle no variance in data
504        Primarily for Scott and FD as the SD and IQR are both 0 in this case
505        """
506        novar_dataset = np.ones(100)
507        novar_resultdict = {'fd': 1, 'scott': 1, 'rice': 1, 'sturges': 1,
508                            'doane': 1, 'sqrt': 1, 'auto': 1, 'stone': 1}
509
510        for estimator, numbins in novar_resultdict.items():
511            a, b = np.histogram(novar_dataset, estimator)
512            assert_equal(len(a), numbins,
513                         err_msg=f"{estimator} estimator, No Variance test")
514
515    def test_limited_variance(self):
516        """
517        Check when IQR is 0, but variance exists, we return a reasonable value.
518        """
519        lim_var_data = np.ones(1000)
520        lim_var_data[:3] = 0
521        lim_var_data[-4:] = 100
522
523        edges_auto = histogram_bin_edges(lim_var_data, 'auto')
524        assert_equal(edges_auto[0], 0)
525        assert_equal(edges_auto[-1], 100.)
526        assert len(edges_auto) < 100
527
528        edges_fd = histogram_bin_edges(lim_var_data, 'fd')
529        assert_equal(edges_fd, np.array([0, 100]))
530
531        edges_sturges = histogram_bin_edges(lim_var_data, 'sturges')
532        assert_equal(edges_sturges, np.linspace(0, 100, 12))
533
534    def test_outlier(self):
535        """
536        Check the FD, Scott and Doane with outliers.
537
538        The FD estimates a smaller binwidth since it's less affected by
539        outliers. Since the range is so (artificially) large, this means more
540        bins, most of which will be empty, but the data of interest usually is
541        unaffected. The Scott estimator is more affected and returns fewer bins,
542        despite most of the variance being in one area of the data. The Doane
543        estimator lies somewhere between the other two.
544        """
545        xcenter = np.linspace(-10, 10, 50)
546        outlier_dataset = np.hstack((np.linspace(-110, -100, 5), xcenter))
547
548        outlier_resultdict = {'fd': 21, 'scott': 5, 'doane': 11, 'stone': 6}
549
550        for estimator, numbins in outlier_resultdict.items():
551            a, b = np.histogram(outlier_dataset, estimator)
552            assert_equal(len(a), numbins)
553
554    def test_scott_vs_stone(self):
555        # Verify that Scott's rule and Stone's rule converges for normally
556        # distributed data
557
558        def nbins_ratio(seed, size):
559            rng = np.random.RandomState(seed)
560            x = rng.normal(loc=0, scale=2, size=size)
561            a, b = len(np.histogram(x, 'stone')[0]), len(np.histogram(x, 'scott')[0])
562            return a / (a + b)
563
564        geom_space = np.geomspace(start=10, stop=100, num=4).round().astype(int)
565        ll = [[nbins_ratio(seed, size) for size in geom_space] for seed in range(10)]
566
567        # the average difference between the two methods decreases as the dataset
568        # size increases.
569        avg = abs(np.mean(ll, axis=0) - 0.5)
570        assert_almost_equal(avg, [0.15, 0.09, 0.08, 0.03], decimal=2)
571
572    def test_simple_range(self):
573        """
574        Straightforward testing with a mixture of linspace data (for
575        consistency). Adding in a 3rd mixture that will then be
576        completely ignored. All test values have been precomputed and
577        the shouldn't change.
578        """
579        # some basic sanity checking, with some fixed data.
580        # Checking for the correct number of bins
581        basic_test = {
582                      50:   {'fd': 8,  'scott': 8,  'rice': 15,
583                             'sturges': 14, 'auto': 14, 'stone': 8},
584                      500:  {'fd': 15, 'scott': 16, 'rice': 32,
585                             'sturges': 20, 'auto': 20, 'stone': 80},
586                      5000: {'fd': 33, 'scott': 33, 'rice': 69,
587                             'sturges': 27, 'auto': 33, 'stone': 80}
588                     }
589
590        for testlen, expectedResults in basic_test.items():
591            # create some sort of non uniform data to test with
592            # (3 peak uniform mixture)
593            x1 = np.linspace(-10, -1, testlen // 5 * 2)
594            x2 = np.linspace(1, 10, testlen // 5 * 3)
595            x3 = np.linspace(-100, -50, testlen)
596            x = np.hstack((x1, x2, x3))
597            for estimator, numbins in expectedResults.items():
598                a, b = np.histogram(x, estimator, range=(-20, 20))
599                msg = f"For the {estimator} estimator"
600                msg += f" with datasize of {testlen}"
601                assert_equal(len(a), numbins, err_msg=msg)
602
603    @pytest.mark.parametrize("bins", ['auto', 'fd', 'doane', 'scott',
604                                      'stone', 'rice', 'sturges'])
605    def test_signed_integer_data(self, bins):
606        # Regression test for gh-14379.
607        a = np.array([-2, 0, 127], dtype=np.int8)
608        hist, edges = np.histogram(a, bins=bins)
609        hist32, edges32 = np.histogram(a.astype(np.int32), bins=bins)
610        assert_array_equal(hist, hist32)
611        assert_array_equal(edges, edges32)
612
613    @pytest.mark.parametrize("bins", ['auto', 'fd', 'doane', 'scott',
614                                      'stone', 'rice', 'sturges'])
615    def test_integer(self, bins):
616        """
617        Test that bin width for integer data is at least 1.
618        """
619        with warnings.catch_warnings():
620            if bins == 'stone':
621                warnings.simplefilter('ignore', RuntimeWarning)
622            assert_equal(
623                np.histogram_bin_edges(np.tile(np.arange(9), 1000), bins),
624                np.arange(9))
625
626    def test_integer_non_auto(self):
627        """
628        Test that the bin-width>=1 requirement *only* applies to auto binning.
629        """
630        assert_equal(
631            np.histogram_bin_edges(np.tile(np.arange(9), 1000), 16),
632            np.arange(17) / 2)
633        assert_equal(
634            np.histogram_bin_edges(np.tile(np.arange(9), 1000), [.1, .2]),
635            [.1, .2])
636
637    def test_simple_weighted(self):
638        """
639        Check that weighted data raises a TypeError
640        """
641        estimator_list = ['fd', 'scott', 'rice', 'sturges', 'auto']
642        for estimator in estimator_list:
643            assert_raises(TypeError, histogram, [1, 2, 3],
644                          estimator, weights=[1, 2, 3])
645
646
647class TestHistogramdd:
648
649    def test_simple(self):
650        x = np.array([[-.5, .5, 1.5], [-.5, 1.5, 2.5], [-.5, 2.5, .5],
651                      [.5,  .5, 1.5], [.5,  1.5, 2.5], [.5,  2.5, 2.5]])
652        H, edges = histogramdd(x, (2, 3, 3),
653                               range=[[-1, 1], [0, 3], [0, 3]])
654        answer = np.array([[[0, 1, 0], [0, 0, 1], [1, 0, 0]],
655                           [[0, 1, 0], [0, 0, 1], [0, 0, 1]]])
656        assert_array_equal(H, answer)
657
658        # Check normalization
659        ed = [[-2, 0, 2], [0, 1, 2, 3], [0, 1, 2, 3]]
660        H, edges = histogramdd(x, bins=ed, density=True)
661        assert_(np.all(H == answer / 12.))
662
663        # Check that H has the correct shape.
664        H, edges = histogramdd(x, (2, 3, 4),
665                               range=[[-1, 1], [0, 3], [0, 4]],
666                               density=True)
667        answer = np.array([[[0, 1, 0, 0], [0, 0, 1, 0], [1, 0, 0, 0]],
668                           [[0, 1, 0, 0], [0, 0, 1, 0], [0, 0, 1, 0]]])
669        assert_array_almost_equal(H, answer / 6., 4)
670        # Check that a sequence of arrays is accepted and H has the correct
671        # shape.
672        z = [np.squeeze(y) for y in np.split(x, 3, axis=1)]
673        H, edges = histogramdd(
674            z, bins=(4, 3, 2), range=[[-2, 2], [0, 3], [0, 2]])
675        answer = np.array([[[0, 0], [0, 0], [0, 0]],
676                           [[0, 1], [0, 0], [1, 0]],
677                           [[0, 1], [0, 0], [0, 0]],
678                           [[0, 0], [0, 0], [0, 0]]])
679        assert_array_equal(H, answer)
680
681        Z = np.zeros((5, 5, 5))
682        Z[list(range(5)), list(range(5)), list(range(5))] = 1.
683        H, edges = histogramdd([np.arange(5), np.arange(5), np.arange(5)], 5)
684        assert_array_equal(H, Z)
685
686    def test_shape_3d(self):
687        # All possible permutations for bins of different lengths in 3D.
688        bins = ((5, 4, 6), (6, 4, 5), (5, 6, 4), (4, 6, 5), (6, 5, 4),
689                (4, 5, 6))
690        r = np.random.rand(10, 3)
691        for b in bins:
692            H, edges = histogramdd(r, b)
693            assert_(H.shape == b)
694
695    def test_shape_4d(self):
696        # All possible permutations for bins of different lengths in 4D.
697        bins = ((7, 4, 5, 6), (4, 5, 7, 6), (5, 6, 4, 7), (7, 6, 5, 4),
698                (5, 7, 6, 4), (4, 6, 7, 5), (6, 5, 7, 4), (7, 5, 4, 6),
699                (7, 4, 6, 5), (6, 4, 7, 5), (6, 7, 5, 4), (4, 6, 5, 7),
700                (4, 7, 5, 6), (5, 4, 6, 7), (5, 7, 4, 6), (6, 7, 4, 5),
701                (6, 5, 4, 7), (4, 7, 6, 5), (4, 5, 6, 7), (7, 6, 4, 5),
702                (5, 4, 7, 6), (5, 6, 7, 4), (6, 4, 5, 7), (7, 5, 6, 4))
703
704        r = np.random.rand(10, 4)
705        for b in bins:
706            H, edges = histogramdd(r, b)
707            assert_(H.shape == b)
708
709    def test_weights(self):
710        v = np.random.rand(100, 2)
711        hist, edges = histogramdd(v)
712        n_hist, edges = histogramdd(v, density=True)
713        w_hist, edges = histogramdd(v, weights=np.ones(100))
714        assert_array_equal(w_hist, hist)
715        w_hist, edges = histogramdd(v, weights=np.ones(100) * 2, density=True)
716        assert_array_equal(w_hist, n_hist)
717        w_hist, edges = histogramdd(v, weights=np.ones(100, int) * 2)
718        assert_array_equal(w_hist, 2 * hist)
719
720    def test_identical_samples(self):
721        x = np.zeros((10, 2), int)
722        hist, edges = histogramdd(x, bins=2)
723        assert_array_equal(edges[0], np.array([-0.5, 0., 0.5]))
724
725    def test_empty(self):
726        a, b = histogramdd([[], []], bins=([0, 1], [0, 1]))
727        assert_array_max_ulp(a, np.array([[0.]]))
728        a, b = np.histogramdd([[], [], []], bins=2)
729        assert_array_max_ulp(a, np.zeros((2, 2, 2)))
730
731    def test_bins_errors(self):
732        # There are two ways to specify bins. Check for the right errors
733        # when mixing those.
734        x = np.arange(8).reshape(2, 4)
735        assert_raises(ValueError, np.histogramdd, x, bins=[-1, 2, 4, 5])
736        assert_raises(ValueError, np.histogramdd, x, bins=[1, 0.99, 1, 1])
737        assert_raises(
738            ValueError, np.histogramdd, x, bins=[1, 1, 1, [1, 2, 3, -3]])
739        assert_(np.histogramdd(x, bins=[1, 1, 1, [1, 2, 3, 4]]))
740
741    def test_inf_edges(self):
742        # Test using +/-inf bin edges works. See #1788.
743        with np.errstate(invalid='ignore'):
744            x = np.arange(6).reshape(3, 2)
745            expected = np.array([[1, 0], [0, 1], [0, 1]])
746            h, e = np.histogramdd(x, bins=[3, [-np.inf, 2, 10]])
747            assert_allclose(h, expected)
748            h, e = np.histogramdd(x, bins=[3, np.array([-1, 2, np.inf])])
749            assert_allclose(h, expected)
750            h, e = np.histogramdd(x, bins=[3, [-np.inf, 3, np.inf]])
751            assert_allclose(h, expected)
752
753    def test_rightmost_binedge(self):
754        # Test event very close to rightmost binedge. See Github issue #4266
755        x = [0.9999999995]
756        bins = [[0., 0.5, 1.0]]
757        hist, _ = histogramdd(x, bins=bins)
758        assert_(hist[0] == 0.0)
759        assert_(hist[1] == 1.)
760        x = [1.0]
761        bins = [[0., 0.5, 1.0]]
762        hist, _ = histogramdd(x, bins=bins)
763        assert_(hist[0] == 0.0)
764        assert_(hist[1] == 1.)
765        x = [1.0000000001]
766        bins = [[0., 0.5, 1.0]]
767        hist, _ = histogramdd(x, bins=bins)
768        assert_(hist[0] == 0.0)
769        assert_(hist[1] == 0.0)
770        x = [1.0001]
771        bins = [[0., 0.5, 1.0]]
772        hist, _ = histogramdd(x, bins=bins)
773        assert_(hist[0] == 0.0)
774        assert_(hist[1] == 0.0)
775
776    def test_finite_range(self):
777        vals = np.random.random((100, 3))
778        histogramdd(vals, range=[[0.0, 1.0], [0.25, 0.75], [0.25, 0.5]])
779        assert_raises(ValueError, histogramdd, vals,
780                      range=[[0.0, 1.0], [0.25, 0.75], [0.25, np.inf]])
781        assert_raises(ValueError, histogramdd, vals,
782                      range=[[0.0, 1.0], [np.nan, 0.75], [0.25, 0.5]])
783
784    def test_equal_edges(self):
785        """ Test that adjacent entries in an edge array can be equal """
786        x = np.array([0, 1, 2])
787        y = np.array([0, 1, 2])
788        x_edges = np.array([0, 2, 2])
789        y_edges = 1
790        hist, edges = histogramdd((x, y), bins=(x_edges, y_edges))
791
792        hist_expected = np.array([
793            [2.],
794            [1.],  # x == 2 falls in the final bin
795        ])
796        assert_equal(hist, hist_expected)
797
798    def test_edge_dtype(self):
799        """ Test that if an edge array is input, its type is preserved """
800        x = np.array([0, 10, 20])
801        y = x / 10
802        x_edges = np.array([0, 5, 15, 20])
803        y_edges = x_edges / 10
804        hist, edges = histogramdd((x, y), bins=(x_edges, y_edges))
805
806        assert_equal(edges[0].dtype, x_edges.dtype)
807        assert_equal(edges[1].dtype, y_edges.dtype)
808
809    def test_large_integers(self):
810        big = 2**60  # Too large to represent with a full precision float
811
812        x = np.array([0], np.int64)
813        x_edges = np.array([-1, +1], np.int64)
814        y = big + x
815        y_edges = big + x_edges
816
817        hist, edges = histogramdd((x, y), bins=(x_edges, y_edges))
818
819        assert_equal(hist[0, 0], 1)
820
821    def test_density_non_uniform_2d(self):
822        # Defines the following grid:
823        #
824        #    0 2     8
825        #   0+-+-----+
826        #    + |     +
827        #    + |     +
828        #   6+-+-----+
829        #   8+-+-----+
830        x_edges = np.array([0, 2, 8])
831        y_edges = np.array([0, 6, 8])
832        relative_areas = np.array([
833            [3, 9],
834            [1, 3]])
835
836        # ensure the number of points in each region is proportional to its area
837        x = np.array([1] + [1] * 3 + [7] * 3 + [7] * 9)
838        y = np.array([7] + [1] * 3 + [7] * 3 + [1] * 9)
839
840        # sanity check that the above worked as intended
841        hist, edges = histogramdd((y, x), bins=(y_edges, x_edges))
842        assert_equal(hist, relative_areas)
843
844        # resulting histogram should be uniform, since counts and areas are proportional
845        hist, edges = histogramdd((y, x), bins=(y_edges, x_edges), density=True)
846        assert_equal(hist, 1 / (8 * 8))
847
848    def test_density_non_uniform_1d(self):
849        # compare to histogram to show the results are the same
850        v = np.arange(10)
851        bins = np.array([0, 1, 3, 6, 10])
852        hist, edges = histogram(v, bins, density=True)
853        hist_dd, edges_dd = histogramdd((v,), (bins,), density=True)
854        assert_equal(hist, hist_dd)
855        assert_equal(edges, edges_dd[0])
856 
codekingpro/portable-devtools · Team Ai