codekingpro/portable-devtools
114k
1import warnings
2
3import pytest
4
5import numpy as np
6from numpy import histogram, histogram_bin_edges, histogramdd
7from numpy.testing import (
8 assert_,
9 assert_allclose,
10 assert_almost_equal,
11 assert_array_almost_equal,
12 assert_array_equal,
13 assert_array_max_ulp,
14 assert_equal,
15 assert_raises,
16 assert_raises_regex,
17)
18
19
20class TestHistogram:
21
22 def setup_method(self):
23 pass
24
25 def teardown_method(self):
26 pass
27
28 def test_simple(self):
29 n = 100
30 v = np.random.rand(n)
31 (a, b) = histogram(v)
32 # check if the sum of the bins equals the number of samples
33 assert_equal(np.sum(a, axis=0), n)
34 # check that the bin counts are evenly spaced when the data is from
35 # a linear function
36 (a, b) = histogram(np.linspace(0, 10, 100))
37 assert_array_equal(a, 10)
38
39 def test_one_bin(self):
40 # Ticket 632
41 hist, edges = histogram([1, 2, 3, 4], [1, 2])
42 assert_array_equal(hist, [2, ])
43 assert_array_equal(edges, [1, 2])
44 assert_raises(ValueError, histogram, [1, 2], bins=0)
45 h, e = histogram([1, 2], bins=1)
46 assert_equal(h, np.array([2]))
47 assert_allclose(e, np.array([1., 2.]))
48
49 def test_density(self):
50 # Check that the integral of the density equals 1.
51 n = 100
52 v = np.random.rand(n)
53 a, b = histogram(v, density=True)
54 area = np.sum(a * np.diff(b))
55 assert_almost_equal(area, 1)
56
57 # Check with non-constant bin widths
58 v = np.arange(10)
59 bins = [0, 1, 3, 6, 10]
60 a, b = histogram(v, bins, density=True)
61 assert_array_equal(a, .1)
62 assert_equal(np.sum(a * np.diff(b)), 1)
63
64 # Test that passing False works too
65 a, b = histogram(v, bins, density=False)
66 assert_array_equal(a, [1, 2, 3, 4])
67
68 # Variable bin widths are especially useful to deal with
69 # infinities.
70 v = np.arange(10)
71 bins = [0, 1, 3, 6, np.inf]
72 a, b = histogram(v, bins, density=True)
73 assert_array_equal(a, [.1, .1, .1, 0.])
74
75 # Taken from a bug report from N. Becker on the numpy-discussion
76 # mailing list Aug. 6, 2010.
77 counts, dmy = np.histogram(
78 [1, 2, 3, 4], [0.5, 1.5, np.inf], density=True)
79 assert_equal(counts, [.25, 0])
80
81 def test_outliers(self):
82 # Check that outliers are not tallied
83 a = np.arange(10) + .5
84
85 # Lower outliers
86 h, b = histogram(a, range=[0, 9])
87 assert_equal(h.sum(), 9)
88
89 # Upper outliers
90 h, b = histogram(a, range=[1, 10])
91 assert_equal(h.sum(), 9)
92
93 # Normalization
94 h, b = histogram(a, range=[1, 9], density=True)
95 assert_almost_equal((h * np.diff(b)).sum(), 1, decimal=15)
96
97 # Weights
98 w = np.arange(10) + .5
99 h, b = histogram(a, range=[1, 9], weights=w, density=True)
100 assert_equal((h * np.diff(b)).sum(), 1)
101
102 h, b = histogram(a, bins=8, range=[1, 9], weights=w)
103 assert_equal(h, w[1:-1])
104
105 def test_arr_weights_mismatch(self):
106 a = np.arange(10) + .5
107 w = np.arange(11) + .5
108 with assert_raises_regex(ValueError, "same shape as"):
109 h, b = histogram(a, range=[1, 9], weights=w, density=True)
110
111 def test_type(self):
112 # Check the type of the returned histogram
113 a = np.arange(10) + .5
114 h, b = histogram(a)
115 assert_(np.issubdtype(h.dtype, np.integer))
116
117 h, b = histogram(a, density=True)
118 assert_(np.issubdtype(h.dtype, np.floating))
119
120 h, b = histogram(a, weights=np.ones(10, int))
121 assert_(np.issubdtype(h.dtype, np.integer))
122
123 h, b = histogram(a, weights=np.ones(10, float))
124 assert_(np.issubdtype(h.dtype, np.floating))
125
126 def test_f32_rounding(self):
127 # gh-4799, check that the rounding of the edges works with float32
128 x = np.array([276.318359, -69.593948, 21.329449], dtype=np.float32)
129 y = np.array([5005.689453, 4481.327637, 6010.369629], dtype=np.float32)
130 counts_hist, xedges, yedges = np.histogram2d(x, y, bins=100)
131 assert_equal(counts_hist.sum(), 3.)
132
133 def test_bool_conversion(self):
134 # gh-12107
135 # Reference integer histogram
136 a = np.array([1, 1, 0], dtype=np.uint8)
137 int_hist, int_edges = np.histogram(a)
138
139 # Should raise a warning on booleans
140 # Ensure that the histograms are equivalent, need to suppress
141 # the warnings to get the actual outputs
142 with pytest.warns(RuntimeWarning, match='Converting input from .*'):
143 hist, edges = np.histogram([True, True, False])
144 # A warning should be issued
145 assert_array_equal(hist, int_hist)
146 assert_array_equal(edges, int_edges)
147
148 def test_weights(self):
149 v = np.random.rand(100)
150 w = np.ones(100) * 5
151 a, b = histogram(v)
152 na, nb = histogram(v, density=True)
153 wa, wb = histogram(v, weights=w)
154 nwa, nwb = histogram(v, weights=w, density=True)
155 assert_array_almost_equal(a * 5, wa)
156 assert_array_almost_equal(na, nwa)
157
158 # Check weights are properly applied.
159 v = np.linspace(0, 10, 10)
160 w = np.concatenate((np.zeros(5), np.ones(5)))
161 wa, wb = histogram(v, bins=np.arange(11), weights=w)
162 assert_array_almost_equal(wa, w)
163
164 # Check with integer weights
165 wa, wb = histogram([1, 2, 2, 4], bins=4, weights=[4, 3, 2, 1])
166 assert_array_equal(wa, [4, 5, 0, 1])
167 wa, wb = histogram(
168 [1, 2, 2, 4], bins=4, weights=[4, 3, 2, 1], density=True)
169 assert_array_almost_equal(wa, np.array([4, 5, 0, 1]) / 10. / 3. * 4)
170
171 # Check weights with non-uniform bin widths
172 a, b = histogram(
173 np.arange(9), [0, 1, 3, 6, 10],
174 weights=[2, 1, 1, 1, 1, 1, 1, 1, 1], density=True)
175 assert_almost_equal(a, [.2, .1, .1, .075])
176
177 def test_exotic_weights(self):
178
179 # Test the use of weights that are not integer or floats, but e.g.
180 # complex numbers or object types.
181
182 # Complex weights
183 values = np.array([1.3, 2.5, 2.3])
184 weights = np.array([1, -1, 2]) + 1j * np.array([2, 1, 2])
185
186 # Check with custom bins
187 wa, wb = histogram(values, bins=[0, 2, 3], weights=weights)
188 assert_array_almost_equal(wa, np.array([1, 1]) + 1j * np.array([2, 3]))
189
190 # Check with even bins
191 wa, wb = histogram(values, bins=2, range=[1, 3], weights=weights)
192 assert_array_almost_equal(wa, np.array([1, 1]) + 1j * np.array([2, 3]))
193
194 # Decimal weights
195 from decimal import Decimal
196 values = np.array([1.3, 2.5, 2.3])
197 weights = np.array([Decimal(1), Decimal(2), Decimal(3)])
198
199 # Check with custom bins
200 wa, wb = histogram(values, bins=[0, 2, 3], weights=weights)
201 assert_array_almost_equal(wa, [Decimal(1), Decimal(5)])
202
203 # Check with even bins
204 wa, wb = histogram(values, bins=2, range=[1, 3], weights=weights)
205 assert_array_almost_equal(wa, [Decimal(1), Decimal(5)])
206
207 def test_no_side_effects(self):
208 # This is a regression test that ensures that values passed to
209 # ``histogram`` are unchanged.
210 values = np.array([1.3, 2.5, 2.3])
211 np.histogram(values, range=[-10, 10], bins=100)
212 assert_array_almost_equal(values, [1.3, 2.5, 2.3])
213
214 def test_empty(self):
215 a, b = histogram([], bins=([0, 1]))
216 assert_array_equal(a, np.array([0]))
217 assert_array_equal(b, np.array([0, 1]))
218
219 def test_error_binnum_type(self):
220 # Tests if right Error is raised if bins argument is float
221 vals = np.linspace(0.0, 1.0, num=100)
222 histogram(vals, 5)
223 assert_raises(TypeError, histogram, vals, 2.4)
224
225 def test_finite_range(self):
226 # Normal ranges should be fine
227 vals = np.linspace(0.0, 1.0, num=100)
228 histogram(vals, range=[0.25, 0.75])
229 assert_raises(ValueError, histogram, vals, range=[np.nan, 0.75])
230 assert_raises(ValueError, histogram, vals, range=[0.25, np.inf])
231
232 def test_invalid_range(self):
233 # start of range must be < end of range
234 vals = np.linspace(0.0, 1.0, num=100)
235 with assert_raises_regex(ValueError, "max must be larger than"):
236 np.histogram(vals, range=[0.1, 0.01])
237
238 def test_bin_edge_cases(self):
239 # Ensure that floating-point computations correctly place edge cases.
240 arr = np.array([337, 404, 739, 806, 1007, 1811, 2012])
241 hist, edges = np.histogram(arr, bins=8296, range=(2, 2280))
242 mask = hist > 0
243 left_edges = edges[:-1][mask]
244 right_edges = edges[1:][mask]
245 for x, left, right in zip(arr, left_edges, right_edges):
246 assert_(x >= left)
247 assert_(x < right)
248
249 def test_last_bin_inclusive_range(self):
250 arr = np.array([0., 0., 0., 1., 2., 3., 3., 4., 5.])
251 hist, edges = np.histogram(arr, bins=30, range=(-0.5, 5))
252 assert_equal(hist[-1], 1)
253
254 def test_bin_array_dims(self):
255 # gracefully handle bins object > 1 dimension
256 vals = np.linspace(0.0, 1.0, num=100)
257 bins = np.array([[0, 0.5], [0.6, 1.0]])
258 with assert_raises_regex(ValueError, "must be 1d"):
259 np.histogram(vals, bins=bins)
260
261 def test_unsigned_monotonicity_check(self):
262 # Ensures ValueError is raised if bins not increasing monotonically
263 # when bins contain unsigned values (see #9222)
264 arr = np.array([2])
265 bins = np.array([1, 3, 1], dtype='uint64')
266 with assert_raises(ValueError):
267 hist, edges = np.histogram(arr, bins=bins)
268
269 def test_object_array_of_0d(self):
270 # gh-7864
271 assert_raises(ValueError,
272 histogram, [np.array(0.4) for i in range(10)] + [-np.inf])
273 assert_raises(ValueError,
274 histogram, [np.array(0.4) for i in range(10)] + [np.inf])
275
276 # these should not crash
277 np.histogram([np.array(0.5) for i in range(10)] + [.500000000000002])
278 np.histogram([np.array(0.5) for i in range(10)] + [.5])
279
280 def test_some_nan_values(self):
281 # gh-7503
282 one_nan = np.array([0, 1, np.nan])
283 all_nan = np.array([np.nan, np.nan])
284
285 # the internal comparisons with NaN give warnings
286 with warnings.catch_warnings():
287 warnings.simplefilter('ignore', RuntimeWarning)
288 # can't infer range with nan
289 assert_raises(ValueError, histogram, one_nan, bins='auto')
290 assert_raises(ValueError, histogram, all_nan, bins='auto')
291
292 # explicit range solves the problem
293 h, b = histogram(one_nan, bins='auto', range=(0, 1))
294 assert_equal(h.sum(), 2) # nan is not counted
295 h, b = histogram(all_nan, bins='auto', range=(0, 1))
296 assert_equal(h.sum(), 0) # nan is not counted
297
298 # as does an explicit set of bins
299 h, b = histogram(one_nan, bins=[0, 1])
300 assert_equal(h.sum(), 2) # nan is not counted
301 h, b = histogram(all_nan, bins=[0, 1])
302 assert_equal(h.sum(), 0) # nan is not counted
303
304 def test_datetime(self):
305 begin = np.datetime64('2000-01-01', 'D')
306 offsets = np.array([0, 0, 1, 1, 2, 3, 5, 10, 20])
307 bins = np.array([0, 2, 7, 20])
308 dates = begin + offsets
309 date_bins = begin + bins
310
311 td = np.dtype('timedelta64[D]')
312
313 # Results should be the same for integer offsets or datetime values.
314 # For now, only explicit bins are supported, since linspace does not
315 # work on datetimes or timedeltas
316 d_count, d_edge = histogram(dates, bins=date_bins)
317 t_count, t_edge = histogram(offsets.astype(td), bins=bins.astype(td))
318 i_count, i_edge = histogram(offsets, bins=bins)
319
320 assert_equal(d_count, i_count)
321 assert_equal(t_count, i_count)
322
323 assert_equal((d_edge - begin).astype(int), i_edge)
324 assert_equal(t_edge.astype(int), i_edge)
325
326 assert_equal(d_edge.dtype, dates.dtype)
327 assert_equal(t_edge.dtype, td)
328
329 def do_signed_overflow_bounds(self, dtype):
330 exponent = 8 * np.dtype(dtype).itemsize - 1
331 arr = np.array([-2**exponent + 4, 2**exponent - 4], dtype=dtype)
332 hist, e = histogram(arr, bins=2)
333 assert_equal(e, [-2**exponent + 4, 0, 2**exponent - 4])
334 assert_equal(hist, [1, 1])
335
336 def test_signed_overflow_bounds(self):
337 self.do_signed_overflow_bounds(np.byte)
338 self.do_signed_overflow_bounds(np.short)
339 self.do_signed_overflow_bounds(np.intc)
340 self.do_signed_overflow_bounds(np.int_)
341 self.do_signed_overflow_bounds(np.longlong)
342
343 def do_precision_lower_bound(self, float_small, float_large):
344 eps = np.finfo(float_large).eps
345
346 arr = np.array([1.0], float_small)
347 range = np.array([1.0 + eps, 2.0], float_large)
348
349 # test is looking for behavior when the bounds change between dtypes
350 if range.astype(float_small)[0] != 1:
351 return
352
353 # previously crashed
354 count, x_loc = np.histogram(arr, bins=1, range=range)
355 assert_equal(count, [0])
356 assert_equal(x_loc.dtype, float_large)
357
358 def do_precision_upper_bound(self, float_small, float_large):
359 eps = np.finfo(float_large).eps
360
361 arr = np.array([1.0], float_small)
362 range = np.array([0.0, 1.0 - eps], float_large)
363
364 # test is looking for behavior when the bounds change between dtypes
365 if range.astype(float_small)[-1] != 1:
366 return
367
368 # previously crashed
369 count, x_loc = np.histogram(arr, bins=1, range=range)
370 assert_equal(count, [0])
371
372 assert_equal(x_loc.dtype, float_large)
373
374 def do_precision(self, float_small, float_large):
375 self.do_precision_lower_bound(float_small, float_large)
376 self.do_precision_upper_bound(float_small, float_large)
377
378 def test_precision(self):
379 # not looping results in a useful stack trace upon failure
380 self.do_precision(np.half, np.single)
381 self.do_precision(np.half, np.double)
382 self.do_precision(np.half, np.longdouble)
383 self.do_precision(np.single, np.double)
384 self.do_precision(np.single, np.longdouble)
385 self.do_precision(np.double, np.longdouble)
386
387 def test_histogram_bin_edges(self):
388 hist, e = histogram([1, 2, 3, 4], [1, 2])
389 edges = histogram_bin_edges([1, 2, 3, 4], [1, 2])
390 assert_array_equal(edges, e)
391
392 arr = np.array([0., 0., 0., 1., 2., 3., 3., 4., 5.])
393 hist, e = histogram(arr, bins=30, range=(-0.5, 5))
394 edges = histogram_bin_edges(arr, bins=30, range=(-0.5, 5))
395 assert_array_equal(edges, e)
396
397 hist, e = histogram(arr, bins='auto', range=(0, 1))
398 edges = histogram_bin_edges(arr, bins='auto', range=(0, 1))
399 assert_array_equal(edges, e)
400
401 def test_small_value_range(self):
402 arr = np.array([1, 1 + 2e-16] * 10)
403 with pytest.raises(ValueError, match="Too many bins for data range"):
404 histogram(arr, bins=10)
405
406 # @requires_memory(free_bytes=1e10)
407 # @pytest.mark.slow
408 @pytest.mark.skip(reason="Bad memory reports lead to OOM in ci testing")
409 def test_big_arrays(self):
410 sample = np.zeros([100000000, 3])
411 xbins = 400
412 ybins = 400
413 zbins = np.arange(16000)
414 hist = np.histogramdd(sample=sample, bins=(xbins, ybins, zbins))
415 assert_equal(type(hist), type((1, 2)))
416
417 def test_gh_23110(self):
418 hist, e = np.histogram(np.array([-0.9e-308], dtype='>f8'),
419 bins=2,
420 range=(-1e-308, -2e-313))
421 expected_hist = np.array([1, 0])
422 assert_array_equal(hist, expected_hist)
423
424 def test_gh_28400(self):
425 e = 1 + 1e-12
426 Z = [0, 1, 1, 1, 1, 1, e, e, e, e, e, e, 2]
427 counts, edges = np.histogram(Z, bins="auto")
428 assert len(counts) < 10
429 assert edges[0] == Z[0]
430 assert edges[-1] == Z[-1]
431
432class TestHistogramOptimBinNums:
433 """
434 Provide test coverage when using provided estimators for optimal number of
435 bins
436 """
437
438 def test_empty(self):
439 estimator_list = ['fd', 'scott', 'rice', 'sturges',
440 'doane', 'sqrt', 'auto', 'stone']
441 # check it can deal with empty data
442 for estimator in estimator_list:
443 a, b = histogram([], bins=estimator)
444 assert_array_equal(a, np.array([0]))
445 assert_array_equal(b, np.array([0, 1]))
446
447 def test_simple(self):
448 """
449 Straightforward testing with a mixture of linspace data (for
450 consistency). All test values have been precomputed and the values
451 shouldn't change
452 """
453 # Some basic sanity checking, with some fixed data.
454 # Checking for the correct number of bins
455 basic_test = {50: {'fd': 4, 'scott': 4, 'rice': 8, 'sturges': 7,
456 'doane': 8, 'sqrt': 8, 'auto': 7, 'stone': 2},
457 500: {'fd': 8, 'scott': 8, 'rice': 16, 'sturges': 10,
458 'doane': 12, 'sqrt': 23, 'auto': 10, 'stone': 9},
459 5000: {'fd': 17, 'scott': 17, 'rice': 35, 'sturges': 14,
460 'doane': 17, 'sqrt': 71, 'auto': 17, 'stone': 20}}
461
462 for testlen, expectedResults in basic_test.items():
463 # Create some sort of non uniform data to test with
464 # (2 peak uniform mixture)
465 x1 = np.linspace(-10, -1, testlen // 5 * 2)
466 x2 = np.linspace(1, 10, testlen // 5 * 3)
467 x = np.concatenate((x1, x2))
468 for estimator, numbins in expectedResults.items():
469 a, b = np.histogram(x, estimator)
470 assert_equal(len(a), numbins, err_msg=f"For the {estimator} estimator "
471 f"with datasize of {testlen}")
472
473 def test_small(self):
474 """
475 Smaller datasets have the potential to cause issues with the data
476 adaptive methods, especially the FD method. All bin numbers have been
477 precalculated.
478 """
479 small_dat = {1: {'fd': 1, 'scott': 1, 'rice': 1, 'sturges': 1,
480 'doane': 1, 'sqrt': 1, 'stone': 1},
481 2: {'fd': 2, 'scott': 1, 'rice': 3, 'sturges': 2,
482 'doane': 1, 'sqrt': 2, 'stone': 1},
483 3: {'fd': 2, 'scott': 2, 'rice': 3, 'sturges': 3,
484 'doane': 3, 'sqrt': 2, 'stone': 1}}
485
486 for testlen, expectedResults in small_dat.items():
487 testdat = np.arange(testlen).astype(float)
488 for estimator, expbins in expectedResults.items():
489 a, b = np.histogram(testdat, estimator)
490 assert_equal(len(a), expbins, err_msg=f"For the {estimator} estimator "
491 f"with datasize of {testlen}")
492
493 def test_incorrect_methods(self):
494 """
495 Check a Value Error is thrown when an unknown string is passed in
496 """
497 check_list = ['mad', 'freeman', 'histograms', 'IQR']
498 for estimator in check_list:
499 assert_raises(ValueError, histogram, [1, 2, 3], estimator)
500
501 def test_novariance(self):
502 """
503 Check that methods handle no variance in data
504 Primarily for Scott and FD as the SD and IQR are both 0 in this case
505 """
506 novar_dataset = np.ones(100)
507 novar_resultdict = {'fd': 1, 'scott': 1, 'rice': 1, 'sturges': 1,
508 'doane': 1, 'sqrt': 1, 'auto': 1, 'stone': 1}
509
510 for estimator, numbins in novar_resultdict.items():
511 a, b = np.histogram(novar_dataset, estimator)
512 assert_equal(len(a), numbins,
513 err_msg=f"{estimator} estimator, No Variance test")
514
515 def test_limited_variance(self):
516 """
517 Check when IQR is 0, but variance exists, we return a reasonable value.
518 """
519 lim_var_data = np.ones(1000)
520 lim_var_data[:3] = 0
521 lim_var_data[-4:] = 100
522
523 edges_auto = histogram_bin_edges(lim_var_data, 'auto')
524 assert_equal(edges_auto[0], 0)
525 assert_equal(edges_auto[-1], 100.)
526 assert len(edges_auto) < 100
527
528 edges_fd = histogram_bin_edges(lim_var_data, 'fd')
529 assert_equal(edges_fd, np.array([0, 100]))
530
531 edges_sturges = histogram_bin_edges(lim_var_data, 'sturges')
532 assert_equal(edges_sturges, np.linspace(0, 100, 12))
533
534 def test_outlier(self):
535 """
536 Check the FD, Scott and Doane with outliers.
537
538 The FD estimates a smaller binwidth since it's less affected by
539 outliers. Since the range is so (artificially) large, this means more
540 bins, most of which will be empty, but the data of interest usually is
541 unaffected. The Scott estimator is more affected and returns fewer bins,
542 despite most of the variance being in one area of the data. The Doane
543 estimator lies somewhere between the other two.
544 """
545 xcenter = np.linspace(-10, 10, 50)
546 outlier_dataset = np.hstack((np.linspace(-110, -100, 5), xcenter))
547
548 outlier_resultdict = {'fd': 21, 'scott': 5, 'doane': 11, 'stone': 6}
549
550 for estimator, numbins in outlier_resultdict.items():
551 a, b = np.histogram(outlier_dataset, estimator)
552 assert_equal(len(a), numbins)
553
554 def test_scott_vs_stone(self):
555 # Verify that Scott's rule and Stone's rule converges for normally
556 # distributed data
557
558 def nbins_ratio(seed, size):
559 rng = np.random.RandomState(seed)
560 x = rng.normal(loc=0, scale=2, size=size)
561 a, b = len(np.histogram(x, 'stone')[0]), len(np.histogram(x, 'scott')[0])
562 return a / (a + b)
563
564 geom_space = np.geomspace(start=10, stop=100, num=4).round().astype(int)
565 ll = [[nbins_ratio(seed, size) for size in geom_space] for seed in range(10)]
566
567 # the average difference between the two methods decreases as the dataset
568 # size increases.
569 avg = abs(np.mean(ll, axis=0) - 0.5)
570 assert_almost_equal(avg, [0.15, 0.09, 0.08, 0.03], decimal=2)
571
572 def test_simple_range(self):
573 """
574 Straightforward testing with a mixture of linspace data (for
575 consistency). Adding in a 3rd mixture that will then be
576 completely ignored. All test values have been precomputed and
577 the shouldn't change.
578 """
579 # some basic sanity checking, with some fixed data.
580 # Checking for the correct number of bins
581 basic_test = {
582 50: {'fd': 8, 'scott': 8, 'rice': 15,
583 'sturges': 14, 'auto': 14, 'stone': 8},
584 500: {'fd': 15, 'scott': 16, 'rice': 32,
585 'sturges': 20, 'auto': 20, 'stone': 80},
586 5000: {'fd': 33, 'scott': 33, 'rice': 69,
587 'sturges': 27, 'auto': 33, 'stone': 80}
588 }
589
590 for testlen, expectedResults in basic_test.items():
591 # create some sort of non uniform data to test with
592 # (3 peak uniform mixture)
593 x1 = np.linspace(-10, -1, testlen // 5 * 2)
594 x2 = np.linspace(1, 10, testlen // 5 * 3)
595 x3 = np.linspace(-100, -50, testlen)
596 x = np.hstack((x1, x2, x3))
597 for estimator, numbins in expectedResults.items():
598 a, b = np.histogram(x, estimator, range=(-20, 20))
599 msg = f"For the {estimator} estimator"
600 msg += f" with datasize of {testlen}"
601 assert_equal(len(a), numbins, err_msg=msg)
602
603 @pytest.mark.parametrize("bins", ['auto', 'fd', 'doane', 'scott',
604 'stone', 'rice', 'sturges'])
605 def test_signed_integer_data(self, bins):
606 # Regression test for gh-14379.
607 a = np.array([-2, 0, 127], dtype=np.int8)
608 hist, edges = np.histogram(a, bins=bins)
609 hist32, edges32 = np.histogram(a.astype(np.int32), bins=bins)
610 assert_array_equal(hist, hist32)
611 assert_array_equal(edges, edges32)
612
613 @pytest.mark.parametrize("bins", ['auto', 'fd', 'doane', 'scott',
614 'stone', 'rice', 'sturges'])
615 def test_integer(self, bins):
616 """
617 Test that bin width for integer data is at least 1.
618 """
619 with warnings.catch_warnings():
620 if bins == 'stone':
621 warnings.simplefilter('ignore', RuntimeWarning)
622 assert_equal(
623 np.histogram_bin_edges(np.tile(np.arange(9), 1000), bins),
624 np.arange(9))
625
626 def test_integer_non_auto(self):
627 """
628 Test that the bin-width>=1 requirement *only* applies to auto binning.
629 """
630 assert_equal(
631 np.histogram_bin_edges(np.tile(np.arange(9), 1000), 16),
632 np.arange(17) / 2)
633 assert_equal(
634 np.histogram_bin_edges(np.tile(np.arange(9), 1000), [.1, .2]),
635 [.1, .2])
636
637 def test_simple_weighted(self):
638 """
639 Check that weighted data raises a TypeError
640 """
641 estimator_list = ['fd', 'scott', 'rice', 'sturges', 'auto']
642 for estimator in estimator_list:
643 assert_raises(TypeError, histogram, [1, 2, 3],
644 estimator, weights=[1, 2, 3])
645
646
647class TestHistogramdd:
648
649 def test_simple(self):
650 x = np.array([[-.5, .5, 1.5], [-.5, 1.5, 2.5], [-.5, 2.5, .5],
651 [.5, .5, 1.5], [.5, 1.5, 2.5], [.5, 2.5, 2.5]])
652 H, edges = histogramdd(x, (2, 3, 3),
653 range=[[-1, 1], [0, 3], [0, 3]])
654 answer = np.array([[[0, 1, 0], [0, 0, 1], [1, 0, 0]],
655 [[0, 1, 0], [0, 0, 1], [0, 0, 1]]])
656 assert_array_equal(H, answer)
657
658 # Check normalization
659 ed = [[-2, 0, 2], [0, 1, 2, 3], [0, 1, 2, 3]]
660 H, edges = histogramdd(x, bins=ed, density=True)
661 assert_(np.all(H == answer / 12.))
662
663 # Check that H has the correct shape.
664 H, edges = histogramdd(x, (2, 3, 4),
665 range=[[-1, 1], [0, 3], [0, 4]],
666 density=True)
667 answer = np.array([[[0, 1, 0, 0], [0, 0, 1, 0], [1, 0, 0, 0]],
668 [[0, 1, 0, 0], [0, 0, 1, 0], [0, 0, 1, 0]]])
669 assert_array_almost_equal(H, answer / 6., 4)
670 # Check that a sequence of arrays is accepted and H has the correct
671 # shape.
672 z = [np.squeeze(y) for y in np.split(x, 3, axis=1)]
673 H, edges = histogramdd(
674 z, bins=(4, 3, 2), range=[[-2, 2], [0, 3], [0, 2]])
675 answer = np.array([[[0, 0], [0, 0], [0, 0]],
676 [[0, 1], [0, 0], [1, 0]],
677 [[0, 1], [0, 0], [0, 0]],
678 [[0, 0], [0, 0], [0, 0]]])
679 assert_array_equal(H, answer)
680
681 Z = np.zeros((5, 5, 5))
682 Z[list(range(5)), list(range(5)), list(range(5))] = 1.
683 H, edges = histogramdd([np.arange(5), np.arange(5), np.arange(5)], 5)
684 assert_array_equal(H, Z)
685
686 def test_shape_3d(self):
687 # All possible permutations for bins of different lengths in 3D.
688 bins = ((5, 4, 6), (6, 4, 5), (5, 6, 4), (4, 6, 5), (6, 5, 4),
689 (4, 5, 6))
690 r = np.random.rand(10, 3)
691 for b in bins:
692 H, edges = histogramdd(r, b)
693 assert_(H.shape == b)
694
695 def test_shape_4d(self):
696 # All possible permutations for bins of different lengths in 4D.
697 bins = ((7, 4, 5, 6), (4, 5, 7, 6), (5, 6, 4, 7), (7, 6, 5, 4),
698 (5, 7, 6, 4), (4, 6, 7, 5), (6, 5, 7, 4), (7, 5, 4, 6),
699 (7, 4, 6, 5), (6, 4, 7, 5), (6, 7, 5, 4), (4, 6, 5, 7),
700 (4, 7, 5, 6), (5, 4, 6, 7), (5, 7, 4, 6), (6, 7, 4, 5),
701 (6, 5, 4, 7), (4, 7, 6, 5), (4, 5, 6, 7), (7, 6, 4, 5),
702 (5, 4, 7, 6), (5, 6, 7, 4), (6, 4, 5, 7), (7, 5, 6, 4))
703
704 r = np.random.rand(10, 4)
705 for b in bins:
706 H, edges = histogramdd(r, b)
707 assert_(H.shape == b)
708
709 def test_weights(self):
710 v = np.random.rand(100, 2)
711 hist, edges = histogramdd(v)
712 n_hist, edges = histogramdd(v, density=True)
713 w_hist, edges = histogramdd(v, weights=np.ones(100))
714 assert_array_equal(w_hist, hist)
715 w_hist, edges = histogramdd(v, weights=np.ones(100) * 2, density=True)
716 assert_array_equal(w_hist, n_hist)
717 w_hist, edges = histogramdd(v, weights=np.ones(100, int) * 2)
718 assert_array_equal(w_hist, 2 * hist)
719
720 def test_identical_samples(self):
721 x = np.zeros((10, 2), int)
722 hist, edges = histogramdd(x, bins=2)
723 assert_array_equal(edges[0], np.array([-0.5, 0., 0.5]))
724
725 def test_empty(self):
726 a, b = histogramdd([[], []], bins=([0, 1], [0, 1]))
727 assert_array_max_ulp(a, np.array([[0.]]))
728 a, b = np.histogramdd([[], [], []], bins=2)
729 assert_array_max_ulp(a, np.zeros((2, 2, 2)))
730
731 def test_bins_errors(self):
732 # There are two ways to specify bins. Check for the right errors
733 # when mixing those.
734 x = np.arange(8).reshape(2, 4)
735 assert_raises(ValueError, np.histogramdd, x, bins=[-1, 2, 4, 5])
736 assert_raises(ValueError, np.histogramdd, x, bins=[1, 0.99, 1, 1])
737 assert_raises(
738 ValueError, np.histogramdd, x, bins=[1, 1, 1, [1, 2, 3, -3]])
739 assert_(np.histogramdd(x, bins=[1, 1, 1, [1, 2, 3, 4]]))
740
741 def test_inf_edges(self):
742 # Test using +/-inf bin edges works. See #1788.
743 with np.errstate(invalid='ignore'):
744 x = np.arange(6).reshape(3, 2)
745 expected = np.array([[1, 0], [0, 1], [0, 1]])
746 h, e = np.histogramdd(x, bins=[3, [-np.inf, 2, 10]])
747 assert_allclose(h, expected)
748 h, e = np.histogramdd(x, bins=[3, np.array([-1, 2, np.inf])])
749 assert_allclose(h, expected)
750 h, e = np.histogramdd(x, bins=[3, [-np.inf, 3, np.inf]])
751 assert_allclose(h, expected)
752
753 def test_rightmost_binedge(self):
754 # Test event very close to rightmost binedge. See Github issue #4266
755 x = [0.9999999995]
756 bins = [[0., 0.5, 1.0]]
757 hist, _ = histogramdd(x, bins=bins)
758 assert_(hist[0] == 0.0)
759 assert_(hist[1] == 1.)
760 x = [1.0]
761 bins = [[0., 0.5, 1.0]]
762 hist, _ = histogramdd(x, bins=bins)
763 assert_(hist[0] == 0.0)
764 assert_(hist[1] == 1.)
765 x = [1.0000000001]
766 bins = [[0., 0.5, 1.0]]
767 hist, _ = histogramdd(x, bins=bins)
768 assert_(hist[0] == 0.0)
769 assert_(hist[1] == 0.0)
770 x = [1.0001]
771 bins = [[0., 0.5, 1.0]]
772 hist, _ = histogramdd(x, bins=bins)
773 assert_(hist[0] == 0.0)
774 assert_(hist[1] == 0.0)
775
776 def test_finite_range(self):
777 vals = np.random.random((100, 3))
778 histogramdd(vals, range=[[0.0, 1.0], [0.25, 0.75], [0.25, 0.5]])
779 assert_raises(ValueError, histogramdd, vals,
780 range=[[0.0, 1.0], [0.25, 0.75], [0.25, np.inf]])
781 assert_raises(ValueError, histogramdd, vals,
782 range=[[0.0, 1.0], [np.nan, 0.75], [0.25, 0.5]])
783
784 def test_equal_edges(self):
785 """ Test that adjacent entries in an edge array can be equal """
786 x = np.array([0, 1, 2])
787 y = np.array([0, 1, 2])
788 x_edges = np.array([0, 2, 2])
789 y_edges = 1
790 hist, edges = histogramdd((x, y), bins=(x_edges, y_edges))
791
792 hist_expected = np.array([
793 [2.],
794 [1.], # x == 2 falls in the final bin
795 ])
796 assert_equal(hist, hist_expected)
797
798 def test_edge_dtype(self):
799 """ Test that if an edge array is input, its type is preserved """
800 x = np.array([0, 10, 20])
801 y = x / 10
802 x_edges = np.array([0, 5, 15, 20])
803 y_edges = x_edges / 10
804 hist, edges = histogramdd((x, y), bins=(x_edges, y_edges))
805
806 assert_equal(edges[0].dtype, x_edges.dtype)
807 assert_equal(edges[1].dtype, y_edges.dtype)
808
809 def test_large_integers(self):
810 big = 2**60 # Too large to represent with a full precision float
811
812 x = np.array([0], np.int64)
813 x_edges = np.array([-1, +1], np.int64)
814 y = big + x
815 y_edges = big + x_edges
816
817 hist, edges = histogramdd((x, y), bins=(x_edges, y_edges))
818
819 assert_equal(hist[0, 0], 1)
820
821 def test_density_non_uniform_2d(self):
822 # Defines the following grid:
823 #
824 # 0 2 8
825 # 0+-+-----+
826 # + | +
827 # + | +
828 # 6+-+-----+
829 # 8+-+-----+
830 x_edges = np.array([0, 2, 8])
831 y_edges = np.array([0, 6, 8])
832 relative_areas = np.array([
833 [3, 9],
834 [1, 3]])
835
836 # ensure the number of points in each region is proportional to its area
837 x = np.array([1] + [1] * 3 + [7] * 3 + [7] * 9)
838 y = np.array([7] + [1] * 3 + [7] * 3 + [1] * 9)
839
840 # sanity check that the above worked as intended
841 hist, edges = histogramdd((y, x), bins=(y_edges, x_edges))
842 assert_equal(hist, relative_areas)
843
844 # resulting histogram should be uniform, since counts and areas are proportional
845 hist, edges = histogramdd((y, x), bins=(y_edges, x_edges), density=True)
846 assert_equal(hist, 1 / (8 * 8))
847
848 def test_density_non_uniform_1d(self):
849 # compare to histogram to show the results are the same
850 v = np.arange(10)
851 bins = np.array([0, 1, 3, 6, 10])
852 hist, edges = histogram(v, bins, density=True)
853 hist_dd, edges_dd = histogramdd((v,), (bins,), density=True)
854 assert_equal(hist, hist_dd)
855 assert_equal(edges, edges_dd[0])
856 