from datetime import datetime import numpy as np import pytest import lottery_predictor.analyze as analyze def test_load_dataframe_by_date() -> None: pb_start = datetime(2015, 10, 4) mm_start = datetime(2025, 4, 5) date_end = datetime(2026, 3, 8) assert int( analyze.load_dataframe_by_dates( game='Powerball', start_date=pb_start, end_date=date_end, ).count()['draw_date'], ) >= 1324 assert int( analyze.load_dataframe_by_dates( game='MegaMillions', start_date=mm_start, end_date=date_end, ).count()['draw_date'], ) >= 96 with pytest.raises(KeyError): analyze.load_dataframe_by_dates(game='SomeNonExistentGameName') def test_load_dataframe_most_recent() -> None: assert len(analyze.load_dataframe_most_recent(game='Powerball')) == 10 assert len(analyze.load_dataframe_most_recent(game='MegaMillions')) == 10 assert len( analyze.load_dataframe_most_recent(game='Powerball', limit=105), ) == 105 assert len( analyze.load_dataframe_most_recent(game='MegaMillions', limit=45), ) == 45 with pytest.raises(ValueError): analyze.load_dataframe_most_recent(game='Powerball', limit=-1) analyze.load_dataframe_most_recent(game='DoesntExist') def test_prepare_split_data() -> None: mm_start = datetime(2025, 4, 5) date_end = datetime(2026, 3, 8) mega = analyze.load_dataframe_by_dates( game='MegaMillions', start_date=mm_start, end_date=date_end, ) x, y1, y2 = analyze.prepare_split_data( data=mega.values, window_size=len(mega) - 1, ) assert isinstance(x, np.ndarray) assert isinstance(y1, np.ndarray) assert isinstance(y2, np.ndarray) def test_make_prediction() -> None: pb_start = datetime(2015, 10, 4) date_end = datetime(2026, 3, 8) power = analyze.load_dataframe_by_dates( game='Powerball', start_date=pb_start, end_date=date_end, ) main, game = analyze.make_prediction(data_frame=power, window_size=10) assert isinstance(main, np.ndarray) assert isinstance(game, np.int64) def test_least_and_most_common_number() -> None: pb_start = datetime(2015, 10, 4) mm_start = datetime(2025, 4, 5) date_end = datetime(2026, 3, 8) pb_df = analyze.load_dataframe_by_dates( game='Powerball', start_date=pb_start, end_date=date_end, ) mm_df = analyze.load_dataframe_by_dates( game='MegaMillions', start_date=mm_start, end_date=date_end, ) assert analyze.get_most_common_number(pb_df, top=5) == [61, 21, 23, 28, 33] assert analyze.get_least_common_number(pb_df, bottom=5) == [ 13, 49, 26, 46, 34, ] assert analyze.get_most_common_number(mm_df, top=5) == [42, 18, 40, 49, 10] assert analyze.get_least_common_number(mm_df, bottom=5) == [ 35, 51, 61, 1, 20, ] assert analyze.get_most_common_number( pb_df, columns=['powerball'], top=1, ) == [4] assert analyze.get_least_common_number( pb_df, columns=['powerball'], bottom=1, ) == [16] assert analyze.get_most_common_number( mm_df, columns=['mega_ball'], top=1, ) == [24] assert analyze.get_least_common_number( mm_df, columns=['mega_ball'], bottom=1, ) == [20] def test_calculate_probabilities() -> None: """tests build_binary_matrix, build_next_targets and calculate_number_probability""" pb_start = datetime(2015, 10, 4) mm_start = datetime(2025, 4, 5) date_end = datetime(2026, 3, 8) pb_df = analyze.load_dataframe_by_dates( game='Powerball', start_date=pb_start, end_date=date_end, ) pb_probabilities = analyze.calculate_probabilities( data_frame=pb_df, max_number=69, ) mm_df = analyze.load_dataframe_by_dates( game='MegaMillions', start_date=mm_start, end_date=date_end, ) mm_probabilities = analyze.calculate_probabilities( data_frame=mm_df, max_number=70, ) assert pb_probabilities[1] == 0.3142857142857143 assert mm_probabilities[1] == 0.04225352112676056 def test_hot_cold_numbers() -> None: """tests get_hot_numbers and get_cold_numbers""" pb_start = datetime(2015, 10, 4) mm_start = datetime(2025, 4, 5) date_end = datetime(2026, 3, 8) pb_df = analyze.load_dataframe_by_dates( game='Powerball', start_date=pb_start, end_date=date_end, ) pb_probs = analyze.calculate_probabilities(data_frame=pb_df, max_number=69) mm_df = analyze.load_dataframe_by_dates( game='MegaMillions', start_date=mm_start, end_date=date_end, ) mm_probs = analyze.calculate_probabilities(data_frame=mm_df, max_number=70) assert (analyze.get_hot_numbers(probabilities=pb_probs) == [ (13, 1.0), (61, 0.7), (21, 0.6714285714285715), (23, 0.6428571428571428), (28, 0.6428571428571428), ]) assert (analyze.get_cold_numbers(probabilities=pb_probs) == [ (49, 0.10000000000000009), (26, 0.11428571428571432), (46, 0.11428571428571432), (34, 0.17142857142857149), (65, 0.18571428571428572), ]) assert (analyze.get_hot_numbers(probabilities=mm_probs) == [ (42, 0.19718309859154928), (18, 0.18309859154929578), (40, 0.18309859154929578), (10, 0.16901408450704225), (49, 0.16901408450704225), ]) assert (analyze.get_cold_numbers(probabilities=mm_probs) == [ (35, 0.028169014084507043), (51, 0.028169014084507043), (1, 0.04225352112676056), (3, 0.04225352112676056), (20, 0.04225352112676056), ])