From 9d3c01f9669f1976f02dd857a564cddda4084a1d Mon Sep 17 00:00:00 2001 From: Chris Smith Date: Tue, 10 Mar 2026 22:45:53 -0400 Subject: [PATCH] Add probability calculation and hot/cold number functions --- lottery_predictor/analyze.py | 52 ++++++++++++++++++++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/lottery_predictor/analyze.py b/lottery_predictor/analyze.py index 76d1b42..76bf85f 100644 --- a/lottery_predictor/analyze.py +++ b/lottery_predictor/analyze.py @@ -1,12 +1,16 @@ from datetime import datetime from collections import Counter +import numpy as np import pandas as pd +from sklearn.ensemble import RandomForestClassifier from sqlalchemy import select from data.database import Base, dal, project_variables import util.drawing as drawutil +RANDOM_SEED = 42 + def load_dataframe( table_name: str, start_date: datetime | None = None, end_date: datetime | None = None @@ -73,3 +77,51 @@ def get_least_common_number( flat_numbers = data_frame[columns].values.flatten() counts = Counter(flat_numbers) return list(reversed([int(num) for num, _ in counts.most_common()[-bottom:]])) + + +def calculate_probabilities( + data_frame: pd.DataFrame, max_number: int, columns: list | None = None + ) -> dict[int, float]: + """ + :param data_frame: A pandas DataFrame containing the data to calculate probabilities for + :param max_number: The maximum number possible in the data_frame + :param columns: The list of column names to use from the data_frame + :returns dict: A dictionary containing the probabilities + """ + if columns is None: + columns = ['main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5'] + if data_frame.empty: + return dict() + + # get all the numbers in the groups + all_numbers = [num for group in data_frame[columns].values for num in group] + # count all the occurrences of each number + counts = Counter(all_numbers) + # calculate the basic probability of each number occurring again + probabilities = {num: counts.get(num, 0) / (max_number + 1) for num in range(1, max_number + 1)} + + # if any calculation is greater than one, use what is to the right of the decimal point as the value + for key, value in probabilities.items(): + if value > 1: + probabilities[key] = value - int(str(value).split('.')[0]) + + # return the probability dict + return probabilities + + +def get_hot_numbers(probabilities: dict[int, float], top: int = 5) -> list[tuple[int, float]]: + """ + :param probabilities: A dictionary containing the probabilities + :param top: The count of hottest items to return, defaults to 5 + :returns list of tuples: A list of the hot numbers and their raw score + """ + return sorted(probabilities.items(), key=lambda item: item[1], reverse=True)[:top] + + +def get_cold_numbers(probabilities: dict[int, float], bottom: int = 5) -> list[tuple[int, float]]: + """ + :param probabilities: A dictionary containing the probabilities + :param bottom: The count of coldest items to return, defaults to 5 + :returns list of tuples: A list of the hot numbers and their raw score + """ + return sorted(probabilities.items(), key=lambda item: item[1])[:bottom]