from collections import Counter from datetime import datetime from itertools import combinations import numpy as np import pandas as pd from sklearn.ensemble import RandomForestRegressor from sqlalchemy import select from data.database import Base, dal import util.drawing as drawutil def load_dataframe( table_name: str, start_date: datetime | None = None, end_date: datetime | None = None ) -> pd.DataFrame: """ :param table_name: the name of the table to load data from :param start_date: the start date for the data :param end_date: the end date for the data :returns: a pandas DataFrame containing the data """ dal.connect() session = dal.Session() game_dict = drawutil.check_table_name(table_name=table_name) if game_dict: target_table = Base.metadata.tables.get(game_dict['db_table_name']) if start_date is not None and end_date is not None: sql_statement = (select(target_table) .where(target_table.columns.draw_date >= start_date) .where(target_table.columns.draw_date <= end_date)) elif start_date is not None and end_date is None: sql_statement = (select(target_table).where(target_table.columns.draw_date >= start_date)) elif start_date is None and end_date is not None: sql_statement = (select(target_table).where(target_table.columns.draw_date <= end_date)) else: sql_statement = select(target_table) return pd.read_sql(sql_statement, session.bind) else: raise ValueError('An invalid table name was provided') def get_most_common_number( data_frame: pd.DataFrame, columns: list | None = None, top: int = 1 ) -> list[int]: """ :param data_frame: a pandas DataFrame containing the data :param columns: a list of column names to use :param top: the number of top (most seen) numbers to return :returns: a list of the most common numbers """ if columns is None: columns = ['main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5'] flat_numbers = data_frame[columns].values.flatten() counts = Counter(flat_numbers) return [int(num) for num, _ in counts.most_common(top)] def get_least_common_number( data_frame: pd.DataFrame, columns: list | None = None, bottom: int = 1 ) -> list[int]: """ :param data_frame: a pandas DataFrame containing the data :param columns: a list of column names to use :param bottom: the number of bottom (least seen) numbers to return :returns: a list of the least common numbers """ if columns is None: columns = ['main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5'] flat_numbers = data_frame[columns].values.flatten() counts = Counter(flat_numbers) return list(reversed([int(num) for num, _ in counts.most_common()[-bottom:]])) def calculate_probabilities( data_frame: pd.DataFrame, max_number: int, columns: list | None = None ) -> dict[int, float]: """ :param data_frame: A pandas DataFrame containing the data to calculate probabilities for :param max_number: The maximum number possible in the data_frame :param columns: The list of column names to use from the data_frame :returns dict: A dictionary containing the probabilities """ if columns is None: columns = ['main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5'] if data_frame.empty: return dict() # get all the numbers in the groups all_numbers = [num for group in data_frame[columns].values for num in group] # count all the occurrences of each number counts = Counter(all_numbers) # calculate the basic probability of each number occurring again probabilities = {num: counts.get(num, 0) / (max_number + 1) for num in range(1, max_number + 1)} # if any calculation is greater than one, use what is to the right of the decimal point as the value for key, value in probabilities.items(): if value > 1: probabilities[key] = value - int(str(value).split('.')[0]) # return the probability dict return probabilities def get_hot_numbers(probabilities: dict[int, float], top: int = 5) -> list[tuple[int, float]]: """ :param probabilities: A dictionary containing the probabilities :param top: The count of hottest items to return, defaults to 5 :returns list of tuples: A list of the hot numbers and their raw score """ return sorted(probabilities.items(), key=lambda item: item[1], reverse=True)[:top] def get_cold_numbers(probabilities: dict[int, float], bottom: int = 5) -> list[tuple[int, float]]: """ :param probabilities: A dictionary containing the probabilities :param bottom: The count of coldest items to return, defaults to 5 :returns list of tuples: A list of the hot numbers and their raw score """ return sorted(probabilities.items(), key=lambda item: item[1])[:bottom] """ def extract_features(row): balls = [row['main_ball1'], row['main_ball2'], row['main_ball3'], row['main_ball4'], row['main_ball5']] pd.Series({ 'sum': sum(balls), 'mean': np.mean(balls), 'std': np.std(balls), 'even_count': len([b for b in balls if b % 2 == 0]), 'range': max(balls) - min(balls) }) # Melt the 5 ball columns into one long column of numbers melted = df.melt(id_vars=['draw_date'], value_vars=['ball1', 'ball2', 'ball3', 'ball4', 'ball5'], value_name='number') # Find the most recent date each number appeared last_seen = melted.groupby('number')['draw_date'].max() # calculate days overdue relative to today current_date = datetime.now() overdue_days = (current_date - last_seen).dt.days # sort to find the "most overdue" at the top most_overdue = overdue_days.sort_values(ascending=False) # print the 10 most overdue print(most_overdue.head(10)) game_ball_last_seen = df.groupby('powerball')['draw_date'].max() # or 'mega-ball' game_ball_overdue = (current_date - game_ball_last_seen).dt.days.soft_values(ascending=False) features = np.apply(extract_features, axis=1) y = np.ones(len(data_frame)) # train a model to recognize what a "winning" set looks like model = RandomForestRegressor(n_estimators=100) model.fit(features, y) # Sample: Generate combinations from the top 15 most frequent/overdue numbers target_numbers = most_overdue.head(15).index.tolist() candidate_sets = list(combinations(target_numbers, 2)) results = [] for s in candidate_sets: # Create features for this hypothetical set s_feat = extract_features({'main_ball1': s[0], 'main_ball2': s[1], 'main_ball3': s[2], 'main_ball4': s[3], 'main_ball5': s[4]}) # Model predicts how closely this matches historical 'winning' patterns score = model.predict(s_feat.values.reshape(1, -1))[0] results.append({'set': s, 'score': score}) # Rank by score ranked_sets = pd.DataFrame(results).sort_values(by='score', ascending=False) # Get frequency counts counts = melted['number'].value_counts(normalize=True) def calc_prob(s): return np.prod([counts.get(num, 0) for num in s]) ranked_sets['statistical_prob'] = ranked_sets['set'].apply(calc_prob) game_counts = df['game_ball'].value_counts(normalize=True) def calc_total_prob(row): # calculate prob for the 5 main balls five_ball_prob = np.prod([counts.get(num, 0) for num in row['set']]) # multiply the prob of the specific game ball game_prob = game_counts.get(row['game_ball'], 0) return five_ball_prob * game_prob ranked_sets['total_prob'] = ranked_sets.apply(calc_total_prob, axis=1) """