202 lines
7.5 KiB
Python
202 lines
7.5 KiB
Python
from collections import Counter
|
|
from datetime import datetime
|
|
from itertools import combinations
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
from sklearn.ensemble import RandomForestRegressor
|
|
from sqlalchemy import select
|
|
|
|
from data.database import Base, dal
|
|
import util.drawing as drawutil
|
|
|
|
|
|
def load_dataframe(
|
|
table_name: str, start_date: datetime | None = None, end_date: datetime | None = None
|
|
) -> pd.DataFrame:
|
|
"""
|
|
:param table_name: the name of the table to load data from
|
|
:param start_date: the start date for the data
|
|
:param end_date: the end date for the data
|
|
|
|
:returns: a pandas DataFrame containing the data
|
|
"""
|
|
|
|
dal.connect()
|
|
session = dal.Session()
|
|
|
|
game_dict = drawutil.check_table_name(table_name=table_name)
|
|
if game_dict:
|
|
target_table = Base.metadata.tables.get(game_dict['db_table_name'])
|
|
if start_date is not None and end_date is not None:
|
|
sql_statement = (select(target_table)
|
|
.where(target_table.columns.draw_date >= start_date)
|
|
.where(target_table.columns.draw_date <= end_date))
|
|
elif start_date is not None and end_date is None:
|
|
sql_statement = (select(target_table).where(target_table.columns.draw_date >= start_date))
|
|
elif start_date is None and end_date is not None:
|
|
sql_statement = (select(target_table).where(target_table.columns.draw_date <= end_date))
|
|
else:
|
|
sql_statement = select(target_table)
|
|
|
|
return pd.read_sql(sql_statement, session.bind)
|
|
else:
|
|
raise ValueError('An invalid table name was provided')
|
|
|
|
|
|
def get_most_common_number(
|
|
data_frame: pd.DataFrame, columns: list | None = None, top: int = 1
|
|
) -> list[int]:
|
|
"""
|
|
:param data_frame: a pandas DataFrame containing the data
|
|
:param columns: a list of column names to use
|
|
:param top: the number of top (most seen) numbers to return
|
|
|
|
:returns: a list of the most common numbers
|
|
"""
|
|
if columns is None:
|
|
columns = ['main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5']
|
|
flat_numbers = data_frame[columns].values.flatten()
|
|
counts = Counter(flat_numbers)
|
|
return [int(num) for num, _ in counts.most_common(top)]
|
|
|
|
|
|
def get_least_common_number(
|
|
data_frame: pd.DataFrame, columns: list | None = None, bottom: int = 1
|
|
) -> list[int]:
|
|
"""
|
|
:param data_frame: a pandas DataFrame containing the data
|
|
:param columns: a list of column names to use
|
|
:param bottom: the number of bottom (least seen) numbers to return
|
|
|
|
:returns: a list of the least common numbers
|
|
"""
|
|
if columns is None:
|
|
columns = ['main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5']
|
|
flat_numbers = data_frame[columns].values.flatten()
|
|
counts = Counter(flat_numbers)
|
|
return list(reversed([int(num) for num, _ in counts.most_common()[-bottom:]]))
|
|
|
|
|
|
def calculate_probabilities(
|
|
data_frame: pd.DataFrame, max_number: int, columns: list | None = None
|
|
) -> dict[int, float]:
|
|
"""
|
|
:param data_frame: A pandas DataFrame containing the data to calculate probabilities for
|
|
:param max_number: The maximum number possible in the data_frame
|
|
:param columns: The list of column names to use from the data_frame
|
|
:returns dict: A dictionary containing the probabilities
|
|
"""
|
|
if columns is None:
|
|
columns = ['main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5']
|
|
if data_frame.empty:
|
|
return dict()
|
|
|
|
# get all the numbers in the groups
|
|
all_numbers = [num for group in data_frame[columns].values for num in group]
|
|
# count all the occurrences of each number
|
|
counts = Counter(all_numbers)
|
|
# calculate the basic probability of each number occurring again
|
|
probabilities = {num: counts.get(num, 0) / (max_number + 1) for num in range(1, max_number + 1)}
|
|
|
|
# if any calculation is greater than one, use what is to the right of the decimal point as the value
|
|
for key, value in probabilities.items():
|
|
if value > 1:
|
|
probabilities[key] = value - int(str(value).split('.')[0])
|
|
|
|
# return the probability dict
|
|
return probabilities
|
|
|
|
|
|
def get_hot_numbers(probabilities: dict[int, float], top: int = 5) -> list[tuple[int, float]]:
|
|
"""
|
|
:param probabilities: A dictionary containing the probabilities
|
|
:param top: The count of hottest items to return, defaults to 5
|
|
:returns list of tuples: A list of the hot numbers and their raw score
|
|
"""
|
|
return sorted(probabilities.items(), key=lambda item: item[1], reverse=True)[:top]
|
|
|
|
|
|
def get_cold_numbers(probabilities: dict[int, float], bottom: int = 5) -> list[tuple[int, float]]:
|
|
"""
|
|
:param probabilities: A dictionary containing the probabilities
|
|
:param bottom: The count of coldest items to return, defaults to 5
|
|
:returns list of tuples: A list of the hot numbers and their raw score
|
|
"""
|
|
return sorted(probabilities.items(), key=lambda item: item[1])[:bottom]
|
|
|
|
|
|
"""
|
|
def extract_features(row):
|
|
balls = [row['main_ball1'], row['main_ball2'], row['main_ball3'], row['main_ball4'], row['main_ball5']]
|
|
pd.Series({
|
|
'sum': sum(balls),
|
|
'mean': np.mean(balls),
|
|
'std': np.std(balls),
|
|
'even_count': len([b for b in balls if b % 2 == 0]),
|
|
'range': max(balls) - min(balls)
|
|
})
|
|
|
|
# Melt the 5 ball columns into one long column of numbers
|
|
melted = df.melt(id_vars=['draw_date'], value_vars=['ball1', 'ball2', 'ball3', 'ball4', 'ball5'], value_name='number')
|
|
|
|
# Find the most recent date each number appeared
|
|
last_seen = melted.groupby('number')['draw_date'].max()
|
|
|
|
# calculate days overdue relative to today
|
|
current_date = datetime.now()
|
|
overdue_days = (current_date - last_seen).dt.days
|
|
|
|
# sort to find the "most overdue" at the top
|
|
most_overdue = overdue_days.sort_values(ascending=False)
|
|
|
|
# print the 10 most overdue
|
|
print(most_overdue.head(10))
|
|
|
|
game_ball_last_seen = df.groupby('powerball')['draw_date'].max() # or 'mega-ball'
|
|
game_ball_overdue = (current_date - game_ball_last_seen).dt.days.soft_values(ascending=False)
|
|
|
|
features = np.apply(extract_features, axis=1)
|
|
y = np.ones(len(data_frame))
|
|
# train a model to recognize what a "winning" set looks like
|
|
model = RandomForestRegressor(n_estimators=100)
|
|
model.fit(features, y)
|
|
|
|
# Sample: Generate combinations from the top 15 most frequent/overdue numbers
|
|
target_numbers = most_overdue.head(15).index.tolist()
|
|
candidate_sets = list(combinations(target_numbers, 2))
|
|
|
|
results = []
|
|
for s in candidate_sets:
|
|
# Create features for this hypothetical set
|
|
s_feat = extract_features({'main_ball1': s[0], 'main_ball2': s[1], 'main_ball3': s[2], 'main_ball4': s[3],
|
|
'main_ball5': s[4]})
|
|
|
|
# Model predicts how closely this matches historical 'winning' patterns
|
|
score = model.predict(s_feat.values.reshape(1, -1))[0]
|
|
results.append({'set': s, 'score': score})
|
|
|
|
# Rank by score
|
|
ranked_sets = pd.DataFrame(results).sort_values(by='score', ascending=False)
|
|
|
|
# Get frequency counts
|
|
counts = melted['number'].value_counts(normalize=True)
|
|
|
|
def calc_prob(s):
|
|
return np.prod([counts.get(num, 0) for num in s])
|
|
|
|
ranked_sets['statistical_prob'] = ranked_sets['set'].apply(calc_prob)
|
|
|
|
game_counts = df['game_ball'].value_counts(normalize=True)
|
|
|
|
def calc_total_prob(row):
|
|
# calculate prob for the 5 main balls
|
|
five_ball_prob = np.prod([counts.get(num, 0) for num in row['set']])
|
|
# multiply the prob of the specific game ball
|
|
game_prob = game_counts.get(row['game_ball'], 0)
|
|
return five_ball_prob * game_prob
|
|
|
|
ranked_sets['total_prob'] = ranked_sets.apply(calc_total_prob, axis=1)
|
|
|
|
"""
|