Files
Lottery_Project/lottery_predictor/analyze.py
T

237 lines
8.3 KiB
Python

import random
from collections import Counter
from datetime import datetime
import numpy as np
import pandas as pd
from sklearn.ensemble import RandomForestRegressor
from sqlalchemy import select
from data.database import Base, dal
from lottery_predictor.config import GAME_INFO
def load_dataframe_by_dates(
game: str, start_date: datetime | None = None,
end_date: datetime | None = None,
) -> pd.DataFrame:
"""
:param game: the name of the game to get data for
:param start_date: the start date for the data
:param end_date: the end date for the data
:returns: a pandas DataFrame containing the data
"""
dal.connect()
session = dal.Session()
game_dict = GAME_INFO[game]
if game_dict:
target_table = Base.metadata.tables.get(game_dict['db_table_name'])
if start_date is not None and end_date is not None:
sql_statement = (select(target_table)
.where(
target_table.columns.draw_date >= start_date,
)
.where(target_table.columns.draw_date <= end_date)
.order_by(target_table.columns.draw_date.desc()))
elif start_date is not None and end_date is None:
sql_statement = (select(target_table)
.where(
target_table.columns.draw_date >= start_date,
)
.order_by(target_table.columns.draw_date.desc()))
elif start_date is None and end_date is not None:
sql_statement = (select(target_table)
.where(target_table.columns.draw_date <= end_date)
.order_by(target_table.columns.draw_date.desc()))
else:
sql_statement = select(target_table)
return pd.read_sql(sql_statement, session.bind)
else:
raise ValueError('An invalid table name was provided')
def load_dataframe_most_recent(game: str, limit: int = 10) -> pd.DataFrame:
"""
:param game: the name of the game to get data for
:param limit: The N most recent draws (default: 10)
:returns: a pandas DataFrame containing the data
"""
dal.connect()
session = dal.Session()
game_dict = GAME_INFO[game]
if game_dict:
target_table = Base.metadata.tables.get(game_dict['db_table_name'])
if limit is not None and limit > 0:
sql_statement = (select(target_table)
.order_by(target_table.columns.draw_date.desc())
.limit(limit))
return pd.read_sql(sql_statement, session.bind)
else:
raise ValueError(
'Limit must be a positive integer greater than zero.',
)
else:
raise ValueError('An invalid table name was provided.')
def prepare_split_data(data: np.ndarray, window_size: int = 10) -> tuple[
np.ndarray, np.ndarray, np.ndarray]:
# Clean the data by removing the draw_date and multiplier columns
clean_data = data[:, 1:7]
# Check to be sure there is enough data for the window_size
if len(clean_data) <= window_size:
raise ValueError(
f"Not enough data! Dataset has {len(clean_data)} rows, "
f"but window_size requires at least {window_size + 1} rows.",
)
# Calculate the indices for all windows at once
indices = np.arange(len(clean_data) - window_size)
# Create X: flattened sliding windows
x = np.array([clean_data[i: i + window_size].flatten() for i in indices])
# Create y_field: first 5 columns (2D array)
y_field = clean_data[window_size:, 0:5]
# Create y_game on the last column ensuring it is a 2D array
y_game = clean_data[window_size:, 5].ravel()
return x, y_field, y_game
def make_prediction(data_frame: pd.DataFrame, window_size: int = 10) -> tuple[
np.ndarray, np.ndarray]:
# Get a random number
state = random.randint(1000, 300_000)
# Convert the data_frame to a numpy array for slicing
data = data_frame.values
# Prepare and split the data
x, y_field, y_game = prepare_split_data(data, window_size=window_size)
# Train the model for the field balls
field_model = RandomForestRegressor(n_estimators=200, random_state=state)
field_model.fit(x, y_field)
# Train the model for the game ball
game_model = RandomForestRegressor(n_estimators=200, random_state=state)
game_model.fit(x, y_game)
# Predict the next draw
clean_data = data[:, 1:7]
current_window = clean_data[-window_size:].flatten().reshape(1, -1)
# Get predictions and round to the nearest whole number
predicted_field = np.sort(
np.round(field_model.predict(current_window)).astype(int),
)
predicted_game = np.round(game_model.predict(current_window)).astype(int)
return predicted_field[0], predicted_game[0]
def get_most_common_number(
data_frame: pd.DataFrame, columns: list | None = None, top: int = 1,
) -> list[int]:
"""
:param data_frame: a pandas DataFrame containing the data
:param columns: a list of column names to use
:param top: the number of top (most seen) numbers to return
:returns: a list of the most common numbers
"""
if columns is None:
columns = [
'main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5',
]
flat_numbers = data_frame[columns].values.flatten()
counts = Counter(flat_numbers)
return [int(num) for num, _ in counts.most_common(top)]
def get_least_common_number(
data_frame: pd.DataFrame, columns: list | None = None, bottom: int = 1,
) -> list[int]:
"""
:param data_frame: a pandas DataFrame containing the data
:param columns: a list of column names to use
:param bottom: the number of bottom (least seen) numbers to return
:returns: a list of the least common numbers
"""
if columns is None:
columns = [
'main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5',
]
flat_numbers = data_frame[columns].values.flatten()
counts = Counter(flat_numbers)
return list(
reversed([int(num) for num, _ in counts.most_common()[-bottom:]]),
)
def calculate_probabilities(
data_frame: pd.DataFrame, max_number: int, columns: list | None = None,
) -> dict[int, float]:
"""
:param data_frame: A pandas DataFrame containing the data to calculate
probabilities for
:param max_number: The maximum number possible in the data_frame
:param columns: The list of column names to use from the data_frame
:returns dict: A dictionary containing the probabilities
"""
if columns is None:
columns = [
'main_ball1', 'main_ball2', 'main_ball3', 'main_ball4', 'main_ball5',
]
if data_frame.empty:
return dict()
# get all the numbers in the groups
all_numbers = [num for group in data_frame[columns].values for num in group]
# count all the occurrences of each number
counts = Counter(all_numbers)
# calculate the basic probability of each number occurring again
probabilities = {num: counts.get(num, 0) / (max_number + 1) for num in
range(1, max_number + 1)}
# if any calculation is greater than one, use what is to the right of the
# decimal point as the value
for key, value in probabilities.items():
if value > 1:
probabilities[key] = value - int(str(value).split('.')[0])
# return the probability dict
return probabilities
def get_hot_numbers(probabilities: dict[int, float], top: int = 5) -> list[
tuple[int, float]]:
"""
:param probabilities: A dictionary containing the probabilities
:param top: The count of hottest items to return, defaults to 5
:returns list of tuples: A list of the hot numbers and their raw score
"""
return sorted(
probabilities.items(), key=lambda item: item[1], reverse=True,
)[:top]
def get_cold_numbers(probabilities: dict[int, float], bottom: int = 5) -> list[
tuple[int, float]]:
"""
:param probabilities: A dictionary containing the probabilities
:param bottom: The count of coldest items to return, defaults to 5
:returns list of tuples: A list of the hot numbers and their raw score
"""
return sorted(probabilities.items(), key=lambda item: item[1])[:bottom]