From 66b50698582feaa0f95f454821e5ee6fc1ccfdb7 Mon Sep 17 00:00:00 2001 From: Chris Smith Date: Tue, 17 Mar 2026 20:47:04 -0400 Subject: [PATCH] Add AI code in comment form for review --- lottery_predictor/analyze.py | 84 +++++++++++++++++++++++++++++++++--- 1 file changed, 79 insertions(+), 5 deletions(-) diff --git a/lottery_predictor/analyze.py b/lottery_predictor/analyze.py index 76bf85f..b1759e2 100644 --- a/lottery_predictor/analyze.py +++ b/lottery_predictor/analyze.py @@ -1,16 +1,15 @@ -from datetime import datetime from collections import Counter +from datetime import datetime +from itertools import combinations import numpy as np import pandas as pd -from sklearn.ensemble import RandomForestClassifier +from sklearn.ensemble import RandomForestRegressor from sqlalchemy import select -from data.database import Base, dal, project_variables +from data.database import Base, dal import util.drawing as drawutil -RANDOM_SEED = 42 - def load_dataframe( table_name: str, start_date: datetime | None = None, end_date: datetime | None = None @@ -125,3 +124,78 @@ def get_cold_numbers(probabilities: dict[int, float], bottom: int = 5) -> list[t :returns list of tuples: A list of the hot numbers and their raw score """ return sorted(probabilities.items(), key=lambda item: item[1])[:bottom] + + +""" +def extract_features(row): + balls = [row['main_ball1'], row['main_ball2'], row['main_ball3'], row['main_ball4'], row['main_ball5']] + pd.Series({ + 'sum': sum(balls), + 'mean': np.mean(balls), + 'std': np.std(balls), + 'even_count': len([b for b in balls if b % 2 == 0]), + 'range': max(balls) - min(balls) + }) + +# Melt the 5 ball columns into one long column of numbers +melted = df.melt(id_vars=['draw_date'], value_vars=['ball1', 'ball2', 'ball3', 'ball4', 'ball5'], value_name='number') + +# Find the most recent date each number appeared +last_seen = melted.groupby('number')['draw_date'].max() + +# calculate days overdue relative to today +current_date = datetime.now() +overdue_days = (current_date - last_seen).dt.days + +# sort to find the "most overdue" at the top +most_overdue = overdue_days.sort_values(ascending=False) + +# print the 10 most overdue +print(most_overdue.head(10)) + +game_ball_last_seen = df.groupby('powerball')['draw_date'].max() # or 'mega-ball' +game_ball_overdue = (current_date - game_ball_last_seen).dt.days.soft_values(ascending=False) + +features = np.apply(extract_features, axis=1) +y = np.ones(len(data_frame)) +# train a model to recognize what a "winning" set looks like +model = RandomForestRegressor(n_estimators=100) +model.fit(features, y) + +# Sample: Generate combinations from the top 15 most frequent/overdue numbers +target_numbers = most_overdue.head(15).index.tolist() +candidate_sets = list(combinations(target_numbers, 2)) + +results = [] +for s in candidate_sets: + # Create features for this hypothetical set + s_feat = extract_features({'main_ball1': s[0], 'main_ball2': s[1], 'main_ball3': s[2], 'main_ball4': s[3], + 'main_ball5': s[4]}) + + # Model predicts how closely this matches historical 'winning' patterns + score = model.predict(s_feat.values.reshape(1, -1))[0] + results.append({'set': s, 'score': score}) + +# Rank by score +ranked_sets = pd.DataFrame(results).sort_values(by='score', ascending=False) + +# Get frequency counts +counts = melted['number'].value_counts(normalize=True) + +def calc_prob(s): + return np.prod([counts.get(num, 0) for num in s]) + +ranked_sets['statistical_prob'] = ranked_sets['set'].apply(calc_prob) + +game_counts = df['game_ball'].value_counts(normalize=True) + +def calc_total_prob(row): + # calculate prob for the 5 main balls + five_ball_prob = np.prod([counts.get(num, 0) for num in row['set']]) + # multiply the prob of the specific game ball + game_prob = game_counts.get(row['game_ball'], 0) + return five_ball_prob * game_prob + +ranked_sets['total_prob'] = ranked_sets.apply(calc_total_prob, axis=1) + +"""