Compare commits

...
1 Commits
Author SHA1 Message Date
chris 66b5069858 Add AI code in comment form for review 2026-03-17 20:47:04 -04:00
+79 -5
View File
@@ -1,16 +1,15 @@
from datetime import datetime
from collections import Counter from collections import Counter
from datetime import datetime
from itertools import combinations
import numpy as np import numpy as np
import pandas as pd import pandas as pd
from sklearn.ensemble import RandomForestClassifier from sklearn.ensemble import RandomForestRegressor
from sqlalchemy import select from sqlalchemy import select
from data.database import Base, dal, project_variables from data.database import Base, dal
import util.drawing as drawutil import util.drawing as drawutil
RANDOM_SEED = 42
def load_dataframe( def load_dataframe(
table_name: str, start_date: datetime | None = None, end_date: datetime | None = None table_name: str, start_date: datetime | None = None, end_date: datetime | None = None
@@ -125,3 +124,78 @@ def get_cold_numbers(probabilities: dict[int, float], bottom: int = 5) -> list[t
:returns list of tuples: A list of the hot numbers and their raw score :returns list of tuples: A list of the hot numbers and their raw score
""" """
return sorted(probabilities.items(), key=lambda item: item[1])[:bottom] return sorted(probabilities.items(), key=lambda item: item[1])[:bottom]
"""
def extract_features(row):
balls = [row['main_ball1'], row['main_ball2'], row['main_ball3'], row['main_ball4'], row['main_ball5']]
pd.Series({
'sum': sum(balls),
'mean': np.mean(balls),
'std': np.std(balls),
'even_count': len([b for b in balls if b % 2 == 0]),
'range': max(balls) - min(balls)
})
# Melt the 5 ball columns into one long column of numbers
melted = df.melt(id_vars=['draw_date'], value_vars=['ball1', 'ball2', 'ball3', 'ball4', 'ball5'], value_name='number')
# Find the most recent date each number appeared
last_seen = melted.groupby('number')['draw_date'].max()
# calculate days overdue relative to today
current_date = datetime.now()
overdue_days = (current_date - last_seen).dt.days
# sort to find the "most overdue" at the top
most_overdue = overdue_days.sort_values(ascending=False)
# print the 10 most overdue
print(most_overdue.head(10))
game_ball_last_seen = df.groupby('powerball')['draw_date'].max() # or 'mega-ball'
game_ball_overdue = (current_date - game_ball_last_seen).dt.days.soft_values(ascending=False)
features = np.apply(extract_features, axis=1)
y = np.ones(len(data_frame))
# train a model to recognize what a "winning" set looks like
model = RandomForestRegressor(n_estimators=100)
model.fit(features, y)
# Sample: Generate combinations from the top 15 most frequent/overdue numbers
target_numbers = most_overdue.head(15).index.tolist()
candidate_sets = list(combinations(target_numbers, 2))
results = []
for s in candidate_sets:
# Create features for this hypothetical set
s_feat = extract_features({'main_ball1': s[0], 'main_ball2': s[1], 'main_ball3': s[2], 'main_ball4': s[3],
'main_ball5': s[4]})
# Model predicts how closely this matches historical 'winning' patterns
score = model.predict(s_feat.values.reshape(1, -1))[0]
results.append({'set': s, 'score': score})
# Rank by score
ranked_sets = pd.DataFrame(results).sort_values(by='score', ascending=False)
# Get frequency counts
counts = melted['number'].value_counts(normalize=True)
def calc_prob(s):
return np.prod([counts.get(num, 0) for num in s])
ranked_sets['statistical_prob'] = ranked_sets['set'].apply(calc_prob)
game_counts = df['game_ball'].value_counts(normalize=True)
def calc_total_prob(row):
# calculate prob for the 5 main balls
five_ball_prob = np.prod([counts.get(num, 0) for num in row['set']])
# multiply the prob of the specific game ball
game_prob = game_counts.get(row['game_ball'], 0)
return five_ball_prob * game_prob
ranked_sets['total_prob'] = ranked_sets.apply(calc_total_prob, axis=1)
"""