Compare commits

..
1 Commits
Author SHA1 Message Date
chris 66b5069858 Add AI code in comment form for review 2026-03-17 20:47:04 -04:00
6 changed files with 85 additions and 11 deletions

No files matched your search

+79 -5
View File
@@ -1,16 +1,15 @@
from datetime import datetime
from collections import Counter from collections import Counter
from datetime import datetime
from itertools import combinations
import numpy as np import numpy as np
import pandas as pd import pandas as pd
from sklearn.ensemble import RandomForestClassifier from sklearn.ensemble import RandomForestRegressor
from sqlalchemy import select from sqlalchemy import select
from data.database import Base, dal, project_variables from data.database import Base, dal
import util.drawing as drawutil import util.drawing as drawutil
RANDOM_SEED = 42
def load_dataframe( def load_dataframe(
table_name: str, start_date: datetime | None = None, end_date: datetime | None = None table_name: str, start_date: datetime | None = None, end_date: datetime | None = None
@@ -125,3 +124,78 @@ def get_cold_numbers(probabilities: dict[int, float], bottom: int = 5) -> list[t
:returns list of tuples: A list of the hot numbers and their raw score :returns list of tuples: A list of the hot numbers and their raw score
""" """
return sorted(probabilities.items(), key=lambda item: item[1])[:bottom] return sorted(probabilities.items(), key=lambda item: item[1])[:bottom]
"""
def extract_features(row):
balls = [row['main_ball1'], row['main_ball2'], row['main_ball3'], row['main_ball4'], row['main_ball5']]
pd.Series({
'sum': sum(balls),
'mean': np.mean(balls),
'std': np.std(balls),
'even_count': len([b for b in balls if b % 2 == 0]),
'range': max(balls) - min(balls)
})
# Melt the 5 ball columns into one long column of numbers
melted = df.melt(id_vars=['draw_date'], value_vars=['ball1', 'ball2', 'ball3', 'ball4', 'ball5'], value_name='number')
# Find the most recent date each number appeared
last_seen = melted.groupby('number')['draw_date'].max()
# calculate days overdue relative to today
current_date = datetime.now()
overdue_days = (current_date - last_seen).dt.days
# sort to find the "most overdue" at the top
most_overdue = overdue_days.sort_values(ascending=False)
# print the 10 most overdue
print(most_overdue.head(10))
game_ball_last_seen = df.groupby('powerball')['draw_date'].max() # or 'mega-ball'
game_ball_overdue = (current_date - game_ball_last_seen).dt.days.soft_values(ascending=False)
features = np.apply(extract_features, axis=1)
y = np.ones(len(data_frame))
# train a model to recognize what a "winning" set looks like
model = RandomForestRegressor(n_estimators=100)
model.fit(features, y)
# Sample: Generate combinations from the top 15 most frequent/overdue numbers
target_numbers = most_overdue.head(15).index.tolist()
candidate_sets = list(combinations(target_numbers, 2))
results = []
for s in candidate_sets:
# Create features for this hypothetical set
s_feat = extract_features({'main_ball1': s[0], 'main_ball2': s[1], 'main_ball3': s[2], 'main_ball4': s[3],
'main_ball5': s[4]})
# Model predicts how closely this matches historical 'winning' patterns
score = model.predict(s_feat.values.reshape(1, -1))[0]
results.append({'set': s, 'score': score})
# Rank by score
ranked_sets = pd.DataFrame(results).sort_values(by='score', ascending=False)
# Get frequency counts
counts = melted['number'].value_counts(normalize=True)
def calc_prob(s):
return np.prod([counts.get(num, 0) for num in s])
ranked_sets['statistical_prob'] = ranked_sets['set'].apply(calc_prob)
game_counts = df['game_ball'].value_counts(normalize=True)
def calc_total_prob(row):
# calculate prob for the 5 main balls
five_ball_prob = np.prod([counts.get(num, 0) for num in row['set']])
# multiply the prob of the specific game ball
game_prob = game_counts.get(row['game_ball'], 0)
return five_ball_prob * game_prob
ranked_sets['total_prob'] = ranked_sets.apply(calc_total_prob, axis=1)
"""
+2 -2
View File
@@ -11,6 +11,6 @@ def generate_random_ticket(max_main: int, max_game: int, num_main: int = 5) -> t
:returns: A tuple of (main_balls, game_ball) :returns: A tuple of (main_balls, game_ball)
""" """
main_balls = random.sample(range(1, max_main + 1), num_main) main_balls = random.sample(range(1, max_main), num_main)
game_ball = random.randint(1, max_game + 1) game_ball = random.randint(1, max_game)
return sorted(main_balls), game_ball return sorted(main_balls), game_ball
Executable → Regular
BIN
View File
Binary file not shown.
+2 -2
View File
@@ -9,9 +9,9 @@ def test_load_dataframe() -> None:
mm_start = datetime(2025, 4, 5) mm_start = datetime(2025, 4, 5)
date_end = datetime(2026, 3, 8) date_end = datetime(2026, 3, 8)
assert int(analyze.load_dataframe( assert int(analyze.load_dataframe(
table_name='PowerballDraw', start_date=pb_start, end_date=date_end).count()['draw_date']) >= 1324 table_name='PowerballDraw', start_date=pb_start, end_date=date_end).count()['draw_date']) == 1324
assert int(analyze.load_dataframe( assert int(analyze.load_dataframe(
table_name='MegaMillionsDraw', start_date=mm_start, end_date=date_end).count()['draw_date']) >= 96 table_name='MegaMillionsDraw', start_date=mm_start, end_date=date_end).count()['draw_date']) == 96
with pytest.raises(ValueError): with pytest.raises(ValueError):
analyze.load_dataframe(table_name='SomNonExistentTableName') analyze.load_dataframe(table_name='SomNonExistentTableName')
Executable → Regular
+1 -1
View File
@@ -144,4 +144,4 @@ def test_load_environment_variables() -> None:
"""test loading environment variables for script use""" """test loading environment variables for script use"""
env_vars = ue.load_environment_variables() env_vars = ue.load_environment_variables()
assert type(env_vars) == dict assert type(env_vars) == dict
assert len(env_vars) == 25 assert len(env_vars) == 24
Executable → Regular
+1 -1
View File
@@ -1,7 +1,7 @@
""" """
util/scrape.py util/scrape.py
Utility functions for scraping web data relative to selected games Utility functions for scaping web data relative to selected games
""" """
import httpx import httpx
from dateutil import parser, utils from dateutil import parser, utils