diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..a15f8ec --- /dev/null +++ b/.gitignore @@ -0,0 +1,24 @@ +# Python +__pycache__/ +*.py[cod] +.venv/ +venv/ +env/ + +# Output from the scripts +*_ans.csv +*.log + +# Secrets and local config +.env +.env.* +*.pem +*.key +secrets.py + +# Editor junk +.vscode/ +.idea/ +*.swp +*~ +.DS_Store diff --git a/README.md b/README.md index d252488..9df2600 100644 --- a/README.md +++ b/README.md @@ -8,9 +8,15 @@ Previously there were records dating back to 1992 that were removed by the websi I am welcoming anyone who wants to to port it to their own state/country to do so. I currently have a website www.wvlotterypredictor.xyz hosting my numbers to be refreshed daily. Collaboration and suggestion to a more accurate algorithm is also welcome. I can be contacted about the project through my site email, on here, or on the Twitter linked on my site. +# How the site works +- Each game has its own script. It pulls every past draw from the WV Lottery results api (the same one wvlottery.com uses for its past draws page), adds the older records from the excel files where there are any, and counts how often each number has been drawn. +- On the site the scripts write their numbers to a csv instead of printing them. cron runs each one about 45 minutes after that game's drawing so the new numbers are already posted. +- The site itself is Django served by gunicorn behind nginx. It reads the csv files for the home page and caches the page in redis, which gets cleared every hour. + # Good luck! # CHANGELOG +- September 26th 2026: wvlottery.com was rebuilt and the old history tables are gone, so every script now pulls past draws from the json results api the new site uses. Added Cash Pop. Updated Mega Millions odds for the 2025 changes. - June 29th 2022: Added support for old data that was previously scrubbed from the lottery site. It is more accurate and includes more data than the previous one. Chance for daily3 is now bugged however due to the large amount of numbers and dates. This will be fixed soon. # TODO: diff --git a/cash25.py b/cash25.py index 1d47273..bd2be46 100755 --- a/cash25.py +++ b/cash25.py @@ -3,7 +3,6 @@ from fractions import Fraction as frac import pandas as pd import requests from collections import Counter -from io import StringIO def get_data(): @@ -11,15 +10,23 @@ def get_data(): df_old = pd.read_excel('./excel_lotto_records/cash25.xlsx') salvaged = df_old[['Date', 'Numbers']] - url = 'https://wvlottery.com/draw-games/cash-25/?game-analyze=cash-25&what-to-search=historysearch&date-range=-1' + url = 'https://gateway.loyalty.wvlottery.com/services/jackpot/api/v1/jackpot-results?gameId=10&jackpotStatus=PAYABLE&size=500&sort=externalId,drawDate,desc&page=' header = { "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.75 Safari/537.36", - "X-Requested-With": "XMLHttpRequest" } - - # Get web data - r = requests.get(url, headers=header) - dfs = list(pd.read_html(StringIO(r.text))) + # Get web data, the new site pulls past draws from json 500 at a time + web = {'Date': [], 'Numbers': []} + page = 0 + while True: + r = requests.get(url + str(page), headers=header) + data = r.json() + for x in data['content']: + web['Date'].append(x['drawingDate']) + web['Numbers'].append('–'.join(str(n['data']) for n in x['resultData'] if n['type'] == 'REGULAR')) + if data['last']: + break + page += 1 + dfs = [pd.DataFrame(web)] pd.set_option('display.max_rows', None) # Specifies no max rows, otherwise only shows 10 records df = pd.concat([dfs[0], salvaged], ignore_index=True) diff --git a/cashpop.py b/cashpop.py new file mode 100644 index 0000000..5ea6ba2 --- /dev/null +++ b/cashpop.py @@ -0,0 +1,48 @@ +# WVLottopy: Matteo DiBiagio +from fractions import Fraction as frac +import pandas as pd +import requests +from collections import Counter + +def get_data(): + url = 'https://gateway.loyalty.wvlottery.com/services/jackpot/api/v1/jackpot-results?gameId=24&jackpotStatus=PAYABLE&size=500&sort=externalId,drawDate,desc&page=' + header = { + "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.75 Safari/537.36", + } + # Get web data, cash pop draws every 15 min so only use the last ~3000 draws (about a month) + web = {'Date': [], 'Numbers': []} + page = 0 + while page < 6: + r = requests.get(url + str(page), headers=header) + data = r.json() + for x in data['content']: + web['Date'].append(x['drawingDate']) + web['Numbers'].append('–'.join(str(n['data']) for n in x['resultData'] if n['type'] == 'REGULAR')) + if data['last']: + break + page += 1 + df2 = pd.DataFrame(web) + date = list(df2['Date']) + nums = list(df2['Numbers'].astype('str')) + return date, nums + +date, nums = get_data() + +# Formatting +hyphenfree = [] +for x in nums: + hyphenfree.append(x.replace('–',', ')) +splitlist = ", ".join(hyphenfree) +sep = splitlist.split(", ") + +for n in range(0, 15): + most_common= Counter(sep).most_common(3) + likely_nums = [v[0] for v in most_common] + frequency = [v[-1] for v in most_common] + +Forecast = likely_nums[0] +Backups = str(" - ".join(likely_nums[1:])) +Chance = frac(1, 15) # Only one number is drawn out of 15 so every pick has the same chance + +print(f"Likely number is . . . {Forecast} Backups: {Backups}\n" +f"With percent chance of winning being {Chance}") \ No newline at end of file diff --git a/daily3.py b/daily3.py index b964ba1..03ee4c2 100755 --- a/daily3.py +++ b/daily3.py @@ -3,18 +3,26 @@ from fractions import Fraction as frac import pandas as pd import requests from collections import Counter -from io import StringIO def get_data(): - url = 'https://wvlottery.com/draw-games/daily-3/?game-analyze=daily-3&what-to-search=historysearch&date-range=-1' + url = 'https://gateway.loyalty.wvlottery.com/services/jackpot/api/v1/jackpot-results?gameId=9&jackpotStatus=PAYABLE&size=500&sort=externalId,drawDate,desc&page=' header = { "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.75 Safari/537.36", - "X-Requested-With": "XMLHttpRequest" } - # Get web data - r = requests.get(url, headers=header) - dfs = list(pd.read_html(StringIO(r.text))) + # Get web data, the new site pulls past draws from json 500 at a time + web = {'Date': [], 'Numbers': []} + page = 0 + while True: + r = requests.get(url + str(page), headers=header) + data = r.json() + for x in data['content']: + web['Date'].append(x['drawingDate']) + web['Numbers'].append('–'.join(str(n['data']) for n in x['resultData'] if n['type'] == 'REGULAR')) + if data['last']: + break + page += 1 + dfs = [pd.DataFrame(web)] pd.set_option('display.max_rows', None) # Specifies no max rows, otherwise only shows 10 records df = dfs[0] diff --git a/daily4.py b/daily4.py index ea8ec3d..db7d631 100755 --- a/daily4.py +++ b/daily4.py @@ -3,7 +3,6 @@ from fractions import Fraction as frac import pandas as pd import requests from collections import Counter -from io import StringIO def get_data(): @@ -11,14 +10,23 @@ def get_data(): df_old = pd.read_excel('./excel_lotto_records/daily4.xlsx') salvaged = df_old[['Date', 'Numbers']] - url = 'https://wvlottery.com/draw-games/daily-4/?game-analyze=daily-4&what-to-search=historysearch&date-range=-1' + url = 'https://gateway.loyalty.wvlottery.com/services/jackpot/api/v1/jackpot-results?gameId=14&jackpotStatus=PAYABLE&size=500&sort=externalId,drawDate,desc&page=' header = { "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.75 Safari/537.36", - "X-Requested-With": "XMLHttpRequest" } - # Get web data - r = requests.get(url, headers=header) - dfs = list(pd.read_html(StringIO(r.text))) + # Get web data, the new site pulls past draws from json 500 at a time + web = {'Date': [], 'Numbers': []} + page = 0 + while True: + r = requests.get(url + str(page), headers=header) + data = r.json() + for x in data['content']: + web['Date'].append(x['drawingDate']) + web['Numbers'].append('–'.join(str(n['data']) for n in x['resultData'] if n['type'] == 'REGULAR')) + if data['last']: + break + page += 1 + dfs = [pd.DataFrame(web)] pd.set_option('display.max_rows', None) # Specifies no max rows, otherwise only shows 10 records df = pd.concat([dfs[0], salvaged], ignore_index=True) diff --git a/lotto_america.py b/lotto_america.py index 746fff6..0ea6d18 100755 --- a/lotto_america.py +++ b/lotto_america.py @@ -3,26 +3,34 @@ from fractions import Fraction as frac import pandas as pd import requests from collections import Counter -from io import StringIO def get_data(): df_old = pd.read_excel('./excel_lotto_records/lotto_america.xlsx') - salvaged = df_old[['Date', 'Numbers', 'SB', 'All Star']] + salvaged = df_old[['Date', 'Numbers', 'SB']] - url = 'https://wvlottery.com/draw-games/lotto-america/?game-analyze=lotto-america&what-to-search=historysearch&date-range=-1' + url = 'https://gateway.loyalty.wvlottery.com/services/jackpot/api/v1/jackpot-results?gameId=16&jackpotStatus=PAYABLE&size=500&sort=externalId,drawDate,desc&page=' header = { "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.75 Safari/537.36", - "X-Requested-With": "XMLHttpRequest" } - - # Get web data - r = requests.get(url, headers=header) - dfs = list(pd.read_html(StringIO(r.text))) + # Get web data, the new site pulls past draws from json 500 at a time + web = {'Date': [], 'Numbers': [], 'SB': []} + page = 0 + while True: + r = requests.get(url + str(page), headers=header) + data = r.json() + for x in data['content']: + web['Date'].append(x['drawingDate']) + web['Numbers'].append('–'.join(str(n['data']) for n in x['resultData'] if n['type'] == 'REGULAR')) + web['SB'].append([n['data'] for n in x['resultData'] if n['type'] == 'SPECIAL'][0]) + if data['last']: + break + page += 1 + dfs = [pd.DataFrame(web)] pd.set_option('display.max_rows', None) # Specifies no max rows, otherwise only shows 10 records df = pd.concat([dfs[0], salvaged], ignore_index=True) - df2 = df[['Date', 'Numbers', 'SB', 'All Star']] + df2 = df[['Date', 'Numbers', 'SB']] date = list(df2['Date']) nums = list(df2['Numbers'].astype('str')) SBs = list(df2['SB'].astype('int')) diff --git a/lottopy_pb.py b/lottopy_pb.py index beea272..c687486 100755 --- a/lottopy_pb.py +++ b/lottopy_pb.py @@ -3,28 +3,36 @@ from fractions import Fraction as frac import pandas as pd import requests from collections import Counter -from io import StringIO def get_data(): # previous records which have since been removed from wvlottery.com's database df_old = pd.read_excel('./excel_lotto_records/lotto.xlsx') - salvaged = df_old[['Date', 'Numbers', 'PB', 'PPX', 'Winners WV Only', 'Payout WV Only']] + salvaged = df_old[['Date', 'Numbers', 'PB']] # Current web records - url = 'https://wvlottery.com/draw-games/powerball/?game-analyze=powerball&what-to-search=historysearch&date-range=-1' + url = 'https://gateway.loyalty.wvlottery.com/services/jackpot/api/v1/jackpot-results?gameId=12&jackpotStatus=PAYABLE&size=500&sort=externalId,drawDate,desc&page=' header = { "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.75 Safari/537.36", - "X-Requested-With": "XMLHttpRequest" } - # Get web data - r = requests.get(url, headers=header) - - dfs = list(pd.read_html(StringIO(r.text))) + # Get web data, the new site pulls past draws from json 500 at a time + web = {'Date': [], 'Numbers': [], 'PB': []} + page = 0 + while True: + r = requests.get(url + str(page), headers=header) + data = r.json() + for x in data['content']: + web['Date'].append(x['drawingDate']) + web['Numbers'].append('–'.join(str(n['data']) for n in x['resultData'] if n['type'] == 'REGULAR')) + web['PB'].append([n['data'] for n in x['resultData'] if n['type'] == 'SPECIAL'][0]) + if data['last']: + break + page += 1 + dfs = [pd.DataFrame(web)] pd.set_option('display.max_rows', None) # Specifies no max rows, otherwise only shows 10 records df = pd.concat([dfs[0], salvaged], ignore_index=True) - df2 = df[['Date', 'Numbers', 'PB', 'PPX', 'Winners WV Only', 'Payout WV Only']] + df2 = df[['Date', 'Numbers', 'PB']] date = list(df2['Date']) nums = list(df2['Numbers'].astype('str')) PBs = list(df2['PB'].astype('int')) diff --git a/megamil.py b/megamil.py index 9c08b2d..e6c6fd9 100755 --- a/megamil.py +++ b/megamil.py @@ -3,7 +3,6 @@ from fractions import Fraction as frac import pandas as pd import requests from collections import Counter -from io import StringIO def get_data(): @@ -11,19 +10,28 @@ def get_data(): df_old = pd.read_excel('./excel_lotto_records/lotto_megamil.xlsx') salvaged = df_old[['Date', 'Numbers', 'MB']] - url = 'https://wvlottery.com/draw-games/mega-millions/?game-analyze=mega-millions&what-to-search=historysearch&date-range=-1' + url = 'https://gateway.loyalty.wvlottery.com/services/jackpot/api/v1/jackpot-results?gameId=20&jackpotStatus=PAYABLE&size=500&sort=externalId,drawDate,desc&page=' header = { "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.75 Safari/537.36", - "X-Requested-With": "XMLHttpRequest" } - - # Get web data - r = requests.get(url, headers=header) - dfs = list(pd.read_html(StringIO(r.text))) + # Get web data, the new site pulls past draws from json 500 at a time + web = {'Date': [], 'Numbers': [], 'MB': []} + page = 0 + while True: + r = requests.get(url + str(page), headers=header) + data = r.json() + for x in data['content']: + web['Date'].append(x['drawingDate']) + web['Numbers'].append('–'.join(str(n['data']) for n in x['resultData'] if n['type'] == 'REGULAR')) + web['MB'].append([n['data'] for n in x['resultData'] if n['type'] == 'SPECIAL'][0]) + if data['last']: + break + page += 1 + dfs = [pd.DataFrame(web)] pd.set_option('display.max_rows', None) # Specifies no max rows, otherwise only shows 10 records df = pd.concat([dfs[0], salvaged], ignore_index=True) - df2 = df[['Date', 'Numbers', 'MB', 'MP']] + df2 = df[['Date', 'Numbers', 'MB']] date = list(df2['Date']) nums = list(df2['Numbers'].astype('str')) MBs = list(df2['MB'].astype('int')) @@ -50,7 +58,7 @@ for n in range(0, 25): sorted_nums = sorted(likely_nums, key=lambda x: (len(x), x)) total_freq = sum(frequency + frequency_mb) -Chance = frac(total_freq, 302575350) # Chance = number call freq / all possible numbers i.e. 11238513 +Chance = frac(total_freq, 290472336) # Chance = number call freq / all possible numbers i.e. 290472336 Forecast = str(" - ".join(sorted_nums)) print(f"Likely numbers are . . . {Forecast} MB: {MB}\n" diff --git a/requirements.txt b/requirements.txt index d37d055..2fd93ab 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,2 +1,3 @@ pandas==1.3.5 -requests==2.26.0 \ No newline at end of file +requests==2.26.0 +openpyxl diff --git a/start.sh b/start.sh index f0731f3..6a6f210 100755 --- a/start.sh +++ b/start.sh @@ -15,4 +15,6 @@ python daily3.py && echo "<--------> Daily4 <-------->" && python daily4.py && echo "<--------> Cash25 <-------->" && -python cash25.py +python cash25.py && +echo "<--------> Cash Pop <-------->" && +python cashpop.py