"""
scrape_rarity.py
Fetches N64 rarity data from rarityguide.com and saves to /tmp/rarity.json
"""
import requests
from bs4 import BeautifulSoup
import json
import re
import warnings
warnings.filterwarnings("ignore") # suppress urllib3/ssl noise
HEADERS = {
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120 Safari/537.36"
}
BASE_URL = "https://www.rarityguide.com/n64_view.php"
# ── helpers ────────────────────────────────────────────────────────────────────
def fetch_html(params):
resp = requests.get(BASE_URL, headers=HEADERS, params=params, timeout=30)
resp.raise_for_status()
return resp.text
def parse_rarity_alt(alt_text):
"""'92 percent (Ultra Rare)' -> (92.0, 'Ultra Rare')"""
if not alt_text:
return None, None
m_pct = re.search(r'(\d+(?:\.\d+)?)\s*percent', alt_text, re.IGNORECASE)
m_lbl = re.search(r'\(([^)]+)\)', alt_text)
if m_pct:
return float(m_pct.group(1)), (m_lbl.group(1).strip() if m_lbl else "")
return None, None
def extract_games(html):
"""Return dict of {title: {rarity_pct, rarity_label}} from one page."""
soup = BeautifulSoup(html, "html.parser")
pct_imgs = soup.find_all("img", src=re.compile(r"percent\d+\.gif"))
page_results = {}
for img in pct_imgs:
alt = img.get("alt", "")
pct, label = parse_rarity_alt(alt)
if pct is None:
continue
tr = img.find_parent("tr")
if not tr:
continue
tds = tr.find_all("td", recursive=False)
# Title is in td[1] (td[0] is a blank selector cell)
if len(tds) < 2:
continue
title = tds[1].get_text(strip=True)
if title:
page_results[title] = {"rarity_pct": pct, "rarity_label": label}
return page_results
def print_raw_rows(html, n=3):
"""Print raw HTML for the first n game rows."""
soup = BeautifulSoup(html, "html.parser")
pct_imgs = soup.find_all("img", src=re.compile(r"percent\d+\.gif"))
print(f"\n--- First {n} raw game
rows ---")
for i, img in enumerate(pct_imgs[:n]):
tr = img.find_parent("tr")
print(f"\n[Data Row {i}]\n{tr}\n")
# ── main ───────────────────────────────────────────────────────────────────────
# Page 1 (FirstRecord omitted = 1)
print("Fetching page 1 (FirstRecord=1)...")
html1 = fetch_html({"SortDirection": "desc", "SortField": "5", "recordsPerPage": "100"})
with open("/tmp/rarity_raw.html", "w", encoding="utf-8") as f:
f.write(html1)
print(f"Saved raw HTML ({len(html1):,} bytes) to /tmp/rarity_raw.html")
# ── structure diagnostics ──────────────────────────────────────────────────────
soup_diag = BeautifulSoup(html1, "html.parser")
all_tables = soup_diag.find_all("table")
print(f"\nTotal tags: {len(all_tables)}")
for i, t in enumerate(all_tables):
print(f" Table {i}: id={t.get('id')!r} class={t.get('class')!r} rows={len(t.find_all('tr'))}")
pct_imgs_diag = soup_diag.find_all("img", src=re.compile(r"percent\d+\.gif"))
print(f"\nPercent
tags found: {len(pct_imgs_diag)}")
print(f" (recordsPerPage=300 is ignored by server; cap is 100 per page)")
# Print first 3 raw data rows
print_raw_rows(html1, n=3)
# ── collect all pages ──────────────────────────────────────────────────────────
all_results = {}
all_results.update(extract_games(html1))
print(f"\nPage 1: {len(all_results)} games collected.")
# Pages 2 and 3 via FirstRecord offset
for first_record in [101, 201]:
print(f"Fetching page with FirstRecord={first_record}...")
phtml = fetch_html({
"FirstRecord": str(first_record),
"recordsPerPage": "100",
"SortDirection": "desc",
"SortField": "5"
})
page_games = extract_games(phtml)
new_games = {k: v for k, v in page_games.items() if k not in all_results}
all_results.update(new_games)
print(f" -> {len(page_games)} on this page, {len(new_games)} new. Running total: {len(all_results)}")
if len(new_games) == 0:
print(" No new games — stopping pagination.")
break
# ── save JSON ──────────────────────────────────────────────────────────────────
with open("/tmp/rarity.json", "w", encoding="utf-8") as f:
json.dump(all_results, f, indent=2, ensure_ascii=False)
print(f"\nSaved {len(all_results)} games to /tmp/rarity.json")
# ── final stats ───────────────────────────────────────────────────────────────
if all_results:
pcts = [v["rarity_pct"] for v in all_results.values()]
max_game = max(all_results, key=lambda k: all_results[k]["rarity_pct"])
min_game = min(all_results, key=lambda k: all_results[k]["rarity_pct"])
from collections import Counter
label_counts = Counter(v["rarity_label"] for v in all_results.values())
print(f"""
=== FINAL STATS ===
Total games : {len(all_results)}
Max rarity : {all_results[max_game]['rarity_pct']:.0f}% ({all_results[max_game]['rarity_label']}) -> {max_game}
Min rarity : {all_results[min_game]['rarity_pct']:.0f}% ({all_results[min_game]['rarity_label']}) -> {min_game}
Avg rarity : {sum(pcts)/len(pcts):.1f}%
Rarity label distribution:""")
for label, count in sorted(label_counts.items(), key=lambda x: -x[1]):
print(f" {label or '(unlabeled)':<20} {count:>4} games")