ELO scraper: discover the embed link and retry transient failures
Read the current OneDrive link from the stat-check.com/elo iframe on each run, falling back to EMBED_URL if the page can't be read or its link 404s. Retry 429/5xx and network errors up to three times with backoff, after a one-off 503 from the Badger token endpoint failed a run. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -12,6 +12,7 @@ The Stat Check page embeds a public OneDrive Excel workbook through a `1drv.ms`
|
|||||||
|
|
||||||
The scraper reproduces the useful part of that flow:
|
The scraper reproduces the useful part of that flow:
|
||||||
|
|
||||||
|
0. Read the current `1drv.ms` embed link from the Stat Check page's iframe. Stat Check replaces this link occasionally, so it is looked up on every run; `EMBED_URL` in `pure_python_elo_scrape.py` is only a fallback for when the page can't be read or its link returns 404.
|
||||||
1. Request an anonymous Microsoft "Badger" token from:
|
1. Request an anonymous Microsoft "Badger" token from:
|
||||||
`https://api-badgerp.svc.ms/v1.0/token`
|
`https://api-badgerp.svc.ms/v1.0/token`
|
||||||
2. Convert the public OneDrive embed URL into Microsoft Graph's `shares/u!...` id format.
|
2. Convert the public OneDrive embed URL into Microsoft Graph's `shares/u!...` id format.
|
||||||
@@ -24,6 +25,8 @@ The scraper reproduces the useful part of that flow:
|
|||||||
|
|
||||||
The Microsoft endpoints used here are undocumented and could change, but this currently works without browser automation.
|
The Microsoft endpoints used here are undocumented and could change, but this currently works without browser automation.
|
||||||
|
|
||||||
|
Requests that fail with HTTP 429, a 5xx status, or a network error are retried up to three times (after 5, 15 and 45 seconds).
|
||||||
|
|
||||||
## Usage
|
## Usage
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
|
|||||||
@@ -4,8 +4,10 @@ import html
|
|||||||
import json
|
import json
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
|
import time
|
||||||
import zipfile
|
import zipfile
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from urllib.error import HTTPError, URLError
|
||||||
from urllib.request import Request, urlopen
|
from urllib.request import Request, urlopen
|
||||||
from xml.etree.ElementTree import iterparse
|
from xml.etree.ElementTree import iterparse
|
||||||
|
|
||||||
@@ -16,6 +18,27 @@ NS = {
|
|||||||
"pkgrel": "http://schemas.openxmlformats.org/package/2006/relationships",
|
"pkgrel": "http://schemas.openxmlformats.org/package/2006/relationships",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Seconds to wait before each retry of a transient failure.
|
||||||
|
RETRY_DELAYS = (5, 15, 45)
|
||||||
|
|
||||||
|
|
||||||
|
def fetch(req: Request) -> bytes:
|
||||||
|
"""Read a URL, retrying 429/5xx responses and network errors."""
|
||||||
|
for attempt, delay in enumerate((*RETRY_DELAYS, None), start=1):
|
||||||
|
try:
|
||||||
|
with urlopen(req, timeout=60) as resp:
|
||||||
|
return resp.read()
|
||||||
|
except HTTPError as e:
|
||||||
|
if delay is None or not (e.code == 429 or e.code >= 500):
|
||||||
|
raise
|
||||||
|
reason = f"HTTP {e.code}"
|
||||||
|
except (URLError, TimeoutError, ConnectionError) as e:
|
||||||
|
if delay is None:
|
||||||
|
raise
|
||||||
|
reason = str(e)
|
||||||
|
print(f" {req.full_url.split('?')[0]}: {reason}; retry {attempt} in {delay}s", file=sys.stderr)
|
||||||
|
time.sleep(delay)
|
||||||
|
|
||||||
|
|
||||||
def metadata_download_url(capture_path: Path) -> str:
|
def metadata_download_url(capture_path: Path) -> str:
|
||||||
capture = json.loads(capture_path.read_text())
|
capture = json.loads(capture_path.read_text())
|
||||||
@@ -29,8 +52,7 @@ def metadata_download_url(capture_path: Path) -> str:
|
|||||||
|
|
||||||
def download(url: str, output: Path) -> None:
|
def download(url: str, output: Path) -> None:
|
||||||
req = Request(url, headers={"User-Agent": "Mozilla/5.0"})
|
req = Request(url, headers={"User-Agent": "Mozilla/5.0"})
|
||||||
with urlopen(req, timeout=60) as resp:
|
output.write_bytes(fetch(req))
|
||||||
output.write_bytes(resp.read())
|
|
||||||
|
|
||||||
|
|
||||||
def shared_strings(zf: zipfile.ZipFile) -> list[str]:
|
def shared_strings(zf: zipfile.ZipFile) -> list[str]:
|
||||||
|
|||||||
@@ -1,14 +1,22 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
import base64
|
import base64
|
||||||
|
import html
|
||||||
import json
|
import json
|
||||||
|
import re
|
||||||
import sys
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from urllib.error import HTTPError
|
||||||
from urllib.parse import quote
|
from urllib.parse import quote
|
||||||
from urllib.request import Request, urlopen
|
from urllib.request import Request
|
||||||
|
|
||||||
from download_and_parse import download, xlsx_to_csv
|
from download_and_parse import download, fetch, xlsx_to_csv
|
||||||
|
|
||||||
|
|
||||||
|
STATCHECK_PAGE_URL = "https://www.stat-check.com/elo"
|
||||||
|
IFRAME_SRC_RE = re.compile(r'<iframe[^>]+src="(https://1drv\.ms/x/[^"]+)"', re.I)
|
||||||
|
|
||||||
|
# Fallback for when the link can't be read from the Stat Check page. Stat Check
|
||||||
|
# replaces the share link from time to time, so this can go stale.
|
||||||
EMBED_URL = (
|
EMBED_URL = (
|
||||||
"https://1drv.ms/x/c/9f33f29504402fa1/"
|
"https://1drv.ms/x/c/9f33f29504402fa1/"
|
||||||
"IQSM-tN0M6OcTaaLY3OtMA2vAYx131Ze6y5N0aHYpoCCJ1g"
|
"IQSM-tN0M6OcTaaLY3OtMA2vAYx131Ze6y5N0aHYpoCCJ1g"
|
||||||
@@ -21,8 +29,16 @@ BADGER_APP_ID = "00000000-0000-0000-0000-0000481710a4"
|
|||||||
|
|
||||||
def request_json(url: str, *, method: str = "GET", data: bytes | None = None, headers: dict[str, str] | None = None) -> dict:
|
def request_json(url: str, *, method: str = "GET", data: bytes | None = None, headers: dict[str, str] | None = None) -> dict:
|
||||||
req = Request(url, data=data, method=method, headers=headers or {})
|
req = Request(url, data=data, method=method, headers=headers or {})
|
||||||
with urlopen(req, timeout=60) as resp:
|
return json.loads(fetch(req).decode("utf-8"))
|
||||||
return json.loads(resp.read().decode("utf-8"))
|
|
||||||
|
|
||||||
|
def discover_embed_url(page_url: str = STATCHECK_PAGE_URL) -> str:
|
||||||
|
req = Request(page_url, headers={"User-Agent": "Mozilla/5.0"})
|
||||||
|
page = fetch(req).decode("utf-8", errors="replace")
|
||||||
|
match = IFRAME_SRC_RE.search(page)
|
||||||
|
if not match:
|
||||||
|
raise RuntimeError(f"No OneDrive workbook iframe found on {page_url}")
|
||||||
|
return html.unescape(match.group(1))
|
||||||
|
|
||||||
|
|
||||||
def badger_token() -> str:
|
def badger_token() -> str:
|
||||||
@@ -73,10 +89,31 @@ def download_url_for_embed(embed_url: str) -> str:
|
|||||||
return data["@content.downloadUrl"]
|
return data["@content.downloadUrl"]
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_download_url() -> str:
|
||||||
|
"""Resolve the workbook download URL, preferring the link currently on the Stat Check page."""
|
||||||
|
try:
|
||||||
|
embed_url = discover_embed_url()
|
||||||
|
except Exception as e:
|
||||||
|
print(f" Could not read link from {STATCHECK_PAGE_URL} ({e}); using fallback EMBED_URL", file=sys.stderr)
|
||||||
|
return download_url_for_embed(EMBED_URL)
|
||||||
|
|
||||||
|
print(f" Using link from {STATCHECK_PAGE_URL}: {embed_url}")
|
||||||
|
if embed_url == EMBED_URL:
|
||||||
|
return download_url_for_embed(embed_url)
|
||||||
|
print(" (differs from EMBED_URL; consider updating the fallback)", file=sys.stderr)
|
||||||
|
try:
|
||||||
|
return download_url_for_embed(embed_url)
|
||||||
|
except HTTPError as e:
|
||||||
|
if e.code != 404:
|
||||||
|
raise
|
||||||
|
print(" Page link returned 404; trying fallback EMBED_URL", file=sys.stderr)
|
||||||
|
return download_url_for_embed(EMBED_URL)
|
||||||
|
|
||||||
|
|
||||||
def main() -> None:
|
def main() -> None:
|
||||||
xlsx_path = Path(sys.argv[1] if len(sys.argv) > 1 else "elo-pure.xlsx")
|
xlsx_path = Path(sys.argv[1] if len(sys.argv) > 1 else "elo-pure.xlsx")
|
||||||
csv_path = Path(sys.argv[2] if len(sys.argv) > 2 else "elo-pure.csv")
|
csv_path = Path(sys.argv[2] if len(sys.argv) > 2 else "elo-pure.csv")
|
||||||
url = download_url_for_embed(EMBED_URL)
|
url = resolve_download_url()
|
||||||
download(url, xlsx_path)
|
download(url, xlsx_path)
|
||||||
xlsx_to_csv(xlsx_path, csv_path)
|
xlsx_to_csv(xlsx_path, csv_path)
|
||||||
print(f"downloaded={xlsx_path.stat().st_size} bytes csv={csv_path.stat().st_size} bytes")
|
print(f"downloaded={xlsx_path.stat().st_size} bytes csv={csv_path.stat().st_size} bytes")
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ from datetime import datetime, timezone
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from download_and_parse import download, xlsx_to_csv
|
from download_and_parse import download, xlsx_to_csv
|
||||||
from pure_python_elo_scrape import EMBED_URL, download_url_for_embed
|
from pure_python_elo_scrape import resolve_download_url
|
||||||
|
|
||||||
HERE = Path(__file__).parent
|
HERE = Path(__file__).parent
|
||||||
ROOT = HERE.parent
|
ROOT = HERE.parent
|
||||||
@@ -46,7 +46,7 @@ def main() -> None:
|
|||||||
json_path = Path(sys.argv[3]) if len(sys.argv) > 3 else ROOT / "public" / "elo-data.json"
|
json_path = Path(sys.argv[3]) if len(sys.argv) > 3 else ROOT / "public" / "elo-data.json"
|
||||||
|
|
||||||
print("Resolving OneDrive download URL…")
|
print("Resolving OneDrive download URL…")
|
||||||
url = download_url_for_embed(EMBED_URL)
|
url = resolve_download_url()
|
||||||
|
|
||||||
print("Downloading XLSX…")
|
print("Downloading XLSX…")
|
||||||
download(url, xlsx_path)
|
download(url, xlsx_path)
|
||||||
|
|||||||
Reference in New Issue
Block a user