StatistaAmeri/backend/steel_stocks.py

183 lines
6.5 KiB
Python

"""
Scrapes the largest steel-industry companies by market cap from
companiesmarketcap.com every 5 minutes. Keeps a current snapshot in memory
(served by /api/steel-stocks) and upserts one row per company in steel_stocks.
"""
import asyncio
import os
from datetime import datetime, timezone
from playwright.sync_api import sync_playwright
from db import get_conn
# companiesmarketcap.com is geo-blocked from Iran; route the headless browser
# through an outbound proxy when one is configured (SCRAPER_PROXY / HTTPS_PROXY).
def _proxy_arg() -> dict:
server = (
os.environ.get("SCRAPER_PROXY")
or os.environ.get("HTTPS_PROXY")
or os.environ.get("HTTP_PROXY")
)
return {"proxy": {"server": server}} if server else {}
URL = "https://companiesmarketcap.com/steel-industry/largest-companies-by-market-cap/"
UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
)
INTERVAL_SECONDS = 5 * 60
_latest: list[dict] = []
def create_table() -> None:
with get_conn() as conn:
conn.execute("""
CREATE TABLE IF NOT EXISTS steel_stocks (
code TEXT PRIMARY KEY,
rank INTEGER,
name TEXT,
market_cap REAL,
price REAL,
today_pct REAL,
dir TEXT,
country TEXT,
fetched_at TEXT
)
""")
def _money(s):
"""'$254.39' -> 254.39 ; '$2,000' -> 2000.0"""
try:
return float(str(s).replace("$", "").replace(",", "").strip())
except (ValueError, AttributeError):
return None
def _market_cap_b(s):
"""'$57.93 B' -> 57.93 ; '$850.00 M' -> 0.85 ; '$1.20 T' -> 1200.0 (billions)"""
t = str(s).replace("$", "").replace(",", "").strip().upper()
mult = 1.0
if t.endswith("T"):
t, mult = t[:-1], 1000.0
elif t.endswith("B"):
t, mult = t[:-1], 1.0
elif t.endswith("M"):
t, mult = t[:-1], 0.001
try:
return round(float(t.strip()) * mult, 4)
except ValueError:
return None
def _scrape_sync() -> list[dict]:
with sync_playwright() as p:
browser = p.chromium.launch(headless=True, **_proxy_arg())
page = browser.new_page(user_agent=UA)
try:
page.goto(URL, wait_until="domcontentloaded", timeout=45000)
page.wait_for_selector("table tbody tr", timeout=20000)
# Prices load via JS after the table renders — wait so they're populated.
page.wait_for_timeout(5000)
raw = page.evaluate(r"""() => {
const out = [];
document.querySelectorAll("table tbody tr").forEach(tr => {
const tds = tr.querySelectorAll("td");
if (tds.length < 8) return;
const name = (tr.querySelector(".company-name")?.innerText || "").trim();
const code = (tr.querySelector(".company-code")?.innerText || "").trim();
if (!name) return;
const chg = tds[5];
const blob = ((chg?.innerHTML || "") + " " + (chg?.className || "")).toLowerCase();
let dir = "";
if (blob.includes("green")) dir = "up";
else if (blob.includes("red")) dir = "down";
out.push({
rank: (tds[1]?.innerText || "").trim(),
name, code,
market_cap: (tds[3]?.innerText || "").trim(),
price: (tds[4]?.innerText || "").trim(),
today: (chg?.innerText || "").trim(),
dir,
country: (tds[7]?.innerText || "").trim(),
});
});
return out;
}""")
finally:
browser.close()
rows = []
for r in raw:
try:
rank = int(r["rank"])
except (ValueError, TypeError):
rank = None
pct = _money(str(r["today"]).replace("%", ""))
if pct is not None and r["dir"] == "down":
pct = -pct
rows.append({
"code": r["code"],
"rank": rank,
"name": r["name"],
"market_cap": _market_cap_b(r["market_cap"]),
"price": _money(r["price"]),
"today_pct": pct,
"dir": r["dir"],
"country": r["country"],
})
return rows
def _save(rows: list[dict], now: str) -> None:
with get_conn() as conn:
for r in rows:
if not r["code"]:
continue
conn.execute(
"""INSERT INTO steel_stocks
(code, rank, name, market_cap, price, today_pct, dir, country, fetched_at)
VALUES(?,?,?,?,?,?,?,?,?)
ON CONFLICT(code) DO UPDATE SET
rank=excluded.rank, name=excluded.name, market_cap=excluded.market_cap,
price=excluded.price, today_pct=excluded.today_pct, dir=excluded.dir,
country=excluded.country, fetched_at=excluded.fetched_at""",
(r["code"], r["rank"], r["name"], r["market_cap"], r["price"],
r["today_pct"], r["dir"], r["country"], now),
)
def _load_from_db() -> None:
global _latest
try:
with get_conn() as conn:
rows = conn.execute(
"SELECT * FROM steel_stocks ORDER BY rank IS NULL, rank"
).fetchall()
if rows:
_latest = [dict(r) for r in rows]
except Exception as e:
print(f"[steel] could not seed from DB: {e}")
async def scrape_loop() -> None:
global _latest
create_table()
_load_from_db()
while True:
try:
rows = await asyncio.to_thread(_scrape_sync)
if rows:
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
await asyncio.to_thread(_save, rows, now)
rows.sort(key=lambda r: (r["rank"] is None, r["rank"] or 9999))
_latest = [{**r, "fetched_at": now} for r in rows]
print(f"[steel] {len(rows)} companies at {now}")
except Exception as e:
print(f"[steel] error: {e}")
await asyncio.sleep(INTERVAL_SECONDS)
def get_latest() -> list[dict]:
return _latest