import math, random
def _generate_sparkline(current_price: float, pct_change: float = 0.0, n_points: int = 40) -> list:
if not current_price or current_price <= 0:
return [0.0] * n_points
pct = (pct_change or 0.0) / 100.0
start_price = current_price / (1.0 + pct) if (1.0 + pct) > 0 else current_price
points = []
trend_step = (current_price - start_price) / max(1, n_points - 1)
volatility = max(abs(pct) * 0.4, 0.007) * current_price
rand_seed = int(abs(current_price) * 1000) % 100000
r = random.Random(rand_seed)
for i in range(n_points - 1):
progress = i / float(n_points - 1)
base = start_price + trend_step * i
wave = math.sin(progress * math.pi * 3.5) * (volatility * 0.6) + math.cos(progress * math.pi * 5.2) * (volatility * 0.4)
noise = (r.random() - 0.5) * volatility * 0.45
val = base + wave + noise
points.append(round(max(val, current_price * 0.2), 2 if current_price > 100 else 4))
points.append(round(current_price, 2 if current_price > 100 else 4))
return points
"""
Scrapes Iranian domestic market prices from TGJU every 60 seconds:
- currency (tgju.org/currency)
- gold (tgju.org/gold-chart) — gold ETFs ("صندوق") excluded
- coin (tgju.org/coin) — bubbles/funds/pre-86 excluded
One headless browser visits all three pages per cycle. The latest snapshot
of each is cached in memory (served by /api/currency, /api/gold, /api/coin)
and only changed rows are persisted (dedup) to keep the DB lean.
Row names are read from
(not | ): on TGJU market tables the first
| is the price, so a | -based selector stored numbers instead of names.
"""
import asyncio
import re
from datetime import datetime, timezone
from playwright.sync_api import sync_playwright
from db import get_conn
UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
)
INTERVAL_SECONDS = 60
URLS = {
"currency": "https://www.tgju.org/currency",
"gold": "https://www.tgju.org/gold-chart",
"coin": "https://www.tgju.org/coin",
}
# Rows to drop per kind (keep the board clean and on-topic).
_EXCLUDE = {
"currency": lambda name: False,
"gold": lambda name: "صندوق" in name, # gold ETFs
"coin": lambda name: any(x in name for x in ("حباب", "تمام سکه", "صندوق", "قبل")),
}
KINDS = ("currency", "gold", "coin")
_TABLES = {"currency": "currency_prices", "gold": "gold_prices", "coin": "coin_prices"}
_latest: dict[str, list[dict]] = {k: [] for k in KINDS}
_last_saved: dict[str, dict[str, float]] = {k: {} for k in KINDS}
_PCT_RE = re.compile(r"\(([\d.]+)\s*%\)")
_EVAL_JS = r"""() => {
const out = [];
document.querySelectorAll("table.data-table.market-table").forEach(table => {
table.querySelectorAll("tbody tr").forEach(tr => {
const th = tr.querySelector("th");
const tds = Array.from(tr.querySelectorAll("td"));
const name = th ? (th.innerText || "").trim() : "";
const price = tds[0] ? (tds[0].innerText || "").trim() : "";
const chgEl = tds[1];
const change = chgEl ? (chgEl.innerText || "").trim() : "";
let dir = "";
if (chgEl) {
const c = chgEl.className || "";
if (c.includes("high")) dir = "up";
else if (c.includes("low")) dir = "down";
}
if (name && price) out.push({ name, price, change, dir });
});
});
return out;
}"""
def create_tables() -> None:
with get_conn() as conn:
for table in _TABLES.values():
conn.execute(f"""
CREATE TABLE IF NOT EXISTS {table} (
id INTEGER PRIMARY KEY AUTOINCREMENT,
name TEXT NOT NULL,
price REAL,
change_val TEXT,
pct REAL,
dir TEXT,
fetched_at TEXT NOT NULL
)
""")
conn.execute(f"CREATE INDEX IF NOT EXISTS idx_{table}_name ON {table}(name)")
def _to_float(raw: str):
try:
return float(raw.replace(",", "").strip())
except (ValueError, AttributeError):
return None
def _parse_pct(change: str, direction: str):
m = _PCT_RE.search(change or "")
if not m:
return 0.0
pct = float(m.group(1))
return -pct if direction == "down" else pct
def _scrape_all() -> dict[str, list[dict]]:
out: dict[str, list[dict]] = {}
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page(user_agent=UA)
try:
for kind, url in URLS.items():
try:
page.goto(url, wait_until="domcontentloaded", timeout=45000)
page.wait_for_selector("table.data-table.market-table tbody tr", timeout=20000)
# TGJU server-renders STALE prices, then live-updates the DOM a few
# seconds later via its own JS. Wait so we read the live values, not
# the cached snapshot (otherwise prices look frozen / dedup sees no change).
page.wait_for_timeout(9000)
raw = page.evaluate(_EVAL_JS)
except Exception as e:
print(f"[tgju] {kind} scrape error: {e}")
out[kind] = []
continue
exclude = _EXCLUDE[kind]
seen: set[str] = set()
rows = []
for r in raw:
name = r["name"]
if exclude(name) or name in seen:
continue # drop off-topic rows + duplicate tables (coins repeat)
price = _to_float(r["price"])
if price is None:
continue
seen.add(name)
rows.append({
"name": name,
"price": price,
"change_val": r["change"],
"pct": _parse_pct(r["change"], r["dir"]),
"dir": r["dir"],
})
out[kind] = rows
finally:
browser.close()
return out
def _save(kind: str, rows: list[dict], now: str) -> None:
if not rows:
return
table = _TABLES[kind]
with get_conn() as conn:
conn.executemany(
f"INSERT INTO {table}(name, price, change_val, pct, dir, fetched_at) "
"VALUES(?,?,?,?,?,?)",
[(r["name"], r["price"], r["change_val"], r["pct"], r["dir"], now) for r in rows],
)
def _load_last_saved() -> None:
for kind, table in _TABLES.items():
try:
with get_conn() as conn:
rows = conn.execute(
f"SELECT name, price FROM {table} "
f"WHERE id IN (SELECT MAX(id) FROM {table} GROUP BY name)"
).fetchall()
for r in rows:
if r["price"] is not None:
_last_saved[kind][r["name"]] = r["price"]
except Exception as e:
print(f"[tgju] could not load last-saved {kind}: {e}")
async def scrape_loop() -> None:
"""Background task: refresh currency + gold + coin every INTERVAL_SECONDS.
The in-memory snapshots always hold the full board; only changed rows persist."""
create_tables()
_load_last_saved()
while True:
try:
data = await asyncio.to_thread(_scrape_all)
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
summary = []
for kind in KINDS:
rows = data.get(kind, [])
if not rows:
continue
_latest[kind] = [{**r, "fetched_at": now} for r in rows]
changed = [r for r in rows if _last_saved[kind].get(r["name"]) != r["price"]]
if changed:
await asyncio.to_thread(_save, kind, changed, now)
for r in changed:
_last_saved[kind][r["name"]] = r["price"]
summary.append(f"{kind}:{len(changed)}/{len(rows)}")
# heartbeat: record that a scrape cycle completed, independent of dedup
try:
with get_conn() as c:
c.execute(
"INSERT INTO app_config(key,value) VALUES('tgju_last_scrape',?) "
"ON CONFLICT(key) DO UPDATE SET value=excluded.value",
(now,),
)
except Exception:
pass
if summary:
print(f"[tgju] {' '.join(summary)} at {now}")
except Exception as e:
print(f"[tgju] error: {e}")
await asyncio.sleep(INTERVAL_SECONDS)
def get_latest() -> list[dict]:
return _latest["currency"]
def get_gold() -> list[dict]:
return _latest["gold"]
def get_coin() -> list[dict]:
return _latest["coin"]
|