diff --git a/backend/app/auto_credit.py b/backend/app/auto_credit.py new file mode 100644 index 0000000..37e8a42 --- /dev/null +++ b/backend/app/auto_credit.py @@ -0,0 +1,153 @@ +"""Auto Credit alternative factor — scrape real Thai vehicle sales from Trading Economics. + +Source: https://tradingeconomics.com/thailand/total-vehicle-sales +Server-rendered HTML (per `har-derived-api-client` guidance, no JSON API — the +HAR capture shows only ads/analytics). Three tables carry the data: + + Table 1 (announcements): "New Car Sales YoY" rows — latest YoY% per month. + Table 2 (related indicators): Auto Exports, Vehicle Production, Passenger + Car Sales current values. + Table 3 (main series): Total Vehicle Sales `[latest, prev, high, low, + range, units, freq, seasonality]`. + +We parse these so the Auto Credit factor can use vehicle-sales momentum (YoY%) +and the current total, which is the real, non-mock input the user asked for. +""" + +from __future__ import annotations + +import html +import re +from dataclasses import dataclass +from typing import Optional +from urllib.request import Request, urlopen + +_USER_AGENT = ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" +) +_URL = "https://tradingeconomics.com/thailand/total-vehicle-sales" + + +class AutoCreditError(Exception): + """Raised when the Trading Economics page cannot be fetched or parsed.""" + + +@dataclass(frozen=True) +class AutoCreditSnapshot: + total_vehicle_sales: Optional[float] = None # latest units + prev_vehicle_sales: Optional[float] = None # prior period units + new_car_sales_yoy: Optional[float] = None # latest % YoY + prev_new_car_sales_yoy: Optional[float] = None + vehicle_production: Optional[float] = None + passenger_car_sales: Optional[float] = None + auto_exports: Optional[float] = None + as_of: str = "" + source: str = "tradingeconomics" + + def to_dict(self) -> dict: + return { + "source": self.source, + "as_of": self.as_of, + "total_vehicle_sales": self.total_vehicle_sales, + "prev_vehicle_sales": self.prev_vehicle_sales, + "new_car_sales_yoy": self.new_car_sales_yoy, + "prev_new_car_sales_yoy": self.prev_new_car_sales_yoy, + "vehicle_production": self.vehicle_production, + "passenger_car_sales": self.passenger_car_sales, + "auto_exports": self.auto_exports, + } + + +def _fetch(url: str = _URL, timeout: float = 30.0) -> str: + req = Request(url, headers={"User-Agent": _USER_AGENT, "Accept": "text/html"}) + try: + with urlopen(req, timeout=timeout) as resp: + raw = resp.read() + except Exception as exc: + raise AutoCreditError(f"failed to fetch {url}: {exc}") from exc + try: + return raw.decode("utf-8") + except UnicodeDecodeError: + return raw.decode("latin-1", "ignore") + + +def _tables(html_text: str) -> list[str]: + return re.findall(r"]*>(.*?)", html_text, re.S) + + +def _cells(row_html: str) -> list[str]: + return [ + html.unescape(re.sub(r"<[^>]+>", "", td)).strip() + for td in re.findall(r"]*>(.*?)", row_html, re.S) + ] + + +def _to_float(text: str) -> Optional[float]: + text = text.replace(",", "").strip().replace("%", "") + if not text or text in ("-", "N/A"): + return None + try: + return float(text) + except ValueError: + return None + + +def parse_auto_credit_html(html_text: str) -> AutoCreditSnapshot: + """Parse the Trading Economics Thailand total-vehicle-sales page.""" + tables = _tables(html_text) + if not tables: + raise AutoCreditError("no tables found in Trading Economics page") + + total = prev = None + prod = pass_sales = exports = None + car_yoy = prev_yoy = None + + for table in tables: + rows = re.findall(r"]*>(.*?)", table, re.S) + for row_html in rows: + cells = _cells(row_html) + if not cells: + continue + joined = " | ".join(cells) + # Main series row: ['', '61244.00', '57765.00', '157529','5338','1980-2026','Units',...] + # Detect it as the row with 8+ cells where cell[6] == 'Units'. + if len(cells) >= 7 and cells[6].strip().lower() == "units" and "Monthly" in joined: + total = _to_float(cells[1]) + prev = _to_float(cells[2]) + # Related indicators table (label, value, prev, unit, period) + elif cells[0].strip().lower() == "vehicle production": + prod = _to_float(cells[1]) + elif cells[0].strip().lower() == "passenger car sales": + pass_sales = _to_float(cells[1]) + elif cells[0].strip().lower() == "auto exports": + exports = _to_float(cells[1]) + # New Car Sales YoY announcements: [date,time,'New Car Sales YoY','May','10.60%',...] + elif cells[0].lower().startswith("202") and "New Car Sales YoY" in joined: + if len(cells) >= 5: + pct = _to_float(cells[4]) + if pct is None: + continue # pending row (e.g. current period not reported yet) + if car_yoy is None: + car_yoy = pct + else: + prev_yoy = car_yoy + car_yoy = pct + + if total is None and car_yoy is None and prod is None: + raise AutoCreditError("no auto-credit values found in Trading Economics page") + + return AutoCreditSnapshot( + total_vehicle_sales=total, + prev_vehicle_sales=prev, + new_car_sales_yoy=car_yoy, + prev_new_car_sales_yoy=prev_yoy, + vehicle_production=prod, + passenger_car_sales=pass_sales, + auto_exports=exports, + ) + + +def fetch_auto_credit(timeout: float = 30.0) -> AutoCreditSnapshot: + html_text = _fetch(timeout=timeout) + return parse_auto_credit_html(html_text) diff --git a/backend/app/refining_energy.py b/backend/app/refining_energy.py new file mode 100644 index 0000000..c57d573 --- /dev/null +++ b/backend/app/refining_energy.py @@ -0,0 +1,161 @@ +"""Refining / Energy alternative factor — EIA Wholesale Spot Petroleum Prices. + +Source: https://www.eia.gov/todayinenergy/prices.php (US Energy Information +Administration, open public data). The page is server-rendered HTML (no hidden +JSON API needed) and is **not** behind a bot challenge, so `urllib` fetch works +directly — unlike RBN Energy which sits behind a Cloudflare challenge (403 via +urllib). + +Table: "Wholesale Spot Petroleum Prices, Close" + Product / Area / Price / PercentChange* + Crude Oil ($/barrel) WTI 87.21 -2.8 / Brent 96.92 +3.1 / Louisian. Light 90.71 -2.7 + Gasoline (RBOB) ($/gal) NY 3.42 / Gulf 3.52 / LA 3.52 + 3:2:1 Crack Spread ($/bbl) Gulf Coast (LLS) 71.10 +5.7 + ... (heating oil, diesel, natural gas) + +The 3:2:1 crack spread and crude benchmarks are the refining-margin proxy the +energy theme scores on. +""" + +from __future__ import annotations + +import html +import re +from dataclasses import dataclass +from typing import Optional +from urllib.request import Request, urlopen + +_USER_AGENT = ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" +) +_URL = "https://www.eia.gov/todayinenergy/prices.php" + + +class RefiningError(Exception): + """Raised when the EIA page cannot be fetched or parsed.""" + + +@dataclass(frozen=True) +class RefiningSnapshot: + crack_spread: Optional[float] = None # 3:2:1 crack spread $/bbl + crack_spread_change: Optional[float] = None # +5.7 + wti_crude: Optional[float] = None + brent_crude: Optional[float] = None + gasoline_gulf: Optional[float] = None + as_of: str = "" + source: str = "eia" + + def to_dict(self) -> dict: + return { + "source": self.source, + "as_of": self.as_of, + "crack_spread": self.crack_spread, + "crack_spread_change": self.crack_spread_change, + "wti_crude": self.wti_crude, + "brent_crude": self.brent_crude, + "gasoline_gulf": self.gasoline_gulf, + } + + +def _fetch(url: str = _URL, timeout: float = 30.0) -> str: + req = Request(url, headers={"User-Agent": _USER_AGENT, "Accept": "text/html"}) + try: + with urlopen(req, timeout=timeout) as resp: + raw = resp.read() + except Exception as exc: + raise RefiningError(f"failed to fetch {url}: {exc}") from exc + try: + return raw.decode("utf-8") + except UnicodeDecodeError: + return raw.decode("latin-1", "ignore") + + +def _to_float(text: str) -> Optional[float]: + text = text.replace(",", "").strip().replace("+", "") + if not text or text in ("-", "N/A"): + return None + try: + return float(text) + except ValueError: + return None + + +def _cells(row_html: str) -> list[str]: + return [ + html.unescape(re.sub(r"<[^>]+>", "", td)).strip() + for td in re.findall(r"]*>(.*?)", row_html, re.S) + ] + + +def parse_refining_html(html_text: str) -> RefiningSnapshot: + """Parse the EIA Today in Energy daily prices page.""" + tables = re.findall(r"]*>(.*?)", html_text, re.S) + if not tables: + raise RefiningError("no tables found in EIA prices page") + + crack = crack_chg = None + wti = brent = gasoline_gulf = None + current_product = "" + + for table in tables: + rows = re.findall(r"]*>(.*?)", table, re.S) + for row_html in rows: + cells = _cells(row_html) + text = " ".join(cells) + low = text.lower() + # Track the product section from its header row (cells[0] non-empty, + # e.g. "Gasoline (RBOB) ($/gallon)" / "Heating Oil ($/gallon)"). + if cells and cells[0].strip(): + c0 = cells[0].lower() + # Only update product section from a product header row (which + # carries a unit like "($/barrel)" / "($/gallon)") — Area-only + # sub-rows like 'Gulf Coast' / 'Brent' must not reset it. + if "($/" in c0 or "($ /" in c0: + if "gasoline" in c0 or "rbbob" in c0: + current_product = "gasoline" + elif "heating oil" in c0: + current_product = "heating_oil" + elif "low-sulfur diesel" in c0 or "diesel" in c0: + current_product = "diesel" + else: + current_product = c0 + if len(cells) >= 2: + area = cells[1].strip().lower() + c0 = cells[0].strip().lower() if cells and cells[0].strip() else "" + # When the row's Area label sits in cells[0] (Product column not + # repeated, e.g. ['Brent','96.92','+3.1']), price is cells[1]; + # when it sits in cells[1] (e.g. ['','Brent','96.92','+3.1']), + # price is cells[2]. + if c0 == "brent" or area == "brent": + brent = (_to_float(cells[1]) if c0 == "brent" else (_to_float(cells[2]) if len(cells) > 2 else None)) + elif c0 == "wti" or area == "wti": + wti = (_to_float(cells[1]) if c0 == "wti" else (_to_float(cells[2]) if len(cells) > 2 else None)) + elif current_product == "gasoline" and (c0 == "gulf coast" or "gulf coast" in area): + gasoline_gulf = (_to_float(cells[1]) if c0 == "gulf coast" else (_to_float(cells[2]) if len(cells) > 2 else None)) + if "3:2:1 crack spread" in low: + # row: '3:2:1 Crack Spread ($/barrel)', 'Gulf Coast (LLS)', '71.10', '+5.7' + for k in range(len(cells)): + if "gulf coast" in cells[k].lower(): + crack = _to_float(cells[k + 1]) if k + 1 < len(cells) else None + crack_chg = _to_float(cells[k + 2]) if k + 2 < len(cells) else None + break + if crack is None and len(cells) >= 3: + crack = _to_float(cells[-2]) + crack_chg = _to_float(cells[-1]) + + if crack is None and wti is None and brent is None: + raise RefiningError("no refining/energy values found in EIA prices page") + + return RefiningSnapshot( + crack_spread=crack, + crack_spread_change=crack_chg, + wti_crude=wti, + brent_crude=brent, + gasoline_gulf=gasoline_gulf, + ) + + +def fetch_refining(timeout: float = 30.0) -> RefiningSnapshot: + html_text = _fetch(timeout=timeout) + return parse_refining_html(html_text) diff --git a/backend/scripts/collect_auto_credit.py b/backend/scripts/collect_auto_credit.py new file mode 100644 index 0000000..ddda905 --- /dev/null +++ b/backend/scripts/collect_auto_credit.py @@ -0,0 +1,38 @@ +"""Collect the Auto Credit factor (Trading Economics vehicle sales) to a JSON snapshot.""" + +from __future__ import annotations + +import argparse +import json +from datetime import datetime, timezone +from pathlib import Path + +from app import auto_credit + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--out", type=Path, + help="write JSON snapshot to this file (default: stdout)") + parser.add_argument("--timeout", type=float, default=30.0) + args = parser.parse_args() + + snap = auto_credit.fetch_auto_credit(timeout=args.timeout) + payload = { + "source": "tradingeconomics", + "retrieved_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + "factor": "auto_credit", + "data": snap.to_dict(), + } + text = json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2) + if args.out: + args.out.parent.mkdir(parents=True, exist_ok=True) + args.out.write_text(text, encoding="utf-8") + print(f"wrote auto_credit snapshot -> {args.out}") + else: + print(text) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/backend/scripts/collect_refining.py b/backend/scripts/collect_refining.py new file mode 100644 index 0000000..b03ac86 --- /dev/null +++ b/backend/scripts/collect_refining.py @@ -0,0 +1,38 @@ +"""Collect the Refining/Energy factor (EIA crack spread) to a JSON snapshot.""" + +from __future__ import annotations + +import argparse +import json +from datetime import datetime, timezone +from pathlib import Path + +from app import refining_energy + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--out", type=Path, + help="write JSON snapshot to this file (default: stdout)") + parser.add_argument("--timeout", type=float, default=30.0) + args = parser.parse_args() + + snap = refining_energy.fetch_refining(timeout=args.timeout) + payload = { + "source": "eia", + "retrieved_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + "factor": "refining_energy", + "data": snap.to_dict(), + } + text = json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2) + if args.out: + args.out.parent.mkdir(parents=True, exist_ok=True) + args.out.write_text(text, encoding="utf-8") + print(f"wrote refining_energy snapshot -> {args.out}") + else: + print(text) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/backend/tests/test_auto_credit.py b/backend/tests/test_auto_credit.py new file mode 100644 index 0000000..4f38442 --- /dev/null +++ b/backend/tests/test_auto_credit.py @@ -0,0 +1,61 @@ +"""Tests for the Auto Credit (Trading Economics) parser.""" + +from __future__ import annotations + +import unittest + +from app import auto_credit + + +def _trading_econ_html() -> str: + announcement = """ + + + +
DateTimeStatisticPeriodActualPrevious
2026-06-2903:45 AMNew Car Sales YoYMay10.60%2.54%
2026-07-2404:20 AMNew Car Sales YoYJun17.26%10.60%
""" + indicators = """ + + + + +
IndicatorActualPreviousUnitPeriod
Auto Exports81526.0059434.00UnitsJun 2026
Vehicle Production120391.00114214.00UnitsJun 2026
Passenger Car Sales22875.0019389.00UnitsJun 2026
""" + main = """ + + +
ActualPreviousHighestLowestDatesUnitFrequencySeasonality
61244.0057765.00157529.005338.001980 - 2026UnitsMonthlyVolume, NSA
""" + return f"{announcement}{indicators}{main}" + + +class AutoCreditParseTest(unittest.TestCase): + def test_parses_auto_credit_values(self) -> None: + snap = auto_credit.parse_auto_credit_html(_trading_econ_html()) + self.assertEqual(snap.total_vehicle_sales, 61244.00) + self.assertEqual(snap.prev_vehicle_sales, 57765.00) + self.assertEqual(snap.vehicle_production, 120391.00) + self.assertEqual(snap.passenger_car_sales, 22875.00) + self.assertEqual(snap.auto_exports, 81526.00) + # New Car Sales YoY — latest (Jun 2026) is the last announcement row + self.assertEqual(snap.new_car_sales_yoy, 17.26) + self.assertEqual(snap.prev_new_car_sales_yoy, 10.60) + + def test_missing_table_raises(self) -> None: + with self.assertRaises(auto_credit.AutoCreditError): + auto_credit.parse_auto_credit_html("no data") + + def test_negative_and_percent_handling(self) -> None: + # negative YoY should parse as negative float + html_text = """ + +
2026-08-2504:00 AMNew Car Sales YoYJul-3.5%17.26%
""" + snap = auto_credit.parse_auto_credit_html(html_text) + self.assertEqual(snap.new_car_sales_yoy, -3.5) + + def test_build_dict(self) -> None: + snap = auto_credit.parse_auto_credit_html(_trading_econ_html()) + d = snap.to_dict() + self.assertEqual(d["source"], "tradingeconomics") + self.assertIn("total_vehicle_sales", d) + + +if __name__ == "__main__": + unittest.main() diff --git a/backend/tests/test_refining_energy.py b/backend/tests/test_refining_energy.py new file mode 100644 index 0000000..df7ca2f --- /dev/null +++ b/backend/tests/test_refining_energy.py @@ -0,0 +1,58 @@ +"""Tests for the Refining/Energy (EIA) parser.""" + +from __future__ import annotations + +import unittest + +from app import refining_energy + + +def _eia_html() -> str: + return """ + +

Wholesale Spot Petroleum Prices, 8/21/26 Close

+ + + + + + + + + +
ProductAreaPricePercentChange*
Crude Oil ($/barrel)WTI87.21-2.8
Brent96.92+3.1
Louisiana Light90.71-2.7
Gasoline (RBOB) ($/gallon)NY Harbor3.42+2.3
Gulf Coast3.52+1.3
3:2:1 Crack Spread ($/barrel)Gulf Coast (LLS)71.10+5.7
Low-Sulfur Diesel ($/gallon)NY Harbor4.47-2.1
+ + """ + + +class RefiningParseTest(unittest.TestCase): + def test_parses_crack_spread_and_crudes(self) -> None: + snap = refining_energy.parse_refining_html(_eia_html()) + self.assertEqual(snap.crack_spread, 71.10) + self.assertEqual(snap.crack_spread_change, 5.7) + self.assertEqual(snap.wti_crude, 87.21) + self.assertEqual(snap.brent_crude, 96.92) + self.assertEqual(snap.gasoline_gulf, 3.52) + + def test_missing_table_raises(self) -> None: + with self.assertRaises(refining_energy.RefiningError): + refining_energy.parse_refining_html("no data") + + def test_negative_change_parses(self) -> None: + html_text = """ +
3:2:1 Crack Spread ($/barrel)Gulf Coast (LLS)60.00-2.5
""" + snap = refining_energy.parse_refining_html(html_text) + self.assertEqual(snap.crack_spread, 60.00) + self.assertEqual(snap.crack_spread_change, -2.5) + + def test_crack_spread_fallback_when_area_missing(self) -> None: + # If the row lacks the "Gulf Coast" area cell, fall back to last two cells. + html_text = """ +
3:2:1 Crack Spread ($/barrel)71.10+5.7
""" + snap = refining_energy.parse_refining_html(html_text) + self.assertEqual(snap.crack_spread, 71.10) + self.assertEqual(snap.crack_spread_change, 5.7) + + +if __name__ == "__main__": + unittest.main()