[verified] Add Auto Credit (Trading Economics) + Refining/Energy (EIA) factor collectors

- auto_credit.py: scrape Trading Economics Thailand total-vehicle-sales HTML -> total sales + new car sales YoY + vehicle production/passenger/exports
- refining_energy.py: scrape EIA prices.php 3:2:1 crack spread (Gulf LLS) + WTI/Brent/gasoline
- Both server-rendered HTML scrapers (har-derived-api-client pattern); EIA used instead of RBN (RBN is Cloudflare-challenged, 403 via urllib; EIA is open US gov data, HTTP 200 direct)
- collect_auto_credit.py / collect_refining.py CLI -> JSON snapshot
- 8 tests; full backend suite 148 OK; live verified (auto 59198/20% YoY; crack 71.10 $/bbl); static scan clean
This commit is contained in:
Kunthawat Greethong
2026-08-25 11:05:13 +07:00
parent 8a6991b7dd
commit f6dd6304cc
6 changed files with 509 additions and 0 deletions

153
backend/app/auto_credit.py Normal file
View File

@@ -0,0 +1,153 @@
"""Auto Credit alternative factor — scrape real Thai vehicle sales from Trading Economics.
Source: https://tradingeconomics.com/thailand/total-vehicle-sales
Server-rendered HTML (per `har-derived-api-client` guidance, no JSON API — the
HAR capture shows only ads/analytics). Three tables carry the data:
Table 1 (announcements): "New Car Sales YoY" rows — latest YoY% per month.
Table 2 (related indicators): Auto Exports, Vehicle Production, Passenger
Car Sales current values.
Table 3 (main series): Total Vehicle Sales `[latest, prev, high, low,
range, units, freq, seasonality]`.
We parse these so the Auto Credit factor can use vehicle-sales momentum (YoY%)
and the current total, which is the real, non-mock input the user asked for.
"""
from __future__ import annotations
import html
import re
from dataclasses import dataclass
from typing import Optional
from urllib.request import Request, urlopen
_USER_AGENT = (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
)
_URL = "https://tradingeconomics.com/thailand/total-vehicle-sales"
class AutoCreditError(Exception):
"""Raised when the Trading Economics page cannot be fetched or parsed."""
@dataclass(frozen=True)
class AutoCreditSnapshot:
total_vehicle_sales: Optional[float] = None # latest units
prev_vehicle_sales: Optional[float] = None # prior period units
new_car_sales_yoy: Optional[float] = None # latest % YoY
prev_new_car_sales_yoy: Optional[float] = None
vehicle_production: Optional[float] = None
passenger_car_sales: Optional[float] = None
auto_exports: Optional[float] = None
as_of: str = ""
source: str = "tradingeconomics"
def to_dict(self) -> dict:
return {
"source": self.source,
"as_of": self.as_of,
"total_vehicle_sales": self.total_vehicle_sales,
"prev_vehicle_sales": self.prev_vehicle_sales,
"new_car_sales_yoy": self.new_car_sales_yoy,
"prev_new_car_sales_yoy": self.prev_new_car_sales_yoy,
"vehicle_production": self.vehicle_production,
"passenger_car_sales": self.passenger_car_sales,
"auto_exports": self.auto_exports,
}
def _fetch(url: str = _URL, timeout: float = 30.0) -> str:
req = Request(url, headers={"User-Agent": _USER_AGENT, "Accept": "text/html"})
try:
with urlopen(req, timeout=timeout) as resp:
raw = resp.read()
except Exception as exc:
raise AutoCreditError(f"failed to fetch {url}: {exc}") from exc
try:
return raw.decode("utf-8")
except UnicodeDecodeError:
return raw.decode("latin-1", "ignore")
def _tables(html_text: str) -> list[str]:
return re.findall(r"<table[^>]*>(.*?)</table>", html_text, re.S)
def _cells(row_html: str) -> list[str]:
return [
html.unescape(re.sub(r"<[^>]+>", "", td)).strip()
for td in re.findall(r"<td[^>]*>(.*?)</td>", row_html, re.S)
]
def _to_float(text: str) -> Optional[float]:
text = text.replace(",", "").strip().replace("%", "")
if not text or text in ("-", "N/A"):
return None
try:
return float(text)
except ValueError:
return None
def parse_auto_credit_html(html_text: str) -> AutoCreditSnapshot:
"""Parse the Trading Economics Thailand total-vehicle-sales page."""
tables = _tables(html_text)
if not tables:
raise AutoCreditError("no tables found in Trading Economics page")
total = prev = None
prod = pass_sales = exports = None
car_yoy = prev_yoy = None
for table in tables:
rows = re.findall(r"<tr[^>]*>(.*?)</tr>", table, re.S)
for row_html in rows:
cells = _cells(row_html)
if not cells:
continue
joined = " | ".join(cells)
# Main series row: ['', '61244.00', '57765.00', '157529','5338','1980-2026','Units',...]
# Detect it as the row with 8+ cells where cell[6] == 'Units'.
if len(cells) >= 7 and cells[6].strip().lower() == "units" and "Monthly" in joined:
total = _to_float(cells[1])
prev = _to_float(cells[2])
# Related indicators table (label, value, prev, unit, period)
elif cells[0].strip().lower() == "vehicle production":
prod = _to_float(cells[1])
elif cells[0].strip().lower() == "passenger car sales":
pass_sales = _to_float(cells[1])
elif cells[0].strip().lower() == "auto exports":
exports = _to_float(cells[1])
# New Car Sales YoY announcements: [date,time,'New Car Sales YoY','May','10.60%',...]
elif cells[0].lower().startswith("202") and "New Car Sales YoY" in joined:
if len(cells) >= 5:
pct = _to_float(cells[4])
if pct is None:
continue # pending row (e.g. current period not reported yet)
if car_yoy is None:
car_yoy = pct
else:
prev_yoy = car_yoy
car_yoy = pct
if total is None and car_yoy is None and prod is None:
raise AutoCreditError("no auto-credit values found in Trading Economics page")
return AutoCreditSnapshot(
total_vehicle_sales=total,
prev_vehicle_sales=prev,
new_car_sales_yoy=car_yoy,
prev_new_car_sales_yoy=prev_yoy,
vehicle_production=prod,
passenger_car_sales=pass_sales,
auto_exports=exports,
)
def fetch_auto_credit(timeout: float = 30.0) -> AutoCreditSnapshot:
html_text = _fetch(timeout=timeout)
return parse_auto_credit_html(html_text)

View File

@@ -0,0 +1,161 @@
"""Refining / Energy alternative factor — EIA Wholesale Spot Petroleum Prices.
Source: https://www.eia.gov/todayinenergy/prices.php (US Energy Information
Administration, open public data). The page is server-rendered HTML (no hidden
JSON API needed) and is **not** behind a bot challenge, so `urllib` fetch works
directly — unlike RBN Energy which sits behind a Cloudflare challenge (403 via
urllib).
Table: "Wholesale Spot Petroleum Prices, <date> Close"
Product / Area / Price / PercentChange*
Crude Oil ($/barrel) WTI 87.21 -2.8 / Brent 96.92 +3.1 / Louisian. Light 90.71 -2.7
Gasoline (RBOB) ($/gal) NY 3.42 / Gulf 3.52 / LA 3.52
3:2:1 Crack Spread ($/bbl) Gulf Coast (LLS) 71.10 +5.7
... (heating oil, diesel, natural gas)
The 3:2:1 crack spread and crude benchmarks are the refining-margin proxy the
energy theme scores on.
"""
from __future__ import annotations
import html
import re
from dataclasses import dataclass
from typing import Optional
from urllib.request import Request, urlopen
_USER_AGENT = (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
)
_URL = "https://www.eia.gov/todayinenergy/prices.php"
class RefiningError(Exception):
"""Raised when the EIA page cannot be fetched or parsed."""
@dataclass(frozen=True)
class RefiningSnapshot:
crack_spread: Optional[float] = None # 3:2:1 crack spread $/bbl
crack_spread_change: Optional[float] = None # +5.7
wti_crude: Optional[float] = None
brent_crude: Optional[float] = None
gasoline_gulf: Optional[float] = None
as_of: str = ""
source: str = "eia"
def to_dict(self) -> dict:
return {
"source": self.source,
"as_of": self.as_of,
"crack_spread": self.crack_spread,
"crack_spread_change": self.crack_spread_change,
"wti_crude": self.wti_crude,
"brent_crude": self.brent_crude,
"gasoline_gulf": self.gasoline_gulf,
}
def _fetch(url: str = _URL, timeout: float = 30.0) -> str:
req = Request(url, headers={"User-Agent": _USER_AGENT, "Accept": "text/html"})
try:
with urlopen(req, timeout=timeout) as resp:
raw = resp.read()
except Exception as exc:
raise RefiningError(f"failed to fetch {url}: {exc}") from exc
try:
return raw.decode("utf-8")
except UnicodeDecodeError:
return raw.decode("latin-1", "ignore")
def _to_float(text: str) -> Optional[float]:
text = text.replace(",", "").strip().replace("+", "")
if not text or text in ("-", "N/A"):
return None
try:
return float(text)
except ValueError:
return None
def _cells(row_html: str) -> list[str]:
return [
html.unescape(re.sub(r"<[^>]+>", "", td)).strip()
for td in re.findall(r"<t[dh][^>]*>(.*?)</t[dh]>", row_html, re.S)
]
def parse_refining_html(html_text: str) -> RefiningSnapshot:
"""Parse the EIA Today in Energy daily prices page."""
tables = re.findall(r"<table[^>]*>(.*?)</table>", html_text, re.S)
if not tables:
raise RefiningError("no tables found in EIA prices page")
crack = crack_chg = None
wti = brent = gasoline_gulf = None
current_product = ""
for table in tables:
rows = re.findall(r"<tr[^>]*>(.*?)</tr>", table, re.S)
for row_html in rows:
cells = _cells(row_html)
text = " ".join(cells)
low = text.lower()
# Track the product section from its header row (cells[0] non-empty,
# e.g. "Gasoline (RBOB) ($/gallon)" / "Heating Oil ($/gallon)").
if cells and cells[0].strip():
c0 = cells[0].lower()
# Only update product section from a product header row (which
# carries a unit like "($/barrel)" / "($/gallon)") — Area-only
# sub-rows like 'Gulf Coast' / 'Brent' must not reset it.
if "($/" in c0 or "($ /" in c0:
if "gasoline" in c0 or "rbbob" in c0:
current_product = "gasoline"
elif "heating oil" in c0:
current_product = "heating_oil"
elif "low-sulfur diesel" in c0 or "diesel" in c0:
current_product = "diesel"
else:
current_product = c0
if len(cells) >= 2:
area = cells[1].strip().lower()
c0 = cells[0].strip().lower() if cells and cells[0].strip() else ""
# When the row's Area label sits in cells[0] (Product column not
# repeated, e.g. ['Brent','96.92','+3.1']), price is cells[1];
# when it sits in cells[1] (e.g. ['','Brent','96.92','+3.1']),
# price is cells[2].
if c0 == "brent" or area == "brent":
brent = (_to_float(cells[1]) if c0 == "brent" else (_to_float(cells[2]) if len(cells) > 2 else None))
elif c0 == "wti" or area == "wti":
wti = (_to_float(cells[1]) if c0 == "wti" else (_to_float(cells[2]) if len(cells) > 2 else None))
elif current_product == "gasoline" and (c0 == "gulf coast" or "gulf coast" in area):
gasoline_gulf = (_to_float(cells[1]) if c0 == "gulf coast" else (_to_float(cells[2]) if len(cells) > 2 else None))
if "3:2:1 crack spread" in low:
# row: '3:2:1 Crack Spread ($/barrel)', 'Gulf Coast (LLS)', '71.10', '+5.7'
for k in range(len(cells)):
if "gulf coast" in cells[k].lower():
crack = _to_float(cells[k + 1]) if k + 1 < len(cells) else None
crack_chg = _to_float(cells[k + 2]) if k + 2 < len(cells) else None
break
if crack is None and len(cells) >= 3:
crack = _to_float(cells[-2])
crack_chg = _to_float(cells[-1])
if crack is None and wti is None and brent is None:
raise RefiningError("no refining/energy values found in EIA prices page")
return RefiningSnapshot(
crack_spread=crack,
crack_spread_change=crack_chg,
wti_crude=wti,
brent_crude=brent,
gasoline_gulf=gasoline_gulf,
)
def fetch_refining(timeout: float = 30.0) -> RefiningSnapshot:
html_text = _fetch(timeout=timeout)
return parse_refining_html(html_text)

View File

@@ -0,0 +1,38 @@
"""Collect the Auto Credit factor (Trading Economics vehicle sales) to a JSON snapshot."""
from __future__ import annotations
import argparse
import json
from datetime import datetime, timezone
from pathlib import Path
from app import auto_credit
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--out", type=Path,
help="write JSON snapshot to this file (default: stdout)")
parser.add_argument("--timeout", type=float, default=30.0)
args = parser.parse_args()
snap = auto_credit.fetch_auto_credit(timeout=args.timeout)
payload = {
"source": "tradingeconomics",
"retrieved_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
"factor": "auto_credit",
"data": snap.to_dict(),
}
text = json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2)
if args.out:
args.out.parent.mkdir(parents=True, exist_ok=True)
args.out.write_text(text, encoding="utf-8")
print(f"wrote auto_credit snapshot -> {args.out}")
else:
print(text)
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,38 @@
"""Collect the Refining/Energy factor (EIA crack spread) to a JSON snapshot."""
from __future__ import annotations
import argparse
import json
from datetime import datetime, timezone
from pathlib import Path
from app import refining_energy
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--out", type=Path,
help="write JSON snapshot to this file (default: stdout)")
parser.add_argument("--timeout", type=float, default=30.0)
args = parser.parse_args()
snap = refining_energy.fetch_refining(timeout=args.timeout)
payload = {
"source": "eia",
"retrieved_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
"factor": "refining_energy",
"data": snap.to_dict(),
}
text = json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2)
if args.out:
args.out.parent.mkdir(parents=True, exist_ok=True)
args.out.write_text(text, encoding="utf-8")
print(f"wrote refining_energy snapshot -> {args.out}")
else:
print(text)
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,61 @@
"""Tests for the Auto Credit (Trading Economics) parser."""
from __future__ import annotations
import unittest
from app import auto_credit
def _trading_econ_html() -> str:
announcement = """
<table><tr><th>Date</th><th>Time</th><th>Statistic</th><th>Period</th><th>Actual</th><th>Previous</th></tr>
<tr><td>2026-06-29</td><td>03:45 AM</td><td>New Car Sales YoY</td><td>May</td><td>10.60%</td><td>2.54%</td></tr>
<tr><td>2026-07-24</td><td>04:20 AM</td><td>New Car Sales YoY</td><td>Jun</td><td>17.26%</td><td>10.60%</td></tr>
</table>"""
indicators = """
<table><tr><th>Indicator</th><th>Actual</th><th>Previous</th><th>Unit</th><th>Period</th></tr>
<tr><td>Auto Exports</td><td>81526.00</td><td>59434.00</td><td>Units</td><td>Jun 2026</td></tr>
<tr><td>Vehicle Production</td><td>120391.00</td><td>114214.00</td><td>Units</td><td>Jun 2026</td></tr>
<tr><td>Passenger Car Sales</td><td>22875.00</td><td>19389.00</td><td>Units</td><td>Jun 2026</td></tr>
</table>"""
main = """
<table><tr><th> </th><th>Actual</th><th>Previous</th><th>Highest</th><th>Lowest</th><th>Dates</th><th>Unit</th><th>Frequency</th><th>Seasonality</th></tr>
<tr><td></td><td>61244.00</td><td>57765.00</td><td>157529.00</td><td>5338.00</td><td>1980 - 2026</td><td>Units</td><td>Monthly</td><td>Volume, NSA</td></tr>
</table>"""
return f"<html><body>{announcement}{indicators}{main}</body></html>"
class AutoCreditParseTest(unittest.TestCase):
def test_parses_auto_credit_values(self) -> None:
snap = auto_credit.parse_auto_credit_html(_trading_econ_html())
self.assertEqual(snap.total_vehicle_sales, 61244.00)
self.assertEqual(snap.prev_vehicle_sales, 57765.00)
self.assertEqual(snap.vehicle_production, 120391.00)
self.assertEqual(snap.passenger_car_sales, 22875.00)
self.assertEqual(snap.auto_exports, 81526.00)
# New Car Sales YoY — latest (Jun 2026) is the last announcement row
self.assertEqual(snap.new_car_sales_yoy, 17.26)
self.assertEqual(snap.prev_new_car_sales_yoy, 10.60)
def test_missing_table_raises(self) -> None:
with self.assertRaises(auto_credit.AutoCreditError):
auto_credit.parse_auto_credit_html("<html><body>no data</body></html>")
def test_negative_and_percent_handling(self) -> None:
# negative YoY should parse as negative float
html_text = """
<table><tr><td>2026-08-25</td><td>04:00 AM</td><td>New Car Sales YoY</td><td>Jul</td><td>-3.5%</td><td>17.26%</td></tr>
</table>"""
snap = auto_credit.parse_auto_credit_html(html_text)
self.assertEqual(snap.new_car_sales_yoy, -3.5)
def test_build_dict(self) -> None:
snap = auto_credit.parse_auto_credit_html(_trading_econ_html())
d = snap.to_dict()
self.assertEqual(d["source"], "tradingeconomics")
self.assertIn("total_vehicle_sales", d)
if __name__ == "__main__":
unittest.main()

View File

@@ -0,0 +1,58 @@
"""Tests for the Refining/Energy (EIA) parser."""
from __future__ import annotations
import unittest
from app import refining_energy
def _eia_html() -> str:
return """
<html><body>
<h2>Wholesale Spot Petroleum Prices, 8/21/26 Close</h2>
<table>
<tr><th>Product</th><th>Area</th><th>Price</th><th>PercentChange*</th></tr>
<tr><td>Crude Oil ($/barrel)</td><td>WTI</td><td>87.21</td><td>-2.8</td></tr>
<tr><td></td><td>Brent</td><td>96.92</td><td>+3.1</td></tr>
<tr><td></td><td>Louisiana Light</td><td>90.71</td><td>-2.7</td></tr>
<tr><td>Gasoline (RBOB) ($/gallon)</td><td>NY Harbor</td><td>3.42</td><td>+2.3</td></tr>
<tr><td></td><td>Gulf Coast</td><td>3.52</td><td>+1.3</td></tr>
<tr><td>3:2:1 Crack Spread ($/barrel)</td><td>Gulf Coast (LLS)</td><td>71.10</td><td>+5.7</td></tr>
<tr><td>Low-Sulfur Diesel ($/gallon)</td><td>NY Harbor</td><td>4.47</td><td>-2.1</td></tr>
</table>
</body></html>
"""
class RefiningParseTest(unittest.TestCase):
def test_parses_crack_spread_and_crudes(self) -> None:
snap = refining_energy.parse_refining_html(_eia_html())
self.assertEqual(snap.crack_spread, 71.10)
self.assertEqual(snap.crack_spread_change, 5.7)
self.assertEqual(snap.wti_crude, 87.21)
self.assertEqual(snap.brent_crude, 96.92)
self.assertEqual(snap.gasoline_gulf, 3.52)
def test_missing_table_raises(self) -> None:
with self.assertRaises(refining_energy.RefiningError):
refining_energy.parse_refining_html("<html>no data</html>")
def test_negative_change_parses(self) -> None:
html_text = """
<table><tr><td>3:2:1 Crack Spread ($/barrel)</td><td>Gulf Coast (LLS)</td><td>60.00</td><td>-2.5</td></tr></table>"""
snap = refining_energy.parse_refining_html(html_text)
self.assertEqual(snap.crack_spread, 60.00)
self.assertEqual(snap.crack_spread_change, -2.5)
def test_crack_spread_fallback_when_area_missing(self) -> None:
# If the row lacks the "Gulf Coast" area cell, fall back to last two cells.
html_text = """
<table><tr><td>3:2:1 Crack Spread ($/barrel)</td><td>71.10</td><td>+5.7</td></tr></table>"""
snap = refining_energy.parse_refining_html(html_text)
self.assertEqual(snap.crack_spread, 71.10)
self.assertEqual(snap.crack_spread_change, 5.7)
if __name__ == "__main__":
unittest.main()