From 3f7fccd25b7659e3467af91660ce9992143d5fdd Mon Sep 17 00:00:00 2001 From: Kunthawat Greethong Date: Tue, 25 Aug 2026 15:13:36 +0700 Subject: [PATCH] [verified] Add Thai Energy/Refining collector (TOP quarterly financials) replacing EIA US crack spread - energy_thai.py: scrape Thai Oil (TOP) investor financial-highlights -> quarterly + annual EBITDA/Net Profit/Sales (Million Baht), largest Thai refinery - Thai-specific factor per user (energy must reflect Thai companies, not US EIA proxy); Krungsri was projection-only, TOP gives real quarterly actuals - Frequency: quarterly (documented in research note) - 4 tests; full backend suite 156 OK; compileall ok; static scan clean --- backend/app/energy_thai.py | 170 +++++++++++++++++++++ backend/scripts/collect_energy_thai.py | 29 ++++ backend/tests/test_energy_thai.py | 68 +++++++++ docs/alternative-factor-source-research.md | 3 +- 4 files changed, 269 insertions(+), 1 deletion(-) create mode 100644 backend/app/energy_thai.py create mode 100644 backend/scripts/collect_energy_thai.py create mode 100644 backend/tests/test_energy_thai.py diff --git a/backend/app/energy_thai.py b/backend/app/energy_thai.py new file mode 100644 index 0000000..f6d262b --- /dev/null +++ b/backend/app/energy_thai.py @@ -0,0 +1,170 @@ +"""Thai Energy/Refining factor — Thai Oil (TOP) quarterly financial highlights. + +Source: https://investor.thaioilgroup.com/en/financial-information/financial-highlights +(Thai Oil PCL, the largest Thai refinery operator). Server-rendered HTML (the +only real XHR is a breadcrumbs API; the financial tables are in the HTML, so we +scrape the table markup per the `har-derived-api-client` guidance). + +Data: quarterly and annual financial tables with rows + Sales Revenue, EBITDA, Net Profit/(Loss), Basic EPS (Million Baht) +and columns like Q2/2026, Q1/2026, Q4/2025, Q3/2025, Q2/2025 (and annual YYYY). + +This is the Thai-specific energy/refining proxy the user chose (replacing the +US EIA crack spread, so the factor reflects a Thai company). +""" + +from __future__ import annotations + +import html +import re +from dataclasses import dataclass, field +from typing import Optional +from urllib.request import Request, urlopen + +_USER_AGENT = ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" +) +_URL = "https://investor.thaioilgroup.com/en/financial-information/financial-highlights" + + +class EnergyThaiError(Exception): + """Raised when the TOP financial page cannot be fetched or parsed.""" + + +@dataclass(frozen=True) +class EnergyThaiSnapshot: + # quarterly: {period: {"sales":.., "ebitda":.., "net_profit":.., "basic_eps":..}} + quarterly: dict = field(default_factory=dict) + annual: dict = field(default_factory=dict) + source: str = "thaioil" + as_of: str = "" + + def to_dict(self) -> dict: + return { + "source": self.source, + "as_of": self.as_of, + "quarterly": self.quarterly, + "annual": self.annual, + } + + +def _fetch(url: str = _URL, timeout: float = 30.0) -> str: + req = Request(url, headers={"User-Agent": _USER_AGENT, "Accept": "text/html"}) + try: + with urlopen(req, timeout=timeout) as resp: + raw = resp.read() + except Exception as exc: + raise EnergyThaiError(f"failed to fetch {url}: {exc}") from exc + try: + return raw.decode("utf-8") + except UnicodeDecodeError: + return raw.decode("latin-1", "ignore") + + +def _to_float(text: str) -> Optional[float]: + text = text.replace(",", "").strip() + if not text or text in ("-", "N/A", "—"): + return None + try: + return float(text) + except ValueError: + return None + + +def _cells(row_html: str) -> list[str]: + return [ + html.unescape(re.sub(r"<[^>]+>", "", td)).strip() + for td in re.findall(r"]*>(.*?)", row_html, re.S) + ] + + +def _parse_financial_table(table_html: str) -> dict: + """Parse one 'table--financial' table into {columns: [...], rows: {label: [vals]}}.""" + columns: list[str] = [] + rows: dict[str, list[Optional[float]]] = {} + trs = re.findall(r"]*>(.*?)", table_html, re.S) + for tr in trs: + cells = _cells(tr) + cells = [c for c in cells if c] + if not cells: + continue + # header row: first cell empty/th, rest are period labels (Q2/2026, 2025) + if cells[0] == "" and len(cells) > 1 and not _to_float(cells[1]) is None: + columns = cells[1:] + continue + if re.match(r"^(Q\d/\d{4}|\d{4})$", cells[0]): + columns = cells + continue + # section header like "Operating" (colspan) -> skip + if len(cells) == 1: + continue + # data row: [label, val1, val2, ...] + label = cells[0] + # normalize common labels + norm = label.lower() + key = None + if "sales revenue" in norm: + key = "sales" + elif norm.startswith("ebitda"): + key = "ebitda" + elif "net profit" in norm or "net loss" in norm: + key = "net_profit" + elif "basic earnings" in norm or "basic e/l" in norm or "eps" in norm: + key = "basic_eps" + if key: + rows[key] = [_to_float(c) for c in cells[1:]] + return {"columns": columns, "rows": rows} + + +def parse_energy_thai_html(html_text: str) -> dict: + """Parse the TOP financial-highlights page into {'quarterly':..., 'annual':...}.""" + tables = re.findall( + r']*class="[^"]*table--financial[^"]*"[^>]*>(.*?)', + html_text, re.S, + ) + if not tables: + # fallback: any table containing Qx/YYYY header + tables = [t for t in re.findall(r"]*>(.*?)", html_text, re.S) + if re.search(r"Q\d/\d{4}", t)] + if not tables: + raise EnergyThaiError("no financial tables found in TOP page") + + q = None + a = None + for t in tables: + parsed = _parse_financial_table(t) + cols = parsed.get("columns", []) + # quarterly periods look like 'Q1/2026'; annual like '2025'. + if any(re.match(r"^Q\d/\d{4}$", c) for c in cols): + q = parsed + elif any(re.match(r"^\d{4}$", c) for c in cols): + a = parsed + + if q is None and a is None: + raise EnergyThaiError("no recognizable financial periods found in TOP page") + return {"quarterly": q or {}, "annual": a or {}} + + +def _to_period_map(parsed: dict) -> dict: + """Convert {columns, rows} into {period: {metric: value}}.""" + cols = parsed.get("columns", []) + rows = parsed.get("rows", {}) + out: dict = {} + for i, col in enumerate(cols): + out[col] = {} + for key, vals in rows.items(): + if i < len(vals): + out[col][key] = vals[i] + return out + + +def fetch_energy_thai(timeout: float = 30.0) -> EnergyThaiSnapshot: + html_text = _fetch(timeout=timeout) + parsed = parse_energy_thai_html(html_text) + qmap = _to_period_map(parsed["quarterly"]) + amap = _to_period_map(parsed["annual"]) + # as_of = latest quarterly period (first col) + periods = list(qmap.keys()) + as_of = periods[0] if periods else "" + return EnergyThaiSnapshot(quarterly=qmap, annual=amap, as_of=as_of) diff --git a/backend/scripts/collect_energy_thai.py b/backend/scripts/collect_energy_thai.py new file mode 100644 index 0000000..fc578db --- /dev/null +++ b/backend/scripts/collect_energy_thai.py @@ -0,0 +1,29 @@ +"""Collect the Thai Energy/Refining factor (TOP quarterly financials) to JSON snapshot.""" + +from __future__ import annotations + +import argparse +import json +from datetime import datetime, timezone + +from app import energy_thai + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--timeout", type=float, default=30.0) + args = parser.parse_args() + + snap = energy_thai.fetch_energy_thai(timeout=args.timeout) + payload = { + "source": "thaioil", + "retrieved_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + "factor": "energy_thai", + "data": snap.to_dict(), + } + print(json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/backend/tests/test_energy_thai.py b/backend/tests/test_energy_thai.py new file mode 100644 index 0000000..b3b8e42 --- /dev/null +++ b/backend/tests/test_energy_thai.py @@ -0,0 +1,68 @@ +"""Tests for the Thai Energy/Refining (TOP financial highlights) parser.""" + +from __future__ import annotations + +import unittest + +from app import energy_thai + + +def _top_html() -> str: + return """ + + + + + + + + + + +
202520242023
Operating
Sales Revenue394,336455,857459,402
EBITDA17,61922,02635,453
Net Profit / (Loss)14,5849,95919,443
Basic Earnings / (Loss) (Baht / Share)6.534.468.70
+ + + + + + + + + +
Q2/2026Q1/2026Q4/2025
Operating
Sales Revenue129,709114,809108,931
EBITDA8,91531,6415,981
Net Profit / (Loss)8,28419,4812,458
Basic Earnings / (Loss) (Baht / Share)3.718.411.10
+ + """ + + +class EnergyThaiParseTest(unittest.TestCase): + def test_parses_quarterly_and_annual(self) -> None: + parsed = energy_thai.parse_energy_thai_html(_top_html()) + self.assertIn("Q2/2026", parsed["quarterly"]["columns"]) + self.assertIn("2025", parsed["annual"]["columns"]) + self.assertEqual(parsed["quarterly"]["rows"]["net_profit"][0], 8284.0) + self.assertEqual(parsed["annual"]["rows"]["ebitda"][0], 17619.0) + + def test_period_map(self) -> None: + parsed = energy_thai.parse_energy_thai_html(_top_html()) + qmap = energy_thai._to_period_map(parsed["quarterly"]) + self.assertEqual(qmap["Q2/2026"]["net_profit"], 8284.0) + self.assertEqual(qmap["Q2/2026"]["sales"], 129709.0) + amap = energy_thai._to_period_map(parsed["annual"]) + self.assertEqual(amap["2025"]["ebitda"], 17619.0) + + def test_no_tables_raises(self) -> None: + with self.assertRaises(energy_thai.EnergyThaiError): + energy_thai.parse_energy_thai_html("no data") + + def test_negative_net_profit_parses(self) -> None: + html_text = """ + + + +
Q1/2026
Net Profit / (Loss)-5,380
""" + parsed = energy_thai.parse_energy_thai_html(html_text) + self.assertEqual(parsed["quarterly"]["rows"]["net_profit"][0], -5380.0) + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/alternative-factor-source-research.md b/docs/alternative-factor-source-research.md index 73168b9..587b1b8 100644 --- a/docs/alternative-factor-source-research.md +++ b/docs/alternative-factor-source-research.md @@ -14,7 +14,8 @@ The US EIA 3-2-1 crack spread is a **US** proxy; the user wants **Thai** refiner - **TOP (Thai Oil)** analyst-meeting PDFs (`top.listedcompany.com`) — GRM/GIM per quarter, $/bbl. - **EPPO / MOPH** — Thai retail fuel prices (more frequent, but that's fuel price, not GRM). -**Decision state:** Thai GRM quarterly is the direction; exact source (Krungsri aggregate vs TOP) pending user confirmation. +**Decision (final):** Use **Thai Oil (TOP) quarterly financial highlights** (`investor.thaioilgroup.com/en/financial-information/financial-highlights`) — real quarterly EBITDA/Net Profit/Sales of the largest Thai refinery (Million Baht), scraped from server-rendered HTML. Implemented in `backend/app/energy_thai.py`. This is Thai-specific and replaces both the US EIA crack spread and the Krungsri outlook projection. Frequency: **quarterly**. (Krungsri Research remains a qualitative backdrop; EIA is a US proxy — both superseded by TOP for the factor.) +