301 lines
9.2 KiB
Python
301 lines
9.2 KiB
Python
#!/usr/bin/env python3
|
|
"""Discover accessible SGX endpoints listed in build.md.
|
|
|
|
This script probes every URL under the `endpoints` object in build.md,
|
|
except FINANCIAL_REPORTS_API_URL, and records whether each endpoint is
|
|
accessible or blocked. Raw responses and metadata are written to
|
|
discovery/cache/; per-endpoint notes are written to discovery/endpoints/;
|
|
and a master index is written to discovery/discovery.md.
|
|
|
|
This script uses only the Python standard library to avoid external
|
|
dependencies.
|
|
"""
|
|
|
|
import json
|
|
import re
|
|
import ssl
|
|
import sys
|
|
import time
|
|
import urllib.error
|
|
import urllib.request
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
|
|
# Paths relative to the repository root.
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
BUILD_MD = ROOT / "build.md"
|
|
CACHE_DIR = ROOT / "discovery" / "cache"
|
|
ENDPOINTS_DIR = ROOT / "discovery" / "endpoints"
|
|
DISCOVERY_MD = ROOT / "discovery" / "discovery.md"
|
|
|
|
# Endpoint explicitly excluded from discovery per build.md.
|
|
EXCLUDED_KEY = "FINANCIAL_REPORTS_API_URL"
|
|
|
|
# Polite delay between requests.
|
|
DELAY_SECONDS = 1.5
|
|
REQUEST_TIMEOUT = 30.0
|
|
|
|
# Heuristic markers that indicate the request was blocked or redirected to an
|
|
# interstitial page instead of the expected API response.
|
|
BLOCKED_MARKERS = (
|
|
b"captcha",
|
|
b"recaptcha",
|
|
b"challenge",
|
|
b"cloudflare",
|
|
b"akamai",
|
|
b"access denied",
|
|
b"accessdenied",
|
|
b"blocked",
|
|
b"waf",
|
|
b"forbidden",
|
|
b"unauthorized",
|
|
)
|
|
|
|
# Hosts we expect to return JSON rather than HTML.
|
|
JSON_HOSTS = ("api.sgx.com", "api2.sgx.com", "api3.sgx.com", "links.sgx.com")
|
|
|
|
|
|
def extract_endpoints(path: Path) -> dict[str, str]:
|
|
"""Extract the endpoints object from build.md.
|
|
|
|
build.md embeds a snapshot of appconfig.json between descriptive text.
|
|
This function locates the outer JSON object by balancing braces and
|
|
returns the `endpoints` mapping.
|
|
"""
|
|
text = path.read_text(encoding="utf-8")
|
|
start_marker = "Snapshot Version :"
|
|
start_idx = text.find(start_marker)
|
|
if start_idx == -1:
|
|
raise ValueError(f"Could not find '{start_marker}' in {path}")
|
|
|
|
json_start = text.find("{", start_idx)
|
|
if json_start == -1:
|
|
raise ValueError(f"Could not find JSON start in {path}")
|
|
|
|
# Balance braces to find the end of the JSON object.
|
|
depth = 0
|
|
json_end = -1
|
|
for i, char in enumerate(text[json_start:], start=json_start):
|
|
if char == "{":
|
|
depth += 1
|
|
elif char == "}":
|
|
depth -= 1
|
|
if depth == 0:
|
|
json_end = i + 1
|
|
break
|
|
if json_end == -1:
|
|
raise ValueError(f"Could not find matching '}}' in {path}")
|
|
|
|
config = json.loads(text[json_start:json_end])
|
|
endpoints = config.get("endpoints")
|
|
if not isinstance(endpoints, dict):
|
|
raise ValueError(f"No 'endpoints' object found in {path}")
|
|
return endpoints
|
|
|
|
|
|
def is_likely_api(url: str) -> bool:
|
|
return any(host in url.lower() for host in JSON_HOSTS)
|
|
|
|
|
|
def looks_blocked(
|
|
status: int, body: bytes, content_type: str, url: str
|
|
) -> tuple[bool, list[str]]:
|
|
reasons: list[str] = []
|
|
if status == 0:
|
|
reasons.append("network/transport error")
|
|
elif status >= 400:
|
|
reasons.append(f"HTTP {status}")
|
|
|
|
lower_body = body[:8192].lower()
|
|
for marker in BLOCKED_MARKERS:
|
|
if marker in lower_body:
|
|
reasons.append(f"blocked marker: {marker.decode()}")
|
|
|
|
if is_likely_api(url) and "text/html" in content_type.lower():
|
|
reasons.append("unexpected HTML for API endpoint")
|
|
|
|
return (bool(reasons), reasons)
|
|
|
|
|
|
def safe_filename(key: str) -> str:
|
|
return re.sub(r"[^A-Za-z0-9_]+", "_", key).strip("_")
|
|
|
|
|
|
def probe_endpoint(key: str, url: str) -> tuple[dict, bytes]:
|
|
meta = {
|
|
"key": key,
|
|
"url": url,
|
|
"timestamp_utc": datetime.now(timezone.utc).isoformat(),
|
|
"status": 0,
|
|
"content_type": "",
|
|
"content_length": 0,
|
|
"elapsed_seconds": 0.0,
|
|
"error": "",
|
|
"blocked": True,
|
|
"block_reasons": [],
|
|
}
|
|
|
|
request = urllib.request.Request(
|
|
url,
|
|
headers={
|
|
"Accept": "*/*",
|
|
"User-Agent": "sgx-scrapper-discovery/0.1",
|
|
},
|
|
)
|
|
|
|
# Use a default SSL context; SGX endpoints are HTTPS.
|
|
context = ssl.create_default_context()
|
|
|
|
try:
|
|
start = time.perf_counter()
|
|
with urllib.request.urlopen(
|
|
request, timeout=REQUEST_TIMEOUT, context=context
|
|
) as response:
|
|
elapsed = time.perf_counter() - start
|
|
body = response.read()
|
|
meta.update(
|
|
{
|
|
"status": response.status,
|
|
"content_type": response.headers.get_content_type()
|
|
+ (
|
|
""
|
|
if not response.headers.get_content_charset()
|
|
else f"; charset={response.headers.get_content_charset()}"
|
|
),
|
|
"content_length": len(body),
|
|
"elapsed_seconds": round(elapsed, 3),
|
|
}
|
|
)
|
|
except urllib.error.HTTPError as exc:
|
|
# HTTPError still contains a response body and status code.
|
|
body = exc.read() if exc.fp else b""
|
|
meta.update(
|
|
{
|
|
"status": exc.code,
|
|
"content_type": exc.headers.get_content_type()
|
|
if exc.headers
|
|
else "",
|
|
"content_length": len(body),
|
|
"error": f"HTTPError: {exc}",
|
|
}
|
|
)
|
|
except urllib.error.URLError as exc:
|
|
meta["error"] = f"URLError: {exc}"
|
|
body = b""
|
|
except Exception as exc:
|
|
meta["error"] = f"{type(exc).__name__}: {exc}"
|
|
body = b""
|
|
|
|
blocked, reasons = looks_blocked(
|
|
meta["status"], body, meta["content_type"], url
|
|
)
|
|
meta["blocked"] = blocked
|
|
meta["block_reasons"] = reasons
|
|
|
|
return meta, body
|
|
|
|
|
|
def write_cache(key: str, body: bytes, meta: dict) -> None:
|
|
CACHE_DIR.mkdir(parents=True, exist_ok=True)
|
|
safe = safe_filename(key)
|
|
body_path = CACHE_DIR / f"{safe}.json"
|
|
meta_path = CACHE_DIR / f"{safe}.meta.json"
|
|
body_path.write_bytes(body)
|
|
meta_path.write_text(
|
|
json.dumps(meta, indent=2, ensure_ascii=False), encoding="utf-8"
|
|
)
|
|
|
|
|
|
def write_endpoint_markdown(key: str, url: str, meta: dict) -> None:
|
|
ENDPOINTS_DIR.mkdir(parents=True, exist_ok=True)
|
|
safe = safe_filename(key)
|
|
md_path = ENDPOINTS_DIR / f"{safe}_meta.md"
|
|
status = meta["status"] or "ERR"
|
|
blocked = "BLOCKED" if meta["blocked"] else "ACCESSIBLE"
|
|
lines = [
|
|
f"# {key}",
|
|
"",
|
|
f"- **URL:** {url}",
|
|
f"- **Status:** {status}",
|
|
f"- **Classification:** {blocked}",
|
|
f"- **Content-Type:** {meta['content_type'] or 'n/a'}",
|
|
f"- **Content-Length:** {meta['content_length']} bytes",
|
|
f"- **Elapsed:** {meta['elapsed_seconds']} s",
|
|
f"- **Probed at:** {meta['timestamp_utc']}",
|
|
]
|
|
if meta["error"]:
|
|
lines.extend(["", f"- **Error:** {meta['error']}"])
|
|
if meta["block_reasons"]:
|
|
lines.extend(
|
|
["", f"- **Block reasons:** {', '.join(meta['block_reasons'])}"]
|
|
)
|
|
lines.extend(
|
|
[
|
|
"",
|
|
"## Notes",
|
|
"",
|
|
"_Add observations about schema, pagination, or usefulness here._",
|
|
"",
|
|
]
|
|
)
|
|
md_path.write_text("\n".join(lines), encoding="utf-8")
|
|
|
|
|
|
def write_master_index(results: list[dict]) -> None:
|
|
lines = [
|
|
"# SGX Endpoint Discovery",
|
|
"",
|
|
f"Generated: {datetime.now(timezone.utc).isoformat()}",
|
|
"",
|
|
"## Summary",
|
|
"",
|
|
f"- **Total probed:** {len(results)}",
|
|
f"- **Accessible:** {sum(1 for r in results if not r['blocked'])}",
|
|
f"- **Blocked:** {sum(1 for r in results if r['blocked'])}",
|
|
"",
|
|
"## Index",
|
|
"",
|
|
"| Key | URL | Status | Class | Content-Type | Size | Notes |",
|
|
"| --- | --- | ------ | ----- | ------------ | ---- | ----- |",
|
|
]
|
|
for r in results:
|
|
notes = "; ".join(r["block_reasons"]) or (
|
|
"OK" if not r["blocked"] else ""
|
|
)
|
|
status = r["status"] or "ERR"
|
|
cls = "ACCESSIBLE" if not r["blocked"] else "BLOCKED"
|
|
lines.append(
|
|
f"| {r['key']} | {r['url']} | {status} | {cls} | "
|
|
f"{r['content_type'] or 'n/a'} | {r['content_length']} | {notes} |"
|
|
)
|
|
lines.append("")
|
|
DISCOVERY_MD.parent.mkdir(parents=True, exist_ok=True)
|
|
DISCOVERY_MD.write_text("\n".join(lines), encoding="utf-8")
|
|
|
|
|
|
def main() -> int:
|
|
if not BUILD_MD.exists():
|
|
print(f"ERROR: {BUILD_MD} not found.", file=sys.stderr)
|
|
return 1
|
|
|
|
endpoints = extract_endpoints(BUILD_MD)
|
|
if EXCLUDED_KEY in endpoints:
|
|
del endpoints[EXCLUDED_KEY]
|
|
|
|
results: list[dict] = []
|
|
for key, url in endpoints.items():
|
|
print(f"Probing {key} ...")
|
|
meta, body = probe_endpoint(key, url)
|
|
write_cache(key, body, meta)
|
|
write_endpoint_markdown(key, url, meta)
|
|
results.append(meta)
|
|
time.sleep(DELAY_SECONDS)
|
|
|
|
write_master_index(results)
|
|
print(f"\nDone. Results: {DISCOVERY_MD}")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|