#!/usr/bin/env python3 """Export Vaneform's cached monthly site estimates. Python 3.10+, stdlib only. Live: VANEFORM_API_KEY in the environment; domains as positional arguments. Replay: --snapshot traffic-api-snapshot-2026-10-09.json (no network or key). Exit 0: all selected rows usable; 1: CSV includes unavailable/error rows; 2: invalid input, missing key or an unwritable output. Never fill gaps with zero. """ import argparse import csv from datetime import datetime, timezone import json import math import os from pathlib import Path import re import sys import time import urllib.error import urllib.request API = "https://vaneform.com/api/v1/domains/" MONTH = re.compile(r"^\d{4}-(0[1-9]|1[0-2])$") FIELDS = ["domain", "month", "visits", "previous_month", "previous_visits", "mom_change_pct", "estimate", "confidence", "as_of", "retrieved_at", "resource_status", "scale_status", "row_status", "source_url", "error_code"] MAX_BODY = 2_000_000 class NoRedirect(urllib.request.HTTPRedirectHandler): def redirect_request(self, req, fp, code, msg, headers, newurl): return None def domain_arg(value): domain = value.strip().lower() labels = domain.split(".") if (len(domain) > 253 or len(labels) < 2 or any(not re.fullmatch(r"[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?", label) for label in labels) or not re.fullmatch(r"[a-z][a-z0-9-]+", labels[-1])): raise argparse.ArgumentTypeError("Use a bare ASCII domain (punycode for IDNs).") return domain def month_arg(value): if not MONTH.fullmatch(value): raise argparse.ArgumentTypeError("Month must be YYYY-MM.") return value def decode_body(raw): if len(raw) > MAX_BODY: raise ValueError("response_too_large") body = json.loads(raw) if not isinstance(body, dict): raise ValueError("invalid_response") return body def fetch_capture(domain): token = os.environ["VANEFORM_API_KEY"] url = API + domain + "/scale" opener = urllib.request.build_opener(NoRedirect()) request = urllib.request.Request(url, headers={ "Authorization": "Bearer " + token, "Accept": "application/json", "User-Agent": "VaneformGuideCSV/1.0"}) for attempt in range(3): status, body = None, {} try: with opener.open(request, timeout=30) as response: status = response.status body = decode_body(response.read(MAX_BODY + 1)) except urllib.error.HTTPError as error: status = error.code try: body = decode_body(error.read(MAX_BODY + 1)) except (ValueError, UnicodeError): body = {"error": {"code": "http_error"}} finally: error.close() except (urllib.error.URLError, TimeoutError, OSError): # Keep network failure distinct from an empty estimate; a rerun is explicit. body = {"error": {"code": "network_error"}} except (ValueError, UnicodeError): body = {"error": {"code": "invalid_response"}} error = body.get("error") code = error.get("code") if isinstance(error, dict) else None retryable = status in (500, 502, 503, 504) or (status == 429 and code == "lookup_rate_exceeded") if retryable and attempt < 2: time.sleep(2 ** attempt) continue return {"domain": domain, "retrieved_at": datetime.now(timezone.utc).isoformat(), "source_url": url, "http_status": status, "body": body} def previous_month(month): year, number = map(int, month.split("-")) return f"{year - 1}-12" if number == 1 else f"{year}-{number - 1:02d}" def numeric(value): return isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value) and value >= 0 def rows_for(capture, selected_month): domain = domain_arg(capture["domain"]) body = capture["body"] if not isinstance(body, dict): raise ValueError("Snapshot body must be an object.") base = dict(domain=domain, retrieved_at=capture["retrieved_at"], source_url=capture["source_url"], resource_status=body.get("status"), month=selected_month) error = body.get("error") if capture["http_status"] != 200 or error: code = error.get("code", "http_error") if isinstance(error, dict) else "http_error" return [dict(base, row_status="request_error", error_code=code)] if body.get("domain") != domain: return [dict(base, row_status="invalid_response", error_code="domain_mismatch")] scale = body.get("scale") if not isinstance(scale, dict): return [dict(base, row_status="invalid_response", error_code="scale_missing")] base.update(scale_status=scale.get("status"), estimate=scale.get("estimate"), confidence=scale.get("confidence"), as_of=scale.get("as_of")) if scale.get("status") not in ("ready", "partial"): return [dict(base, row_status="scale_unavailable")] if scale.get("estimate") != "site": return [dict(base, row_status="unsupported_estimate")] if (scale.get("confidence") not in ("low", "medium", "high") or not isinstance(scale.get("as_of"), str) or not scale["as_of"]): return [dict(base, row_status="invalid_response", error_code="provenance_missing")] series = scale.get("monthly_visits") if not isinstance(series, list) or not series: # The scalar visits field is deliberately not assigned a guessed month. return [dict(base, row_status="history_unavailable")] values = {} for point in series: if (not isinstance(point, dict) or not isinstance(point.get("month"), str) or not MONTH.fullmatch(point["month"]) or point["month"] in values or not numeric(point.get("visits"))): return [dict(base, row_status="invalid_response", error_code="invalid_history")] values[point["month"]] = point["visits"] if selected_month and selected_month not in values: return [dict(base, row_status="month_unavailable")] rows = [] for month in ([selected_month] if selected_month else sorted(values)): prior = previous_month(month) baseline = values.get(prior) growth = round((values[month] / baseline - 1) * 100, 2) if baseline is not None and baseline > 0 else None rows.append(dict(base, month=month, visits=values[month], previous_month=prior, previous_visits=baseline, mom_change_pct=growth, row_status="ok")) return rows def csv_value(value): if value is None: return "" # Quote dangerous text for spreadsheet imports; numeric growth remains numeric. if isinstance(value, str) and value.lstrip().startswith(("=", "+", "-", "@")): return "'" + value return value def main(): parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) parser.add_argument("domains", nargs="*", type=domain_arg) parser.add_argument("--snapshot", type=Path, help="Replay the supplied capture JSON; no network calls") parser.add_argument("--month", type=month_arg, help="Select one recorded month, never the nearest month") parser.add_argument("--output", required=True, type=Path, help="New CSV path; refuses to overwrite") args = parser.parse_args() if args.snapshot and args.domains: parser.error("Choose a snapshot or live domains, not both.") if not args.snapshot and not args.domains: parser.error("Provide domains or --snapshot.") if not args.snapshot and not os.environ.get("VANEFORM_API_KEY", "").strip(): parser.error("Set VANEFORM_API_KEY in your environment.") if args.output.exists(): parser.error("Output exists. Choose a new path to preserve your earlier snapshot.") try: if args.snapshot: captures = json.loads(args.snapshot.read_text(encoding="utf-8")) if not isinstance(captures, list) or not captures: raise ValueError("Snapshot must contain a nonempty capture array.") else: captures = [fetch_capture(domain) for domain in dict.fromkeys(args.domains)] rows = [row for capture in captures for row in rows_for(capture, args.month)] with args.output.open("x", encoding="utf-8", newline="") as output: writer = csv.DictWriter(output, fieldnames=FIELDS, lineterminator="\n") writer.writeheader() writer.writerows({key: csv_value(row.get(key)) for key in FIELDS} for row in rows) except (OSError, ValueError, KeyError, TypeError, argparse.ArgumentTypeError): # Do not echo credentials, arbitrary server bodies or local file contents. print("Export failed: check the snapshot structure and output path.", file=sys.stderr) return 2 unavailable = sum(row["row_status"] != "ok" for row in rows) print(f"Wrote {len(rows)} rows; {unavailable} unavailable/error rows.") return 1 if unavailable else 0 if __name__ == "__main__": sys.exit(main())