[object Object]

← back to Ga Allsites

snapshot before restart: preserve in-flight work

6ef7b5c65bd0eb27dd3a3f06b406b5f90ca008f9 · 2026-09-04 22:38:51 -0700 · Steve

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_011gbmuMHgCLDEPM4TpfAr4G

Files touched

Diff

commit 6ef7b5c65bd0eb27dd3a3f06b406b5f90ca008f9
Author: Steve <steve@designerwallcoverings.com>
Date:   Fri Sep 4 22:38:51 2026 -0700

    snapshot before restart: preserve in-flight work
    
    Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
    Claude-Session: https://claude.ai/code/session_011gbmuMHgCLDEPM4TpfAr4G
---
 authoritative_sweep.py         | 319 +++++++++++++++++++++++++++++++++++++++++
 gate1_create_fleet_property.py |  49 +++++++
 2 files changed, 368 insertions(+)

diff --git a/authoritative_sweep.py b/authoritative_sweep.py
new file mode 100644
index 0000000..77ae246
--- /dev/null
+++ b/authoritative_sweep.py
@@ -0,0 +1,319 @@
+#!/usr/bin/env python3
+"""
+Authoritative GA4 drift audit.
+Step 1: Build measurement_id -> {property_name, stream_uri} map from Admin API (ground truth).
+Step 2: Enumerate fleet sites and their served G- IDs.
+Step 3: Cross-classify each (site, served_id) pair.
+READ-ONLY — no writes.
+"""
+from __future__ import annotations
+import json, os, re, subprocess, sys
+from pathlib import Path
+from urllib.parse import urlparse
+
+KEY = Path.home() / ".config/ga-analytics-agent/service-account.json"
+ACCOUNT_ID_FILE = Path.home() / ".config/ga-analytics-agent/account_id"
+
+# ---------------------------------------------------------------------------
+# Step 1: Build authoritative map from Admin API
+# ---------------------------------------------------------------------------
+
+def build_authoritative_map():
+    """Returns {measurement_id: {property_name, stream_uri, property_id}}"""
+    from google.analytics.admin import AnalyticsAdminServiceClient
+    import google.oauth2.service_account as sa
+
+    creds = sa.Credentials.from_service_account_file(
+        str(KEY),
+        scopes=["https://www.googleapis.com/auth/analytics.readonly"]
+    )
+    client = AnalyticsAdminServiceClient(credentials=creds)
+    account_id = ACCOUNT_ID_FILE.read_text().strip()
+
+    auth_map = {}  # measurement_id -> dict
+    properties = client.list_properties(request={"filter": f"parent:{account_id}"})
+    prop_list = list(properties)
+    print(f"  Found {len(prop_list)} properties under {account_id}", flush=True)
+
+    for p in prop_list:
+        prop_id = p.name.split("/")[-1]
+        try:
+            streams = client.list_data_streams(parent=p.name)
+            for s in streams:
+                if s.type_.name == "WEB_DATA_STREAM":
+                    mid = s.web_stream_data.measurement_id
+                    uri = s.web_stream_data.default_uri
+                    if mid:
+                        auth_map[mid] = {
+                            "property_name": p.display_name,
+                            "property_id": prop_id,
+                            "stream_uri": uri,
+                        }
+        except Exception as e:
+            print(f"  WARN: {p.display_name} ({prop_id}) stream error: {e}", flush=True)
+
+    return auth_map
+
+
+# ---------------------------------------------------------------------------
+# Step 2: Fleet site list + live curl to extract G- IDs
+# ---------------------------------------------------------------------------
+
+# Known front-facing fleet sites to check.
+# Format: (label, url)
+FLEET_SITES = [
+    # DW brand microsites (Kamatera-hosted)
+    ("designerwallcoverings.com",         "https://www.designerwallcoverings.com"),
+    ("grassclothwallcoverings.com",       "https://www.grassclothwallcoverings.com"),
+    ("grassclothwallpaper.com",           "https://www.grassclothwallpaper.com"),
+    ("silkwallcoverings.com",             "https://www.silkwallcoverings.com"),
+    ("silkwallpaper.com",                 "https://www.silkwallpaper.com"),
+    ("corkwallcovering.com",              "https://www.corkwallcovering.com"),
+    ("corkwallpaper.com",                 "https://www.corkwallpaper.com"),
+    ("linenwallcovering.com",             "https://www.linenwallcovering.com"),
+    ("linenwallpaper.com",                "https://www.linenwallpaper.com"),
+    ("jutewallpaper.com",                 "https://www.jutewallpaper.com"),
+    ("jutewallcovering.com",              "https://www.jutewallcovering.com"),
+    ("raffiawallcovering.com",            "https://www.raffiawallcovering.com"),
+    ("raffiawallpaper.com",               "https://www.raffiawallpaper.com"),
+    ("micawallcovering.com",              "https://www.micawallcovering.com"),
+    ("micawallpaper.com",                 "https://www.micawallpaper.com"),
+    ("mylarcovering.com",                 "https://www.mylarcovering.com"),
+    ("metsecwallcovering.com",            "https://www.metsecwallcovering.com"),
+    ("suedewall.com",                     "https://www.suedewall.com"),
+    ("novasuede-website.com",             "https://www.novasuede.com"),
+    ("glassbeadedwallcovering.com",       "https://www.glassbeadedwallcovering.com"),
+    ("flockedwallcovering.com",           "https://www.flockedwallcovering.com"),
+    ("goldleafwallpaper.com",             "https://www.goldleafwallpaper.com"),
+    ("woodveneer-wallcovering.com",       "https://www.woodveneerwallcovering.com"),
+    ("losangelesfabrics.com",             "https://www.losangelesfabrics.com"),
+    ("wallpaperchicago.com",              "https://www.wallpaperchicago.com"),
+    ("philipperomano.com",                "https://www.philipperomano.com"),
+    ("carnegie-microsite",                "https://carnegie.designerwallcoverings.com"),
+    ("all.designerwallcoverings.com",     "https://all.designerwallcoverings.com"),
+    # Decade/era microsites
+    ("1800swallpaper.com",                "https://www.1800swallpaper.com"),
+    ("1810swallpaper.com",                "https://www.1810swallpaper.com"),
+    ("1820swallpaper.com",                "https://www.1820swallpaper.com"),
+    ("1830swallpaper.com",                "https://www.1830swallpaper.com"),
+    ("1840swallpaper.com",                "https://www.1840swallpaper.com"),
+    ("1850swallpaper.com",                "https://www.1850swallpaper.com"),
+    ("1860swallpaper.com",                "https://www.1860swallpaper.com"),
+    ("1870swallpaper.com",                "https://www.1870swallpaper.com"),
+    ("1880swallpaper.com",                "https://www.1880swallpaper.com"),
+    ("1890swallpaper.com",                "https://www.1890swallpaper.com"),
+    ("1900swallpaper.com",                "https://www.1900swallpaper.com"),
+    ("1910swallpaper.com",                "https://www.1910swallpaper.com"),
+    ("1920swallpaper.com",                "https://www.1920swallpaper.com"),
+    ("1930swallpaper.com",                "https://www.1930swallpaper.com"),
+    ("1940swallpaper.com",                "https://www.1940swallpaper.com"),
+    ("1950swallpaper.com",                "https://www.1950swallpaper.com"),
+    ("1960swallpaper.com",                "https://www.1960swallpaper.com"),
+    ("1970swallpaper.com",                "https://www.1970swallpaper.com"),
+    ("1980swallpaper.com",                "https://www.1980swallpaper.com"),
+    # Other standalone storefronts
+    ("fabricut.designerwallcoverings.com","https://fabricut.designerwallcoverings.com"),
+    ("abramsagency.com",                  "https://abramsagency.com"),
+]
+
+GTAG_PATTERN = re.compile(r'G-[A-Z0-9]{6,12}')
+
+def curl_site_for_gtag(url: str, timeout: int = 12) -> list[str]:
+    """Curl a site's home page, extract all G-XXXXXXX measurement IDs."""
+    try:
+        result = subprocess.run(
+            ["curl", "-s", "-L", "--max-time", str(timeout),
+             "-A", "Mozilla/5.0 (compatible; GA-audit/1.0)",
+             "--compressed", url],
+            capture_output=True, text=True, timeout=timeout + 3
+        )
+        html = result.stdout
+        ids = list(set(GTAG_PATTERN.findall(html)))
+        return ids
+    except Exception as e:
+        return [f"ERROR:{e}"]
+
+
+# ---------------------------------------------------------------------------
+# Step 3: Classification logic
+# ---------------------------------------------------------------------------
+
+def extract_host(uri: str) -> str:
+    """Extract hostname (no www) from a URI string like https://example.com"""
+    if not uri:
+        return ""
+    if not uri.startswith("http"):
+        uri = "https://" + uri
+    h = urlparse(uri).hostname or ""
+    return h.lstrip("www.")
+
+def classify(site_label: str, site_url: str, served_id: str, auth_map: dict) -> dict:
+    """
+    Returns classification dict with keys:
+      status: OK | CROSS-ATTRIBUTION | TRUE-ORPHAN | SUBDOMAIN-ROLLUP
+      ...
+    """
+    site_host = extract_host(site_url)
+
+    if served_id not in auth_map:
+        return {
+            "status": "TRUE-ORPHAN",
+            "served_id": served_id,
+            "site": site_label,
+            "site_url": site_url,
+            "registered_uri": None,
+            "note": "ID not found in GA4 account — data goes nowhere",
+        }
+
+    prop = auth_map[served_id]
+    reg_uri = prop["stream_uri"]
+    reg_host = extract_host(reg_uri)
+
+    # Exact match
+    if reg_host == site_host:
+        return {"status": "OK", "served_id": served_id, "site": site_label,
+                "registered_uri": reg_uri}
+
+    # Subdomain-rollup: carnegie.designerwallcoverings.com -> designerwallcoverings.com
+    if site_host.endswith("." + reg_host) or reg_host.endswith("." + site_host):
+        return {
+            "status": "SUBDOMAIN-ROLLUP",
+            "served_id": served_id,
+            "site": site_label,
+            "site_host": site_host,
+            "registered_uri": reg_uri,
+            "property_name": prop["property_name"],
+            "note": "Subdomain fires into parent-domain property — intentional",
+        }
+
+    # Check if the site host is a subdomain of the registered host (e.g. www.foo.com -> foo.com)
+    # and vice versa — treat as OK if one is www. prefix of the other
+    if reg_host.replace("www.", "") == site_host.replace("www.", ""):
+        return {"status": "OK", "served_id": served_id, "site": site_label,
+                "registered_uri": reg_uri}
+
+    # Different domain entirely = cross-attribution
+    return {
+        "status": "CROSS-ATTRIBUTION",
+        "served_id": served_id,
+        "site": site_label,
+        "site_url": site_url,
+        "site_host": site_host,
+        "registered_uri": reg_uri,
+        "registered_host": reg_host,
+        "property_name": prop["property_name"],
+        "note": f"Traffic from {site_host} lands in property registered for {reg_host}",
+    }
+
+
+# ---------------------------------------------------------------------------
+# Main
+# ---------------------------------------------------------------------------
+
+def main():
+    print("=== GA4 AUTHORITATIVE DRIFT SWEEP ===\n", flush=True)
+    print("Step 1: Building authoritative map from Admin API...", flush=True)
+    auth_map = build_authoritative_map()
+    print(f"  Authoritative map: {len(auth_map)} measurement IDs across account\n", flush=True)
+
+    # Save map for inspection
+    out_path = Path("/Users/macstudio3/Projects/ga-allsites/authoritative_map.json")
+    out_path.write_text(json.dumps(auth_map, indent=2))
+    print(f"  Map saved: {out_path}\n", flush=True)
+
+    print("Step 2: Curling fleet sites for live G- IDs...\n", flush=True)
+
+    results = []
+    no_ga_sites = []
+    curl_errors = []
+
+    for (label, url) in FLEET_SITES:
+        ids = curl_site_for_gtag(url)
+        error_ids = [i for i in ids if i.startswith("ERROR:")]
+        real_ids  = [i for i in ids if not i.startswith("ERROR:")]
+
+        if error_ids:
+            curl_errors.append((label, url, error_ids[0]))
+            print(f"  [{label}] CURL ERROR: {error_ids[0]}", flush=True)
+            continue
+
+        if not real_ids:
+            no_ga_sites.append((label, url))
+            print(f"  [{label}] no G- ID found", flush=True)
+            continue
+
+        for mid in real_ids:
+            classification = classify(label, url, mid, auth_map)
+            results.append(classification)
+            status = classification["status"]
+            reg_uri = classification.get("registered_uri", "N/A")
+            print(f"  [{label}] {mid} -> {status} (registered: {reg_uri})", flush=True)
+
+    print("\n\nStep 3: Summary\n", flush=True)
+
+    ok_list        = [r for r in results if r["status"] == "OK"]
+    cross_list     = [r for r in results if r["status"] == "CROSS-ATTRIBUTION"]
+    orphan_list    = [r for r in results if r["status"] == "TRUE-ORPHAN"]
+    rollup_list    = [r for r in results if r["status"] == "SUBDOMAIN-ROLLUP"]
+
+    total_checked = len(FLEET_SITES) - len(curl_errors)
+    total_with_ga = len(results)
+
+    print(f"Sites attempted:        {len(FLEET_SITES)}")
+    print(f"Sites with GA (pairs):  {total_with_ga}")
+    print(f"Sites no GA found:      {len(no_ga_sites)}")
+    print(f"Curl errors:            {len(curl_errors)}")
+    print(f"")
+    print(f"OK (correct):           {len(ok_list)}")
+    print(f"SUBDOMAIN-ROLLUP:       {len(rollup_list)}")
+    print(f"CROSS-ATTRIBUTION:      {len(cross_list)}")
+    print(f"TRUE-ORPHAN:            {len(orphan_list)}")
+
+    if cross_list:
+        print("\n=== CROSS-ATTRIBUTION FINDINGS ===")
+        for r in cross_list:
+            print(f"  SITE:          {r['site']} ({r['site_url']})")
+            print(f"  SERVED ID:     {r['served_id']}")
+            print(f"  PROPERTY:      {r['property_name']}")
+            print(f"  REGISTERED:    {r['registered_uri']}")
+            print(f"  NOTE:          {r['note']}")
+            print()
+
+    if orphan_list:
+        print("\n=== TRUE-ORPHAN FINDINGS ===")
+        for r in orphan_list:
+            print(f"  SITE:          {r['site']} ({r['site_url']})")
+            print(f"  SERVED ID:     {r['served_id']}")
+            print(f"  NOTE:          {r['note']}")
+            print()
+
+    if rollup_list:
+        print("\n=== SUBDOMAIN ROLLUPS (informational, not alarmed) ===")
+        for r in rollup_list:
+            print(f"  {r['site']} -> {r['registered_uri']} ({r.get('property_name','')})")
+
+    if no_ga_sites:
+        print("\n=== SITES WITH NO GA ID FOUND ===")
+        for (label, url) in no_ga_sites:
+            print(f"  {label}: {url}")
+
+    if curl_errors:
+        print("\n=== CURL ERRORS (could not check) ===")
+        for (label, url, err) in curl_errors:
+            print(f"  {label}: {url} -> {err}")
+
+    # Save full results
+    results_path = Path("/Users/macstudio3/Projects/ga-allsites/sweep_results.json")
+    results_path.write_text(json.dumps({
+        "auth_map_size": len(auth_map),
+        "results": results,
+        "no_ga": [(l, u) for l, u in no_ga_sites],
+        "errors": [(l, u, e) for l, u, e in curl_errors],
+    }, indent=2))
+    print(f"\nFull results saved: {results_path}")
+
+    return 0
+
+
+if __name__ == "__main__":
+    sys.exit(main())
diff --git a/gate1_create_fleet_property.py b/gate1_create_fleet_property.py
new file mode 100644
index 0000000..623a494
--- /dev/null
+++ b/gate1_create_fleet_property.py
@@ -0,0 +1,49 @@
+import os, pathlib, json
+os.environ['GOOGLE_APPLICATION_CREDENTIALS']=str(pathlib.Path.home()/".config"/"ga-analytics-agent"/"service-account.json")
+from google.analytics.admin import AnalyticsAdminServiceClient
+from google.analytics.admin_v1alpha import types as t
+c=AnalyticsAdminServiceClient()
+ACCT="accounts/371997142"  # Steve Abrams APPS (where all DW properties live)
+
+# Idempotency guard: never create a duplicate.
+existing=None
+for a in c.list_account_summaries():
+    for ps in a.property_summaries:
+        if ps.display_name.strip().lower()=="dw fleet microsites":
+            existing=ps.property
+if existing:
+    # fetch its measurement id
+    mid=""
+    for ds in c.list_data_streams(parent=existing):
+        w=getattr(ds,"web_stream_data",None)
+        if w and w.measurement_id: mid=w.measurement_id
+    print(f"ALREADY EXISTS: {existing}  MEASUREMENT_ID: {mid} -- not creating a duplicate.")
+    raise SystemExit(0)
+
+# 1) property
+prop=t.Property(display_name="DW Fleet Microsites", time_zone="America/Los_Angeles",
+                currency_code="USD", industry_category=t.IndustryCategory.SHOPPING, parent=ACCT)
+p=c.create_property(property=prop); pid=p.name.split("/")[-1]
+print(f"PROPERTY_CREATED: {p.name}  display='{p.display_name}'")
+
+# 2) web data stream -> measurement id
+ds=t.DataStream(display_name="DW Fleet Microsites Web",
+    type_=t.DataStream.DataStreamType.WEB_DATA_STREAM,
+    web_stream_data=t.DataStream.WebStreamData(default_uri="https://designerwallcoverings.com"))
+s=c.create_data_stream(parent=p.name, data_stream=ds); mid=s.web_stream_data.measurement_id
+print(f"DATA_STREAM_CREATED: {s.name}")
+print(f"MEASUREMENT_ID: {mid}")
+
+# 3) event-scoped 'vendor' custom dimension
+dim=t.CustomDimension(parameter_name="vendor", display_name="Vendor",
+    description="Brand vendor slug (e.g. kravet, carnegie, thibaut)",
+    scope=t.CustomDimension.DimensionScope.EVENT)
+cd=c.create_custom_dimension(parent=p.name, custom_dimension=dim)
+print(f"CUSTOM_DIMENSION_CREATED: param='{cd.parameter_name}' scope={cd.scope.name}")
+
+print("\n=== GATE 1 COMPLETE ===")
+print(json.dumps({"property":p.name,"property_id":pid,"measurement_id":mid,"account":ACCT}, indent=2))
+# persist for GATE 2
+out=pathlib.Path.home()/ "Projects"/"ga-allsites"/"gate1_result.json"
+out.write_text(json.dumps({"property":p.name,"property_id":pid,"measurement_id":mid}, indent=2))
+print(f"\nsaved -> {out}")

← 2499b12 auto-data-snapshot: 2026-09-01T22:59:22 (1 data files) — etl  ·  back to Ga Allsites  ·  auto-data-snapshot: 2026-09-09T10:04:38 (1 data files) — etl d4c5798 →