← back to Wallco Ai

scripts/backfill-colorname-drift.py

250 lines

#!/usr/bin/env python3
"""backfill-colorname-drift.py — Path B surgical backfill for the color-name vs
dominant-hex drift Steve flagged on /design/10171.

Re-derives the TITLE color word from the row's dominant_hex via the SAME
colorway_name(hex) used by refresh_designs_snapshot.py, and swaps the wrong
color word in-place while preserving the title shape:

  "<Color> <Suffix> No.<id>"   →  "<NewColor> <Suffix> No.<id>"
  "... in <Color>"             →  "... in <NewColor>"

Category " · <token>" handling: per /dtd verdict (2026-06-01, Claude+Codex 2/0
for option B), the category color token is rewritten ONLY when its leading
color word chromatically DISAGREES with the dominant_hex. The curated luxe
slugs (antique-tan, bordeaux-velvet, midnight-navy …) are color-faithful, so
this is a no-op on the current live data — preserving the de Gournay naming.

This script operates on a data/designs.json snapshot file (the surface
server.js reads title/category from for /api/design/:id and /api/designs). It
is a SURGICAL backfill — NOT a full snapshot rebuild. It does NOT touch
is_published, user_removed, or any non-drifted row. NO catalog id-space churn.

Usage:
  # dry-run report (no write):
  python3 scripts/backfill-colorname-drift.py --in data/designs.json --dry

  # write a corrected snapshot to a new file:
  python3 scripts/backfill-colorname-drift.py --in IN.json --out OUT.json

The >50% shrink guard does NOT apply (row count is invariant — this only
rewrites title strings on a subset of rows, never adds/removes rows; the script
asserts the row count is unchanged).
"""
import sys, json, re, colorsys, argparse

# ── colorway logic (verbatim from refresh_designs_snapshot.py @ 2026-06-01) ──
COLORWAYS = {
    'neutral': {'vlight': 'Alabaster', 'light': 'Oyster', 'mid': 'Greige',
                'dark': 'Pewter', 'vdark': 'Charcoal'},
    'bands': [
        (15,  {'dark': 'Oxblood',     'mid': 'Claret',   'light': 'Blush'}),
        (35,  {'dark': 'Terracotta',  'mid': 'Amber',    'light': 'Apricot'}),
        (55,  {'dark': 'Ochre',       'mid': 'Honey',    'light': 'Chamois'}),
        (80,  {'dark': 'Loden',       'mid': 'Olive',    'light': 'Chartreuse'}),
        (155, {'dark': 'Bottle Green','mid': 'Sage',     'light': 'Celadon'}),
        (200, {'dark': 'Verdigris',   'mid': 'Teal',     'light': 'Eau de Nil'}),
        (250, {'dark': 'Prussian',    'mid': 'Sapphire', 'light': 'Delft'}),
        (290, {'dark': 'Aubergine',   'mid': 'Amethyst', 'light': 'Wisteria'}),
        (330, {'dark': 'Damson',      'mid': 'Plum',     'light': 'Mauve'}),
        (361, {'dark': 'Garnet',      'mid': 'Rosewood', 'light': 'Rose'}),
    ],
}

def hex_to_hls(hex_):
    if not hex_ or not hex_.startswith('#'):
        return None
    h = hex_.lstrip('#')
    if len(h) == 3:
        h = ''.join(c * 2 for c in h)
    if len(h) < 6 or any(c not in '0123456789abcdefABCDEF' for c in h[:6]):
        return None
    r, g, b = int(h[0:2], 16) / 255, int(h[2:4], 16) / 255, int(h[4:6], 16) / 255
    return colorsys.rgb_to_hls(r, g, b)

def colorway_name(hex_):
    hls = hex_to_hls(hex_)
    if hls is None:
        return None
    h, l, s = hls
    if s < 0.10:
        n = COLORWAYS['neutral']
        if l > 0.85: return n['vlight']
        if l > 0.65: return n['light']
        if l > 0.42: return n['mid']
        if l > 0.22: return n['dark']
        return n['vdark']
    deg = h * 360
    for maxd, shades in COLORWAYS['bands']:
        if deg < maxd:
            return shades['light' if l > 0.62 else 'dark' if l < 0.34 else 'mid']
    return 'Rose'

def zone(hex_):
    hls = hex_to_hls(hex_)
    if hls is None:
        return None
    h, l, s = hls
    if s < 0.12: return 'Neutral'
    d = h * 360
    if d >= 345 or d < 15: return 'Red'
    if d < 60:  return 'Warm'
    if d < 170: return 'Green'
    if d < 260: return 'Blue'
    if d < 290: return 'Violet'
    return 'Pink'

# full vocabulary of color words colorway_name can emit (single + multi-word)
COLOR_VOCAB = set()
for _maxd, _sh in COLORWAYS['bands']:
    COLOR_VOCAB.update(_sh.values())
COLOR_VOCAB.update(COLORWAYS['neutral'].values())
# longest-first so "Bottle Green"/"Eau de Nil" match before "Green"/"Nil"
COLOR_VOCAB_RE = '|'.join(re.escape(c) for c in sorted(COLOR_VOCAB, key=len, reverse=True))

SUFFIXES = (r'Studio|Atelier|Folio|Salon|Reverie|House|Manor|Origin|Drift|'
            r'Daydream|Halcyon|Vespers|Idyll|Repose|Slumber|Haze|Lullaby')
# "<Color> <Suffix> No.<id>" — color is the LEADING token (may be 2-3 words)
TITLE_LEAD_RE = re.compile(r'^(' + COLOR_VOCAB_RE + r')\s+(' + SUFFIXES + r')\s+No\.(\d+)\s*$', re.I)
# "... in <Color>" — color is the TRAILING token
TITLE_IN_RE   = re.compile(r'^(.*\bin\s+)(' + COLOR_VOCAB_RE + r')\s*$', re.I)

# ── Category color-token disagreement test (per /dtd verdict, option B) ──────
# To avoid zone()-boundary artifacts that flagged color-FAITHFUL curated slugs
# (greenhouse-glass on a teal hex, deep-plum on a purple hex), the token and the
# hex are judged on the SAME scale: the colorway_name BAND family. A token only
# counts as a genuine mismatch when its literal color word maps to a band family
# that differs from the family colorway_name(hex) lands in. Non-color luxe
# modifiers ("deep","greenhouse","antique","manor","smoke" …) are intentionally
# NOT colors and never trigger — they qualify the real color word, they aren't it.
_WORD_TO_NAME = {  # curated/common color word → a canonical colorway_name word
    'crimson':'Oxblood','scarlet':'Oxblood','oxblood':'Oxblood','garnet':'Garnet',
    'claret':'Claret','rosewood':'Rosewood','bordeaux':'Oxblood','ruby':'Garnet',
    'terracotta':'Terracotta','amber':'Amber','apricot':'Apricot','ochre':'Ochre',
    'honey':'Honey','chamois':'Chamois','saffron':'Ochre','mustard':'Ochre',
    'olive':'Olive','loden':'Loden','chartreuse':'Chartreuse','sage':'Sage',
    'celadon':'Celadon','forest':'Bottle Green','moss':'Sage','emerald':'Bottle Green',
    'verdigris':'Verdigris','teal':'Teal','delft':'Delft','prussian':'Prussian',
    'sapphire':'Sapphire','navy':'Prussian','indigo':'Prussian','cobalt':'Sapphire',
    'aubergine':'Aubergine','amethyst':'Amethyst','wisteria':'Wisteria',
    'damson':'Damson','plum':'Plum','mauve':'Mauve','violet':'Amethyst',
    'rose':'Rose','blush':'Blush',
    'alabaster':'Alabaster','oyster':'Oyster','greige':'Greige','pewter':'Pewter',
    'charcoal':'Charcoal',
}
def _band_family(name):
    """Return the band key (max_deg) a colorway_name word belongs to, or
    'neutral'; None if unknown. Lets token-color and hex-color be compared on
    one scale, eliminating coarse-zone boundary artifacts.

    The two RED bands straddle the 360°/0° hue wraparound — band 15
    (Oxblood/Claret/Blush) and band 361 (Garnet/Rosewood/Rose) are perceptually
    ONE red family. They are merged to key 15 so a curated dark-red slug
    ("bordeaux-velvet") is NOT flagged as disagreeing with a colorway_name that
    merely landed on the other side of the seam ("Garnet")."""
    if not name:
        return None
    nl = name.strip()
    if nl in COLORWAYS['neutral'].values():
        return 'neutral'
    for maxd, shades in COLORWAYS['bands']:
        if nl in shades.values():
            return 15 if maxd == 361 else maxd  # merge the wraparound red band
    return None

def fix_title(title, hexd):
    """Return (new_title, old_color, new_color) or None if no change."""
    new_color = colorway_name(hexd)
    if not new_color or not title:
        return None
    t = title.strip()
    m = TITLE_LEAD_RE.match(t)
    if m:
        old = m.group(1)
        if old.lower() == new_color.lower():
            return None
        return (f"{new_color} {m.group(2)} No.{m.group(3)}", old, new_color)
    m = TITLE_IN_RE.match(t)
    if m:
        old = m.group(2)
        if old.lower() == new_color.lower():
            return None
        return (f"{m.group(1)}{new_color}", old, new_color)
    return None

def fix_category(cat, hexd):
    """Per /dtd option B: rewrite the ' · <token>' color token ONLY when the
    token's literal color word lands in a DIFFERENT colorway band family than
    colorway_name(hex). Curated, color-faithful slugs are preserved. Returns
    (new_cat, old_token, new_token) or None."""
    if not cat or ' · ' not in cat:
        return None
    prefix, tok = cat.split(' · ', 1)
    words = [w for w in tok.strip().lower().split('-') if w]
    color_word = next((w for w in words if w in _WORD_TO_NAME), None)
    if color_word is None:
        return None  # no literal color in the slug → preserve as-is
    hex_name = colorway_name(hexd)
    tok_fam = _band_family(_WORD_TO_NAME[color_word])
    hex_fam = _band_family(hex_name)
    if tok_fam is None or hex_fam is None:
        return None
    if tok_fam == 'neutral' or hex_fam == 'neutral' or tok_fam == hex_fam:
        return None  # same family (or neutral) → genuine agreement, preserve
    new_tok = hex_name.lower().replace(' ', '-')
    return (f"{prefix} · {new_tok}", tok.strip(), new_tok)

def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--in', dest='inp', required=True)
    ap.add_argument('--out', dest='out')
    ap.add_argument('--dry', action='store_true')
    args = ap.parse_args()

    data = json.load(open(args.inp))
    if isinstance(data, dict):
        sys.exit("ERROR: expected a JSON array of design rows, got an object")
    n0 = len(data)

    title_changed = cat_changed = 0
    examples = []
    for d in data:
        # ONLY touch live-facing published, non-removed rows (the audited set).
        if not d.get('is_published') or d.get('user_removed'):
            continue
        hexd = d.get('dominant_hex') or ''
        ft = fix_title(d.get('title') or '', hexd)
        if ft:
            new_title, old_c, new_c = ft
            if not args.dry:
                d['title'] = new_title
            title_changed += 1
            if len(examples) < 20:
                examples.append((d['id'], hexd, old_c, new_c))
        fc = fix_category(d.get('category') or '', hexd)
        if fc:
            new_cat, _ot, _nt = fc
            if not args.dry:
                d['category'] = new_cat
            cat_changed += 1

    assert len(data) == n0, "row count changed — backfill must be row-invariant"

    print(f"rows in snapshot:                 {n0}")
    print(f"TITLE color words re-derived:     {title_changed}")
    print(f"CATEGORY color tokens re-derived: {cat_changed}  (option B: only on chromatic disagreement)")
    print("sample title fixes (id | hex | old -> new):")
    for e in examples:
        print(f"  {e[0]:>6} | {e[1]:<9} | {e[2]} -> {e[3]}")

    if args.dry:
        print("\n[DRY RUN — no file written]")
        return
    if not args.out:
        sys.exit("ERROR: --out required when not --dry")
    json.dump(data, open(args.out, 'w'))
    print(f"\nwrote corrected snapshot → {args.out}")

if __name__ == '__main__':
    main()