#!/usr/bin/env python3 """Reconstructs what Rishi's matcha cost in the past, from Internet Archive snapshots. Used in article 15. A shop's product feed only ever returns today's price. To find out whether a tea got more expensive — and when — you have to read the old price tag off an archived copy of the page. Known failure modes in this kind of price archaeology, and the guard applied to each: 1. A product page carries the prices of **other products too** (related-item carousels, "you may also like" rails). Scraping the first price or the first weight on the page mixes them in. The tell here was a 100 g product whose size appeared to flip to 50 g and then 30 g between snapshots. The guard: read only the block that carries **this page's own handle**. 2. Shopify stores prices in **cents** in some places and dollars in others (2100 vs "21.00"). Reading cents as dollars multiplies by 100. 3. Themes change. The same shop exposes its product as JSON-LD in older snapshots and as a `var meta` block in newer ones. Handle both, and confirm the handle matches in either case. 4. Snapshots fail routinely. **Record a failure as missing, never as zero** — a zero price will quietly become a 100% discount in any later average. 5. Take the net weight from the variant's own label, not from a SKU code. Codes look like they encode size until the one time they don't. """ import ssl, re, csv, json, gzip, time, socket, urllib.request, urllib.parse socket.setdefaulttimeout(25) UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE PRODUCTS = { 'Everyday Matcha': 'organic-everday-matcha-tin', 'Ceremonial Matcha': 'organic-ceremonial-matcha-tin', 'Teahouse Matcha': 'organic-teahouse-ceremonial-matcha', 'Barista Matcha': 'organic-barista-matcha', } SNAPSHOTS = ['20230922', '20240304', '20240816', '20241201', '20250323', '20251110', '20260326', '20260714'] def snap(ts, handle): url = f'https://web.archive.org/web/{ts}000000id_/https://rishi-tea.com/products/{handle}' for attempt in range(2): try: r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=25, context=CTX) raw = r.read() if r.headers.get('Content-Encoding') == 'gzip': raw = gzip.decompress(raw) return raw.decode('utf-8', 'ignore') except Exception: time.sleep(2) return None def _from_meta(html, handle): """Newer theme: the `var meta` product block. Reject it unless the handle matches.""" m = re.search(r'var\s+meta\s*=\s*(\{.*?\});\s*\n', html, re.S) if not m: return None try: p = (json.loads(m.group(1)).get('product') or {}) except Exception: return None if p.get('handle') != handle or not p.get('variants'): return None v = p['variants'][0] cents = v.get('price') price = round(cents / 100, 2) if isinstance(cents, (int, float)) else None # Shopify stores this in cents g = re.search(r'(\d{1,4})\s*Grams?', str(v.get('public_title') or v.get('name') or ''), re.I) return price, (int(g.group(1)) if g else None), ('meta' if len(p['variants']) == 1 else f"meta; {len(p['variants'])} variants, first used") def _from_jsonld(html, handle): """Older theme: the JSON-LD Product. Accept only the one whose url carries the handle.""" for m in re.finditer(r']+application/ld\+json[^>]*>(.*?)', html, re.S): try: d = json.loads(m.group(1)) except Exception: continue for n in (d if isinstance(d, list) else [d]): if not (isinstance(n, dict) and n.get('@type') == 'Product'): continue if handle not in str(n.get('url') or ''): continue offers = n.get('offers') offers = offers if isinstance(offers, list) else ([offers] if offers else []) for o in offers: if not isinstance(o, dict) or o.get('price') is None: continue g = re.search(r'(\d{1,4})\s*Grams?', str(o.get('name') or ''), re.I) return (round(float(o['price']), 2), (int(g.group(1)) if g else None), 'json-ld' if len(offers) == 1 else f'json-ld; {len(offers)} offers, first used') return None def parse(html, handle): """Either theme: read only the block belonging to this product.""" for fn in (_from_meta, _from_jsonld): r = fn(html, handle) if r and r[0] is not None: return r return None, None, 'product block not found for this handle' if __name__ == '__main__': rows = [] for name, handle in PRODUCTS.items(): print(f"\n=== {name} ===", flush=True) for ts in SNAPSHOTS: html = snap(ts, handle) if html is None: rows.append({'product': name, 'date': f'{ts[:4]}-{ts[4:6]}', 'price': '', 'grams': '', 'per_g': '', 'note': 'snapshot fetch failed'}) print(f" {ts[:4]}-{ts[4:6]} fetch failed (recorded as missing, not as zero)", flush=True) continue price, grams, note = parse(html, handle) rows.append({'product': name, 'date': f'{ts[:4]}-{ts[4:6]}', 'price': price or '', 'grams': grams or '', 'per_g': round(price / grams, 4) if price and grams else '', 'note': note}) pg = f"${price/grams:.3f}/g" if price and grams else '' print(f" {ts[:4]}-{ts[4:6]} ${price if price else '?':<8} {str(grams)+'g' if grams else '?':<6} {pg:<11} {note}", flush=True) time.sleep(0.4) path = 'data/rishi_price_history_2026-08-26.csv' with open(path, 'w', newline='', encoding='utf-8') as f: w = csv.DictWriter(f, fieldnames=['product', 'date', 'price', 'grams', 'per_g', 'note']) w.writeheader(); w.writerows(rows) print(f"\n{len(rows)} rows -> {path}", flush=True)