#!/usr/bin/env python3 """Reconstructs what Jade Leaf's matcha cost, from Internet Archive snapshots of the seller's own product pages. Used in article 23. A shop's product feed only ever returns today's price. To find out whether a tin was repriced — and when — you have to read the old page out of the archive. The question here is narrow: through a period when Japanese tea leaf prices rose sharply, did this shop's shelf prices move? Known failure modes in this kind of listing archaeology, and the guard applied to each: 1. A product page carries **other products' prices too** (related-item rails, "you may also like"). Reading the first price on the page mixes them in. Only the structured block whose own handle matches the page being measured is read. 2. Shopify reports prices in **cents** in the `var meta` block and in dollars in JSON-LD (2699 vs "26.99"). Reading cents as dollars multiplies by 100. 3. **A variant is not a product.** This shop sells the same tea in five sizes at once. Prices are recorded per variant with the variant's own label, and never averaged across sizes. 4. Themes change. The same shop exposes JSON-LD in some snapshots and a `var meta` block in others. Both are handled, and the handle is confirmed to match either way. 5. Snapshots fail routinely. **A failure is recorded as missing, never as zero** — a zero price quietly becomes a 100% discount in any later average. 6. **A capture window is not the present.** The last archived price is the last price the archive saw, not today's. Today's price is collected separately from the live shop and joined on, so a gap at the end of the window is never read as "unchanged since". 7. Handles are renamed. A shop that moves a product to a new URL leaves the old URL's history stranded. Every handle the shop has used for a product is queried, and which handle produced each row is recorded. """ import ssl, re, csv, json, gzip, time, socket, urllib.request, urllib.parse socket.setdefaulttimeout(30) UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE SHOP = 'www.jadeleafmatcha.com' STAMP = '2026-08-27' # Guard 7: both the current handle and any earlier handle the shop redirected from. PRODUCTS = { 'Organic Ceremonial Matcha - Teahouse Edition': [ 'organic-teahouse-edition-ceremonial-matcha', 'organic-teahouse-edition-ceremonial-matcha-tins-pouches'], 'Organic Ceremonial Matcha - Barista Edition': [ 'organic-barista-edition-ceremonial-matcha'], 'Organic Culinary Matcha': ['organic-culinary-matcha'], 'Organic Ingredient Matcha': ['ingredient-matcha', 'organic-ingredient-matcha'], } # Scope: the four powders on this shop's own Pure Matcha shelf that have been listed long enough # to have a price history. The stick packs and the sweetened latte mixes are deliberately out of # scope here — the question this file answers is what happened to the price of the tea, and a mix # is nine parts sugar, so its price history measures something else. def cdx(pattern, limit=400): q = {'url': pattern, 'output': 'json', 'fl': 'timestamp,original,statuscode', 'collapse': 'timestamp:6', 'limit': str(limit), 'filter': 'statuscode:200'} u = 'https://web.archive.org/cdx/search/cdx?' + urllib.parse.urlencode(q) for _ in range(3): try: r = urllib.request.urlopen(urllib.request.Request(u, headers=UA), timeout=90, context=CTX) d = json.loads(r.read().decode('utf-8', 'ignore') or '[]') return d[1:] if d else [] except Exception: time.sleep(3) return [] def snap(ts, handle): url = f'https://web.archive.org/web/{ts}id_/https://{SHOP}/products/{handle}' for _ in range(3): try: r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=60, context=CTX) raw = r.read() if r.headers.get('Content-Encoding') == 'gzip': raw = gzip.decompress(raw) return raw.decode('utf-8', 'ignore') except Exception: time.sleep(3) return None def canonical_handle(page): """The handle of the page the snapshot actually is, from its own canonical link. Needed because the archive redirects and because a renamed product answers on its old URL.""" for pat in (r']+rel="canonical"[^>]+href="([^"]+)"', r']+property="og:url"[^>]+content="([^"]+)"'): m = re.search(pat, page) if m: return m.group(1).rstrip('/').rsplit('/', 1)[-1].split('?')[0] return '' def _from_meta(page, handle, known): """The `var meta` product block. Guard 1 says never read a price block that belongs to a different product. The obvious check — `meta.product.handle == handle` — silently rejects every snapshot before this shop's theme started emitting a handle at all, which is all of 2022 and 2023, and reports them as "price block not found". So the page's own canonical link is used to confirm which product page the snapshot is, and the handle field is used only when the theme provides it. """ m = re.search(r'var\s+meta\s*=\s*(\{.*?\});\s*\n', page, re.S) if not m: return None try: p = (json.loads(m.group(1)).get('product') or {}) except Exception: return None if not p.get('variants'): return None ident = p.get('handle') or canonical_handle(page) if ident and ident not in known: return None out = [] for v in p['variants']: cents = v.get('price') # guard 2: cents, not dollars out.append({'variant': str(v.get('public_title') or v.get('name') or ''), 'price': round(cents / 100, 2) if isinstance(cents, (int, float)) else None, 'sku': v.get('sku') or '', 'product_id': p.get('id') or ''}) return out, ('meta' if p.get('handle') else 'meta (identified by canonical link)') def _from_jsonld(page, handle): """Older theme: the JSON-LD Product whose url carries this handle (guard 1).""" for m in re.finditer(r']+application/ld\+json[^>]*>(.*?)', page, re.S): try: d = json.loads(m.group(1)) except Exception: continue for n in (d if isinstance(d, list) else [d]): if not (isinstance(n, dict) and n.get('@type') == 'Product'): continue if handle not in str(n.get('url') or ''): continue offers = n.get('offers') offers = offers if isinstance(offers, list) else ([offers] if offers else []) out = [{'variant': str(o.get('name') or ''), 'price': round(float(o['price']), 2), 'sku': o.get('sku') or '', 'product_id': ''} for o in offers if isinstance(o, dict) and o.get('price') is not None] if out: return out, 'json-ld' return None if __name__ == '__main__': rows = [] for name, handles in PRODUCTS.items(): stamps = [] for h in handles: for r in cdx(f'{SHOP}/products/{h}'): stamps.append((r[0], h)) stamps.sort() print(f"\n=== {name} — {len(stamps)} snapshots across {len(handles)} handle(s) ===", flush=True) if not stamps: rows.append({'product': name, 'handle': handles[0], 'snapshot': '', 'variant': '', 'price': '', 'sku': '', 'note': 'no snapshots'}) continue for ts, h in stamps: date = f'{ts[:4]}-{ts[4:6]}-{ts[6:8]}' page = snap(ts, h) if page is None: # guard 5 rows.append({'product': name, 'handle': h, 'snapshot': date, 'variant': '', 'price': '', 'sku': '', 'note': 'snapshot fetch failed'}) print(f" {date} [{h[:28]}] fetch failed (recorded as missing, not as zero)", flush=True) continue parsed = _from_meta(page, h, set(handles)) or _from_jsonld(page, h) if not parsed: rows.append({'product': name, 'handle': h, 'snapshot': date, 'variant': '', 'price': '', 'sku': '', 'note': 'price block not found for this handle'}) print(f" {date} [{h[:28]}] price block not found", flush=True) continue variants, src = parsed for v in variants: # guard 3 rows.append({'product': name, 'handle': h, 'snapshot': date, 'variant': v['variant'], 'price': v['price'], 'sku': v['sku'], 'shopify_product_id': v.get('product_id', ''), 'note': src}) print(f" {date} [{h[:28]}] " + ' '.join(f"{v['variant'] or '-'}=${v['price']}" for v in variants[:5]), flush=True) time.sleep(0.4) path = f'data/jadeleaf_price_history_{STAMP}.csv' with open(path, 'w', newline='', encoding='utf-8') as f: w = csv.DictWriter(f, fieldnames=['product', 'handle', 'snapshot', 'variant', 'price', 'sku', 'shopify_product_id', 'note'], extrasaction='ignore') w.writeheader(); w.writerows(rows) ok = sum(1 for r in rows if r.get('price') not in ('', None)) fail = sum(1 for r in rows if 'failed' in (r.get('note') or '')) miss = sum(1 for r in rows if 'not found' in (r.get('note') or '')) print(f"\n{len(rows)} rows -> {path}") print(f" priced rows {ok} fetch failures {fail} price block missing {miss}") print(" Guard 6: the last archived price is the last price the archive saw. Today's shelf " "price is in data/jadeleaf_listings_" + STAMP + ".csv and must be joined on before " "any statement about what a price is 'still' at.")