#!/usr/bin/env python3
"""Reconstructs what Jade Leaf's matcha cost, from Internet Archive snapshots of the seller's own
product pages. Used in article 23.
A shop's product feed only ever returns today's price. To find out whether a tin was repriced —
and when — you have to read the old page out of the archive. The question here is narrow: through
a period when Japanese tea leaf prices rose sharply, did this shop's shelf prices move?
Known failure modes in this kind of listing archaeology, and the guard applied to each:
1. A product page carries **other products' prices too** (related-item rails, "you may also
like"). Reading the first price on the page mixes them in. Only the structured block whose
own handle matches the page being measured is read.
2. Shopify reports prices in **cents** in the `var meta` block and in dollars in JSON-LD
(2699 vs "26.99"). Reading cents as dollars multiplies by 100.
3. **A variant is not a product.** This shop sells the same tea in five sizes at once. Prices
are recorded per variant with the variant's own label, and never averaged across sizes.
4. Themes change. The same shop exposes JSON-LD in some snapshots and a `var meta` block in
others. Both are handled, and the handle is confirmed to match either way.
5. Snapshots fail routinely. **A failure is recorded as missing, never as zero** — a zero price
quietly becomes a 100% discount in any later average.
6. **A capture window is not the present.** The last archived price is the last price the
archive saw, not today's. Today's price is collected separately from the live shop and
joined on, so a gap at the end of the window is never read as "unchanged since".
7. Handles are renamed. A shop that moves a product to a new URL leaves the old URL's history
stranded. Every handle the shop has used for a product is queried, and which handle produced
each row is recorded.
"""
import ssl, re, csv, json, gzip, time, socket, urllib.request, urllib.parse
socket.setdefaulttimeout(30)
UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'}
CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
SHOP = 'www.jadeleafmatcha.com'
STAMP = '2026-08-27'
# Guard 7: both the current handle and any earlier handle the shop redirected from.
PRODUCTS = {
'Organic Ceremonial Matcha - Teahouse Edition': [
'organic-teahouse-edition-ceremonial-matcha',
'organic-teahouse-edition-ceremonial-matcha-tins-pouches'],
'Organic Ceremonial Matcha - Barista Edition': [
'organic-barista-edition-ceremonial-matcha'],
'Organic Culinary Matcha': ['organic-culinary-matcha'],
'Organic Ingredient Matcha': ['ingredient-matcha', 'organic-ingredient-matcha'],
}
# Scope: the four powders on this shop's own Pure Matcha shelf that have been listed long enough
# to have a price history. The stick packs and the sweetened latte mixes are deliberately out of
# scope here — the question this file answers is what happened to the price of the tea, and a mix
# is nine parts sugar, so its price history measures something else.
def cdx(pattern, limit=400):
q = {'url': pattern, 'output': 'json', 'fl': 'timestamp,original,statuscode',
'collapse': 'timestamp:6', 'limit': str(limit), 'filter': 'statuscode:200'}
u = 'https://web.archive.org/cdx/search/cdx?' + urllib.parse.urlencode(q)
for _ in range(3):
try:
r = urllib.request.urlopen(urllib.request.Request(u, headers=UA), timeout=90, context=CTX)
d = json.loads(r.read().decode('utf-8', 'ignore') or '[]')
return d[1:] if d else []
except Exception:
time.sleep(3)
return []
def snap(ts, handle):
url = f'https://web.archive.org/web/{ts}id_/https://{SHOP}/products/{handle}'
for _ in range(3):
try:
r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=60, context=CTX)
raw = r.read()
if r.headers.get('Content-Encoding') == 'gzip':
raw = gzip.decompress(raw)
return raw.decode('utf-8', 'ignore')
except Exception:
time.sleep(3)
return None
def canonical_handle(page):
"""The handle of the page the snapshot actually is, from its own canonical link. Needed
because the archive redirects and because a renamed product answers on its old URL."""
for pat in (r']+rel="canonical"[^>]+href="([^"]+)"',
r']+property="og:url"[^>]+content="([^"]+)"'):
m = re.search(pat, page)
if m:
return m.group(1).rstrip('/').rsplit('/', 1)[-1].split('?')[0]
return ''
def _from_meta(page, handle, known):
"""The `var meta` product block.
Guard 1 says never read a price block that belongs to a different product. The obvious check
— `meta.product.handle == handle` — silently rejects every snapshot before this shop's theme
started emitting a handle at all, which is all of 2022 and 2023, and reports them as "price
block not found". So the page's own canonical link is used to confirm which product page the
snapshot is, and the handle field is used only when the theme provides it.
"""
m = re.search(r'var\s+meta\s*=\s*(\{.*?\});\s*\n', page, re.S)
if not m:
return None
try:
p = (json.loads(m.group(1)).get('product') or {})
except Exception:
return None
if not p.get('variants'):
return None
ident = p.get('handle') or canonical_handle(page)
if ident and ident not in known:
return None
out = []
for v in p['variants']:
cents = v.get('price') # guard 2: cents, not dollars
out.append({'variant': str(v.get('public_title') or v.get('name') or ''),
'price': round(cents / 100, 2) if isinstance(cents, (int, float)) else None,
'sku': v.get('sku') or '', 'product_id': p.get('id') or ''})
return out, ('meta' if p.get('handle') else 'meta (identified by canonical link)')
def _from_jsonld(page, handle):
"""Older theme: the JSON-LD Product whose url carries this handle (guard 1)."""
for m in re.finditer(r'', page, re.S):
try:
d = json.loads(m.group(1))
except Exception:
continue
for n in (d if isinstance(d, list) else [d]):
if not (isinstance(n, dict) and n.get('@type') == 'Product'):
continue
if handle not in str(n.get('url') or ''):
continue
offers = n.get('offers')
offers = offers if isinstance(offers, list) else ([offers] if offers else [])
out = [{'variant': str(o.get('name') or ''), 'price': round(float(o['price']), 2),
'sku': o.get('sku') or '', 'product_id': ''}
for o in offers if isinstance(o, dict) and o.get('price') is not None]
if out:
return out, 'json-ld'
return None
if __name__ == '__main__':
rows = []
for name, handles in PRODUCTS.items():
stamps = []
for h in handles:
for r in cdx(f'{SHOP}/products/{h}'):
stamps.append((r[0], h))
stamps.sort()
print(f"\n=== {name} — {len(stamps)} snapshots across {len(handles)} handle(s) ===", flush=True)
if not stamps:
rows.append({'product': name, 'handle': handles[0], 'snapshot': '', 'variant': '',
'price': '', 'sku': '', 'note': 'no snapshots'})
continue
for ts, h in stamps:
date = f'{ts[:4]}-{ts[4:6]}-{ts[6:8]}'
page = snap(ts, h)
if page is None: # guard 5
rows.append({'product': name, 'handle': h, 'snapshot': date, 'variant': '',
'price': '', 'sku': '', 'note': 'snapshot fetch failed'})
print(f" {date} [{h[:28]}] fetch failed (recorded as missing, not as zero)", flush=True)
continue
parsed = _from_meta(page, h, set(handles)) or _from_jsonld(page, h)
if not parsed:
rows.append({'product': name, 'handle': h, 'snapshot': date, 'variant': '',
'price': '', 'sku': '', 'note': 'price block not found for this handle'})
print(f" {date} [{h[:28]}] price block not found", flush=True)
continue
variants, src = parsed
for v in variants: # guard 3
rows.append({'product': name, 'handle': h, 'snapshot': date,
'variant': v['variant'], 'price': v['price'], 'sku': v['sku'],
'shopify_product_id': v.get('product_id', ''), 'note': src})
print(f" {date} [{h[:28]}] " +
' '.join(f"{v['variant'] or '-'}=${v['price']}" for v in variants[:5]), flush=True)
time.sleep(0.4)
path = f'data/jadeleaf_price_history_{STAMP}.csv'
with open(path, 'w', newline='', encoding='utf-8') as f:
w = csv.DictWriter(f, fieldnames=['product', 'handle', 'snapshot', 'variant', 'price',
'sku', 'shopify_product_id', 'note'], extrasaction='ignore')
w.writeheader(); w.writerows(rows)
ok = sum(1 for r in rows if r.get('price') not in ('', None))
fail = sum(1 for r in rows if 'failed' in (r.get('note') or ''))
miss = sum(1 for r in rows if 'not found' in (r.get('note') or ''))
print(f"\n{len(rows)} rows -> {path}")
print(f" priced rows {ok} fetch failures {fail} price block missing {miss}")
print(" Guard 6: the last archived price is the last price the archive saw. Today's shelf "
"price is in data/jadeleaf_listings_" + STAMP + ".csv and must be joined on before "
"any statement about what a price is 'still' at.")