#!/usr/bin/env python3
"""Reconstructs what Naoki's matcha was called, cost and claimed, from Internet Archive snapshots. Used in article 16.
A shop's product feed only ever returns today's listing. To find out whether a tin was renamed,
repriced, or reworded — and when — you have to read the old page out of the archive.
Known failure modes in this kind of listing archaeology, and the guard applied to each:
1. A product page carries **other products' prices and names too** (related-item rails,
"customers also bought"). Scraping the first price or first title on the page mixes them in.
The guard: read only the structured block whose own handle matches the page being measured.
2. Shopify reports prices in **cents** in the `var meta` block and in dollars in JSON-LD
(2499 vs "24.99"). Reading cents as dollars multiplies by 100.
3. A Shopify variant's `grams` field is **shipping weight, not net weight**. At this shop a
1.4 oz tin reports 45 g and a 3.5 oz tin reports 100 or 109 g. Net weight comes from the
variant's own label, which states ounces.
4. Themes change. The same shop exposes JSON-LD in older snapshots and a `var meta` block in
newer ones. Handle both, and confirm the handle matches in either case.
5. Snapshots fail routinely. **Record a failure as missing, never as zero** — a zero price
quietly becomes a 100% discount in any later average.
6. Word counts must separate **what the seller wrote** from what the page framework carries.
A term appearing only in an image `alt` attribute or a related-product rail is not the
seller describing this product. Count the product's own name and its own description
separately from everything else on the page.
"""
import ssl, re, csv, json, gzip, time, socket, html, urllib.request, urllib.parse
socket.setdefaulttimeout(30)
UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'}
CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
SHOP = 'naokimatcha.com'
PRODUCTS = {
'Superior Blend Matcha': 'superior-blend-matcha',
'Organic First Spring Blend Matcha': 'organic-first-spring-blend-matcha',
'Barista Blend Matcha': 'barista-blend-matcha',
'Barista Pro Blend Matcha': 'barista-pro-blend-matcha',
'Fragrant Yame Matcha': 'fragrant-yame-matcha',
'Chiran Harvest Matcha': 'chiran-harvest-matcha',
'Ujitawara Special Matcha': 'ujitawara-special',
'Organic All Purpose Culinary Matcha': 'all-purpose-culinary-matcha',
}
# The word this article is about, plus the neighbours it is confused with.
TERMS = ['ceremonial', 'superior', 'premium', 'grade a', 'culinary']
def cdx(pattern, limit=400):
q = {'url': pattern, 'output': 'json', 'fl': 'timestamp,original,statuscode',
'collapse': 'timestamp:6', 'limit': str(limit), 'filter': 'statuscode:200'}
u = 'https://web.archive.org/cdx/search/cdx?' + urllib.parse.urlencode(q)
for a in range(3):
try:
r = urllib.request.urlopen(urllib.request.Request(u, headers=UA), timeout=90, context=CTX)
d = json.loads(r.read().decode('utf-8', 'ignore') or '[]')
return d[1:] if d else []
except Exception:
time.sleep(3)
return []
def snap(ts, handle):
url = f'https://web.archive.org/web/{ts}id_/https://{SHOP}/products/{handle}'
for attempt in range(3):
try:
r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=45, context=CTX)
raw = r.read()
if r.headers.get('Content-Encoding') == 'gzip':
raw = gzip.decompress(raw)
return raw.decode('utf-8', 'ignore')
except Exception:
time.sleep(3)
return None
def _from_meta(page, handle):
"""Newer theme: the `var meta` product block. Reject it unless the handle matches."""
m = re.search(r'var\s+meta\s*=\s*(\{.*?\});\s*\n', page, re.S)
if not m:
return None
try:
p = (json.loads(m.group(1)).get('product') or {})
except Exception:
return None
if p.get('handle') != handle or not p.get('variants'):
return None
out = []
for v in p['variants']:
cents = v.get('price')
out.append({'variant': str(v.get('public_title') or v.get('name') or ''),
'price': round(cents / 100, 2) if isinstance(cents, (int, float)) else None,
'sku': v.get('sku') or ''}) # Shopify stores this in cents
return p.get('type') or '', out, 'meta'
def _from_jsonld(page, handle):
"""Older theme: the JSON-LD Product. Accept only the one whose url carries the handle."""
for m in re.finditer(r'', page, re.S):
try:
d = json.loads(m.group(1))
except Exception:
continue
for n in (d if isinstance(d, list) else [d]):
if not (isinstance(n, dict) and n.get('@type') == 'Product'):
continue
if handle not in str(n.get('url') or ''):
continue
offers = n.get('offers')
offers = offers if isinstance(offers, list) else ([offers] if offers else [])
out = []
for o in offers:
if isinstance(o, dict) and o.get('price') is not None:
out.append({'variant': str(o.get('name') or ''),
'price': round(float(o['price']), 2), 'sku': o.get('sku') or ''})
if out:
return '', out, 'json-ld'
return None
def product_name(page, handle):
"""The name the seller gave this product, taken from its own structured block."""
m = re.search(r'var\s+meta\s*=\s*(\{.*?\});\s*\n', page, re.S)
if m:
try:
p = json.loads(m.group(1)).get('product') or {}
if p.get('handle') == handle:
# `meta` carries an id and type but not always the title; fall through if absent
pass
except Exception:
pass
for m in re.finditer(r'', page, re.S):
try:
d = json.loads(m.group(1))
except Exception:
continue
for n in (d if isinstance(d, list) else [d]):
if isinstance(n, dict) and n.get('@type') == 'Product' and handle in str(n.get('url') or ''):
if n.get('name'):
return html.unescape(str(n['name'])).strip(), 'json-ld name'
m = re.search(r']*>(.*?)', page, re.S)
if m:
return html.unescape(re.sub(r'\s+', ' ', m.group(1))).strip(), 'title tag'
return '', 'not found'
def seller_description(page, handle):
"""The seller's own description of THIS product, separated from the rest of the page.
Guard 6: a term in an image alt attribute or a related-product rail is not the seller
describing this product. Prefer the structured description; fall back to the meta
description, which Shopify fills from the same field.
"""
for m in re.finditer(r'', page, re.S):
try:
d = json.loads(m.group(1))
except Exception:
continue
for n in (d if isinstance(d, list) else [d]):
if isinstance(n, dict) and n.get('@type') == 'Product' and handle in str(n.get('url') or ''):
if n.get('description'):
return html.unescape(re.sub(r'<[^>]+>', ' ', str(n['description']))), 'json-ld description'
m = re.search(r' {path}", flush=True)