#!/usr/bin/env python3 """Reconstructs what Naoki's matcha was called, cost and claimed, from Internet Archive snapshots. Used in article 16. A shop's product feed only ever returns today's listing. To find out whether a tin was renamed, repriced, or reworded — and when — you have to read the old page out of the archive. Known failure modes in this kind of listing archaeology, and the guard applied to each: 1. A product page carries **other products' prices and names too** (related-item rails, "customers also bought"). Scraping the first price or first title on the page mixes them in. The guard: read only the structured block whose own handle matches the page being measured. 2. Shopify reports prices in **cents** in the `var meta` block and in dollars in JSON-LD (2499 vs "24.99"). Reading cents as dollars multiplies by 100. 3. A Shopify variant's `grams` field is **shipping weight, not net weight**. At this shop a 1.4 oz tin reports 45 g and a 3.5 oz tin reports 100 or 109 g. Net weight comes from the variant's own label, which states ounces. 4. Themes change. The same shop exposes JSON-LD in older snapshots and a `var meta` block in newer ones. Handle both, and confirm the handle matches in either case. 5. Snapshots fail routinely. **Record a failure as missing, never as zero** — a zero price quietly becomes a 100% discount in any later average. 6. Word counts must separate **what the seller wrote** from what the page framework carries. A term appearing only in an image `alt` attribute or a related-product rail is not the seller describing this product. Count the product's own name and its own description separately from everything else on the page. """ import ssl, re, csv, json, gzip, time, socket, html, urllib.request, urllib.parse socket.setdefaulttimeout(30) UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE SHOP = 'naokimatcha.com' PRODUCTS = { 'Superior Blend Matcha': 'superior-blend-matcha', 'Organic First Spring Blend Matcha': 'organic-first-spring-blend-matcha', 'Barista Blend Matcha': 'barista-blend-matcha', 'Barista Pro Blend Matcha': 'barista-pro-blend-matcha', 'Fragrant Yame Matcha': 'fragrant-yame-matcha', 'Chiran Harvest Matcha': 'chiran-harvest-matcha', 'Ujitawara Special Matcha': 'ujitawara-special', 'Organic All Purpose Culinary Matcha': 'all-purpose-culinary-matcha', } # The word this article is about, plus the neighbours it is confused with. TERMS = ['ceremonial', 'superior', 'premium', 'grade a', 'culinary'] def cdx(pattern, limit=400): q = {'url': pattern, 'output': 'json', 'fl': 'timestamp,original,statuscode', 'collapse': 'timestamp:6', 'limit': str(limit), 'filter': 'statuscode:200'} u = 'https://web.archive.org/cdx/search/cdx?' + urllib.parse.urlencode(q) for a in range(3): try: r = urllib.request.urlopen(urllib.request.Request(u, headers=UA), timeout=90, context=CTX) d = json.loads(r.read().decode('utf-8', 'ignore') or '[]') return d[1:] if d else [] except Exception: time.sleep(3) return [] def snap(ts, handle): url = f'https://web.archive.org/web/{ts}id_/https://{SHOP}/products/{handle}' for attempt in range(3): try: r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=45, context=CTX) raw = r.read() if r.headers.get('Content-Encoding') == 'gzip': raw = gzip.decompress(raw) return raw.decode('utf-8', 'ignore') except Exception: time.sleep(3) return None def _from_meta(page, handle): """Newer theme: the `var meta` product block. Reject it unless the handle matches.""" m = re.search(r'var\s+meta\s*=\s*(\{.*?\});\s*\n', page, re.S) if not m: return None try: p = (json.loads(m.group(1)).get('product') or {}) except Exception: return None if p.get('handle') != handle or not p.get('variants'): return None out = [] for v in p['variants']: cents = v.get('price') out.append({'variant': str(v.get('public_title') or v.get('name') or ''), 'price': round(cents / 100, 2) if isinstance(cents, (int, float)) else None, 'sku': v.get('sku') or ''}) # Shopify stores this in cents return p.get('type') or '', out, 'meta' def _from_jsonld(page, handle): """Older theme: the JSON-LD Product. Accept only the one whose url carries the handle.""" for m in re.finditer(r']+application/ld\+json[^>]*>(.*?)', page, re.S): try: d = json.loads(m.group(1)) except Exception: continue for n in (d if isinstance(d, list) else [d]): if not (isinstance(n, dict) and n.get('@type') == 'Product'): continue if handle not in str(n.get('url') or ''): continue offers = n.get('offers') offers = offers if isinstance(offers, list) else ([offers] if offers else []) out = [] for o in offers: if isinstance(o, dict) and o.get('price') is not None: out.append({'variant': str(o.get('name') or ''), 'price': round(float(o['price']), 2), 'sku': o.get('sku') or ''}) if out: return '', out, 'json-ld' return None def product_name(page, handle): """The name the seller gave this product, taken from its own structured block.""" m = re.search(r'var\s+meta\s*=\s*(\{.*?\});\s*\n', page, re.S) if m: try: p = json.loads(m.group(1)).get('product') or {} if p.get('handle') == handle: # `meta` carries an id and type but not always the title; fall through if absent pass except Exception: pass for m in re.finditer(r']+application/ld\+json[^>]*>(.*?)', page, re.S): try: d = json.loads(m.group(1)) except Exception: continue for n in (d if isinstance(d, list) else [d]): if isinstance(n, dict) and n.get('@type') == 'Product' and handle in str(n.get('url') or ''): if n.get('name'): return html.unescape(str(n['name'])).strip(), 'json-ld name' m = re.search(r']*>(.*?)', page, re.S) if m: return html.unescape(re.sub(r'\s+', ' ', m.group(1))).strip(), 'title tag' return '', 'not found' def seller_description(page, handle): """The seller's own description of THIS product, separated from the rest of the page. Guard 6: a term in an image alt attribute or a related-product rail is not the seller describing this product. Prefer the structured description; fall back to the meta description, which Shopify fills from the same field. """ for m in re.finditer(r']+application/ld\+json[^>]*>(.*?)', page, re.S): try: d = json.loads(m.group(1)) except Exception: continue for n in (d if isinstance(d, list) else [d]): if isinstance(n, dict) and n.get('@type') == 'Product' and handle in str(n.get('url') or ''): if n.get('description'): return html.unescape(re.sub(r'<[^>]+>', ' ', str(n['description']))), 'json-ld description' m = re.search(r' {path}", flush=True)