#!/usr/bin/env python3 """Reads what Kettl publishes about every tea it sells, from the seller's own pages. Used in article 24. The question this answers is narrow: for each product Kettl lists, which of its own shelves does it sit on, what does it disclose about who grew the tea and when, and what does a gram cost. Known failure modes in this kind of listing data, and the guard applied to each. Read this before trusting any figure the script produces: 1. **A product feed is not the page.** `products.json` returns `body_html`, but this shop puts its producer / cultivar / area / year block in a theme text block that the feed never returns. Every disclosure field here is read from the rendered product page. 2. A Shopify variant's `grams` field is **shipping weight, not net weight**. Net weight is taken from the size the seller states — in the variant name ("20g", "1kg") or in the page's own `Packaging:` row ("20g tin") — and the source of every weight is recorded. 3. **What counts as matcha is decided by the shop's own shelves and product types**, never by the word "matcha" in a product name. This shop sells matcha bowls, matcha whisks, matcha chocolate and matcha classes, all of which carry the word. 4. **Wholesale and bulk lines distort every per-gram figure.** This shop sells the same teas in 1 kg wholesale bags alongside 20 g tins. They are collected and flagged so that retail and wholesale are never mixed into one median. 5. Retry failed fetches. A page that fails to load is missing, never "the seller does not disclose this". Failures are written to the CSV as a note, not dropped. 6. Sets, subscriptions and classes have no net weight. They are marked and excluded from per-gram statistics rather than silently dropped. 7. **A shop can run a sale.** The feed's `price` is what the shop charges today and `compare_at_price` is its list price. Both are recorded. 8. **This shop's shelves are named after producers**, and a tea can sit on a producer shelf and a category shelf at once. Membership is recorded per shelf, never collapsed into one label. """ import ssl, re, csv, json, time, html, urllib.request, collections SHOP = 'kettl.co' STAMP = '2026-08-28' UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE # Guard 3: the shop's own matcha shelves. Everything else that has "matcha" in its name — bowls, # whisks, chocolate, classes — is not matcha and must not enter a per-gram figure. MATCHA_SHELVES = {'matcha-green-tea', 'kettl-matcha-sen-mon-ten', 'furukawa-family-matcha', 'the-yasuharu-maeno-matcha-collection', 'wholesale-matcha-houjicha-powder-bulk'} # Guard 4 WHOLESALE_SHELVES = {'wholesale-matcha-houjicha-powder-bulk', 'wholesale-loose-tea-bulk', 'wholesale-tea', 'wholesale-retail-tea-and-wares', 'wholesale-tea-bag-bulk'} # Guard 8: shelves named after a person or family who grew or made the thing. PRODUCER_SHELVES = {'the-tsuji-family-collection', 'the-yamaguchi-family-collection', 'the-yasuharu-maeno-matcha-collection', 'furukawa-family-matcha', 'tsuguto-hattori-hand-picked-tea-collection', 'takaaki-yoshida', 'yoshiaki-hiruma-collection', 'shigeyoshi-morioka-collection', 'the-hiroshi-kobayashi-collection'} # Guard 6 NON_WEIGHT_TYPES = {'Ceramics', 'ceramics', 'Class', 'class', 'Incense', 'incense', 'Merchandise', 'Subscription', 'Subscription matcha', 'Tea Accessory', 'Tea Pot', 'Tea Tools', 'Tea Ware', 'Tea tools', 'Chocolate', 'Kombucha', 'kombucha'} # The rows this shop puts in its own spec block, in the order it writes them. SPEC_KEYS = ['Producer', 'Cultivar', 'Production Area', 'Production Year', 'Packaging', 'Region', 'Harvest', 'Area', 'Farm', 'Steaming', 'Elevation'] def fetch(url, tries=4): for a in range(tries): try: r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=60, context=CTX) return r.read().decode('utf-8', 'ignore') except Exception as e: if a == tries - 1: print(f' FETCH FAILED after {tries} tries: {url[:90]} {e}') return None time.sleep(2 * (a + 1)) def jget(url): t = fetch(url) try: return json.loads(t) if t else None except Exception: return None def txt(s): return re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', s or ''))).strip() def spec_rows(page): """The shop's producer/cultivar/area/year block (guard 1). It is a single

whose lines are separated by
, e.g. Producer: Shinya Yamaguchi
Cultivar: Blend
Production Area: Hoshinomura, Yame Only paragraphs that actually contain one of the shop's own keys are read, so that ordinary prose containing a colon is not mistaken for a specification. """ out = {} for m in re.finditer(r'

(.*?)

', page, re.S): block = m.group(1) if not any(f'{k}:' in block for k in SPEC_KEYS): continue for line in re.split(r'', block): s = txt(line) km = re.match(r'([A-Z][A-Za-z ]{2,20}):\s*(.+)$', s) if km and km.group(1).strip() in SPEC_KEYS: out.setdefault(km.group(1).strip(), km.group(2).strip()) return out # Guard 2: net weight from what the seller states, never from Shopify's `grams`. def net_grams(variant_title, packaging): for src, label in ((variant_title or '', 'variant name'), (packaging or '', 'Packaging row')): s = src.lower().replace(',', '') m = re.search(r'(\d+(?:\.\d+)?)\s*kg\b', s) if m: return float(m.group(1)) * 1000, f'seller-stated "{m.group(0)}" in the {label}' m = re.search(r'(\d+(?:\.\d+)?)\s*g\b', s) if m: return float(m.group(1)), f'seller-stated "{m.group(0)}" in the {label}' m = re.search(r'(\d+(?:\.\d+)?)\s*oz\b', s) if m: return round(float(m.group(1)) * 28.3495, 1), f'{m.group(1)} oz in the {label}, converted' return '', 'no size stated by the seller' if __name__ == '__main__': cols = (jget(f'https://{SHOP}/collections.json?limit=250') or {}).get('collections') or [] member, failed = collections.defaultdict(list), [] for c in cols: got = (jget(f'https://{SHOP}/collections/{c["handle"]}/products.json?limit=250') or {}).get('products') if got is None: # guard 5 failed.append(c['handle']); print(f' shelf {c["handle"]}: fetch failed — unknown, not empty') continue for p in got: member[p['handle']].append(c['handle']) time.sleep(0.15) print(f"{len(cols)} shelves read, {len(failed)} failed\n") feed = (jget(f'https://{SHOP}/products.json?limit=250') or {}).get('products') or [] print(f"{len(feed)} products in the feed\n") rows = [] for p in feed: handle, title = p['handle'], p['title'] page = fetch(f'https://{SHOP}/products/{handle}') if page is None: rows.append({'product': title, 'handle': handle, 'note': 'product page fetch failed'}) continue spec = spec_rows(page) shelves = member.get(handle, []) ptype = p.get('product_type') or '' on_matcha = sorted(set(shelves) & MATCHA_SHELVES) is_wholesale = bool(set(shelves) & WHOLESALE_SHELVES) or ptype.lower().startswith('wholesale') producer_shelf = sorted(set(shelves) & PRODUCER_SHELVES) # guard 3, in two parts. # (a) Classification comes from the shop's own fields, never from the word "matcha" in a # product name — this shop sells matcha bowls, whisks, chocolate and classes. # (b) **This shop's two category systems disagree, so one of them is not enough.** Its # `matcha-green-tea` shelf carries both hojicha powders, a subscription and a class; # its wholesale shelf is literally named `wholesale-matcha-houjicha-powder-bulk` and # holds both teas. Taking the shelf alone counts 49 matcha; requiring the shelf and # the product_type to agree counts 45. Every disagreement is recorded rather than # silently resolved, so the four can be inspected. type_is_matcha = ptype.strip().lower() in ('matcha', 'wholesale matcha') is_matcha = bool(on_matcha) and type_is_matcha shelf_type_disagree = bool(on_matcha) != type_is_matcha no_weight = ptype in NON_WEIGHT_TYPES # guard 6 for v in p['variants']: vt = str(v.get('title') or '') grams, wsrc = ('', 'not a weighed product') if no_weight else net_grams(vt, spec.get('Packaging')) try: price = float(v.get('price')) except (TypeError, ValueError): price = None try: listp = float(v['compare_at_price']) if v.get('compare_at_price') else None except (TypeError, ValueError): listp = None rows.append({ 'collected': STAMP, 'product': title, 'handle': handle, 'variant': vt, 'product_type': ptype, 'shelves': '; '.join(shelves), 'is_matcha': int(is_matcha), 'matcha_shelves': '; '.join(on_matcha), 'shelf_type_disagree': int(shelf_type_disagree), 'is_wholesale': int(is_wholesale), # guard 4 'producer_shelf': '; '.join(producer_shelf), # guard 8 'price_today_usd': price, 'list_price_usd': listp if listp else '', 'on_sale': int(bool(listp and price and listp > price)), # guard 7 'net_grams': grams, 'weight_source': wsrc, 'per_gram': round(price / grams, 4) if (price and grams) else '', 'shopify_grams_field': v.get('grams'), # kept only to show it disagrees 'sku': v.get('sku') or '', 'available': int(bool(v.get('available'))), 'producer': spec.get('Producer', ''), 'cultivar': spec.get('Cultivar', ''), 'production_area': spec.get('Production Area', '') or spec.get('Area', ''), 'production_year': spec.get('Production Year', '') or spec.get('Harvest', ''), 'packaging': spec.get('Packaging', ''), 'spec_row_count': len(spec), 'spec_keys': '; '.join(sorted(spec)), 'note': '', }) pg = f"${price/grams:.3f}/g" if (price and grams) else '—' if is_matcha: print(f" {title[:34]:<36}{vt[:12]:<14}{pg:<11}" f"prod={spec.get('Producer','—')[:20]:<22}cv={spec.get('Cultivar','—')[:14]:<16}" f"area={spec.get('Production Area','—')[:20]:<22}yr={spec.get('Production Year','—')[:6]}") time.sleep(0.25) out_cols = list(rows[0].keys()) for r in rows: for k in out_cols: r.setdefault(k, '') path = f'data/kettl_listings_{STAMP}.csv' with open(path, 'w', newline='', encoding='utf-8') as f: w = csv.DictWriter(f, fieldnames=out_cols, extrasaction='ignore') w.writeheader(); w.writerows(rows) print(f"\n{len(rows)} variant rows -> {path}") uniq = {r['handle']: r for r in rows if r.get('product')} matcha = [r for r in uniq.values() if r.get('is_matcha')] retail = [r for r in matcha if not r.get('is_wholesale')] dis = [r for r in uniq.values() if r.get('shelf_type_disagree')] print(f"\n=== {len(uniq)} products; {len(matcha)} are matcha on BOTH the shop's matcha shelf " f"and its matcha product type ({len(retail)} retail, {len(matcha)-len(retail)} wholesale) ===") print(f" {len(dis)} products where the two disagree (counted as matcha by one and not the other):") for r in dis: print(f" {r['product'][:44]:<46}type={r['product_type']:<20}matcha shelf={'yes' if r['matcha_shelves'] else 'no'}") print("\n=== Disclosure, across the matcha products ===") for f_ in ('producer', 'cultivar', 'production_area', 'production_year'): n = sum(1 for r in matcha if (r.get(f_) or '').strip()) print(f" {f_:<18}{n:>3}/{len(matcha)}") print("\n=== Producers named ===") for k, v in collections.Counter(r['producer'] for r in matcha if r['producer']).most_common(): print(f" {v:>3} {k}") print("\n=== Cultivars named ===") for k, v in collections.Counter(r['cultivar'] for r in matcha if r['cultivar']).most_common(): print(f" {v:>3} {k}") print("\n=== Production years ===") for k, v in collections.Counter(r['production_year'] for r in matcha if r['production_year']).most_common(): print(f" {v:>3} {k}")