#!/usr/bin/env python3
"""Reads what Kettl publishes about every tea it sells, from the seller's own pages. Used in article 24.
The question this answers is narrow: for each product Kettl lists, which of its own shelves does it
sit on, what does it disclose about who grew the tea and when, and what does a gram cost.
Known failure modes in this kind of listing data, and the guard applied to each. Read this before
trusting any figure the script produces:
1. **A product feed is not the page.** `products.json` returns `body_html`, but this shop puts its
producer / cultivar / area / year block in a theme text block that the feed never returns.
Every disclosure field here is read from the rendered product page.
2. A Shopify variant's `grams` field is **shipping weight, not net weight**. Net weight is taken
from the size the seller states — in the variant name ("20g", "1kg") or in the page's own
`Packaging:` row ("20g tin") — and the source of every weight is recorded.
3. **What counts as matcha is decided by the shop's own shelves and product types**, never by the
word "matcha" in a product name. This shop sells matcha bowls, matcha whisks, matcha chocolate
and matcha classes, all of which carry the word.
4. **Wholesale and bulk lines distort every per-gram figure.** This shop sells the same teas in
1 kg wholesale bags alongside 20 g tins. They are collected and flagged so that retail and
wholesale are never mixed into one median.
5. Retry failed fetches. A page that fails to load is missing, never "the seller does not disclose
this". Failures are written to the CSV as a note, not dropped.
6. Sets, subscriptions and classes have no net weight. They are marked and excluded from per-gram
statistics rather than silently dropped.
7. **A shop can run a sale.** The feed's `price` is what the shop charges today and
`compare_at_price` is its list price. Both are recorded.
8. **This shop's shelves are named after producers**, and a tea can sit on a producer shelf and a
category shelf at once. Membership is recorded per shelf, never collapsed into one label.
"""
import ssl, re, csv, json, time, html, urllib.request, collections
SHOP = 'kettl.co'
STAMP = '2026-08-28'
UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'}
CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
# Guard 3: the shop's own matcha shelves. Everything else that has "matcha" in its name — bowls,
# whisks, chocolate, classes — is not matcha and must not enter a per-gram figure.
MATCHA_SHELVES = {'matcha-green-tea', 'kettl-matcha-sen-mon-ten', 'furukawa-family-matcha',
'the-yasuharu-maeno-matcha-collection', 'wholesale-matcha-houjicha-powder-bulk'}
# Guard 4
WHOLESALE_SHELVES = {'wholesale-matcha-houjicha-powder-bulk', 'wholesale-loose-tea-bulk',
'wholesale-tea', 'wholesale-retail-tea-and-wares', 'wholesale-tea-bag-bulk'}
# Guard 8: shelves named after a person or family who grew or made the thing.
PRODUCER_SHELVES = {'the-tsuji-family-collection', 'the-yamaguchi-family-collection',
'the-yasuharu-maeno-matcha-collection', 'furukawa-family-matcha',
'tsuguto-hattori-hand-picked-tea-collection', 'takaaki-yoshida',
'yoshiaki-hiruma-collection', 'shigeyoshi-morioka-collection',
'the-hiroshi-kobayashi-collection'}
# Guard 6
NON_WEIGHT_TYPES = {'Ceramics', 'ceramics', 'Class', 'class', 'Incense', 'incense', 'Merchandise',
'Subscription', 'Subscription matcha', 'Tea Accessory', 'Tea Pot', 'Tea Tools',
'Tea Ware', 'Tea tools', 'Chocolate', 'Kombucha', 'kombucha'}
# The rows this shop puts in its own spec block, in the order it writes them.
SPEC_KEYS = ['Producer', 'Cultivar', 'Production Area', 'Production Year', 'Packaging',
'Region', 'Harvest', 'Area', 'Farm', 'Steaming', 'Elevation']
def fetch(url, tries=4):
for a in range(tries):
try:
r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=60, context=CTX)
return r.read().decode('utf-8', 'ignore')
except Exception as e:
if a == tries - 1:
print(f' FETCH FAILED after {tries} tries: {url[:90]} {e}')
return None
time.sleep(2 * (a + 1))
def jget(url):
t = fetch(url)
try:
return json.loads(t) if t else None
except Exception:
return None
def txt(s):
return re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', s or ''))).strip()
def spec_rows(page):
"""The shop's producer/cultivar/area/year block (guard 1).
It is a single
whose lines are separated by
, e.g.
Producer: Shinya Yamaguchi
Cultivar: Blend
Production Area: Hoshinomura, Yame
Only paragraphs that actually contain one of the shop's own keys are read, so that ordinary
prose containing a colon is not mistaken for a specification.
"""
out = {}
for m in re.finditer(r'
(.*?)
', page, re.S):
block = m.group(1)
if not any(f'{k}:' in block for k in SPEC_KEYS):
continue
for line in re.split(r'
', block):
s = txt(line)
km = re.match(r'([A-Z][A-Za-z ]{2,20}):\s*(.+)$', s)
if km and km.group(1).strip() in SPEC_KEYS:
out.setdefault(km.group(1).strip(), km.group(2).strip())
return out
# Guard 2: net weight from what the seller states, never from Shopify's `grams`.
def net_grams(variant_title, packaging):
for src, label in ((variant_title or '', 'variant name'), (packaging or '', 'Packaging row')):
s = src.lower().replace(',', '')
m = re.search(r'(\d+(?:\.\d+)?)\s*kg\b', s)
if m:
return float(m.group(1)) * 1000, f'seller-stated "{m.group(0)}" in the {label}'
m = re.search(r'(\d+(?:\.\d+)?)\s*g\b', s)
if m:
return float(m.group(1)), f'seller-stated "{m.group(0)}" in the {label}'
m = re.search(r'(\d+(?:\.\d+)?)\s*oz\b', s)
if m:
return round(float(m.group(1)) * 28.3495, 1), f'{m.group(1)} oz in the {label}, converted'
return '', 'no size stated by the seller'
if __name__ == '__main__':
cols = (jget(f'https://{SHOP}/collections.json?limit=250') or {}).get('collections') or []
member, failed = collections.defaultdict(list), []
for c in cols:
got = (jget(f'https://{SHOP}/collections/{c["handle"]}/products.json?limit=250') or {}).get('products')
if got is None: # guard 5
failed.append(c['handle']); print(f' shelf {c["handle"]}: fetch failed — unknown, not empty')
continue
for p in got:
member[p['handle']].append(c['handle'])
time.sleep(0.15)
print(f"{len(cols)} shelves read, {len(failed)} failed\n")
feed = (jget(f'https://{SHOP}/products.json?limit=250') or {}).get('products') or []
print(f"{len(feed)} products in the feed\n")
rows = []
for p in feed:
handle, title = p['handle'], p['title']
page = fetch(f'https://{SHOP}/products/{handle}')
if page is None:
rows.append({'product': title, 'handle': handle, 'note': 'product page fetch failed'})
continue
spec = spec_rows(page)
shelves = member.get(handle, [])
ptype = p.get('product_type') or ''
on_matcha = sorted(set(shelves) & MATCHA_SHELVES)
is_wholesale = bool(set(shelves) & WHOLESALE_SHELVES) or ptype.lower().startswith('wholesale')
producer_shelf = sorted(set(shelves) & PRODUCER_SHELVES)
# guard 3, in two parts.
# (a) Classification comes from the shop's own fields, never from the word "matcha" in a
# product name — this shop sells matcha bowls, whisks, chocolate and classes.
# (b) **This shop's two category systems disagree, so one of them is not enough.** Its
# `matcha-green-tea` shelf carries both hojicha powders, a subscription and a class;
# its wholesale shelf is literally named `wholesale-matcha-houjicha-powder-bulk` and
# holds both teas. Taking the shelf alone counts 49 matcha; requiring the shelf and
# the product_type to agree counts 45. Every disagreement is recorded rather than
# silently resolved, so the four can be inspected.
type_is_matcha = ptype.strip().lower() in ('matcha', 'wholesale matcha')
is_matcha = bool(on_matcha) and type_is_matcha
shelf_type_disagree = bool(on_matcha) != type_is_matcha
no_weight = ptype in NON_WEIGHT_TYPES # guard 6
for v in p['variants']:
vt = str(v.get('title') or '')
grams, wsrc = ('', 'not a weighed product') if no_weight else net_grams(vt, spec.get('Packaging'))
try:
price = float(v.get('price'))
except (TypeError, ValueError):
price = None
try:
listp = float(v['compare_at_price']) if v.get('compare_at_price') else None
except (TypeError, ValueError):
listp = None
rows.append({
'collected': STAMP, 'product': title, 'handle': handle, 'variant': vt,
'product_type': ptype, 'shelves': '; '.join(shelves),
'is_matcha': int(is_matcha), 'matcha_shelves': '; '.join(on_matcha),
'shelf_type_disagree': int(shelf_type_disagree),
'is_wholesale': int(is_wholesale), # guard 4
'producer_shelf': '; '.join(producer_shelf), # guard 8
'price_today_usd': price, 'list_price_usd': listp if listp else '',
'on_sale': int(bool(listp and price and listp > price)), # guard 7
'net_grams': grams, 'weight_source': wsrc,
'per_gram': round(price / grams, 4) if (price and grams) else '',
'shopify_grams_field': v.get('grams'), # kept only to show it disagrees
'sku': v.get('sku') or '', 'available': int(bool(v.get('available'))),
'producer': spec.get('Producer', ''), 'cultivar': spec.get('Cultivar', ''),
'production_area': spec.get('Production Area', '') or spec.get('Area', ''),
'production_year': spec.get('Production Year', '') or spec.get('Harvest', ''),
'packaging': spec.get('Packaging', ''),
'spec_row_count': len(spec), 'spec_keys': '; '.join(sorted(spec)),
'note': '',
})
pg = f"${price/grams:.3f}/g" if (price and grams) else '—'
if is_matcha:
print(f" {title[:34]:<36}{vt[:12]:<14}{pg:<11}"
f"prod={spec.get('Producer','—')[:20]:<22}cv={spec.get('Cultivar','—')[:14]:<16}"
f"area={spec.get('Production Area','—')[:20]:<22}yr={spec.get('Production Year','—')[:6]}")
time.sleep(0.25)
out_cols = list(rows[0].keys())
for r in rows:
for k in out_cols:
r.setdefault(k, '')
path = f'data/kettl_listings_{STAMP}.csv'
with open(path, 'w', newline='', encoding='utf-8') as f:
w = csv.DictWriter(f, fieldnames=out_cols, extrasaction='ignore')
w.writeheader(); w.writerows(rows)
print(f"\n{len(rows)} variant rows -> {path}")
uniq = {r['handle']: r for r in rows if r.get('product')}
matcha = [r for r in uniq.values() if r.get('is_matcha')]
retail = [r for r in matcha if not r.get('is_wholesale')]
dis = [r for r in uniq.values() if r.get('shelf_type_disagree')]
print(f"\n=== {len(uniq)} products; {len(matcha)} are matcha on BOTH the shop's matcha shelf "
f"and its matcha product type ({len(retail)} retail, {len(matcha)-len(retail)} wholesale) ===")
print(f" {len(dis)} products where the two disagree (counted as matcha by one and not the other):")
for r in dis:
print(f" {r['product'][:44]:<46}type={r['product_type']:<20}matcha shelf={'yes' if r['matcha_shelves'] else 'no'}")
print("\n=== Disclosure, across the matcha products ===")
for f_ in ('producer', 'cultivar', 'production_area', 'production_year'):
n = sum(1 for r in matcha if (r.get(f_) or '').strip())
print(f" {f_:<18}{n:>3}/{len(matcha)}")
print("\n=== Producers named ===")
for k, v in collections.Counter(r['producer'] for r in matcha if r['producer']).most_common():
print(f" {v:>3} {k}")
print("\n=== Cultivars named ===")
for k, v in collections.Counter(r['cultivar'] for r in matcha if r['cultivar']).most_common():
print(f" {v:>3} {k}")
print("\n=== Production years ===")
for k, v in collections.Counter(r['production_year'] for r in matcha if r['production_year']).most_common():
print(f" {v:>3} {k}")