# Reads what is in each set only from the contents list the seller actually publishes. # NOTE - flagging items from the whole description produces false positives. Jade Leaf's "All you # need is some matcha, a small bowl" was read as "includes a bowl" when it means "bring your own # bowl". So the search is confined to the contents list itself. # Sellers who publish no contents list are kept as unlisted, which is itself one of the findings. import re, json, urllib.request, ssl, html, time, csv UA={'User-Agent':'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX=ssl.create_default_context(); CTX.check_hostname=False; CTX.verify_mode=ssl.CERT_NONE BASE={'rockys':'https://www.rockysmatcha.com','jadeleaf':'https://jadeleafmatcha.com', 'matchaeologist':'https://matchaeologist.com','rishi':'https://rishi-tea.com','matchaful':'https://matchaful.com', 'mizuba':'https://mizubatea.com','tezumi':'https://tezumi.com','nekohama':'https://nekohama.co','yunomi':'https://yunomi.life'} # How a contents list starts. The wording differs from seller to seller. LEAD=r'(set includes|kit includes|bundle includes|this bundle includes|the kit contains|the set contains|includes|contains|comes with|what.s inside|in the box)\s*:?' # Identify each item inside the contents list. WARNING - order matters here: the stand patterns # must be consumed first, or "whisk stand" will set the whisk flag. ITEM=[('stand', r'\b(whisk (?:stand|holder|shaper)|chasen naoshi|kusenaoshi|naoshi)\b'), ('whisk_e',r'\belectric whisk\b|\bwhisk frother\b|\bmilk frother\b'), ('whisk', r'\b(chasen|bamboo (?:matcha )?whisk|matcha whisk|whisk)\b'), ('bowl', r'\b(chawan|matcha (?:tea )?bowl|whisking bowl|tea bowl|katakuchi|pouring bowl)\b'), ('scoop', r'\b(chashaku|bamboo (?:tea )?scoop|matcha scoop|tea scoop|scoop|tea spoon)\b'), ('sifter',r'\b(sifter|chakoshi|strainer|sieve|furui)\b'), ('matcha',r'\b(\d+\s*(?:g|gram|grams|oz|servings?)\b[^.;\n]{0,40}matcha|matcha\b[^.;\n]{0,30}\b\d+\s*g\b|ceremonial|culinary|latte mix|tin\b|can\b|pouch)\b'), ('cloth', r'\b(chakin|cloth|napkin)\b'), ('guide', r'\b(handbook|guide|instructions?|booklet|card)\b'), ('extra', r'\b(furoshiki|gift box|tray|caddy|mat|coaster|sticker)\b')] def bodies(base): out={} for page in range(1,15): ps=None for a in range(3): try: r=urllib.request.urlopen(urllib.request.Request(f'{base}/products.json?limit=250&page={page}',headers=UA),timeout=30,context=CTX) ps=json.loads(r.read().decode('utf-8','ignore'))['products']; break except Exception: time.sleep(1.5*(a+1)) if not ps: break for p in ps: out[str(p['id'])]=re.sub(r'\s+',' ',html.unescape(re.sub(r'<[^>]+>',' | ',p.get('body_html') or ''))) time.sleep(0.3) return out cur=list(csv.DictReader(open('data/matcha_sets_curated_2026-08-25.csv'))) inc=[r for r in cur if r['verdict'] in ('included','electric')] cache={} rows=[] for r in inc: if r['shop'] not in cache: cache[r['shop']]=bodies(BASE[r['shop']]); print(f" {r['shop']} {len(cache[r['shop']])} descriptions") b=cache[r['shop']].get(r['pid'],'') m=re.search(LEAD,b,re.I) if not m: rows.append({**{k:r[k] for k in ['shop','title','variant','price_usd','verdict']},'listed':0, 'items':'','n_items':0,'raw':''}); continue seg=b[m.end():m.end()+420] seg=re.split(r'(?:\*|Limit one|Money Back|For guidelines|Note:)',seg)[0] found=[]; work=seg for name,pat in ITEM: if re.search(pat,work,re.I): found.append(name); work=re.sub(pat,' ',work,flags=re.I) rows.append({**{k:r[k] for k in ['shop','title','variant','price_usd','verdict']},'listed':1, 'items':'+'.join(found),'n_items':len(found),'raw':re.sub(r'\s*\|\s*',' / ',seg)[:200]}) with open('data/matcha_set_contents_2026-08-25.csv','w',newline='') as f: w=csv.DictWriter(f,fieldnames=list(rows[0])); w.writeheader(); w.writerows(rows) lst=[r for r in rows if r['listed']] print(f"\nWith a published contents list: {len(lst)}/{len(rows)}") seen=set() for r in sorted(rows,key=lambda x:(x['shop'],float(x['price_usd']))): k=(r['shop'],r['title']) if k in seen: continue seen.add(k) print(f"\n {r['shop']:<14}${float(r['price_usd']):>7.2f} {'['+r['items']+']' if r['listed'] else 'WARNING no contents list'} {r['title'][:52]}") if r['listed']: print(f" raw: {r['raw'][:150]}") print(f"\n-> data/matcha_set_contents_2026-08-25.csv")