# Price database for matcha sets (set/kit/bundle) and, from the same sellers, the individual # pieces sold separately: whisk, bowl, scoop, sifter and whisk stand. # This collects exactly the territory that article 11's chasen script deliberately excluded, # by running the opposite filter. # Known failure modes in this kind of listing data, and the guard applied to each: # 1. read the currency from Shopify.currency rather than assuming USD; # 2. always wrap word forms in \b; 3. decide the category from the title first; # 4. de-duplicate on product_id; 5. read every row by eye before computing anything. import urllib.request, json, ssl, re, csv, html, time UA={'User-Agent':'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120 Safari/537.36'} CTX=ssl.create_default_context(); CTX.check_hostname=False; CTX.verify_mode=ssl.CERT_NONE SHOPS=[('rockys','https://www.rockysmatcha.com'),('jadeleaf','https://jadeleafmatcha.com'), ('matchaeologist','https://matchaeologist.com'),('ippodo_us','https://ippodotea.com'),('kettl','https://kettl.co'), ('rishi','https://rishi-tea.com'),('matchaful','https://matchaful.com'),('hojicha_co','https://hojichaco.com'), ('mizuba','https://mizubatea.com'),('tezumi','https://tezumi.com'),('nekohama','https://nekohama.co'), ('yunomi','https://yunomi.life')] def cur_of(base): try: h=urllib.request.urlopen(urllib.request.Request(base+'/',headers=UA),timeout=20,context=CTX).read().decode('utf-8','ignore') m=re.search(r'Shopify\.currency\s*=\s*\{"active":"(\w{3})"',h) return m.group(1) if m else 'USD' except Exception: return 'USD' def fetch(base): # WARNING - letting a temporary failure pass silently as zero results reads as "this shop # does not stock it". That actually happened with four shops. out,page=[],1 while page<=14: ps=None for attempt in range(3): try: r=urllib.request.urlopen(urllib.request.Request(f'{base}/products.json?limit=250&page={page}',headers=UA),timeout=30,context=CTX) ps=json.loads(r.read().decode('utf-8','ignore')).get('products',[]); break except Exception as e: if attempt==2: print(f" WARNING {base} page{page} fetch failed: {str(e)[:60]}") time.sleep(1.5*(attempt+1)) if ps is None or not ps: break out+=ps; page+=1 time.sleep(0.3) return out # --- Category, decided from the title alone and evaluated top-down, with 'set' taking priority. --- # NOTE - \bkit\b does not match "kitchen": the character after "kit" is "c", so there is no word # boundary there. CAT=[ ('set', r'\b(set|sets|kit|kits|bundle|bundles|starter|gift box|giftbox|collection|combo|duo|trio|essentials)\b'), ('frother',r'\b(frother|frothers|electric)\b'), ('stand', r'\b(kusenaoshi|kuse naoshi|whisk stand|whisk holder|whisk shaper|chasen naoshi|naoshi)\b'), ('sifter', r'\b(sifter|sifters|sieve|strainer|furui|shifter)\b'), ('scoop', r'\b(chashaku|scoop|scoops|spoon|spoons|ladle)\b'), ('bowl', r'\b(chawan|bowl|bowls|katakuchi|pitcher)\b'), ('whisk', r'\b(chasen|whisk|whisks)\b'), ('cloth', r'\b(chakin|cloth|napkin|towel)\b'), ('tin', r'\b(tin|tins|canister|caddy|natsume|container)\b'), ] # What the set contains, detected from the title plus the opening of the description. COMP={ 'has_whisk': r'\b(chasen|whisk|whisks)\b', 'has_bowl': r'\b(chawan|bowl|bowls)\b', 'has_scoop': r'\b(chashaku|scoop|scoops|spoon)\b', 'has_sifter':r'\b(sifter|sieve|strainer|furui)\b', 'has_stand': r'\b(kusenaoshi|naoshi|whisk stand|whisk holder|whisk shaper)\b', 'has_matcha':r'\b(matcha|tea)\b', 'has_cloth': r'\b(chakin|cloth)\b', 'has_tin': r'\b(tin|canister|caddy|natsume)\b', } def classify(title_l): for name,pat in CAT: if re.search(pat,title_l): return name return 'other' sets_rows,comp_rows=[],[] # WARNING - de-duplicate on (shop, product_id, variant_id). Keying on title plus price deletes # genuinely different products that happen to share both. seen=set() for shop,base in SHOPS: cur=cur_of(base); ns=nc=0 for p in fetch(base): ti=p.get('title',''); tl=ti.lower() cat=classify(tl) if cat=='other': continue body=re.sub(r'\s+',' ',html.unescape(re.sub(r'<[^>]+>',' ',(p.get('body_html') or '')))).lower() ptype=(p.get('product_type') or '') base_row={'shop':shop,'cur':cur,'pid':p.get('id',''),'ptype':ptype[:30],'title':ti[:110]} for v in p.get('variants',[]): pr=v.get('price') if not pr: continue key=(shop,p.get('id'),v.get('id')) if key in seen: continue seen.add(key) vt=(v.get('title') or ''); vt_l=vt.lower() # Bulk and wholesale variants are not the price of a single item. if re.search(r'\b\d{2,}\s*(units?|pcs|pieces|packs?)\b|wholesale|case of',vt_l): continue r=dict(base_row); r['variant']=vt[:40]; r['price']=pr if cat=='set': full=tl+' '+body[:1200]+' '+vt_l for k,pat in COMP.items(): r[k]=1 if re.search(pat,full) else 0 r['n_comp']=sum(r[k] for k in COMP) sets_rows.append(r); ns+=1 else: r['cat']=cat; comp_rows.append(r); nc+=1 print(f" {shop:<15} sets{ns:>3} / pieces{nc:>3} ({cur})") time.sleep(0.4) SF=['shop','cur','pid','ptype','title','variant','price']+list(COMP)+['n_comp'] with open('data/matcha_sets_2026-08-25.csv','w',newline='') as f: w=csv.DictWriter(f,fieldnames=SF); w.writeheader(); w.writerows(sets_rows) CF=['shop','cur','pid','ptype','title','variant','price','cat'] with open('data/matcha_components_2026-08-25.csv','w',newline='') as f: w=csv.DictWriter(f,fieldnames=CF); w.writeheader(); w.writerows(comp_rows) print(f"\nSets {len(sets_rows)} -> data/matcha_sets_2026-08-25.csv") print(f"Pieces {len(comp_rows)} -> data/matcha_components_2026-08-25.csv")