# Turns the raw set data into the curated file, and computes "set price vs. sum of the parts". # WARNING - every one of the 73 candidates was read by eye, description included, and the verdict # is written into this file explicitly rather than being left to an automatic rule. # NOTE - a word-form trap worth knowing about: \bwhisk\b also matches "whisk stand", a rest that # holds the whisk, and will count it as a whisk unless the stand patterns are consumed first. import csv, re, statistics as st JPY_USD = 160.0 # The rate used across this site since article 01. Always quote it with the collection date. # --- Exclusions confirmed by reading each listing. Keyed on the start of the title rather than # the product id, and every one carries its reason. --- EXCLUDE = { 'Original Edition Sticker Collection': 'Stickers. The "Chasen" in the description is the name of the artwork.', "rocky's matcha Daily Ceremonial Bundle": 'Matcha only. The "whisk" in the description is preparation advice.', "rocky's matcha Ceremonial Blend Bundle": 'Matcha only. The "whisk" in the description is preparation advice.', 'Twin Cloud Glass Chawan Set': 'Two bowls. The description says they pair well with a chasen; no whisk is included.', 'Mizuba Matcha Tasting Trio': 'Three matchas. The "whisk" in the description is preparation advice.', 'Matcha Moment® Essential Tea Set': 'Contents cannot be established from the seller\'s own description.', 'Hello Homebody Matcha Set': 'Bowl, cup and whisk stand. The whisk itself is not included.', 'Kiwakoto Suikaen': 'Bowl plus a whisk *stand*. The whisk itself is not included (the whisk-stand word-form error).', 'Latte Matcha Bundle': 'The description does not mention a whisk.', } # Sets built around an electric frother are counted separately from bamboo-whisk sets. ELECTRIC = ['Modern Matcha Starter Set','Cafe Style Matcha Latte Mix Starter Bundle','Culinary Matcha Starter Bundle'] def usd(p, cur): return float(p)/JPY_USD if cur=='JPY' else float(p) sets=[r for r in csv.DictReader(open('data/matcha_sets_2026-08-25.csv')) if r['has_whisk']=='1'] comps=list(csv.DictReader(open('data/matcha_components_2026-08-25.csv'))) def verdict(r): for k,why in EXCLUDE.items(): if r['title'].startswith(k) or k in r['title']: return 'excluded', why if r['shop']=='rishi' and r['title'].startswith('Matcha Tin Duo Sampler') and 'whisk' not in r['variant'].lower(): return 'excluded','This variant ships without a whisk (only the "Tins & Whisk" variant includes one).' for k in ELECTRIC: if r['title'].startswith(k): return 'electric','Electric frother, not a bamboo whisk.' if 'WAREHOUSE' in r['title'].upper(): return 'outlet','Outlet stock; the same product is also listed at full price.' return 'included','' out=[] for r in sets: v,why=verdict(r) out.append({'shop':r['shop'],'cur':r['cur'],'pid':r['pid'],'title':r['title'],'variant':r['variant'], 'price_raw':r['price'],'price_usd':round(usd(r['price'],r['cur']),2),'verdict':v,'reason':why, **{k:r[k] for k in ['has_whisk','has_bowl','has_scoop','has_sifter','has_stand','has_tin','has_cloth']}}) with open('data/matcha_sets_curated_2026-08-25.csv','w',newline='') as f: w=csv.DictWriter(f,fieldnames=list(out[0])); w.writeheader(); w.writerows(out) inc=[r for r in out if r['verdict']=='included'] print(f"=== Verdicts ===") for v in ['included','electric','outlet','excluded']: n=[r for r in out if r['verdict']==v]; print(f" {v:<10}{len(n):>3}") print(f"\n=== Price distribution of the {len(inc)} included sets (converted at ¥{JPY_USD:.0f}/USD) ===") p=sorted(r['price_usd'] for r in inc) print(f" Cheapest ${p[0]:.2f} / median ${st.median(p):.2f} / dearest ${p[-1]:.2f} (n={len(p)})") print(f" Shops: {len(set(r['shop'] for r in inc))}") # --- Within one seller: set price vs. the sum of the same seller's individual pieces --- CAT={'has_bowl':'bowl','has_scoop':'scoop','has_sifter':'sifter','has_stand':'stand'} med={} for r in comps: med.setdefault((r['shop'],r['cat']),[]).append(usd(r['price'],r['cur'])) med={k:st.median(v) for k,v in med.items()} print(f"\n=== Set price vs. sum of the parts, within one seller (parts at that seller's median) ===") print(f" {'shop':<14}{'set':>9}{'parts':>10}{'diff':>9} breakdown") rows=[] for r in sorted(inc,key=lambda x:(x['shop'],x['price_usd'])): parts=[('whisk',med.get((r['shop'],'whisk')))] for k,c in CAT.items(): if r[k]=='1': parts.append((c,med.get((r['shop'],c)))) # A shop that does not sell the pieces separately cannot be compared - which is itself a finding. if any(v is None for _,v in parts): continue tot=sum(v for _,v in parts) rows.append((r,tot)) print(f" {r['shop']:<14}${r['price_usd']:>8.2f}${tot:>9.2f}{(r['price_usd']-tot):>+9.2f} {'+'.join(c for c,_ in parts)} {r['title'][:34]}") if rows: d=[(x['price_usd']-t)/t for x,t in rows] print(f"\n Median difference across the {len(rows)} comparable sets = {st.median(d)*100:+.1f}%") print(f"\n-> data/matcha_sets_curated_2026-08-25.csv")