#!/usr/bin/env python3 """Per-gram price database for genmaicha and the teas it is made from (sencha, bancha). Used in article 13. The question: genmaicha is roughly half rice by volume. Is it cheaper by that much? Measured within a single seller at a time, so that differences between shops cannot masquerade as differences between teas. Known failure modes in this kind of listing data, and the guard applied to each. Read this before trusting any figure the script produces - it is the fastest way to see what the classifier can still get wrong: 1. Read the currency from Shopify.currency rather than assuming USD (Yunomi prices in yen). 2. Match spelled-out units such as "100 Grams" as well; \bg\b alone misses them. 3. Handle all three multiplication formats: "10 x 100g", "1 oz x 6 boxes", "6 boxes (1 oz each)". 4. Wrap word forms in \b, and exclude compounds that reverse the meaning (e.g. whisk / whisk stand). 5. Decide the category from the title first. When reading the body, check that the sentence describes this product and not a neighbouring one. 6. De-duplicate on product_id + variant_id, so that two different products sharing a name survive. 7. Retry failed fetches. Zero results must never be read as "this shop does not sell it". 8. Read every row by eye before computing anything from it. """ import re, json, ssl, csv, html, time, urllib.request UA={'User-Agent':'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX=ssl.create_default_context(); CTX.check_hostname=False; CTX.verify_mode=ssl.CERT_NONE SHOPS=[('rishi','https://rishi-tea.com'),('mizuba','https://mizubatea.com'),('tezumi','https://tezumi.com'), ('yunomi','https://yunomi.life'),('kettl','https://kettl.co'),('ippodo_us','https://ippodotea.com'), ('hojicha_co','https://hojichaco.com'),('matchaful','https://matchaful.com')] def cur_of(base): try: h=urllib.request.urlopen(urllib.request.Request(base+'/',headers=UA),timeout=20,context=CTX).read().decode('utf-8','ignore') m=re.search(r'Shopify\.currency\s*=\s*\{"active":"(\w{3})"',h) return m.group(1) if m else 'USD' except Exception: return 'USD' def fetch(base): out,page=[],1 while page<=15: ps=None for a in range(3): try: r=urllib.request.urlopen(urllib.request.Request(f'{base}/products.json?limit=250&page={page}',headers=UA),timeout=30,context=CTX) ps=json.loads(r.read().decode('utf-8','ignore')).get('products',[]); break except Exception as e: if a==2: print(f" WARNING {base} page{page} fetch failed: {str(e)[:50]}") time.sleep(1.5*(a+1)) if not ps: break out+=ps; page+=1; time.sleep(0.3) return out # --- Tea type, decided from the title first and evaluated top-down. Genmaicha with matcha # added to it is treated as its own category, not as plain genmaicha. The Japanese # strings below are match patterns for Japanese-language product names: do not translate them. KIND=[('genmaicha_matcha', r'(?=.*\bgenmaicha\b)(?=.*\bmatcha\b)|抹茶入'), ('genmaicha', r'\bgenmaicha\b|\bgenmai[- ]?cha\b|玄米茶'), # Deep-steamed sencha (fukamushicha) is sencha. It is sold on the same product page as # genmaicha often enough that an earlier run confused the two. ('sencha', r'\bsencha\b|\bfukamushi(?:cha)?\b|\bshincha\b|煎茶'), ('bancha', r'\bbancha\b|番茶'), ('kukicha', r'\bkukicha\b|茎茶'), ('hojicha', r'\bhojicha\b|\bhoujicha\b|焙じ茶|ほうじ茶'), ('gyokuro', r'\bgyokuro\b|玉露')] # Drop anything that is not the leaf itself: teaware, confectionery, classes, assortments. EXCL=(r'\b(set|sets|kit|bundle|sampler|gift box|teapot|kyusu|cup|mug|bowl|chawan|tin only|canister|strainer|whisk|scoop|' r'class|tasting|workshop|candy|chocolate|cookie|ice cream|soap|candle|towel|' # Packaging and supplies, which are not tea. An empty aluminium bag was once counted as sencha. r'nylon-?poly|aluminum|aluminium|silver bag|zipper bag|packaging|label|sticker|filter paper|empty)\b') # Net weight: handles spelled-out units and multiplication formats. def grams(text): t=text.lower().replace(',', '') # "10 x 100g", "6 x 1 oz", "500 Tea Bags x 2g", "7 sachets x 2.5g" m=re.search(r'(\d+)\s*(?:tea\s*bags?|bags?|sachets?|pyramids?|sticks?|packets?|units?|pcs)?\s*(?:x|×)\s*(\d+(?:\.\d+)?)\s*(g\b|gram|grams|gr\b|oz\b|ounce|ounces)', t) if m: return _g(float(m.group(2)), m.group(3))*int(m.group(1)) # "1 oz x 6 boxes" (unit stated first) m=re.search(r'(\d+(?:\.\d+)?)\s*(g\b|gram|grams|oz\b|ounce|ounces)\s*(?:x|×)\s*(\d+)', t) if m: return _g(float(m.group(1)), m.group(2))*int(m.group(3)) # "6 boxes (1 oz each)" m=re.search(r'(\d+)\s*(?:boxes|bags|packs|units|pcs)\D{0,12}(\d+(?:\.\d+)?)\s*(g\b|gram|grams|oz\b|ounce|ounces)', t) if m: return _g(float(m.group(2)), m.group(3))*int(m.group(1)) # A single quantity on its own m=re.search(r'(\d+(?:\.\d+)?)\s*(g\b|gram\b|grams\b|oz\b|ounce\b|ounces\b|lb\b|lbs\b|pound\b|pounds\b|kg\b)', t) if m: return _g(float(m.group(1)), m.group(2)) return None def _g(n,u): u=u.strip() if u.startswith('oz') or u.startswith('ounce'): return n*28.3495 if u.startswith('lb') or u.startswith('pound'): return n*453.592 if u.startswith('kg'): return n*1000 return n # Does the description state the rice proportion? This is the core question of the article. RICE=r'(\d{1,2})\s*%\s*(?:roasted\s*)?(?:brown\s*)?rice|rice\D{0,20}(\d{1,2})\s*%|half\s+rice|equal parts' def _match(s): for name,pat in KIND: if re.search(pat,s,re.I): return name return None def kind_of(title, variant=''): # Where one product page sells several teas (Yunomi Pyramids and similar), the variant # name is what identifies the tea actually in the box. v = _match(variant) if variant else None if v: return v # If the title lists more than one tea type, it is a multi-flavour product. When the # variant names none of them, the tea cannot be identified, so the row is dropped. # An earlier version followed the title and counted these under the wrong tea. hits={n for n,pat in KIND if re.search(pat,title,re.I)} hits-= {'genmaicha'} if 'genmaicha_matcha' in hits else set() # matcha-added also matches genmaicha; resolve the overlap if len(hits)>1: return None return _match(title) def form_of(title, variant): s=(title+' '+variant).lower() if re.search(r'\bpowder\b|\bpaudā\b|粉末',s): return 'powder' if re.search(r'\btea ?bags?\b|\bsachets?\b|\bpyramids?\b|\bteabags?\b',s): return 'bag' return 'leaf' rows=[]; seen=set() for shop,base in SHOPS: cur=cur_of(base); n=0 for p in fetch(base): ti=p.get('title',''); tl=ti.lower() if re.search(EXCL,tl): continue body=re.sub(r'\s+',' ',html.unescape(re.sub(r'<[^>]+>',' ',(p.get('body_html') or '')))) rm=re.search(RICE,body,re.I) for v in p.get('variants',[]): pr=v.get('price') if not pr or float(pr)<=0: continue key=(shop,p.get('id'),v.get('id')) if key in seen: continue seen.add(key) vt=(v.get('title') or '') if re.search(r'\bwholesale\b|\bcase of\b|\bfoodservice\b',vt,re.I): continue if re.search(EXCL,vt,re.I): continue k=kind_of(ti, vt) if not k: continue g=grams(vt) or grams(ti) rows.append({'shop':shop,'cur':cur,'pid':p.get('id'),'vid':v.get('id'),'kind':k, 'form':form_of(ti,vt), 'title':ti[:110],'variant':vt[:48],'price':pr,'grams':round(g,2) if g else '', 'rice_claim':(rm.group(0)[:40] if rm else '')}) n+=1 print(f" {shop:<12} {n:>4} variants ({cur})") time.sleep(0.4) F=['shop','cur','pid','vid','kind','form','title','variant','price','grams','rice_claim'] with open('data/genmaicha_prices_2026-08-25.csv','w',newline='') as f: w=csv.DictWriter(f,fieldnames=F); w.writeheader(); w.writerows(rows) print(f"\nTotal {len(rows)} rows -> data/genmaicha_prices_2026-08-25.csv") import collections print("By tea type:", dict(collections.Counter(r['kind'] for r in rows))) print("Rows with a usable net weight:", sum(1 for r in rows if r['grams']), "/", len(rows))