#!/usr/bin/env python3 """Measures whether gyokuro sellers state the number of days the tea was shaded. Used in article 14. The Japan Tea Central Public Interest Incorporated Association defines gyokuro as leaf from a covered garden, shaded almost completely from sunlight for about twenty days from the point the first-flush buds start to open. Anything shaded for less than twenty days is kabuse tea instead. The shading period is therefore the one number that makes gyokuro gyokuro. This script asks whether the people selling it print that number. Known failure modes in this kind of listing data, and the guard applied to each. Read this before trusting any figure the script produces: - Wrap word forms in \b, and watch for compounds that reverse the meaning (shade / shade net / shade-grown). - Decide the category from the title first. When reading the body, check that the sentence describes this product and not a neighbouring one. - Retry failed fetches. Zero results must never be read as "this shop does not sell it". - Read every row by eye before computing anything from it. """ import re, json, ssl, csv, html, time, urllib.request UA={'User-Agent':'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX=ssl.create_default_context(); CTX.check_hostname=False; CTX.verify_mode=ssl.CERT_NONE SHOPS=[('rishi','https://rishi-tea.com'),('mizuba','https://mizubatea.com'),('tezumi','https://tezumi.com'), ('yunomi','https://yunomi.life'),('kettl','https://kettl.co'),('ippodo_us','https://ippodotea.com')] def fetch(base): out,page=[],1 while page<=15: ps=None for a in range(3): try: r=urllib.request.urlopen(urllib.request.Request(f'{base}/products.json?limit=250&page={page}',headers=UA),timeout=30,context=CTX) ps=json.loads(r.read().decode('utf-8','ignore')).get('products',[]); break except Exception: time.sleep(1.5*(a+1)) if not ps: break out+=ps; page+=1; time.sleep(0.3) return out # The Japanese strings in the patterns below match Japanese-language product names and # descriptions. They are matching rules, not prose: do not translate them. IS_GYOKURO=r'\bgyokuro\b|玉露' # Kabuse tea and karigane (stem tea) are not gyokuro leaf, so they are labelled separately. IS_KABUSE=r'\bkabuse\b|\bkabusecha\b|かぶせ' IS_KARIGANE=r'\bkarigane\b|\bshiraore\b|\bkukicha\b|\bkuki[- ]?cha\b|雁ヶ音|白折|茎茶' EXCL=(r'\b(set|kit|bundle|sampler|kyusu|cup|bowl|canister|gift box|shiboridashi|hohin|houhin|' r'tea\s*pot|teapot|tableware|strainer|tray)\b') # A stated shading period with a number in it: "20 days", "20-day", "three weeks", "20日" and so on. SHADE_DAYS=[ r'(\d{1,3})\s*(?:\+|plus)?\s*days?\s*(?:of\s*)?(?:shad|cover|shelter)', r'(?:shad|cover|shelter)\w*\s*(?:for\s*)?(?:about\s*|approximately\s*|over\s*|around\s*)?(\d{1,3})\s*(?:\+|plus)?\s*days?', r'(\d{1,2})\s*weeks?\s*(?:of\s*)?(?:shad|cover)', r'(?:shad|cover)\w*\s*(?:for\s*)?(?:about\s*)?(\d{1,2})\s*weeks?', r'(\d{1,3})\s*日\s*(?:間\s*)?(?:被覆|覆|遮光)', ] # Wording that mentions shading but gives no number of days. SHADE_MENTION=r'\bshade[- ]?grown\b|\bshaded\b|\bshading\b|\bcovered\b|\bcovering\b|\bunder shade\b|覆下|被覆|遮光' def shade_days(text): for pat in SHADE_DAYS: m=re.search(pat,text,re.I) if m: n=int(m.group(1)) if 'week' in m.group(0).lower(): n*=7 if 3<=n<=90: return n, m.group(0)[:60] return None, '' rows=[]; seen=set() for shop,base in SHOPS: n=0 for p in fetch(base): ti=p.get('title',''); tl=ti.lower() if not re.search(IS_GYOKURO,tl,re.I): continue if re.search(EXCL,tl,re.I): continue kind='gyokuro' if re.search(IS_KARIGANE,tl,re.I): kind='karigane' # stem tea cut from gyokuro elif re.search(IS_KABUSE,tl,re.I): kind='kabuse' # under twenty days, so a different tea body=re.sub(r'\s+',' ',html.unescape(re.sub(r'<[^>]+>',' ',(p.get('body_html') or '')))) d,ev=shade_days(body) if p.get('id') in seen: continue seen.add(p.get('id')) rows.append({'shop':shop,'pid':p.get('id'),'kind':kind,'title':ti[:100], 'shade_days':d if d else '','evidence':ev, 'mentions_shade':1 if re.search(SHADE_MENTION,body,re.I) else 0, 'body_len':len(body)}) n+=1 print(f" {shop:<11} {n:>3} products") time.sleep(0.4) with open('data/gyokuro_shade_2026-08-25.csv','w',newline='') as f: w=csv.DictWriter(f,fieldnames=['shop','pid','kind','title','shade_days','evidence','mentions_shade','body_len']) w.writeheader(); w.writerows(rows) g=[r for r in rows if r['kind']=='gyokuro'] print(f"\nGyokuro: {len(g)} products, of which {sum(1 for r in g if r['shade_days'])} state the shading period") print(f" Mention shading but give no number of days: {sum(1 for r in g if r['mentions_shade'] and not r['shade_days'])}") print(f" Do not mention shading at all: {sum(1 for r in g if not r['mentions_shade'])}") print(f"-> data/gyokuro_shade_2026-08-25.csv ({len(rows)} rows in total, including kabuse and karigane)")