#!/usr/bin/env python3 """What sellers say sencha is, across every sencha listing reachable from fifteen shops. Used in article 18. Google's AI Overview for `sencha tea` states, as settled fact, that sencha comes in three steaming grades — asamushi (light), chuumushi (medium), fukamushi (deep) — and that each tastes different. It also gives one brewing recipe and one caffeine comparison. None of that is measured against the shelf. This script does the measuring: for each listing, does the seller say which steaming grade it is, which cultivar, which harvest, where it grew, how to brew it, and what a gram costs. Guards 1-16 are inherited from hojicha_disclosure.py unchanged, because the failure modes are the same ones and each was found the hard way. Read that file's header for the reasoning; the short version of each is repeated below. 1. **Currency.** One of these shops prices in yen; the shop's own `Shopify.currency` is read first. 2. **Units spelled out.** `\s*g\b` does not match "100 Grams". 3. **Multiplication is written three ways**: "10 x 100g", "1 oz x 6 boxes", "6 boxes (1 oz each)". 4. **Partial words.** Every term is matched with word boundaries; phrases as phrases. 5. **A set is not the thing.** One price covering several teas is not a sencha price; candidates are decided by hand and the calls published. 6. **Sweetened mixes are not tea.** 7. **Classification comes from the seller** (`product_type`), recorded alongside our own. 8. **Retry, and never read a failure as a zero.** 9. **Scope.** Terms are counted in the seller's own title and description. 10. **A negation is not a claim.** "not sencha" puts the word in the text while denying it. 11. **Synonyms are one descriptor, not two.** 12. **Two kinds of temperature.** A sencha page can give a brewing temperature and a steaming or firing temperature. Only water in the 50-100 C range is read as brewing. 13. **The brand name is not an ingredient.** Shop wording is stripped before reading tea words. 14. **A tea word in the title is a candidate, not a set.** 15. **Compound names.** `fukamushicha` and `fukamushi sencha` are deep-steamed sencha and count; `ichibancha` numbers a harvest and is counted as first flush, not as a steaming grade. 16. **Form comes from the seller's category field first**, then the title. New in this script, because sencha is measured on different axes than hojicha: 17. **Steaming grade is the claim under test.** asamushi / chuumushi / fukamushi are matched as a family each, including the -cha and -sencha compounds and the English glosses, because sellers write "deep steamed", "fukamushi" and "fukamushicha" for the same thing. """ import urllib.request, json, ssl, re, csv, html, time, collections UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE # Guard 13: the brand name is not an ingredient. "rocky's matcha Houjicha 100g" is plain hojicha # sold by a shop with matcha in its name, and matching tea words against the raw title flagged all # four of Rocky's hojicha as blends and dropped them out of the price medians. Brand wording is # removed from the title before any tea name is read off it. BRAND_WORDS = { 'rockys': [r"rocky'?[’']?s matcha", r'rockys matcha'], 'ujido': [r'ujido'], 'jadeleaf': [r'jade leaf matcha', r'jade leaf'], 'matchaeologist': [r'matchaeologist'], 'ippodo_us': [r'ippodo'], 'kettl': [r'kettl'], 'rishi': [r'rishi'], 'matchaful': [r'matchaful'], 'hojicha_co': [r'hojicha co\.?'], 'yunomi': [r'yunomi'], 'mizuba': [r'mizuba'], 'nami': [r'nami matcha', r'\bnami\b'], 'arteao': [r'art of tea'], 'moongoat': [r'moon ?goat', r'matchabot'], 'tezumi': [r'tezumi'], } # Guard 14: a set is one price covering several different teas, and no feed field says so. Titles # naming two or more teas are candidates only; each is decided by hand and the calls are published, # because "Hojicha (Dark Roasted Sencha)" names its base leaf and "Matcha & Hojicha Duo" is two # products in a box, and nothing in the wording separates them. TEA_NAMES = [r'genmaicha', r'gyokuro', r'kukicha', r'bancha', r'black tea', r'matcha', r'ho[ju]?jicha', r'houjicha', r'oolong'] SET_CALLS = { # decided by hand: one price covering more than one tea, or two teas mixed 'shincha & kumpu sencha set': True, 'single cultivar sencha tasting set': True, 'sencha lover loose leaf gift set': True, '#0781.tx yunomi tea house blend: 久久能智神 kukunochinokami - spring sencha infused with matcha': True, } # Decided by hand: the product is a different tea, and its category field is empty so guard 19 # cannot see it. Sencha appears in the title only as the leaf it was made from. OTHER_TEA_CALLS = { # a genmaicha whose title names sencha only as the leaf it was made from '#0633.k6 shogyokuen: premium ichibancha genmaicha from uji, kyoto (made with high grade spring sencha)', } # Kept, and recorded here so the call is visible rather than silent: Kettl's "Kukicha Sencha" is # stems rather than leaf, which article 17 treats as its own tea. It stays in because the seller # sells it as a sencha and its category field says only `Loose Tea`; it is one listing in 300. SHOPS = [('rockys','https://www.rockysmatcha.com'), ('ujido','https://ujido.com'), ('jadeleaf','https://jadeleafmatcha.com'), ('matchaeologist','https://matchaeologist.com'), ('ippodo_us','https://ippodotea.com'), ('kettl','https://kettl.co'), ('rishi','https://rishi-tea.com'), ('matchaful','https://matchaful.com'), ('hojicha_co','https://hojichaco.com'), ('yunomi','https://yunomi.life'), ('mizuba','https://mizubatea.com'), ('nami','https://namimatcha.com'), ('arteao','https://arteao.com'), ('moongoat','https://moongoat.com'), ('tezumi','https://tezumi.com')] # Guard 4: phrases are matched as phrases; single words get word boundaries. # Guard 17: the steaming grade is the claim under test. Sellers write the same thing three ways — # the Japanese word, the word with -cha or -sencha attached, and the English gloss — so each grade # is one family. Guard 15 applies here too: `ichibancha` contains `bancha` and numbers a harvest, # so it is read as first flush below and never as a grade. STEAM = { 'asamushi': [r'asamushi\w*', r'asa mushi', r'light[- ]steam\w*', r'lightly steamed', r'shallow[- ]steam\w*'], 'chuumushi': [r'chuumushi\w*', r'chumushi\w*', r'chu mushi', r'medium[- ]steam\w*', r'standard steam\w*'], 'fukamushi': [r'fukamushi\w*', r'fuka mushi', r'deep[- ]steam\w*', r'deeply steamed', r'long[- ]steam\w*'], 'steamed at all': [r'steam\w*'], # the process itself, however vaguely stated } # What plant, and when it was picked. The site already has an article on cultivars, so this is the # axis a reader can act on. CULTIVAR = { 'yabukita': [r'yabukita'], 'saemidori': [r'saemidori', r'sae midori'], 'okumidori': [r'okumidori'], 'asatsuyu': [r'asatsuyu'], 'samidori': [r'samidori'], 'gokou': [r'gokou', r'gokoh'], 'tsuyuhikari': [r'tsuyuhikari'], 'kanayamidori': [r'kanayamidori'], 'okuyutaka': [r'okuyutaka'], 'asanoka': [r'asanoka'], 'sayamakaori': [r'sayamakaori'], 'yamakai': [r'yamakai'], 'meiryoku': [r'meiryoku'], 'zairai': [r'zairai', r'native seedling'], 'koshun': [r'koshun'], 'harumidori': [r'harumidori'], 'fukumidori': [r'fukumidori'], 'benifuuki': [r'benifuuki', r'benifuki'], } HARVEST = { 'first flush': [r'first flush', r'ichibancha', r'first harvest', r'first picking'], 'shincha': [r'shincha', r'shin cha', r'new tea'], 'second flush':[r'second flush', r'nibancha', r'second harvest', r'second picking'], 'autumn/late': [r'autumn\w*', r'aki ?bancha', r'third flush', r'sanbancha', r'late harvest'], 'a year': [r'\b20[12]\d\b'], 'a month': [r'\b(?:april|may|june|july|august|september|october)\b'], } PROCESS = { # what does the seller say about how it was finished? 'shaded': [r'shade[- ]grown', r'shaded', r'kabuse\w*'], 'hand picked': [r'hand[- ]pick\w*', r'hand[- ]harvest\w*', r'temomi'], 'single origin':[r'single[- ]origin', r'single[- ]farm', r'single[- ]estate', r'single[- ]garden'], 'organic': [r'organic', r'jas'], 'harvest date': [r'harvest(?:ed)? (?:on|date)', r'picked (?:on|in) \w+ 20\d\d'], } CAFFEINE = { 'low caffeine': [r'low[- ]caffeine', r'lower in caffeine', r'less caffeine', r'low in caffeine'], 'caffeine free': [r'caffeine[- ]free', r'no caffeine', r'zero caffeine'], 'mg figure': [r'\d+\s*mg'], 'evening/sleep': [r'before bed', r'evening', r'nighttime', r'night ?time', r'bedtime', r'sleep'], 'kids/pregnancy':[r'children', r'kids', r'pregnan\w+'], } # Guard 11: synonyms are one descriptor, not two. Counting 'toasty' and 'toasted' separately, or # 'smoky' and 'smoke', splits one thing the seller said into two smaller numbers and understates # how common it is. Each family emits its first name once, so a listing is counted once per family. # Guard 11 (synonyms are one descriptor) applied to sencha's own vocabulary rather than hojicha's. # The families are chosen to match the words the AI Overview uses — "fresh, grassy, and vegetal, # with subtle notes of seaweed, pine, or a mild sweetness balanced by gentle astringency" — so that # its description can be counted against the shelf rather than paraphrased. FLAVOUR = { 'umami': ['umami', 'brothy', 'savoury', 'savory', 'broth'], 'grassy': ['grassy', 'grass'], 'vegetal': ['vegetal', 'vegetable', 'spinach', 'kale', 'greens'], 'seaweed': ['seaweed', 'marine', 'oceanic', 'nori', 'kombu', 'sea'], 'sweet': ['sweet', 'sweetness'], 'astringent': ['astringent', 'astringency', 'brisk'], 'bitter': ['bitter', 'bitterness'], 'floral': ['floral', 'flowery'], 'fresh': ['fresh', 'refreshing', 'crisp'], 'fruity': ['fruity', 'melon', 'citrus', 'peach', 'apple'], 'nutty': ['nutty', 'nut'], 'buttery': ['buttery', 'butter', 'creamy'], 'pine': ['pine', 'piney'], 'roasted': ['roasted', 'toasty', 'toasted'], 'mellow': ['mellow', 'smooth', 'round'], } # Title only, for the same reason: "no sugar added" in a description is not a sweetened product. SWEET = [r'latte mix', r'sweet', r'sweetened', r'sugar', r'powdered drink mix'] # Guard 7 in practice: the seller's own product_type decides what is not tea. Reading these off # the title instead lets a tote bag called "Hojicha Hanko Tote" into a tea price distribution. # Guard 18: "sencha cup" is a teacup. Senchawan is a standard vessel size, so ceramics shops name # dozens of cups after the tea, and Tezumi files them under a product_type that is itself the word # for a teacup — `Yunomi (Teacups)`. Reading only the title let 31 pieces of pottery, one ceramic # hand mill and one empty packaging bag into a tea dataset. Same lesson as guard 7: the seller's # category field decides, and the title only fills in where the field is empty. EXCLUDE_TYPES = {'merchandise', 'class', 'chocolate', 'kombucha', 'subscription', 'apparel', 'accessories', 'teaware', 'gift card', 'coffee', 'yunomi (teacups)', 'kitchen tools & utensils'} EXCLUDE_TYPE_PAT = [r'kitchenware', r'teaware', r'teacup', r'ceramic', r'pottery'] # Guard 19: a listing whose own category names a different tea is that tea, whatever its title says. # `Green Tea - Hojicha` on a listing titled "... Hojicha (Dark Roasted Sencha)" is a hojicha. OTHER_TEA_TYPE = [r'hojicha', r'houjicha', r'matcha', r'genmaicha', r'gyokuro', r'oolong', r'black tea'] EXCLUDE_TITLE = [r'button', r'sticker', r'tote', r'kombucha', r'chocolate', r'\bclass\b', r'subscription', r'gift card', r'teapot', r'kyusu', r'chasen', r'whisk', r'tumbler', r'canister', r'case of \d+', r'candy', r'konpeito', r'cookie', # guard 18, title side, for the shops that leave product_type empty r'sencha cups?', r'senchawan', r'tea cups?', r'hand mill', r'nylon-poly bag', r'tableware', r'\bkiln\b', r'pottery'] # Anything added to the leaf: a flavoured hojicha is not a plain one, and its price is not # comparable. Kept and flagged rather than dropped (guard 5). # Read from the title only. A tea that has something added says so in its name; a description # that says "notes of vanilla" is describing a flavour, not listing an ingredient, and counting # the second as the first turned seven plain hojichas into flavoured ones on the first pass. FLAVOURED = [r'cacao', r'cinnamon', r'chamomile', r'yuzu', r'sansho', r'ginger', r'genmai\w*', r'tea flower', r'flower blend', r'whisky', r'whiskey', r'cedar', r'pepper', r'rose', r'citrus', r'chai', r'vanilla', r'oak barrel', r'kombucha', r'hibiscus', r'mint', r'lemon', r'peach', r'sakura'] # What the seller tells you to do with it. Temperatures appear in both scales and sellers do not # agree with each other, so both are captured rather than one being assumed. # Guard 12: a hojicha description contains two kinds of temperature, and only one of them is the # water. "roasted ... at 200C degrees for about two hours" is the roaster, not the kettle. The test # is physical rather than lexical: water brews between 50 and 100 C. A lexical test is not usable # here because 'roasted' appears in 121 of the 143 descriptions and would discard real brewing # temperatures that merely sit near it. BREW_RANGE = {'temp_c': (50, 100), 'temp_f': (120, 212)} BREW = { 'temp_c': r'(\d{2,3})\s*°?\s*C\b', 'temp_f': r'(\d{2,3})\s*°?\s*F\b', 'grams': r'(\d+(?:\.\d+)?)\s*(?:g|grams?)\b\s*(?:of\s+)?(?:tea|leaves|leaf|powder)', 'ml': r'(\d{2,4})\s*(?:ml|milliliters?)\b', 'seconds': r'(\d+)\s*(?:seconds?|sec)\b', 'minutes': r'(\d+)\s*(?:minutes?|mins?)\b', } LATTE = [r'latte', r'milk', r'oat milk', r'steamed milk', r'froth\w*'] # Where the leaf was grown. Prefecture and the few growing districts sold as origins in their own # right (Uji straddles Kyoto; Yame is in Fukuoka; Sayama is in Saitama), so a listing can hit both # a district and its prefecture — the columns record the district and the prefecture separately # rather than summing them. REGION = { 'kyoto': [r'kyoto'], 'uji': [r'uji'], 'shizuoka': [r'shizuoka'], 'kagoshima': [r'kagoshima'], 'fukuoka': [r'fukuoka'], 'yame': [r'yame'], 'saitama': [r'saitama'], 'sayama': [r'sayama'], 'mie': [r'mie'], 'nara': [r'nara'], 'kumamoto': [r'kumamoto'], 'miyazaki': [r'miyazaki'], 'shiga': [r'shiga'], 'aichi': [r'aichi'], 'nagasaki': [r'nagasaki'], 'kochi': [r'kochi'], 'shimane': [r'shimane'], 'gifu': [r'gifu'], 'wakayama': [r'wakayama'], 'niigata': [r'niigata'], 'ibaraki': [r'ibaraki'], 'okinawa': [r'okinawa'], 'tokushima': [r'tokushima'], 'saga': [r'saga'], } MULT = {'kg':1000, 'g':1, 'oz':28.3495, 'lb':453.592, 'gram':1, 'grams':1, 'ounce':28.3495, 'ounces':28.3495} def get(url, tries=4, as_json=True): for a in range(tries): try: r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=40, context=CTX) raw = r.read().decode('utf-8', 'ignore') return json.loads(raw) if as_json else raw except Exception as e: if a == tries - 1: print(f' FETCH FAILED after {tries} tries: {url[:80]} {e}') return None time.sleep(2 * (a + 1)) def shop_currency(base): """Guard 1: read the shop's own currency rather than assuming dollars.""" h = get(base + '/', as_json=False) if not h: return None for pat in (r'Shopify\.currency\s*=\s*\{"active":"(\w{3})"', r'"currency"\s*:\s*"(\w{3})"'): m = re.search(pat, h) if m: return m.group(1) return None def grams(s): """Guard 2 and 3: full unit spellings, and all three ways a pack size gets written.""" s = (s or '').lower().replace(',', '') u = r'(kg|g|grams?|oz|ounces?|lb)' m = re.search(r'(\d+)\s*[x×]\s*(\d+\.?\d*)\s*' + u + r'\b', s) # 10 x 100g if m: return float(m.group(1)) * float(m.group(2)) * MULT[m.group(3)] m = re.search(r'(\d+\.?\d*)\s*' + u + r'\b\s*[x×]\s*(\d+)', s) # 1 oz x 6 if m: return float(m.group(1)) * MULT[m.group(2)] * float(m.group(3)) m = re.search(r'(\d+)\s*(?:boxes|bags|packs|tins|units)\b[^)]*?(\d+\.?\d*)\s*' + u + r'\b', s) if m: return float(m.group(1)) * float(m.group(2)) * MULT[m.group(3)] m = re.search(r'(\d+\.?\d*)\s*' + u + r'\b', s) # plain if m: return float(m.group(1)) * MULT[m.group(2)] return None def txt(h): return re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', h or ''))).strip() NEGATION = re.compile(r'(?:\bnot|\bnon[- ]?|\brather than|\binstead of|\bunlike)\s*$', re.I) def hit(blob, pats): """Guard 4 and 10: phrases as phrases, single words with boundaries, and a match discarded when the words just before it deny it.""" for p in pats: for m in re.finditer(r'(?5} products, {len(cand):>3} match sencha') for p in cand: title = p['title'] body = txt(p.get('body_html')) blob = (title + ' ' + body).lower() ptype = (p.get('product_type') or '').strip().lower() if (ptype in EXCLUDE_TYPES or any(re.search(x, ptype, re.I) for x in EXCLUDE_TYPE_PAT) or any(re.search(x, ptype, re.I) for x in OTHER_TEA_TYPE) or hit(title.lower(), EXCLUDE_TITLE) or title.strip().lower() in OTHER_TEA_CALLS): skipped.append((shop, title, p.get('product_type') or '(no type)')) continue r0 = {'shop': shop, 'cur': cur or '', 'title': title, 'product_type': p.get('product_type') or '', 'tags': ';'.join(p.get('tags') or []), 'body_chars': len(body)} # Guards 5, 13 and 14: a set is flagged, not silently counted as plain hojicha — but the # brand name is removed first, and a title naming another tea is only a candidate. btitle = title.lower() for bp in BRAND_WORDS.get(shop, []): btitle = re.sub(bp, ' ', btitle, flags=re.I) other = [t for t in TEA_NAMES if re.search(r'(? {path}') uniq = {(r['shop'], r['title']): r for r in rows} print(f'\n=== {len(uniq)} distinct listings across {len({r["shop"] for r in rows})} shops ===') def share(label, col): n = sum(1 for r in uniq.values() if r[col]) print(f' {label:<20}{n:>4} / {len(uniq)} ({n/len(uniq)*100:>4.1f}%)') print("\nSteaming — the AI Overview's three grades, measured against the shelf:") for k in STEAM: share(k, 'steam_' + k.replace(' ', '_')) share('-- ANY named grade', 'steam_graded') print('\nCultivar named:') cc = collections.Counter(t for r in uniq.values() for t in r['cultivar'].split(';') if t) for k, v in cc.most_common(12): print(f' {k:<20}{v:>4}') share('-- ANY cultivar', 'cultivar_named') print('\nHarvest:') for k in HARVEST: share(k, 'harv_' + k.replace(' ', '_').replace('/', '_')) print('\nProcess claims:') for k in PROCESS: share(k, 'proc_' + k.replace(' ', '_')) print('\nCaffeine claims:') for k in CAFFEINE: n = sum(1 for r in uniq.values() if r['caf_' + k.replace(' ', '_').replace('/', '_')]) print(f' {k:<14}{n:>4} / {len(uniq)} ({n/len(uniq)*100:>4.1f}%)') if set_review: print(f'\n=== titles naming another tea, called BY HAND as not-a-set ({len(set_review)}) ===') print(' check each one; anything genuinely sold as several teas belongs in SET_CALLS') for sh, ti, ot in set_review: print(f' [{sh}] {ti[:78]} <- {ot}') named_reg = sum(1 for r in uniq.values() if r['region_named']) print(f'\nRegion named: {named_reg}/{len(uniq)} ({named_reg/len(uniq)*100:.1f}%)') rc = collections.Counter(t for r in uniq.values() for t in r['region'].split(';') if t) for k, v in rc.most_common(12): print(f' {k:<12} {v:>4}') print('\nFlavour vocabulary (top):') c = collections.Counter(t for r in uniq.values() for t in r['flavour_terms'].split(';') if t) for t, n in c.most_common(14): print(f' {t:<12}{n:>4} ({n/len(uniq)*100:>4.1f}%)')