#!/usr/bin/env python3 """Six Japanese teas, one ruler. Used in article 19. Google's AI Overview for `japanese green tea` names five types — sencha, matcha, gyokuro, genmaicha, hojicha — gives each a one-line flavour description, and makes one claim about caffeine. It gives no prices. This script collects every listing of all six types (kukicha is added because the AI Overview names it in the caffeine claim) from the same shops on the same day, so that the types can be compared on a single basis. ⚠️ **Why a fresh collection rather than joining the existing datasets.** This site has already priced each of these teas, but on different days with different exclusion rules. Joining those files would reproduce the mistake we made once already, where our own price index said $0.33 a gram for hojicha and an article said $0.42 because one included flavoured teas and the other did not. Every number in article 19 comes from this one run. Guards 1-19 are inherited from hojicha_disclosure.py and sencha_disclosure.py unchanged, because the failure modes are the same. The short version of each: 1. Currency is read from the shop, not assumed. 2. Units are matched spelled out too. 3. Multipacks are written three ways. 4. Word boundaries; phrases as phrases. 5. A multi-tea set is not one tea's price. 6. Sweetened mixes are not tea. 7. Classification comes from the seller's own field. 8. A fetch failure is not a zero. 9. Terms are counted in the seller's own text. 10. A negation is not a claim. 11. Synonyms are one descriptor, not two. 12. Roasting and brewing temperatures differ. 13. A brand name is not an ingredient. 14. A tea word in a title is a candidate. 15. Compound names are handled explicitly. 16. Form comes from the category field. 17. Steaming grade is a family, not a word. 18. "Sencha cup" is a teacup. 19. A listing whose category names another tea is that tea. 20. **One tea per listing.** A listing can match more than one tea name — a genmaicha is green tea plus rice, a hojicha can be roasted sencha, a matcha-iri genmaicha is two teas in one bag. The type is decided in a fixed order of specificity (matcha-blends first, then genmaicha, then hojicha, then the shaded and stem teas, then plain sencha last) rather than by whichever name the regex happens to hit first, and every listing carries the runner-up so the call can be checked. """ import urllib.request, json, ssl, re, csv, html, time, collections UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE # Guard 13: the brand name is not an ingredient. "rocky's matcha Houjicha 100g" is plain hojicha # sold by a shop with matcha in its name, and matching tea words against the raw title flagged all # four of Rocky's hojicha as blends and dropped them out of the price medians. Brand wording is # removed from the title before any tea name is read off it. BRAND_WORDS = { 'rockys': [r"rocky'?[’']?s matcha", r'rockys matcha'], 'ujido': [r'ujido'], 'jadeleaf': [r'jade leaf matcha', r'jade leaf'], 'matchaeologist': [r'matchaeologist'], 'ippodo_us': [r'ippodo'], 'kettl': [r'kettl'], 'rishi': [r'rishi'], 'matchaful': [r'matchaful'], 'hojicha_co': [r'hojicha co\.?'], 'yunomi': [r'yunomi'], 'mizuba': [r'mizuba'], 'nami': [r'nami matcha', r'\bnami\b'], 'arteao': [r'art of tea'], # 'matchabot' is a product line, not a tea word, but stripping it removes the only mention of # matcha from eleven listings that are matcha. It is replaced rather than deleted. 'moongoat': [r'moon ?goat'], 'tezumi': [r'tezumi'], } # Guard 14: a set is one price covering several different teas, and no feed field says so. Titles # naming two or more teas are candidates only; each is decided by hand and the calls are published, # because "Hojicha (Dark Roasted Sencha)" names its base leaf and "Matcha & Hojicha Duo" is two # products in a box, and nothing in the wording separates them. # Guard 20: one tea per listing, decided in a fixed order of specificity. A genmaicha is green tea # with rice in it and will also match 'sencha' or 'bancha'; a hojicha can be roasted sencha; a # matcha-iri genmaicha is two teas in one bag. Whichever name a regex hits first is not an answer, # so the order below is the answer, and the runner-up is recorded next to every call. TEA_ORDER = [ # genmaimatcha / matcha-iri genmaicha is genmaicha with matcha dusted on it, so it is filed # with genmaicha rather than matcha — the rice is the thing that defines it. ('genmaicha', [r'genmaicha', r'genmai ?cha', r'genmai ?matcha', r'matcha ?iri ?genmaicha', r'玄米茶', r'玄米抹茶']), ('hojicha', [r'ho[uj]?jicha', r'houjicha', r'h[ōo]jicha', r'ほうじ茶', r'焙じ茶']), ('gyokuro', [r'gyokuro', r'玉露']), ('kukicha', [r'kukicha', r'kuki ?cha', r'karigane', r'boucha', r'b[ōo]cha', r'shiraore', r'stem tea', r'茎茶', r'雁ヶ音', r'白折']), # Guard 15 again: the word sits inside compounds. toubancha (winter), harubancha (spring), # akibancha (autumn), natsubancha (summer) and iribancha are all bancha, and a boundary match # finds none of them. ichibancha/nibancha/sanbancha number a harvest and stay out. ('bancha', [r'bancha', r'iri ?bancha', r'kyo ?bancha', r'tou ?bancha', r'haru ?bancha', r'aki ?bancha', r'natsu ?bancha', r'番茶']), ('sencha', [r'sencha', r'煎茶']), # Matcha goes last on purpose. A kukicha "infused with matcha" or a karigane "with Uji matcha" # is a stem tea with matcha dusted on it, and putting matcha earlier filed four of them — the # cheapest four in the Japanese matcha set — as matcha. ('matcha', [r'matcha', r'抹茶']), ] MATCH_ANY = r'sencha|煎茶|matcha|抹茶|gyokuro|玉露|genmaicha|玄米茶|ho[uj]?jicha|houjicha|h[ōo]jicha|ほうじ茶|kukicha|karigane|boucha|茎茶|bancha|番茶' TEA_NAMES = [r'genmaicha', r'gyokuro', r'kukicha', r'bancha', r'black tea', r'matcha', r'ho[ju]?jicha', r'houjicha', r'oolong', r'sencha'] SET_CALLS = { # multi-tea sets, decided by hand from the candidate list printed at the end 'yunomi pyramids - japanese tea bags - fukamushicha, genmaicha, hojicha': True, } OTHER_TEA_CALLS = set() SHOPS = [('rockys','https://www.rockysmatcha.com'), ('ujido','https://ujido.com'), ('jadeleaf','https://jadeleafmatcha.com'), ('matchaeologist','https://matchaeologist.com'), ('ippodo_us','https://ippodotea.com'), ('kettl','https://kettl.co'), ('rishi','https://rishi-tea.com'), ('matchaful','https://matchaful.com'), ('hojicha_co','https://hojichaco.com'), ('yunomi','https://yunomi.life'), ('mizuba','https://mizubatea.com'), ('nami','https://namimatcha.com'), ('arteao','https://arteao.com'), ('moongoat','https://moongoat.com'), ('tezumi','https://tezumi.com'), # Guard 23: a generalist retailer that stocks all seven types, added after an # evaluation pointed out that five of the fourteen US sellers were matcha # specialists -- and that this shop publishes per-ounce unit prices, which made # the claim that nobody publishes prices false. ('artfultea','https://artfultea.com')] # Guard 4: phrases are matched as phrases; single words get word boundaries. # Guard 17: the steaming grade is the claim under test. Sellers write the same thing three ways — # the Japanese word, the word with -cha or -sencha attached, and the English gloss — so each grade # is one family. Guard 15 applies here too: `ichibancha` contains `bancha` and numbers a harvest, # so it is read as first flush below and never as a grade. STEAM = { 'asamushi': [r'asamushi\w*', r'asa mushi', r'light[- ]steam\w*', r'lightly steamed', r'shallow[- ]steam\w*'], 'chuumushi': [r'chuumushi\w*', r'chumushi\w*', r'chu mushi', r'medium[- ]steam\w*', r'standard steam\w*'], 'fukamushi': [r'fukamushi\w*', r'fuka mushi', r'deep[- ]steam\w*', r'deeply steamed', r'long[- ]steam\w*'], 'steamed at all': [r'steam\w*'], # the process itself, however vaguely stated } # What plant, and when it was picked. The site already has an article on cultivars, so this is the # axis a reader can act on. CULTIVAR = { 'yabukita': [r'yabukita'], 'saemidori': [r'saemidori', r'sae midori'], 'okumidori': [r'okumidori'], 'asatsuyu': [r'asatsuyu'], 'samidori': [r'samidori'], 'gokou': [r'gokou', r'gokoh'], 'tsuyuhikari': [r'tsuyuhikari'], 'kanayamidori': [r'kanayamidori'], 'okuyutaka': [r'okuyutaka'], 'asanoka': [r'asanoka'], 'sayamakaori': [r'sayamakaori'], 'yamakai': [r'yamakai'], 'meiryoku': [r'meiryoku'], 'zairai': [r'zairai', r'native seedling'], 'koshun': [r'koshun'], 'harumidori': [r'harumidori'], 'fukumidori': [r'fukumidori'], 'benifuuki': [r'benifuuki', r'benifuki'], } HARVEST = { 'first flush': [r'first flush', r'ichibancha', r'first harvest', r'first picking'], 'shincha': [r'shincha', r'shin cha', r'new tea'], 'second flush':[r'second flush', r'nibancha', r'second harvest', r'second picking'], 'autumn/late': [r'autumn\w*', r'aki ?bancha', r'third flush', r'sanbancha', r'late harvest'], 'a year': [r'\b20[12]\d\b'], 'a month': [r'\b(?:april|may|june|july|august|september|october)\b'], } PROCESS = { # what does the seller say about how it was finished? 'shaded': [r'shade[- ]grown', r'shaded', r'kabuse\w*'], 'hand picked': [r'hand[- ]pick\w*', r'hand[- ]harvest\w*', r'temomi'], 'single origin':[r'single[- ]origin', r'single[- ]farm', r'single[- ]estate', r'single[- ]garden'], 'organic': [r'organic', r'jas'], 'harvest date': [r'harvest(?:ed)? (?:on|date)', r'picked (?:on|in) \w+ 20\d\d'], } CAFFEINE = { 'low caffeine': [r'low[- ]caffeine', r'lower in caffeine', r'less caffeine', r'low in caffeine'], 'caffeine free': [r'caffeine[- ]free', r'no caffeine', r'zero caffeine'], 'mg figure': [r'\d+\s*mg'], 'evening/sleep': [r'before bed', r'evening', r'nighttime', r'night ?time', r'bedtime', r'sleep'], 'kids/pregnancy':[r'children', r'kids', r'pregnan\w+'], } # Guard 11: synonyms are one descriptor, not two. Counting 'toasty' and 'toasted' separately, or # 'smoky' and 'smoke', splits one thing the seller said into two smaller numbers and understates # how common it is. Each family emits its first name once, so a listing is counted once per family. # Guard 11 (synonyms are one descriptor) applied to sencha's own vocabulary rather than hojicha's. # The families are chosen to match the words the AI Overview uses — "fresh, grassy, and vegetal, # with subtle notes of seaweed, pine, or a mild sweetness balanced by gentle astringency" — so that # its description can be counted against the shelf rather than paraphrased. # Guard 11 across all six teas at once. The families must cover the vocabulary the AI Overview # uses for every type it names — "grassy" and "vegetal" for sencha, "creamy" for matcha, "savory, # brothy, oceanic" for gyokuro, "nutty" for genmaicha, "caramel or cocoa" for hojicha — or the # comparison is between its words and a list that cannot contain them. The sencha-only version of # this list had no chocolate family, which returned 0 of 139 hojicha listings for "cocoa" when the # hojicha study, using a list that had one, found 13 of 143. FLAVOUR = { 'umami': ['umami', 'brothy', 'savoury', 'savory', 'broth'], 'grassy': ['grassy', 'grass'], 'vegetal': ['vegetal', 'vegetable', 'spinach', 'kale', 'greens'], 'seaweed': ['seaweed', 'marine', 'oceanic', 'nori', 'kombu', 'sea'], 'sweet': ['sweet', 'sweetness'], 'astringent': ['astringent', 'astringency', 'brisk'], 'bitter': ['bitter', 'bitterness'], 'floral': ['floral', 'flowery'], 'fresh': ['fresh', 'refreshing', 'crisp'], 'fruity': ['fruity', 'melon', 'citrus', 'peach', 'apple'], 'nutty': ['nutty', 'nut', 'chestnut', 'hazelnut', 'almond'], 'creamy': ['creamy', 'buttery', 'butter', 'milky', 'velvety'], 'pine': ['pine', 'piney'], 'roasted': ['roasted', 'toasty', 'toasted'], 'smoky': ['smoky', 'smoke'], 'chocolate': ['chocolate', 'cocoa', 'cacao'], 'caramel': ['caramel', 'toffee', 'butterscotch'], 'woody': ['woody', 'wood'], 'grain': ['grain', 'grainy', 'malt', 'malty', 'biscuit', 'barley', 'popcorn', 'rice'], 'mellow': ['mellow', 'smooth', 'round'], 'honey': ['honey'], 'vanilla': ['vanilla'], } # Title only, for the same reason: "no sugar added" in a description is not a sweetened product. SWEET = [r'latte mix', r'sweet', r'sweetened', r'sugar', r'powdered drink mix', r'lemonade', r'drink mix', r'energy', r'granola', r'smoothie', r'instant'] # Guard 7 in practice: the seller's own product_type decides what is not tea. Reading these off # the title instead lets a tote bag called "Hojicha Hanko Tote" into a tea price distribution. # Guard 18: "sencha cup" is a teacup. Senchawan is a standard vessel size, so ceramics shops name # dozens of cups after the tea, and Tezumi files them under a product_type that is itself the word # for a teacup — `Yunomi (Teacups)`. Reading only the title let 31 pieces of pottery, one ceramic # hand mill and one empty packaging bag into a tea dataset. Same lesson as guard 7: the seller's # category field decides, and the title only fills in where the field is empty. EXCLUDE_TYPES = {'merchandise', 'class', 'chocolate', 'kombucha', 'subscription', 'apparel', 'accessories', 'teaware', 'gift card', 'coffee', 'yunomi (teacups)', 'kitchen tools & utensils', 'hat', 'car tag', 'wear (merch)', 'cups', 'apparel & accessories'} # Guard 18, third occurrence. A matcha bowl is not matcha. Tezumi files 610 pieces of pottery # under the product_type `Chawan (Matcha Bowls)`, another 40 under `Matcha Bowl`, 30 tea caddies # under `Natsume` and 7 under `Tea Caddies and Canisters`. Reading only for 'teacup' and # 'ceramic' let all 687 of them into a tea dataset and inflated the matcha share of the American # shelf from a third to five-sixths. The lesson each time is the same: the vessel is named after # the drink, so the seller's category field has to be read for the vessel word too. EXCLUDE_TYPE_PAT = [r'kitchenware', r'teaware', r'teacup', r'ceramic', r'pottery', r'merch', r'\bwear\b', r'\bhat\b', r'apparel', r'\btag\b', r'\bbowls?\b', r'chawan', r'natsume', r'caddies', r'caddy', r'canister', r'tea ?sets?', r'\btrays?\b', r'\bwhisk', r'chasen', r'chashaku', r'scoop'] # Guard 19, six-tea variant. In the sencha study this list excluded any listing whose category # named a different tea. Here five of those names ARE the study, so excluding them would delete # almost the whole dataset — it did, on the first run: 1,411 listings, including every matcha. # What is left is the teas outside the six, which still need excluding. OTHER_TEA_TYPE = [r'oolong', r'black tea', r'white tea', r'pu-?erh', r'herbal', r'rooibos', r'chai'] # The seller's category is instead used as evidence FOR the type, checked against the title call. TYPE_HINT = [('genmaicha', r'genmaicha'), ('hojicha', r'ho[uj]?jicha|houjicha'), ('matcha', r'matcha'), ('gyokuro', r'gyokuro'), ('kukicha', r'kukicha|karigane'), ('bancha', r'bancha'), ('sencha', r'sencha')] EXCLUDE_TITLE = [r'button', r'sticker', r'tote', r'kombucha', r'chocolate', r'\bclass\b', r'subscription', r'gift card', r'teapot', r'kyusu', r'chasen', r'whisk', r'tumbler', r'canister', r'case of \d+', r'candy', r'konpeito', r'cookie', # guard 18, title side, for the shops that leave product_type empty r'sencha cups?', r'senchawan', r'tea cups?', r'hand mill', r'nylon-poly bag', r'tableware', r'\bkiln\b', r'pottery', # guard 18, packaging side: a shop that sells tea also sells the bags to put it # in, and those are priced per bag against a gram weight that is the bag's # capacity, not its contents. One slipped into the sencha study; four into this one. r'aluminum foil', r'aluminium foil', r'gusset', r'\bpouch(?:es)? \(', r'\bbags?,', r'packaging', r'\blabel(?:s)?\b', r'gift ?wrap', r'furoshiki', r'\bspoon\b', r'matcha bowls?', r'chawan', r'natsume', r'\bcaddy\b', r'\bcaddies\b'] # Anything added to the leaf: a flavoured hojicha is not a plain one, and its price is not # comparable. Kept and flagged rather than dropped (guard 5). # Read from the title only. A tea that has something added says so in its name; a description # that says "notes of vanilla" is describing a flavour, not listing an ingredient, and counting # the second as the first turned seven plain hojichas into flavoured ones on the first pass. # genmai is not a flavouring here: roasted rice is what makes a genmaicha, and genmaicha is one of # the six types under study, so it is classified rather than excluded. FLAVOURED = [r'cacao', r'cinnamon', r'chamomile', r'yuzu', r'sansho', r'ginger', r'tea flower', r'flower blend', r'whisky', r'whiskey', r'cedar', r'pepper', r'rose', r'citrus', r'chai', r'vanilla', r'oak barrel', r'kombucha', r'hibiscus', r'mint', r'lemon', r'peach', r'sakura'] # What the seller tells you to do with it. Temperatures appear in both scales and sellers do not # agree with each other, so both are captured rather than one being assumed. # Guard 12: a hojicha description contains two kinds of temperature, and only one of them is the # water. "roasted ... at 200C degrees for about two hours" is the roaster, not the kettle. The test # is physical rather than lexical: water brews between 50 and 100 C. A lexical test is not usable # here because 'roasted' appears in 121 of the 143 descriptions and would discard real brewing # temperatures that merely sit near it. BREW_RANGE = {'temp_c': (50, 100), 'temp_f': (120, 212)} BREW = { 'temp_c': r'(\d{2,3})\s*°?\s*C\b', 'temp_f': r'(\d{2,3})\s*°?\s*F\b', 'grams': r'(\d+(?:\.\d+)?)\s*(?:g|grams?)\b\s*(?:of\s+)?(?:tea|leaves|leaf|powder)', 'ml': r'(\d{2,4})\s*(?:ml|milliliters?)\b', 'seconds': r'(\d+)\s*(?:seconds?|sec)\b', 'minutes': r'(\d+)\s*(?:minutes?|mins?)\b', } LATTE = [r'latte', r'milk', r'oat milk', r'steamed milk', r'froth\w*'] # Where the leaf was grown. Prefecture and the few growing districts sold as origins in their own # right (Uji straddles Kyoto; Yame is in Fukuoka; Sayama is in Saitama), so a listing can hit both # a district and its prefecture — the columns record the district and the prefecture separately # rather than summing them. REGION = { 'kyoto': [r'kyoto'], 'uji': [r'uji'], 'shizuoka': [r'shizuoka'], 'kagoshima': [r'kagoshima'], 'fukuoka': [r'fukuoka'], 'yame': [r'yame'], 'saitama': [r'saitama'], 'sayama': [r'sayama'], 'mie': [r'mie'], 'nara': [r'nara'], 'kumamoto': [r'kumamoto'], 'miyazaki': [r'miyazaki'], 'shiga': [r'shiga'], 'aichi': [r'aichi'], 'nagasaki': [r'nagasaki'], 'kochi': [r'kochi'], 'shimane': [r'shimane'], 'gifu': [r'gifu'], 'wakayama': [r'wakayama'], 'niigata': [r'niigata'], 'ibaraki': [r'ibaraki'], 'okinawa': [r'okinawa'], 'tokushima': [r'tokushima'], 'saga': [r'saga'], } MULT = {'kg':1000, 'g':1, 'oz':28.3495, 'lb':453.592, 'gram':1, 'grams':1, 'ounce':28.3495, 'ounces':28.3495} def get(url, tries=4, as_json=True): for a in range(tries): try: r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=40, context=CTX) raw = r.read().decode('utf-8', 'ignore') return json.loads(raw) if as_json else raw except Exception as e: if a == tries - 1: print(f' FETCH FAILED after {tries} tries: {url[:80]} {e}') return None time.sleep(2 * (a + 1)) def shop_currency(base): """Guard 1: read the shop's own currency rather than assuming dollars.""" h = get(base + '/', as_json=False) if not h: return None for pat in (r'Shopify\.currency\s*=\s*\{"active":"(\w{3})"', r'"currency"\s*:\s*"(\w{3})"'): m = re.search(pat, h) if m: return m.group(1) return None def grams(s): """Guard 2 and 3: full unit spellings, and the ways a pack size gets written. Guard 21: Yunomi writes multipacks as "10 cans (30g / 1 oz) - Total 300g". The old rules read the 30 and divided a ten-can price by one can's weight, putting four matcha listings out by a factor of ten — one of them at ¥7,000 a gram. When the seller states a total, the total wins. """ s = (s or '').lower().replace(',', '') u = r'(kg|g|grams?|oz|ounces?|lb)' m = re.search(r'total[^0-9]{0,6}(\d+(?:\.\d+)?)\s*' + u + r'\b', s) # "- Total 300g" if m: return float(m.group(1)) * MULT[m.group(2)] m = re.search(r'(\d+)\s*[x×]\s*(\d+\.?\d*)\s*' + u + r'\b', s) # 10 x 100g if m: return float(m.group(1)) * float(m.group(2)) * MULT[m.group(3)] m = re.search(r'(\d+\.?\d*)\s*' + u + r'\b[^0-9x×]{0,12}[x×]\s*(\d+)', s) # "1 oz x 6", "40g can x 10" if m: return float(m.group(1)) * MULT[m.group(2)] * float(m.group(3)) m = re.search(r'(\d+)\s*(?:boxes|bags|packs|tins|units)\b[^)]*?(\d+\.?\d*)\s*' + u + r'\b', s) if m: return float(m.group(1)) * float(m.group(2)) * MULT[m.group(3)] m = re.search(r'(\d+\.?\d*)\s*' + u + r'\b', s) # plain if m: return float(m.group(1)) * MULT[m.group(2)] return None def txt(h): return re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', h or ''))).strip() NEGATION = re.compile(r'(?:\bnot|\bnon[- ]?|\brather than|\binstead of|\bunlike)\s*$', re.I) def hit(blob, pats): """Guard 4 and 10: phrases as phrases, single words with boundaries, and a match discarded when the words just before it deny it.""" for p in pats: for m in re.finditer(r'(?5} products, {len(cand):>3} match a Japanese tea') for p in cand: title = p['title'] body = txt(p.get('body_html')) blob = (title + ' ' + body).lower() ptype = (p.get('product_type') or '').strip().lower() if (ptype in EXCLUDE_TYPES or any(re.search(x, ptype, re.I) for x in EXCLUDE_TYPE_PAT) or any(re.search(x, ptype, re.I) for x in OTHER_TEA_TYPE) or hit(title.lower(), EXCLUDE_TITLE) or title.strip().lower() in OTHER_TEA_CALLS): skipped.append((shop, title, p.get('product_type') or '(no type)')) continue r0 = {'shop': shop, 'cur': cur or '', 'title': title, 'product_type': p.get('product_type') or '', 'tags': ';'.join(p.get('tags') or []), 'body_chars': len(body)} # Guards 5, 13 and 14: a set is flagged, not silently counted as plain hojicha — but the # brand name is removed first, and a title naming another tea is only a candidate. btitle = title.lower() for bp in BRAND_WORDS.get(shop, []): btitle = re.sub(bp, ' ', btitle, flags=re.I) other = [t for t in TEA_NAMES if re.search(r'(?= 2) r0['is_blend'] = int(SET_CALLS.get(title.strip().lower(), False)) if r0['set_candidate'] and title.strip().lower() not in SET_CALLS: set_review.append((shop, title, ','.join(other))) # Guard 20: assign one type, in order of specificity, from the brand-stripped title # first and the description only as a fallback. The runner-up is kept so the call can # be disagreed with. # Guard 20: the title decides, and the seller's own category is the only fallback. # Reading the description as a fallback filed a stem tea "infused with matcha" and a # bancha whose page merely mentioned matcha as matcha listings. matched = [k for k, pats in TEA_ORDER if hit(btitle, pats)] if not matched: matched = [k for k, pats in TEA_ORDER if hit(ptype, pats)] r0['tea'] = matched[0] if matched else '' r0['tea_runners_up'] = ';'.join(matched[1:]) r0['tea_from_title'] = int(bool([k for k, pats in TEA_ORDER if hit(btitle, pats)])) # Guard 7: the seller's own category, recorded next to our call so the two can be # compared rather than assumed to agree. r0['tea_by_type_field'] = next((k for k, pat in TYPE_HINT if re.search(pat, ptype, re.I)), '') r0['type_field_agrees'] = int(bool(r0['tea_by_type_field']) and r0['tea_by_type_field'] == r0['tea']) # genmai rice is what makes a genmaicha, so it must not also flag it as flavoured r0['is_sweetened'] = int(hit(title.lower(), SWEET)) # Roasted matcha is the trade's name for hojicha powder, so a surviving 'matcha' after # the brand is stripped means the powder form, not the leaf. # Guard 16: form comes from the seller's own category field first. Rocky's calls its # houjicha product_type 'Matcha' and grinds it — "slow-roasted ... before being ground # into powder" — but the word powder is nowhere in the title, so reading the title # alone filed four ground teas as loose leaf and mixed them into the leaf medians. # Guard 16, sencha variant: the category field decides, then the title. The hojicha # script also read a surviving 'matcha' in the title as powder, because roasted matcha # is that trade's word for hojicha powder. Here it would misfire — a leaf sencha # "infused with matcha" is still leaf — so that clause is dropped. r0['is_powder'] = int(bool(re.search(r'(? {ex_path}') path = 'data/jgt_survey_2026-08-26.csv' cols = list(rows[0].keys()) with open(path, 'w', newline='', encoding='utf-8') as f: w = csv.DictWriter(f, fieldnames=cols, extrasaction='ignore') w.writeheader(); w.writerows(rows) print(f'\n{len(rows)} rows -> {path}') uniq = {(r['shop'], r['title']): r for r in rows} print(f'\n=== {len(uniq)} distinct listings across {len({r["shop"] for r in rows})} shops ===') def share(label, col, sub=None): v = sub if sub is not None else list(uniq.values()) n = sum(1 for r in v if r[col]) print(f' {label:<20}{n:>4} / {len(v)} ({n/len(v)*100:>4.1f}%)') print('\n=== the six types ===') byt = collections.Counter(r['tea'] for r in uniq.values()) for k, _ in TEA_ORDER: print(f' {k:<12}{byt.get(k,0):>4}') if byt.get(''): print(f" {'(none)':<12}{byt['']:>4}") print('\n=== price per gram, by type (plain only, retail 200g or under) ===') import statistics as _st for cur in ('USD', 'JPY'): print(f' -- {cur}') for k, _ in TEA_ORDER: per = {} for r in rows: if (r['tea'] != k or r['cur'] != cur or not r['per_g'] or not r['grams'] or r['is_blend'] or r['is_sweetened'] or r['is_flavoured']): continue if float(r['grams']) > 200: continue per.setdefault((r['shop'], r['title']), []).append(float(r['per_g'])) v = sorted(_st.median(x) for x in per.values()) if len(v) >= 3: print(f' {k:<12} n={len(v):<4} median {_st.median(v):>9.3f} {v[0]:.3f} - {v[-1]:.3f}') print('\n=== disclosure, by type ===') for k, _ in TEA_ORDER: sub = [r for r in uniq.values() if r['tea'] == k] if len(sub) < 5: continue print(f' -- {k} (n={len(sub)})') for lab, col in [('region', 'region_named'), ('cultivar', 'cultivar_named'), ('organic', 'proc_organic'), ('shaded', 'proc_shaded'), ('first flush', 'harv_first_flush'), ('low caffeine', 'caf_low_caffeine'), ('a caffeine mg', 'caf_mg_figure')]: share(' ' + lab, col, sub) if set_review: print(f'\n=== titles naming another tea, called BY HAND as not-a-set ({len(set_review)}) ===') print(' check each one; anything genuinely sold as several teas belongs in SET_CALLS') for sh, ti, ot in set_review: print(f' [{sh}] {ti[:78]} <- {ot}') named_reg = sum(1 for r in uniq.values() if r['region_named']) print(f'\nRegion named: {named_reg}/{len(uniq)} ({named_reg/len(uniq)*100:.1f}%)') rc = collections.Counter(t for r in uniq.values() for t in r['region'].split(';') if t) for k, v in rc.most_common(12): print(f' {k:<12} {v:>4}') print('\nFlavour vocabulary (top):') c = collections.Counter(t for r in uniq.values() for t in r['flavour_terms'].split(';') if t) for t, n in c.most_common(14): print(f' {t:<12}{n:>4} ({n/len(uniq)*100:>4.1f}%)')