#!/usr/bin/env python3 """What sellers say hojicha is, across every hojicha listing reachable from fifteen shops. Used in article 17. Hojicha is usually described as bancha — late-harvest, coarse leaf — roasted until brown. This script does not test that claim against a reference book. It tests it against the shelf: for each listing, which base tea does the seller name, what does it say about the roast, what does it claim about caffeine, and what does a gram cost. Known failure modes in this kind of listing data, and the guard applied to each. Read this before trusting any figure the script produces: 1. **Currency.** One of these shops prices in yen. Reading ¥1,800 as $1,800 puts a single listing two orders of magnitude out and moves every median it touches. The shop's own `Shopify.currency` is read before any price is used. 2. **Units spelled out.** `\\s*g\\b` does not match "100 Grams" — the g is followed by an r, so there is no word boundary. Sizes are matched against the full spellings too. 3. **Multiplication is written three ways**: "10 x 100g", "1 oz x 6 boxes" (count last) and "6 boxes (1 oz each)". Missing any one multiplies or divides a unit price by the pack count. 4. **Partial words.** Searching for `stem` inside running text also matches "system"; `kuki` matches "kukicha" but also any longer word containing it. Every term is matched with word boundaries, and compound terms are matched as phrases. 5. **A set is not the thing.** One price covering several different teas is not a hojicha price. But a tea word in the title does not make a set: it can be the brand (guard 13) or the base leaf the seller is naming. Titles mentioning another tea are candidates, each candidate is decided by hand and the calls are published, and sets are kept and flagged rather than dropped. 6. **Sweetened mixes are not tea.** A latte mix that is mostly sugar sits an order of magnitude below real leaf per gram. Flagged on its own column. 7. **Classification comes from the seller.** Shopify exposes `product_type`; it is recorded alongside the classification this script makes, so the two can be compared rather than assumed to agree. 8. **Retry, and never read a failure as a zero.** A shop that times out has not been shown to sell no hojicha. 10. **A negation is not a claim.** "Real Roasted Matcha, not Sencha" and "this kind of non-tencha made matcha" both put the word in the text while denying it. Matched terms are rejected when the words immediately before them negate: not, non-, rather than, instead of, unlike. 9. **Scope.** Terms are counted in the seller's own title and description. A term that appears only in a theme block outside the product feed is not captured; a sample of pages is checked by hand against the feed to see whether that matters for these shops. """ import urllib.request, json, ssl, re, csv, html, time, collections UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE # Guard 13: the brand name is not an ingredient. "rocky's matcha Houjicha 100g" is plain hojicha # sold by a shop with matcha in its name, and matching tea words against the raw title flagged all # four of Rocky's hojicha as blends and dropped them out of the price medians. Brand wording is # removed from the title before any tea name is read off it. BRAND_WORDS = { 'rockys': [r"rocky'?[’']?s matcha", r'rockys matcha'], 'ujido': [r'ujido'], 'jadeleaf': [r'jade leaf matcha', r'jade leaf'], 'matchaeologist': [r'matchaeologist'], 'ippodo_us': [r'ippodo'], 'kettl': [r'kettl'], 'rishi': [r'rishi'], 'matchaful': [r'matchaful'], 'hojicha_co': [r'hojicha co\.?'], 'yunomi': [r'yunomi'], 'mizuba': [r'mizuba'], 'nami': [r'nami matcha', r'\bnami\b'], 'arteao': [r'art of tea'], 'moongoat': [r'moon ?goat', r'matchabot'], 'tezumi': [r'tezumi'], } # Guard 14: a set is one price covering several different teas, and no feed field says so. Titles # naming two or more teas are candidates only; each is decided by hand and the calls are published, # because "Hojicha (Dark Roasted Sencha)" names its base leaf and "Matcha & Hojicha Duo" is two # products in a box, and nothing in the wording separates them. TEA_NAMES = [r'genmaicha', r'fukamushicha', r'gyokuro', r'kukicha', r'bancha', r'black tea', r'matcha', r'sencha'] SET_CALLS = { # title -> True when the listing is several teas sold together 'yunomi walatte tasting set - japanese matcha / hojicha / genmaicha powders': True, 'yunomi pyramids - japanese tea bags - fukamushicha, genmaicha, hojicha': True, '#0007.k6 obubu tea culinary powder sampler: matcha, sencha, genmaicha, hojicha, black tea powder (7 types)': True, 'daily matcha and hojicha set': True, 'matcha & hojicha duo': True, } SHOPS = [('rockys','https://www.rockysmatcha.com'), ('ujido','https://ujido.com'), ('jadeleaf','https://jadeleafmatcha.com'), ('matchaeologist','https://matchaeologist.com'), ('ippodo_us','https://ippodotea.com'), ('kettl','https://kettl.co'), ('rishi','https://rishi-tea.com'), ('matchaful','https://matchaful.com'), ('hojicha_co','https://hojichaco.com'), ('yunomi','https://yunomi.life'), ('mizuba','https://mizubatea.com'), ('nami','https://namimatcha.com'), ('arteao','https://arteao.com'), ('moongoat','https://moongoat.com'), ('tezumi','https://tezumi.com')] # Guard 4: phrases are matched as phrases; single words get word boundaries. BASE_TEA = { # what leaf does the seller say went into the roaster? # Guard 15: iribancha and kyobancha ARE bancha — the word sits inside a compound, so a # boundary match misses them. ichibancha/nibancha/sanbancha are NOT: those number a harvest, # and they are already counted as first flush or late harvest, so they stay excluded. 'bancha': [r'bancha', r'iribancha', r'iri bancha', r'kyobancha', r'kyo bancha'], 'sencha': [r'sencha'], 'kukicha': [r'kukicha', r'kuki cha'], 'karigane': [r'karigane'], 'stems': [r'stems?', r'twigs?'], 'tencha': [r'tencha'], 'gyokuro': [r'gyokuro'], 'yanagi': [r'yanagi'], 'aracha': [r'aracha'], 'first flush': [r'first flush', r'ichibancha', r'first harvest'], 'late harvest': [r'late harvest', r'second flush', r'third flush', r'nibancha', r'sanbancha'], } ROAST = { # what does the seller say about the roasting itself? 'roast level': [r'light roast', r'medium roast', r'dark roast', r'deep roast', r'heavy roast'], 'charcoal': [r'charcoal'], 'direct fire': [r'direct[- ]fire[d]?', r'jika ?bi'], 'hoiro/drum': [r'hoiro', r'drum[- ]roast\w*', r'rotary'], 'roasted here': [r'roasted in[- ]house', r'house[- ]roasted', r'we roast', r'roasted by us'], 'roast date': [r'roast(?:ed)? (?:on|date)', r'roasted \d{4}'], } CAFFEINE = { 'low caffeine': [r'low[- ]caffeine', r'lower in caffeine', r'less caffeine', r'low in caffeine'], 'caffeine free': [r'caffeine[- ]free', r'no caffeine', r'zero caffeine'], 'mg figure': [r'\d+\s*mg'], 'evening/sleep': [r'before bed', r'evening', r'nighttime', r'night ?time', r'bedtime', r'sleep'], 'kids/pregnancy':[r'children', r'kids', r'pregnan\w+'], } # Guard 11: synonyms are one descriptor, not two. Counting 'toasty' and 'toasted' separately, or # 'smoky' and 'smoke', splits one thing the seller said into two smaller numbers and understates # how common it is. Each family emits its first name once, so a listing is counted once per family. FLAVOUR = { 'roasted': ['roasted'], 'toasty': ['toasty', 'toasted'], 'nutty': ['nutty'], 'caramel': ['caramel'], 'chocolate': ['chocolate', 'cocoa'], 'smoky': ['smoky', 'smoke'], 'sweet': ['sweet'], 'honey': ['honey'], 'earthy': ['earthy'], 'woody': ['woody', 'wood'], 'grain': ['grain', 'grainy', 'malt', 'malty', 'biscuit', 'barley', 'popcorn'], 'coffee': ['coffee'], 'bitter': ['bitter', 'astringent'], 'creamy': ['creamy'], 'umami': ['umami'], 'vanilla': ['vanilla'], } # Title only, for the same reason: "no sugar added" in a description is not a sweetened product. SWEET = [r'latte mix', r'sweet', r'sweetened', r'sugar', r'powdered drink mix'] # Guard 7 in practice: the seller's own product_type decides what is not tea. Reading these off # the title instead lets a tote bag called "Hojicha Hanko Tote" into a tea price distribution. EXCLUDE_TYPES = {'merchandise', 'class', 'chocolate', 'kombucha', 'subscription', 'apparel', 'accessories', 'teaware', 'gift card', 'coffee'} EXCLUDE_TITLE = [r'button', r'sticker', r'tote', r'kombucha', r'chocolate', r'\bclass\b', r'subscription', r'gift card', r'teapot', r'kyusu', r'chasen', r'whisk', r'tumbler', r'canister', r'case of \d+', r'candy', r'konpeito', r'cookie'] # Anything added to the leaf: a flavoured hojicha is not a plain one, and its price is not # comparable. Kept and flagged rather than dropped (guard 5). # Read from the title only. A tea that has something added says so in its name; a description # that says "notes of vanilla" is describing a flavour, not listing an ingredient, and counting # the second as the first turned seven plain hojichas into flavoured ones on the first pass. FLAVOURED = [r'cacao', r'cinnamon', r'chamomile', r'yuzu', r'sansho', r'ginger', r'genmai\w*', r'whisky', r'whiskey', r'cedar', r'pepper', r'rose', r'citrus', r'chai', r'vanilla', r'oak barrel', r'kombucha', r'hibiscus', r'mint', r'lemon', r'peach', r'sakura'] # What the seller tells you to do with it. Temperatures appear in both scales and sellers do not # agree with each other, so both are captured rather than one being assumed. # Guard 12: a hojicha description contains two kinds of temperature, and only one of them is the # water. "roasted ... at 200C degrees for about two hours" is the roaster, not the kettle. The test # is physical rather than lexical: water brews between 50 and 100 C. A lexical test is not usable # here because 'roasted' appears in 121 of the 143 descriptions and would discard real brewing # temperatures that merely sit near it. BREW_RANGE = {'temp_c': (50, 100), 'temp_f': (120, 212)} BREW = { 'temp_c': r'(\d{2,3})\s*°?\s*C\b', 'temp_f': r'(\d{2,3})\s*°?\s*F\b', 'grams': r'(\d+(?:\.\d+)?)\s*(?:g|grams?)\b\s*(?:of\s+)?(?:tea|leaves|leaf|powder)', 'ml': r'(\d{2,4})\s*(?:ml|milliliters?)\b', 'seconds': r'(\d+)\s*(?:seconds?|sec)\b', 'minutes': r'(\d+)\s*(?:minutes?|mins?)\b', } LATTE = [r'latte', r'milk', r'oat milk', r'steamed milk', r'froth\w*'] # Where the leaf was grown. Prefecture and the few growing districts sold as origins in their own # right (Uji straddles Kyoto; Yame is in Fukuoka; Sayama is in Saitama), so a listing can hit both # a district and its prefecture — the columns record the district and the prefecture separately # rather than summing them. REGION = { 'kyoto': [r'kyoto'], 'uji': [r'uji'], 'shizuoka': [r'shizuoka'], 'kagoshima': [r'kagoshima'], 'fukuoka': [r'fukuoka'], 'yame': [r'yame'], 'saitama': [r'saitama'], 'sayama': [r'sayama'], 'mie': [r'mie'], 'nara': [r'nara'], 'kumamoto': [r'kumamoto'], 'miyazaki': [r'miyazaki'], 'shiga': [r'shiga'], 'aichi': [r'aichi'], 'nagasaki': [r'nagasaki'], 'kochi': [r'kochi'], 'shimane': [r'shimane'], 'gifu': [r'gifu'], 'wakayama': [r'wakayama'], 'niigata': [r'niigata'], 'ibaraki': [r'ibaraki'], 'okinawa': [r'okinawa'], 'tokushima': [r'tokushima'], 'saga': [r'saga'], } MULT = {'kg':1000, 'g':1, 'oz':28.3495, 'lb':453.592, 'gram':1, 'grams':1, 'ounce':28.3495, 'ounces':28.3495} def get(url, tries=4, as_json=True): for a in range(tries): try: r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=40, context=CTX) raw = r.read().decode('utf-8', 'ignore') return json.loads(raw) if as_json else raw except Exception as e: if a == tries - 1: print(f' FETCH FAILED after {tries} tries: {url[:80]} {e}') return None time.sleep(2 * (a + 1)) def shop_currency(base): """Guard 1: read the shop's own currency rather than assuming dollars.""" h = get(base + '/', as_json=False) if not h: return None for pat in (r'Shopify\.currency\s*=\s*\{"active":"(\w{3})"', r'"currency"\s*:\s*"(\w{3})"'): m = re.search(pat, h) if m: return m.group(1) return None def grams(s): """Guard 2 and 3: full unit spellings, and all three ways a pack size gets written.""" s = (s or '').lower().replace(',', '') u = r'(kg|g|grams?|oz|ounces?|lb)' m = re.search(r'(\d+)\s*[x×]\s*(\d+\.?\d*)\s*' + u + r'\b', s) # 10 x 100g if m: return float(m.group(1)) * float(m.group(2)) * MULT[m.group(3)] m = re.search(r'(\d+\.?\d*)\s*' + u + r'\b\s*[x×]\s*(\d+)', s) # 1 oz x 6 if m: return float(m.group(1)) * MULT[m.group(2)] * float(m.group(3)) m = re.search(r'(\d+)\s*(?:boxes|bags|packs|tins|units)\b[^)]*?(\d+\.?\d*)\s*' + u + r'\b', s) if m: return float(m.group(1)) * float(m.group(2)) * MULT[m.group(3)] m = re.search(r'(\d+\.?\d*)\s*' + u + r'\b', s) # plain if m: return float(m.group(1)) * MULT[m.group(2)] return None def txt(h): return re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', h or ''))).strip() NEGATION = re.compile(r'(?:\bnot|\bnon[- ]?|\brather than|\binstead of|\bunlike)\s*$', re.I) def hit(blob, pats): """Guard 4 and 10: phrases as phrases, single words with boundaries, and a match discarded when the words just before it deny it.""" for p in pats: for m in re.finditer(r'(?5} products, {len(cand):>3} match hojicha') for p in cand: title = p['title'] body = txt(p.get('body_html')) blob = (title + ' ' + body).lower() ptype = (p.get('product_type') or '').strip().lower() if ptype in EXCLUDE_TYPES or hit(title.lower(), EXCLUDE_TITLE): skipped.append((shop, title, p.get('product_type') or '(no type)')) continue r0 = {'shop': shop, 'cur': cur or '', 'title': title, 'product_type': p.get('product_type') or '', 'tags': ';'.join(p.get('tags') or []), 'body_chars': len(body)} # Guards 5, 13 and 14: a set is flagged, not silently counted as plain hojicha — but the # brand name is removed first, and a title naming another tea is only a candidate. btitle = title.lower() for bp in BRAND_WORDS.get(shop, []): btitle = re.sub(bp, ' ', btitle, flags=re.I) other = [t for t in TEA_NAMES if re.search(r'(? {path}') uniq = {(r['shop'], r['title']): r for r in rows} print(f'\n=== {len(uniq)} distinct listings across {len({r["shop"] for r in rows})} shops ===') print('\nBase tea named by the seller (share of listings):') for k in BASE_TEA: n = sum(1 for r in uniq.values() if r['base_' + k.replace(' ', '_')]) print(f' {k:<14}{n:>4} / {len(uniq)} ({n/len(uniq)*100:>4.1f}%)') named = sum(1 for r in uniq.values() if r['base_named']) print(f' -- any base tea named: {named}/{len(uniq)} ({named/len(uniq)*100:.1f}%)') print('\nRoast disclosure:') for k in ROAST: n = sum(1 for r in uniq.values() if r['roast_' + k.replace(' ', '_').replace('/', '_')]) print(f' {k:<14}{n:>4} / {len(uniq)} ({n/len(uniq)*100:>4.1f}%)') print('\nCaffeine claims:') for k in CAFFEINE: n = sum(1 for r in uniq.values() if r['caf_' + k.replace(' ', '_').replace('/', '_')]) print(f' {k:<14}{n:>4} / {len(uniq)} ({n/len(uniq)*100:>4.1f}%)') if set_review: print(f'\n=== titles naming another tea, called BY HAND as not-a-set ({len(set_review)}) ===') print(' check each one; anything genuinely sold as several teas belongs in SET_CALLS') for sh, ti, ot in set_review: print(f' [{sh}] {ti[:78]} <- {ot}') named_reg = sum(1 for r in uniq.values() if r['region_named']) print(f'\nRegion named: {named_reg}/{len(uniq)} ({named_reg/len(uniq)*100:.1f}%)') rc = collections.Counter(t for r in uniq.values() for t in r['region'].split(';') if t) for k, v in rc.most_common(12): print(f' {k:<12} {v:>4}') print('\nFlavour vocabulary (top):') c = collections.Counter(t for r in uniq.values() for t in r['flavour_terms'].split(';') if t) for t, n in c.most_common(14): print(f' {t:<12}{n:>4} ({n/len(uniq)*100:>4.1f}%)')