#!/usr/bin/env python3
"""Reads what Jade Leaf Matcha publishes about every product it sells, from the seller's own pages.
Used in article 23.
The question this answers is narrow: for each product Jade Leaf lists, which of the seller's own
shelves does it sit on, what does the seller say is in it, what does the seller disclose about
where the tea came from, and what does a gram cost.
Known failure modes in this kind of listing data, and the guard applied to each. Read this before
trusting any figure the script produces:
1. **A product feed is not the page.** `products.json` returns `body_html`, but this shop puts its
specification rows (grade, origin, caffeine) in a theme block and its ingredient and sugar
figures in a JSON-LD block. Neither appears in the feed. Every disclosure field here is read
from the product page, and the feed is used only for the variant and price list.
2. A Shopify variant's `grams` field is **shipping weight, not net weight**. Net weight is parsed
from the size the seller states in the variant name ("30g Tin", "5.3oz Pouch", "1lb Pouch"),
and the source of every weight is recorded in `weight_source`.
3. **Whether a product is matcha or a sweetened mix is decided by the seller's own shelves**
(`/collections/*`), never by words in the product name. "Matcha Latte Mix" and "Ceremonial
Matcha" both have "matcha" in the name; the shop files them on different shelves.
4. Where a term appears matters. A word in a product's name is the seller naming the product; the
same word in a category heading or an image alt attribute is not. Each place is counted in its
own column and never summed into one "mentions" figure.
5. Retry failed fetches. A page that fails to load is missing, never "the seller does not
disclose this". Failures are written to the CSV as a note, not dropped.
6. Bundles, sets and stick packs have no single net weight. They are collected and marked, and
excluded from per-gram statistics rather than silently dropped.
7. **A shop can run a sale.** The feed's `price` is what the shop charges today and
`compare_at_price` is the list price it charges when the sale ends. Both are recorded, and the
per-gram figures are computed from both, so a sale week cannot be reported as a price cut.
8. Multi-count packs ("10 Count", "7 Ct") are a count of sachets, not a weight. They are marked
and excluded from per-gram figures unless the seller states the gram weight of a sachet.
"""
import ssl, re, csv, json, time, html, urllib.request, collections
SHOP = 'www.jadeleafmatcha.com'
STAMP = '2026-08-27'
UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'}
CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
# The shop's own shelves, grouped by what the shop uses them to mean. Guard 3: a product's
# classification comes from these, not from its name.
#
# The shop keeps two kinds of shelf and they must not be mixed. `pure-matcha` and
# `cafe-style-matcha-mixes` are the shop's own split between powder that is only tea and powder
# that is tea plus sugar. `matcha` is a navigation shelf and carries products from both sides —
# including a sweetened latte mix — so counting it as evidence of purity puts mixes in the pure
# column. It is recorded but never used to classify.
PURE_SHELVES = {'pure-matcha', 'pure-matcha-a', 'pure-matcha-products', 'ceremonial',
'limited-edition-ceremonial-matcha'}
MIX_SHELVES = {'cafe-style-matcha-mixes', 'matcha-drink-mixes', 'flavored-matcha',
'premium-matcha-latte-infusions'}
NAV_SHELVES = {'matcha', 'shop-all', 'all-products', 'best-sellers', 'sale', 'new-collection'}
# Cultivar names checked against the whole product page, so that "this listing names no cultivar"
# means the page names none anywhere.
#
# A list written from memory reports the names it already knew and silently misses the rest: the
# first run of this script missed Hoshun, which this shop names on its best-selling culinary
# powder. So the list below is only a seed. The names this shop actually uses are harvested from
# its own "Tea Cultivars:" rows in a first pass and unioned in before anything is counted, and the
# harvested set is printed so the two can be compared.
CULTIVAR_SEED = ['yabukita', 'okumidori', 'samidori', 'saemidori', 'asahi', 'tsuyuhikari',
'kanayamidori', 'okuyutaka', 'asanoka', 'gokou', 'uji hikari', 'narino',
'sayamakaori', 'yamakai', 'meiryoku', 'seimei', 'kirari', 'okuharuka',
'goko', 'komakage', 'ujimidori', 'hoshun', 'fukumidori', 'harumidori']
TERMS = ['ceremonial', 'culinary', 'premium', 'first harvest', 'organic', 'latte', 'sweetened']
def fetch(url, tries=4):
for a in range(tries):
try:
r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=60, context=CTX)
return r.read().decode('utf-8', 'ignore')
except Exception as e:
if a == tries - 1:
print(f' FETCH FAILED after {tries} tries: {url[:90]} {e}')
return None
time.sleep(2 * (a + 1))
def jget(url):
t = fetch(url)
try:
return json.loads(t) if t else None
except Exception:
return None
def txt(s):
return re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', s or ''))).strip()
def spec_rows(page):
"""The seller's specification rows from the theme block (guard 1). Returns Key -> Value."""
out = {}
for m in re.finditer(r'
(.*?)
', page, re.S):
body = m.group(1)
body = re.sub(r'', ' ', body, flags=re.S) # icons carry inline CSS text
s = txt(body)
km = re.match(r'([A-Za-z][A-Za-z /&]{2,28}):\s*(.+)$', s)
if km:
out.setdefault(km.group(1).strip(), km.group(2).strip())
return out
def rte_labels(page):
"""Bold "Label:" lines inside the seller's rich-text description, e.g. "Tea Cultivars:"."""
out = {}
for m in re.finditer(r'<(?:strong|b)>\s*([A-Za-z][A-Za-z /&]{2,28}):\s*(?:strong|b)>(.{0,400}?)',
page, re.S):
out.setdefault(m.group(1).strip(), txt(m.group(2)))
return out
# The shop writes this row's label two ways. Reading only the plural form drops the products
# that name exactly one cultivar — which is the whole of the culinary and ingredient end of the
# range, so the miss lands entirely on one side of the comparison being made.
CULTIVAR_LABELS = ('Tea Cultivars', 'Tea Cultivar', 'Cultivars', 'Cultivar')
def cultivar_row(rte):
for k in CULTIVAR_LABELS:
if rte.get(k):
return rte[k]
return ''
def cultivar_row_label(rte):
for k in CULTIVAR_LABELS:
if rte.get(k):
return k
return ''
def servings_stated(variant_title, page):
"""How many servings the seller says a pack holds, taken from the variant name it is printed
in. Used only to convert a per-serving figure the seller publishes into a share of the pack;
never inferred when the seller does not state it."""
m = re.search(r'\(?\s*(\d+)\s*servings?\s*\)?', (variant_title or ''), re.I)
return int(m.group(1)) if m else ''
def qa_block(page):
"""The seller's own answers in the product Q&A block. These carry figures that appear nowhere
else on the page — the gram weight of one sachet, the caffeine in one packet, which products
the seller has and has not had tested. Taken as one block of text rather than parsed into
pairs, because the theme's markup for a question and for its answer is the same."""
flat = txt(re.sub(r'', ' ', page, flags=re.S))
i = flat.find('Q&A')
if i < 0:
return ''
end = flat.find('From The Jade Leaf Kitchen', i)
block = flat[i + 3:end if end > i else i + 2500].strip()
# Products with no questions still render the tab, and what follows it is the site-wide brand
# banner. Returning that as "the seller's answers" would invent disclosure that does not exist,
# so a block that opens with the banner is reported as empty.
if block.startswith('Jade Leaf Matcha Jade Leaf Matcha'):
return ''
return block
def jsonld_product(page):
"""The seller's own structured data: ingredient string, sugar per serving, tasting notes.
Neither the feed nor the visible page carries these as text."""
best = {}
for m in re.finditer(r'', page, re.S):
raw = m.group(1)
if '"ingredients"' not in raw and '"nutrition"' not in raw:
continue
try:
d = json.loads(raw)
except Exception:
# fall back to reading the two fields directly rather than discarding the block
d = {}
im = re.search(r'"ingredients"\s*:\s*"([^"]*)"', raw)
sm = re.search(r'"sugarContent"\s*:\s*"([^"]*)"', raw)
vm = re.search(r'"servingSize"\s*:\s*"([^"]*)"', raw)
if im: d['ingredients'] = im.group(1)
if sm or vm:
d['nutrition'] = {'sugarContent': sm.group(1) if sm else '',
'servingSize': vm.group(1) if vm else ''}
if isinstance(d, list):
d = next((x for x in d if isinstance(x, dict) and 'ingredients' in x), d[0] if d else {})
if not isinstance(d, dict):
continue
cand = {
'ingredients': d.get('ingredients', ''),
'sugar_per_serving': ((d.get('nutrition') or {}).get('sugarContent') or ''),
'serving_size': ((d.get('nutrition') or {}).get('servingSize') or ''),
'country_of_origin': d.get('countryOfOrigin', ''),
'rating_value': ((d.get('aggregateRating') or {}).get('ratingValue') or ''),
'rating_count': ((d.get('aggregateRating') or {}).get('ratingCount') or ''),
}
for p in (d.get('additionalProperty') or []):
if isinstance(p, dict) and p.get('name') == 'Tasting Notes':
cand['tasting_notes'] = p.get('value', '')
if cand['ingredients'] or cand['sugar_per_serving']:
best = cand
return best
# Guard 2 / guard 8: net weight is read from the size the seller states in the variant name.
# Ounce figures are the seller's own roundings where the seller also prints a gram figure.
OZ_TO_G = {'0.7': 20, '1.06': 30, '3.2': 90.7, '3.5': 100, '5.3': 150, '5.8': 164.4, '16': 453.6}
def net_grams(variant_title):
v = (variant_title or '').lower()
m = re.search(r'(\d+(?:\.\d+)?)\s*g\b', v)
if m:
return float(m.group(1)), f'seller-stated "{m.group(0)}"'
m = re.search(r'(\d+(?:\.\d+)?)\s*(?:oz|ounce)', v)
if m:
g = OZ_TO_G.get(m.group(1))
return (g, f'seller-stated {m.group(1)} oz') if g else ('', f'{m.group(1)} oz — no gram figure published')
if re.search(r'\b1\s*lb\b', v):
return 453.6, 'seller-stated 1 lb'
if re.search(r'\b(\d+)\s*(?:count|ct)\b', v):
return '', 'sachet count, not a weight (guard 8)'
return '', 'no size in variant name'
def count(text, term):
return len(re.findall(r'(? price)),
'net_grams': grams, 'weight_source': wsrc,
'per_gram_today': round(price / grams, 4) if (price and grams) else '',
'per_gram_list': round(listp / grams, 4) if (listp and grams) else '',
'shopify_grams_field': v.get('grams'), # kept only to show it disagrees
'sku': v.get('sku') or '', 'available': int(bool(v.get('available'))),
'grade': spec.get('Matcha Grade', ''), 'origin': spec.get('Product Origin', ''),
'used_for': spec.get('Used For', ''), 'caffeine': spec.get('Caffeine Level', ''),
'flavor_profile': spec.get('Flavor Profile', ''),
'cultivars_row': cultivar_row(rte),
'cultivars_row_label': cultivar_row_label(rte),
'ingredients': ld.get('ingredients', ''),
'sugar_per_serving': ld.get('sugar_per_serving', ''),
'serving_size': ld.get('serving_size', ''),
'country_of_origin': ld.get('country_of_origin', ''),
'rating_value': ld.get('rating_value', ''), 'rating_count': ld.get('rating_count', ''),
'tasting_notes': ld.get('tasting_notes', ''),
'spec_row_count': len(spec),
'cultivars_named_on_page': '; '.join(cultivars_on_page),
'seller_qa': qa,
'description': desc, 'note': '',
})
for t in TERMS: # guard 4: one column per place
k = t.replace(' ', '_')
rows[-1][f'name_{k}'] = count(title, t)
rows[-1][f'spec_{k}'] = count(' '.join(spec.values()) + ' ' + ' '.join(rte.values()), t)
rows[-1][f'desc_{k}'] = count(desc, t)
rows[-1][f'shelf_{k}'] = count(shelf_blob, t)
rows[-1][f'alt_{k}'] = count(alts, t)
pg = f"${price/grams:.3f}/g" if (price and grams) else '—'
print(f" {title[:40]:<42}{pg:<11}sugar={str(ld.get('sugar_per_serving','—'))[:6]:<8}"
f"grade={spec.get('Matcha Grade','—')[:22]:<24}origin={spec.get('Product Origin','—')[:22]:<24}"
f"pure={int(bool(on_pure))} mix={int(bool(on_mix))}")
time.sleep(0.3)
cols_out = list(rows[0].keys())
for r in rows:
for k in cols_out:
r.setdefault(k, '')
path = f'data/jadeleaf_listings_{STAMP}.csv'
with open(path, 'w', newline='', encoding='utf-8') as f:
w = csv.DictWriter(f, fieldnames=cols_out, extrasaction='ignore')
w.writeheader(); w.writerows(rows)
print(f"\n{len(rows)} variant rows -> {path}")
uniq = {r['handle']: r for r in rows if r.get('product')}
pure = [r for r in uniq.values() if r.get('on_pure_shelf')]
mix = [r for r in uniq.values() if r.get('on_mix_shelf')]
print(f"\n=== The seller's own shelves ({len(uniq)} products) ===")
print(f" on a pure-matcha shelf : {len(pure)}")
print(f" on a mix shelf : {len(mix)}")
print(f" on both : {sum(1 for r in uniq.values() if r.get('on_pure_shelf') and r.get('on_mix_shelf'))}")
print(f" on neither : {sum(1 for r in uniq.values() if not r.get('on_pure_shelf') and not r.get('on_mix_shelf'))}")
print(f" on the 'matcha' navigation shelf, which carries both: "
f"{sum(1 for r in uniq.values() if r.get('on_nav_matcha_shelf'))}")
print(" -- pure --"); [print(' ', r['product']) for r in pure]
print(" -- mixes --"); [print(' ', r['product']) for r in mix]
print("\n=== What the seller says is in each product (its own structured data) ===")
for r in uniq.values():
print(f" {r['product'][:44]:<46}serving {str(r.get('serving_size') or '—'):<8}"
f"sugar {str(r.get('sugar_per_serving') or '—'):<8}"
f"{(r.get('ingredients') or '— none published')[:80]}")
print("\n=== Disclosure, per product ===")
for f_ in ('grade', 'origin', 'caffeine', 'cultivars_row'):
n = sum(1 for r in uniq.values() if (r.get(f_) or '').strip())
print(f" {f_:<16}{n:>3}/{len(uniq)}")
labels = collections.Counter(r.get('cultivars_row_label') for r in uniq.values()
if r.get('cultivars_row_label'))
print(f" cultivar row label spellings used by the shop: {dict(labels)}")
print("\n=== Cultivars named anywhere on the product page ===")
for r in uniq.values():
if r.get('cultivars_named_on_page'):
print(f" {r['product'][:46]:<48}{r['cultivars_named_on_page']}")