#!/usr/bin/env python3
"""Reads what Naoki Matcha publishes about each tin it sells, from the seller's own pages. Used in article 16.
The question this answers is narrow: for each product, what is its name, what does the seller
disclose about it, where does the seller use the words that buyers search for, and what does a
gram cost.
Known failure modes in this kind of listing data, and the guard applied to each. Read this
before trusting any figure the script produces:
1. **A product feed is not the page.** `products.json` returns `body_html`, but this shop puts
its specification table — origin, grade, harvest, cultivar — in a theme block that never
appears in the feed. Counting terms in `body_html` alone reports a word as absent when the
page shows it. Every field here is read from the rendered product page.
2. A Shopify variant's `grams` field is **shipping weight, not net weight**. At this shop a
1.4 oz tin reports 45 g and a 3.5 oz tin reports 100 or 109 g for the same stated size. Net
weight is taken from the size the seller states, and the source of each weight is recorded.
3. Classification comes from **the seller's own categories** (`/collections/*`), not from words
in the product name. A tin with "matcha" in its name may be a bundle, an accessory or a
sweetened mix.
4. Where a term appears matters. A word in a product's *name* is the seller naming the product;
the same word in a category heading, a spec row or an image alt attribute is not. Each is
counted in its own column, never summed into one "mentions" figure.
5. Retry failed fetches. A page that fails to load is missing, never "the seller does not
disclose this".
6. Bundles and sets have no single net weight and no single grade. They are collected and
marked, and excluded from per-gram statistics rather than silently dropped.
"""
import ssl, re, csv, json, time, html, urllib.request, collections
SHOP = 'naokimatcha.com'
UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'}
CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
# Guard 2: the seller prints "Net Wt. 1.4 oz (40g)" on the tin, so these are the seller's own
# roundings, not ounce-to-gram arithmetic. Anything not listed here is left blank, not guessed.
STATED_GRAMS = {'1.4': 40, '1.75': 50, '3.5': 100}
# The words this article is about. Counted per field, never summed (guard 4).
TERMS = ['ceremonial', 'superior', 'premium', 'culinary', 'first harvest', 'single cultivar']
# Cultivar names are checked against the whole rendered page, not just the spec table, so that
# "this listing names no cultivar" means the page names none anywhere. The list is published
# here rather than described, so the check can be re-run and disagreed with.
CULTIVARS = ['yabukita', 'okumidori', 'samidori', 'saemidori', 'asahi', 'tsuyuhikari',
'kanayamidori', 'okuyutaka', 'asanoka', 'gokou', 'uji hikari', 'narino',
'sayamakaori', 'yamakai', 'meiryoku', 'seimei', 'kirari', 'okuharuka']
def fetch(url, tries=4):
for a in range(tries):
try:
r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=45, context=CTX)
return r.read().decode('utf-8', 'ignore')
except Exception as e:
if a == tries - 1:
print(f' FETCH FAILED after {tries} tries: {url[:80]} {e}')
return None
time.sleep(2 * (a + 1))
def jget(url):
t = fetch(url)
try:
return json.loads(t) if t else None
except Exception:
return None
def txt(s):
return re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', s or ''))).strip()
def accordions(page):
"""Every collapsible block on the product page, as {heading: body}. The spec table lives here
and not in the product feed (guard 1)."""
out = {}
for m in re.finditer(r'
(.*?)
', body_html_block, re.S): line = txt(p) m = re.match(r'([A-Za-z][A-Za-z /]{2,24}):\s*(.+)$', line) if m: fields.setdefault(m.group(1).strip(), m.group(2).strip()) return fields def count(text, term): return len(re.findall(r'(? {path}") print("\n=== Where the seller uses each term (counted per place, never summed) ===") uniq = {r['handle']: r for r in rows if r.get('product')} for t in TERMS: k = t.replace(' ', '_') n = sum(1 for r in uniq.values() if r.get(f'name_{k}')) s = sum(1 for r in uniq.values() if r.get(f'spec_{k}')) d = sum(1 for r in uniq.values() if r.get(f'desc_{k}')) c = sum(1 for r in uniq.values() if r.get(f'cat_{k}')) a = sum(1 for r in uniq.values() if r.get(f'alt_{k}')) print(f" {t:<17} name {n:>2}/{len(uniq)} spec row {s:>2} description {d:>2} " f"category {c:>2} image alt {a:>2}") print("\n=== Cultivars named anywhere on the product page ===") for r in uniq.values(): if r.get('product'): print(f" {r['product'][:44]:<46}{r.get('cultivars_named_on_page') or '— none'}")