# Measures, across every brand, how often a listing discloses something a buyer can verify
# before paying for it.
import urllib.request, json, ssl, re, csv, html, time
UA={'User-Agent':'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120 Safari/537.36'}
CTX=ssl.create_default_context(); CTX.check_hostname=False; CTX.verify_mode=ssl.CERT_NONE
SHOPS=[('encha','https://encha.com'),('rockys','https://www.rockysmatcha.com'),('ujido','https://ujido.com'),
 ('nekohama','https://nekohama.co'),('jadeleaf','https://jadeleafmatcha.com'),('matchaeologist','https://matchaeologist.com'),
 ('ippodo_us','https://ippodotea.com'),('naoki','https://naokimatcha.com'),('kettl','https://kettl.co'),
 ('rishi','https://rishi-tea.com'),('matchaful','https://matchaful.com'),('dona','https://dona.com')]
def fetch(base):
    out,page=[],1
    while page<=10:
        try:
            r=urllib.request.urlopen(urllib.request.Request(f'{base}/products.json?limit=250&page={page}',headers=UA),timeout=30,context=CTX)
            ps=json.loads(r.read().decode('utf-8','ignore')).get('products',[])
        except Exception as e: break
        if not ps: break
        out+=ps; page+=1
    return out
def txt(p):
    s=' '.join([p.get('title',''),p.get('body_html','') or '',p.get('product_type','') or '',' '.join(p.get('tags',[]) or [])])
    s=html.unescape(re.sub(r'<[^>]+>',' ',s))
    return re.sub(r'\s+',' ',s).lower()
# The six things a buyer can check before purchase.
CHECKS={
 'harvest':  r'first harvest|1st harvest|ichiban|first flush|spring harvest|early harvest|first[- ]harvest|second harvest|nibancha|first spring|spring blend|shincha|first pick|first[- ]crop|early spring|april harvest|may harvest|harvested in',
 'cultivar': r'okumidori|yabukita|samidori|sakimidori|asahi|uji ?hikari|gokou|saemidori|kanaya|tsuyuhikari|narino|komakage|ujihikari|okuyutaka|meiryoku|yamanoibuki|hon ?usu|koshun|seimei|hoshun|sayamakaori|okuhikari|fushun|\bcultivars?\b',
 'region':   r'\buji\b|nishio|yame|kagoshima|shizuoka|kyoto|fukuoka|aichi|mie\b|wazuka|gokasho|shirakawa|tsujiki|kirishima|chiran|sayama|miyazaki|nagasaki|shiga|nara\b|ujitawara|kohata|asahina|tsuchiyama|shibushi|hoshino',
 'stone':    r'stone[- ]?ground|stone[- ]?mill|ishiusu|granite mill',
 'shade':    r'shade[- ]?grown|shaded|tencha|kabuse|covered cultivation|20 days|21 days|shading',
 'date':     r'best before|best by|harvest date|packed on|milled on|production date|expiry|expiration|use by',
}
EXCLUDE=r'''(?x)
 ^\s*(virtual\s+)?class\s*\||subscription
|whisk|chasen|chashaku|ladle|bowl|chawan|scoop|sifter|mug|frother|strainer|pitcher|glass|cup\b|bottle|straw|spoon
|ceremony\s+set|teaware|ritual\s+set|essentials\s+set|homebody|starter\s+set|starter\s+bundle|gift\s+set|tea\s+set|complete\s+matcha
|\bkits?\b|keto|collagen|dad\s+hat|logo\s+hat|tote|car\s+tag|apparel|hoodie|shirt|sticker|towel|cloth|holder|stand|book|candle
|canister|gift\s+card|gift\s+voucher|voucher|tumbler|e-?gift
|genmaimatcha|genmaicha|hojicha|houjicha|sencha|gyokuro|kukicha|bancha|sobacha|oolong
|packets?\b|stick\s+packs|\bsticks\b|single\s+serve
|sweetened|sweet\s+matcha|latte\s+mix|latte\s+concentrate|reishi|iced\s+tea
|variety\s+pack|multipack|collection\s+trio|\(\s*\d+\s*x|case\s+of
|warehouse
|chai|lemonade|granola|honey|syrup|cookie|chocolate|bundle|sampler|discovery\s+set
'''
GRADE=r'ceremonial'
PREM=r'\bpremium\b'
rows=[]
for shop,base in SHOPS:
    ps=fetch(base); n=0
    for p in ps:
        t=txt(p); ti_low=p.get('title','').lower()
        if 'matcha' not in t: continue
        if re.search(EXCLUDE, ti_low): continue
        # Drop by product type: tea bags and loose leaf are not matcha powder.
        ptype=(p.get('product_type','') or '').lower()
        if re.search(r'sachet|loose ?leaf|tea ?bag', ptype): continue
        # Drop only when the product is itself a blend with another tea, or a flavoured item.
        # A passing mention of another tea is not grounds for exclusion.
        if re.search(r'sencha meets|blend of sencha|sencha and matcha|with sencha|flavored_drink_concentrate|vanilla bean|\blavender\b', t): continue
        r={'shop':shop,'title':p.get('title','')[:80],'ceremonial':1 if re.search(GRADE,t) else 0,'premium':1 if re.search(PREM,t) else 0}
        for k,pat in CHECKS.items(): r[k]=1 if re.search(pat,t) else 0
        r['checkable']=sum(r[k] for k in CHECKS)
        r['body_len']=len(t)
        b=p.get('title','')
        b=re.sub(r'\(\s*\d+\.?\d*\s*(g|kg|oz|lb|ml)[^)]*\)','',b,flags=re.I)   # (40g Can)
        b=re.sub(r'\s*[-–|]\s*\d+\.?\d*\s*(g|kg|oz|lb|ml)\b[^-–|]*','',b,flags=re.I) # - 50g Pouch / | 100g
        b=re.sub(r'\b\d+\.?\d*\s?(g|kg|oz|lb|ml)\b','',b,flags=re.I)
        r['base']=re.sub(r'[^a-z0-9]+',' ',b.lower()).strip()
        rows.append(r); n+=1
    print(f"  {shop:<16} {n:>3} matcha products (of {len(ps)} total)")
    time.sleep(0.4)
# Collapse size variants into one product. The disclosures are identical, so keep the first.
seen=set(); uniq=[]
for r in rows:
    k=(r['shop'],r['base'])
    if k in seen: continue
    seen.add(k); uniq.append(r)
print(f"\nCollapsing size variants: {len(rows)} -> {len(uniq)} products")
with open('data/label_claims_2026-08-25.csv','w',newline='') as f:
    w=csv.DictWriter(f,fieldnames=['shop','base','title','ceremonial','premium']+list(CHECKS)+['checkable','body_len']); w.writeheader(); w.writerows(uniq)
rows=uniq
print(f"\nTotal {len(rows)} products -> data/label_claims_2026-08-25.csv")
