# Per-gram price database for hojicha. Same method as the matcha database: every product # listed in each shop's Shopify products.json, not a selection. import urllib.request, json, ssl, re, csv, html, time UA={'User-Agent':'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120 Safari/537.36'} CTX=ssl.create_default_context(); CTX.check_hostname=False; CTX.verify_mode=ssl.CERT_NONE SHOPS=[('rockys','https://www.rockysmatcha.com','US'),('ujido','https://ujido.com','US'), ('jadeleaf','https://jadeleafmatcha.com','US'),('matchaeologist','https://matchaeologist.com','US'), ('ippodo_us','https://ippodotea.com','US'),('kettl','https://kettl.co','US'), ('rishi','https://rishi-tea.com','US'),('matchaful','https://matchaful.com','US'), ('hojicha_co','https://hojichaco.com','US'),('yunomi','https://yunomi.life','JP'), ('mizuba','https://mizubatea.com','US'),('nami','https://namimatcha.com','US'), ('arteao','https://arteao.com','US'),('moongoat','https://moongoat.com','US'),('tezumi','https://tezumi.com','US')] def shop_currency(base): """Read the shop's actual currency. Assuming USD turns ¥18,000 into $18,000.""" try: h=urllib.request.urlopen(urllib.request.Request(base+'/',headers=UA),timeout=20,context=CTX).read().decode('utf-8','ignore') m=re.search(r'Shopify\.currency\s*=\s*\{"active":"(\w{3})"',h) if m: return m.group(1) m=re.search(r'"currency"\s*:\s*"(\w{3})"',h) if m: return m.group(1) except Exception: pass return None def fetch(base): out,page=[],1 while page<=12: try: r=urllib.request.urlopen(urllib.request.Request(f'{base}/products.json?limit=250&page={page}',headers=UA),timeout=30,context=CTX) ps=json.loads(r.read().decode('utf-8','ignore')).get('products',[]) except Exception: break if not ps: break out+=ps; page+=1 return out def grams(s): s=s.lower().replace(',','') MULT={'kg':1000,'g':1,'oz':28.3495,'lb':453.592} # There are three multiplication formats. Missing any one of them moves the per-gram # price by a whole factor. # (a) "10 x 100g" = count x quantity m=re.search(r'(\d+)\s*[x×]\s*(\d+\.?\d*)\s*(kg|g|oz|lb)\b',s) if m: return float(m.group(1))*float(m.group(2))*MULT[m.group(3)] # (b) "1 oz x 6 boxes" = quantity x count, with the unit stated first m=re.search(r'(\d+\.?\d*)\s*(kg|g|oz|lb)\b\s*[x×]\s*(\d+)',s) if m: return float(m.group(1))*MULT[m.group(2)]*float(m.group(3)) # (c) "6 boxes (1 oz each)" m=re.search(r'(\d+)\s*(?:boxes|bags|packs|tins|units)\b[^)]*?(\d+\.?\d*)\s*(kg|g|oz|lb)\b',s) if m: return float(m.group(1))*float(m.group(2))*MULT[m.group(3)] m=re.search(r'(\d+\.?\d*)\s*(?:kg|kilograms?|kilos?)\b',s) if m: return float(m.group(1))*1000 m=re.search(r'(\d+\.?\d*)\s*(?:grams?|gr|g)\b',s) if m: return float(m.group(1)) m=re.search(r'(\d+\.?\d*)\s*(?:oz|ounces?)\b',s) if m: return float(m.group(1))*28.3495 m=re.search(r'(\d+\.?\d*)\s*(?:lbs?|pounds?)\b',s) if m: return float(m.group(1))*453.592 return None # Excluded: teaware, gifts, sets, tea bags, sweetened products and other tea types. EX=r'whisk|chasen|chashaku|bowl|chawan|scoop|sifter|mug|kit\b|gift|set\b|teaware|tumbler|canister|tin only|apparel|hat|tote|book|candle|spoon|strainer|pitcher|glass|cup\b|bottle|straw|voucher|card\b|subscription|class\b|bundle|sampler|discovery' EX2=r'cocoa|case of|wholesale|tea ?bag|sachet|stick pack|single serve|latte mix|sweetened|syrup|chocolate|cookie|ice cream|mochi|candy|kit ?kat' rows=[] for shop,base,cur in SHOPS: actual=shop_currency(base) if actual and actual!=cur: print(f" WARNING {shop}: assumed {cur}, actually {actual} - corrected") cur=actual n=0 for p in fetch(base): ti=p.get('title','') if not re.search(r'hojicha|houjicha',ti.lower()): continue if re.search(EX,ti.lower()) or re.search(EX2,ti.lower()): continue body=html.unescape(re.sub(r'<[^>]+>',' ',(p.get('body_html') or ''))) body=re.sub(r'\s+',' ',body).lower() # Powder or leaf? tl=ti.lower() if re.search(r'\bstems?\b|stem tea|kukicha|tenbone|karigane|loose ?leaf|tea ?leaves|\bbags?\b',tl): form='leaf' # the title says stem tea or loose leaf elif 'powder' in tl: form='powder' # "powder" in the title is definitive elif re.search(r'\bpowder',body[:300]) and not re.search(r'matcha (tea )?powder',body[:300]): form='powder' # from the body, excluding matcha descriptions else: form='leaf' for v in p.get('variants',[]): vt=v.get('title','') or '' price=v.get('price') if not price: continue g=grams(vt) or grams(ti) per=round(float(price)/g,4) if g else None rows.append({'shop':shop,'cur':cur,'title':ti[:90],'variant':vt[:40],'price':price, 'grams':round(g,2) if g else '','per_g':per if per else '','form':form, 'body_len':len(body)}) n+=1 print(f" {shop:<14} {n:>3} variants") time.sleep(0.4) with open('data/hojicha_prices_2026-08-25.csv','w',newline='') as f: w=csv.DictWriter(f,fieldnames=['shop','cur','title','variant','price','grams','per_g','form','body_len']) w.writeheader(); w.writerows(rows) print(f"\nTotal {len(rows)} variants -> data/hojicha_prices_2026-08-25.csv")