#!/usr/bin/env python3 """Collects every buyer review a shop publishes through the Okendo widget. Used in article 15. Three of the brands in this project's dataset expose review data this way. The subscriber id sits in the HTML of any product page as `subscriberId":"..."`: rishi 292e3335-9bc5-448d-a5d3-b4ee44383a84 rockys 82189acd-e210-4b2b-9783-87f5be65e402 naoki 3ab06853-ec1d-4146-a192-b1e5b039cbd2 Known failure modes in this kind of review data, and the guard applied to each. Read this before trusting any figure the script produces: 1. `nextUrl` comes back as a **relative path**. Prefix it with the API base or paging dies silently after the first hundred rows. 2. Rows whose productId contains `site-reviews` are **store-wide reviews about the shopping experience**, not reviews of a product. Counting them as product reviews mixes delivery complaints into a tea's rating. At Rishi they are 250 of 9,453. They are kept, but in a separate `scope` column. 3. The product name is `productName`, not `product.name`. Reading the wrong key collapses every row onto one product. 4. A shop's catalogue is not the same as its category labels. A product with "matcha" in its name may be a leaf tea or a sachet; check the shop's own `product_type` before treating it as powder. 5. Retry failed fetches. Zero results must never be read as "this shop has no reviews". 6. `isVerified` is the key to spotting a change in *who writes the reviews*. When a shop moves from volunteered reviews to review requests sent to every buyer, ratings fall without the product changing. Always report it alongside the ratings, by year. """ import json, ssl, time, csv, urllib.request, collections BASE = 'https://api.okendo.io/v1' UA = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/120'} CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE SHOPS = { 'rishi': '292e3335-9bc5-448d-a5d3-b4ee44383a84', 'rockys': '82189acd-e210-4b2b-9783-87f5be65e402', 'naoki': '3ab06853-ec1d-4146-a192-b1e5b039cbd2', } def get(url): if url.startswith('/'): url = BASE + url for attempt in range(4): try: r = urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=45, context=CTX) return json.loads(r.read().decode('utf-8', 'ignore')) except Exception as e: if attempt == 3: print(f" FAILED after 4 attempts: {url[:90]} {e}") raise time.sleep(2 * (attempt + 1)) def pull(shop, sid, max_pages=150): url = f'{BASE}/stores/{sid}/reviews?limit=100' rows, pages = [], 0 while url and pages < max_pages: d = get(url) batch = d.get('reviews') or [] if not batch: break for r in batch: pid = str(r.get('productId') or '') reviewer = r.get('reviewer') or {} rows.append({ 'shop': shop, 'review_id': r.get('reviewId'), 'product_id': pid, 'scope': 'site' if 'site-reviews' in pid else 'product', 'product_name': r.get('productName') or '', 'rating': r.get('rating'), 'title': (r.get('title') or '').replace('\n', ' '), 'body': (r.get('body') or '').replace('\n', ' '), 'date': r.get('dateCreated') or '', 'is_verified': 1 if reviewer.get('isVerified') else 0, 'is_incentivized': 1 if r.get('isIncentivized') else 0, 'is_recommended': r.get('isRecommended'), 'country': ((reviewer.get('location') or {}).get('country') or {}).get('code', ''), 'helpful': r.get('helpfulCount'), 'status': r.get('status'), }) url = d.get('nextUrl') pages += 1 time.sleep(0.2) return rows if __name__ == '__main__': out = [] for shop, sid in SHOPS.items(): rows = pull(shop, sid) out += rows prod = [r for r in rows if r['scope'] == 'product'] site = [r for r in rows if r['scope'] == 'site'] yrs = sorted({r['date'][:4] for r in rows if r['date'][:4].isdigit()}) print(f" {shop:<8}{len(rows):>6} reviews (product {len(prod)} / store-wide {len(site)}) " f"{yrs[0] if yrs else '?'}-{yrs[-1] if yrs else '?'} products {len({r['product_name'] for r in prod})}") path = 'data/reviews/reviews_okendo_full_2026-08-26.csv' with open(path, 'w', newline='', encoding='utf-8') as f: w = csv.DictWriter(f, fieldnames=list(out[0].keys())) w.writeheader(); w.writerows(out) print(f"\nTotal {len(out)} -> {path}") # Ratings and the share of verified buyers, side by side. If both move in the same year, # the rating change is at least partly a change in who is writing, not what they bought. print("\nBy year (rating vs share of verified buyers):") for shop in SHOPS: g = [r for r in out if r['shop'] == shop and r['scope'] == 'product' and r['date'][:4].isdigit()] by = collections.defaultdict(list) for r in g: by[r['date'][:4]].append(r) print(f" {shop}") for y in sorted(by): v = by[y] if len(v) < 15: continue rt = [x['rating'] for x in v if x['rating'] is not None] low = sum(1 for x in rt if x <= 3) / len(rt) * 100 ver = sum(x['is_verified'] for x in v) / len(v) * 100 print(f" {y} n={len(v):>4} mean {sum(rt)/len(rt):.2f} <=3 stars {low:>5.1f}% verified {ver:>5.1f}%")