#!/usr/bin/env python3 """ Crypto Card Geography Index: scoring. Measures which products AI engines name when someone asks how to spend crypto as local money, and whether the answer changes with the asker's country. python3 score.py --raw raw.jsonl --outdir results/ Corrections applied here, never in raw.jsonl: - text is normalised first: markdown citation links reduced to anchor text and bare URLs dropped, because ChatGPT embeds citation URLs inline and vendor domains would otherwise score as product mentions - names that are also ordinary words are matched case-sensitively - chains and tokens are tracked separately from products, since neither is something a person signs up for """ import argparse, csv, json, os, re from collections import Counter, defaultdict from urllib.parse import urlparse # --- the study's subject: cards you spend with --- CARDS = ["Avici", "KAST", "Kast Card", "Ether.fi", "EtherFi", "Coinbase Card", "MetaMask Card", "Gnosis Pay", "Plasma One", "RedotPay", "Wirex", "Wirex Card", "Nexo Card", "Crypto.com", "Bybit Card", "Binance Card", "Bleap", "Bleap Mastercard", "Oobit", "Kripicard", "Bitypay", "GetPlu", "Bitrefill", "Bitget Wallet", "CoinGate", "BitPay", "Coinbase Commerce", "MoonPay", "BVNK", "Rain", "Holyheld", "Baanx", # found by reading the corpus after the first scoring pass "Tria", "Kolo", "Sphere", "Pulsar", "Bitnob", "Cardtonic", "Jupiter Global", "Bit2Me", "Yellow Card"] # --- exchanges and wallets used to cash out --- VENUES = ["Binance", "Binance P2P", "Coinbase", "Kraken", "Bybit", "OKX", "Bitget", "KuCoin", "Indodax", "Coins.ph", "Tokocrypto", "Upbit", "Bithumb", "Paribu", "BtcTurk", "Luno", "Quidax", "Busha", "Roqqu", "Bitso", "MetaMask", "Trust Wallet", "Phantom"] # --- traditional comparators --- TRADFI = ["Wise", "TransferWise", "Revolut", "Payoneer", "PayPal", "Remitly", "Western Union", "WorldRemit", "OFX", "Chase", "Charles Schwab", "Capital One", "Stripe", "Airwallex", "N26", "Monzo", "Niyo", "Niyo Global", "Fi Money", "Jupiter", "Zolve", "BookMyForex", "Nubank", "Mercado Pago", "GCash", "Maya", "Alipay", "Papara", "Toss", "Wio", "Grey", "Eversend", "Chipper", "Cardsoon", "Barter", "Flutterwave", "Raenest", "Cleva", "Lemon", "Lemon Cash", "Timo", "Techcombank", "Geegpay", "Toss Bank", "KakaoBank"] # --- market-native products, found by reading the corpus rather than listed # up front. The first scoring pass reported Brazil and Vietnam as having no # local products at all, which was a threshold artefact: the engines name # plenty, they just fell below a 5-mention cutoff and were absent from the # lists. Nothing here was hand-picked as "important"; it is everything the # corpus named that is a real bank, exchange, wallet or card. LOCAL = [ # Brazil "Nomad", "Ripio", "Foxbit", "Mercado Bitcoin", "NovaDAX", "Bitomat", "Remessa Online", "Conta Global", "Zypto", "Banco24Horas", "Inter", "Bradesco", "C6", "Nubank", "Mercado Pago", # Indonesia "BCA", "CIMB Niaga", "Pintu", "Jenius", "OVO", "GoPay", "Mandiri", "Maybank", "Danamon", "BNI", "Jago", "Karta", "Tokocrypto", # India "CoinDCX", "WazirX", "ZebPay", "Mudrex", "Cryptomus", "THORWallet Card", "SBM Niyo", "ICICI", "IDFC FIRST", "Tangem Pay", # Korea "Kaia", "Coinone", "Korbit", "DaWinKS", "Hana", "KEB Hana", "Shinhan", "Woori", "Travel Wallet", "DeCard", # Nigeria "GTBank", "Zenith", "Paystack", "Kuda", "OPay", "Moniepoint", "Bitmama", "CoinCola", "GrabrFi", "Nummus", "KudiX", "SpendFigo", "Cardex", "Airtm", "Figo", "UBA", "FirstBank", "Remita", "Verve", # Philippines "PDAX", "PayMaya", "BDO", "BPI", "GoTyme", "UnionBank", "SeaBank", "Metrobank", "PEXX", "Kite Stable", # Turkiye "Midas", "Coinsfera", "Adonis Exchange", "IZIPAY", "Fizen", "OKX TR", # Vietnam "BitcoinVN", "MoMo", "ZaloPay", "Vietcombank", "BIDV", "BenPay", "VPBank", "Remitano", # United States "Gemini", "Coinsbee", ] # Local-currency stablecoins. A separate category because their existence at # all is a finding: the engines know about currency-specific tokens. LOCAL_TOKENS = ["BRZ", "BRL1", "IDRT", "Rupiah Token", "KRW1", "KRWQ", "KRWT"] # Domestic payment rails. Not products, but naming one is proof the engine # knows which country it is answering. RAILS = ["Pix", "PIX", "UPI", "QRIS", "InstaPay", "PESONet", "NAPAS", "NIP", "GPN", "TED"] # --- chains and tokens: infrastructure, not products --- INFRA = ["Solana", "Base", "Polygon", "Tron", "TRON", "Arbitrum", "Optimism", "Stellar", "Avalanche", "Ethereum", "BSC", "USDC", "USDT", "DAI", "BUSD", "FDUSD", "BTC", "ETH", "Pix", "PIX", "UPI"] # Names that are also ordinary English words, or too short to be safe. CASE_SENSITIVE = {"Base", "Rain", "Lemon", "Maya", "Grey", "Barter", "Chase", "Circle", "Jupiter", "Toss", "Luno", "Pix", "PIX", "KAST", "Inter", "Hana", "Midas", "Karta", "Gemini", "Nomad", "Figo", "Verve", "Zenith", "TED", "C6"} # Two names for one product. Counting both inflates the card totals: Wirex and # "Wirex Card" co-occurred in all 17 of the latter's responses. ALIASES = { "KAST": "Kast", "Kast Card": "Kast", "Ether.fi": "Ether.fi", "EtherFi": "Ether.fi", "Wirex Card": "Wirex", "Nexo Card": "Nexo", "Bleap Mastercard": "Bleap", "Bybit Card": "Bybit Card", "Lemon Cash": "Lemon", "TransferWise": "Wise", "Niyo Global": "Niyo", } CRYPTO_GENERIC = ["stablecoin", "stablecoins", "usdc", "usdt", "crypto", "cryptocurrency", "on-chain", "onchain", "web3", "tether"] MD_LINK = re.compile(r"\[([^\]]*)\]\(([^)]*)\)") BARE_URL = re.compile(r"https?://\S+|\(\s*[a-z0-9.-]+\.\w{2,}(?:/\S*)?\s*\)") CCTLD = {"NG": ".ng", "BR": ".br", "ID": ".id", "PH": ".ph", "TR": ".tr", "KR": ".kr", "VN": ".vn", "IN": ".in", "US": ".us"} def normalize(t): return BARE_URL.sub(" ", MD_LINK.sub(lambda m: m.group(1), t)) # "Regulatory Grey Area" appeared repeatedly and is not the Nigerian fintech. NEGATIVE_AFTER = {"Grey": r"(?!\s+[Aa]rea)"} def build(name): flags = 0 if name in CASE_SENSITIVE else re.I tail = NEGATIVE_AFTER.get(name, "") return re.compile( rf"(?= a and m.end() <= b for a, b in claimed): continue claimed.append((m.start(), m.end())) canon = ALIASES.get(n, n) found[canon] = KIND[n] break return found def domain(u): try: d = urlparse(u).netloc.lower() return d[4:] if d.startswith("www.") else d except Exception: return "" def main(): ap = argparse.ArgumentParser() ap.add_argument("--raw", default="raw.jsonl") ap.add_argument("--outdir", default="results") a = ap.parse_args() os.makedirs(a.outdir, exist_ok=True) # one record per (prompt, market, engine, run); a success supersedes an error best = {} for line in open(a.raw): try: r = json.loads(line) except Exception: continue k = (r["prompt_id"], r["market"], r["engine"], r["run"]) if k not in best or (best[k].get("error") and not r.get("error")): best[k] = r recs = [r for r in best.values() if not r.get("error") and r.get("text")] print(f"loaded {len(recs)} usable responses") for r in recs: r["_t"] = normalize(r["text"]) r["_p"] = products(r["_t"]) r["_crypto"] = bool(CRYPTO_RE.search(r["_t"])) r["_cards"] = [n for n, k in r["_p"].items() if k == "card"] MK = ["IN", "BR", "NG", "ID", "PH", "TR", "KR", "VN", "US"] ENG = ["chatgpt", "claude", "perplexity", "google_aio"] # how concentrated is each product in one market: the data decides what is local prod_mkt = defaultdict(Counter) for r in recs: for n in r["_p"]: prod_mkt[n][r["market"]] += 1 local_of = {} for n, c in prod_mkt.items(): tot = sum(c.values()) top, hits = c.most_common(1)[0] # 3 responses rather than 5: a five-response floor hid genuinely local # products and produced a false null for Brazil and Vietnam in the # first pass. Chains and tokens are excluded because naming Solana says # nothing about which country the engine thinks it is answering. if tot >= 3 and hits / tot >= 0.7 and KIND.get(n) != "infra": local_of[n] = top # ---- surface rates ---- with open(f"{a.outdir}/surface_rates.csv", "w", newline="") as f: w = csv.writer(f) w.writerow(["market", "engine", "stage", "responses", "mentioned_crypto", "crypto_rate", "named_a_card", "card_rate"]) for m in MK: for e in ENG: for st in (1, 2): v = [r for r in recs if r["market"] == m and r["engine"] == e and r["stage"] == st] if not v: continue c = sum(1 for r in v if r["_crypto"]) k = sum(1 for r in v if r["_cards"]) w.writerow([m, e, st, len(v), c, round(c / len(v), 4), k, round(k / len(v), 4)]) # ---- share of voice per market ---- with open(f"{a.outdir}/share_of_voice.csv", "w", newline="") as f: w = csv.writer(f) w.writerow(["market", "product", "kind", "responses", "share_of_market", "market_specific_to"]) for m in MK: v = [r for r in recs if r["market"] == m] c = Counter() for r in v: for n in r["_p"]: c[n] += 1 tot = sum(c.values()) or 1 for n, k in c.most_common(): w.writerow([m, n, KIND[n], k, round(k / tot, 4), local_of.get(n, "")]) # ---- localisation: does the answer name a product specific to the asker ---- with open(f"{a.outdir}/localisation.csv", "w", newline="") as f: w = csv.writer(f) w.writerow(["market", "engine", "responses", "named_local_product", "local_rate", "cited_local_domain", "local_source_rate"]) for m in MK: for e in ENG: v = [r for r in recs if r["market"] == m and r["engine"] == e] if not v: continue loc = sum(1 for r in v if any(local_of.get(n) == m for n in r["_p"])) tld = CCTLD.get(m, "") src = sum(1 for r in v if any( domain(s.get("url", "")).endswith(tld) for s in r.get("sources", []))) w.writerow([m, e, len(v), loc, round(loc / len(v), 4), src, round(src / len(v), 4)]) # ---- roster visibility ---- with open(f"{a.outdir}/roster.csv", "w", newline="") as f: w = csv.writer(f) w.writerow(["product", "kind", "responses", "stage1", "stage2", "markets", "market_list"]) c = Counter(); s1 = Counter(); s2 = Counter(); mk = defaultdict(set) for r in recs: for n in r["_p"]: c[n] += 1 (s1 if r["stage"] == 1 else s2)[n] += 1 mk[n].add(r["market"]) canon = [] for n in RE: cn = ALIASES.get(n, n) if cn not in canon: canon.append(cn) for n in sorted(canon, key=lambda x: -c[x]): w.writerow([n, KIND.get(n, ""), c[n], s1[n], s2[n], len(mk[n]), ",".join(sorted(mk[n]))]) # ---- sources ---- with open(f"{a.outdir}/sources.csv", "w", newline="") as f: w = csv.writer(f) w.writerow(["market", "engine", "domain", "citations", "is_local_tld"]) c = Counter() for r in recs: for s in r.get("sources", []): d = domain(s.get("url", "")) if d: c[(r["market"], r["engine"], d)] += 1 for (m, e, d), n in sorted(c.items(), key=lambda x: -x[1]): w.writerow([m, e, d, n, d.endswith(CCTLD.get(m, "~"))]) # ---- summary ---- with open(f"{a.outdir}/summary.txt", "w") as f: f.write(f"responses analysed: {len(recs)}\n") f.write("markets: " + ", ".join(MK) + " (US = control)\n") f.write("Claude has no coverage in NG or VN; those cells are absent, not zero.\n\n") f.write("STAGE 1: a crypto card was named in this share of responses\n") f.write(f" {'market':<8}" + "".join(f"{e:>12}" for e in ENG) + "\n") for m in MK: row = "" for e in ENG: v = [r for r in recs if r["market"] == m and r["engine"] == e and r["stage"] == 1] row += f"{'n/a':>12}" if not v else \ f"{sum(1 for r in v if r['_cards'])}/{len(v):<10}" f.write(f" {m:<8}{row}\n") f.write("\nSTAGE 1: crypto mentioned at all, even generically\n") f.write(f" {'market':<8}" + "".join(f"{e:>12}" for e in ENG) + "\n") for m in MK: row = "" for e in ENG: v = [r for r in recs if r["market"] == m and r["engine"] == e and r["stage"] == 1] row += f"{'n/a':>12}" if not v else \ f"{sum(1 for r in v if r['_crypto'])}/{len(v):<10}" f.write(f" {m:<8}{row}\n") f.write("\nMOST-NAMED PRODUCT IN EACH MARKET (all stages)\n") for m in MK: v = [r for r in recs if r["market"] == m] c = Counter() for r in v: for n, k in r["_p"].items(): if k != "infra": c[n] += 1 top = ", ".join(f"{n} {x}" for n, x in c.most_common(4)) f.write(f" {m:<4} {top}\n") f.write("\nPRODUCTS THE DATA SAYS ARE MARKET-SPECIFIC\n") by = defaultdict(list) for n, m in local_of.items(): if KIND[n] != "infra": by[m].append(n) for m in MK: if by[m]: f.write(f" {m:<4} {', '.join(sorted(by[m]))}\n") f.write("\nROSTER: cards, by responses naming them\n") c = Counter(); s1 = Counter() for r in recs: for n, k in r["_p"].items(): if k == "card": c[n] += 1 if r["stage"] == 1: s1[n] += 1 canon_cards = [] for n in CARDS: cn = ALIASES.get(n, n) if cn not in canon_cards: canon_cards.append(cn) for n in sorted(canon_cards, key=lambda x: -c[x]): f.write(f" {n:<20}{c[n]:>5} responses stage 1: {s1[n]}\n") print(f"written to {a.outdir}/") if __name__ == "__main__": main()