""" token-counting-cost-forecast.py - count tokens the way models bill them, and forecast what a pipeline will cost. Three jobs: python token-counting-cost-forecast.py count FILE [FILE ...] token counts for each file under six public tokenizers, with characters per token and the error of the "characters / 4" rule of thumb python token-counting-cost-forecast.py forecast --prompt FILE --output-tokens 250 --requests 200000 \ --price-in 1.00 --price-out 1.00 [--tokenizer o200k_base] tokens per request, tokens per period and cost per period. Prices are in dollars per million tokens and come from YOUR provider's pricing page; nothing here assumes one. Without --tokenizer the forecast runs under all six and prints the range, which is the honest answer when your provider's own tokenizer is not public. python token-counting-cost-forecast.py reproduce [--save-samples DIR] rebuilds the article's measurement table from the public sources: three Project Gutenberg texts (1342 English, 22367 German, 23962 Chinese; characters 5,000 to 25,000 of each body), the first 20,000 characters of CPython's json/decoder.py + json/encoder.py from the running interpreter, and one ESPN public NFL scoreboard response (compact and indent=2). Requires: Python 3.10+, tiktoken 0.14.0 and tokenizers 0.22.2 (pip install tiktoken tokenizers). The Hugging Face tokenizer files download on first use: Qwen/Qwen2.5-7B-Instruct, deepseek-ai/DeepSeek-V3, mistralai/Mistral-7B-Instruct-v0.3 and microsoft/Phi-3.5-mini-instruct. Counts exclude special tokens (no BOS/EOS, no chat template), so a real request adds a few more. """ import argparse import json import sys import urllib.request TIKTOKEN = ["cl100k_base", "o200k_base"] HF = {"Qwen2.5": "Qwen/Qwen2.5-7B-Instruct", "DeepSeek-V3": "deepseek-ai/DeepSeek-V3", "Mistral-7B-v0.3": "mistralai/Mistral-7B-Instruct-v0.3", "Phi-3.5-mini": "microsoft/Phi-3.5-mini-instruct"} NAMES = TIKTOKEN + list(HF) _cache = {} def tokenizer(name): if name not in _cache: if name in TIKTOKEN: import tiktoken enc = tiktoken.get_encoding(name) _cache[name] = lambda s: len(enc.encode(s, disallowed_special=())) elif name in HF: from tokenizers import Tokenizer tok = Tokenizer.from_pretrained(HF[name]) _cache[name] = lambda s: len(tok.encode(s, add_special_tokens=False).ids) else: sys.exit("unknown tokenizer %r; choose from %s" % (name, ", ".join(NAMES))) return _cache[name] def count_text(text, names=NAMES): return {n: tokenizer(n)(text) for n in names} def cost(tokens_in, tokens_out, requests, price_in, price_out): """Dollars for `requests` calls at the given $ per million tokens.""" return requests * (tokens_in * price_in + tokens_out * price_out) / 1e6 def cmd_count(args): for path in args.files: text = open(path, encoding="utf-8").read() est = len(text) / 4 print("%s: %d characters, chars/4 estimate %d tokens" % (path, len(text), round(est))) for n, t in count_text(text).items(): print(" %-16s %8d tokens %5.2f chars/token estimate off by %+6.1f%%" % (n, t, len(text) / t, 100 * (est / t - 1))) def cmd_forecast(args): text = open(args.prompt, encoding="utf-8").read() names = [args.tokenizer] if args.tokenizer else NAMES rows = [] for n in names: t = tokenizer(n)(text) rows.append((n, t, cost(t, args.output_tokens, args.requests, args.price_in, args.price_out))) print("prompt %s: %d characters; %d output tokens per request; %d requests" % (args.prompt, len(text), args.output_tokens, args.requests)) print("prices: $%.4f in, $%.4f out, per million tokens (from your provider's pricing page)" % (args.price_in, args.price_out)) for n, t, c in rows: print(" %-16s %7d input tokens/request %12.1f million input tokens $%12.2f" % (n, t, t * args.requests / 1e6, c)) if len(rows) > 1: lo, hi = min(r[2] for r in rows), max(r[2] for r in rows) print("range across tokenizers: $%.2f to $%.2f (x%.2f)" % (lo, hi, hi / lo)) def fetch(url): # urllib's default agent on purpose: ESPN answers custom User-Agent strings with 403 with urllib.request.urlopen(urllib.request.Request(url), timeout=60) as r: return r.read().decode("utf-8") def gutenberg_body(n): raw = fetch("https://www.gutenberg.org/cache/epub/%d/pg%d.txt" % (n, n)).replace(chr(13) + chr(10), chr(10)) # Gutenberg serves CRLF a, b = raw.index("*** START OF"), raw.index("*** END OF") return raw[raw.index("\n", a) + 1:b].strip() def cmd_reproduce(args): import json as jsonmod import os lib = os.path.dirname(jsonmod.__file__) code = open(os.path.join(lib, "decoder.py"), encoding="utf-8").read() + open(os.path.join(lib, "encoder.py"), encoding="utf-8").read() src = args.scoreboard board = json.loads(fetch(src) if src.startswith("http") else open(src, encoding="utf-8").read()) if "espn_payload" in board: # a file saved with its pull metadata board = board["espn_payload"] samples = { "english": gutenberg_body(1342)[5000:25000], "german": gutenberg_body(22367)[5000:25000], "chinese": gutenberg_body(23962)[5000:25000], "python code": code[:20000], "json compact": json.dumps(board, separators=(",", ":"), ensure_ascii=False), "json indent=2": json.dumps(board, indent=2, ensure_ascii=False), } if args.save_samples: os.makedirs(args.save_samples, exist_ok=True) for k, s in samples.items(): name = k.replace(" ", "_").replace("=", "") + ".txt" with open(os.path.join(args.save_samples, name), "w", encoding="utf-8", newline="") as f: f.write(s) print("content,tokenizer,characters,tokens,chars_per_token,chars4_error_pct") for k, s in samples.items(): for n, t in count_text(s).items(): print("%s,%s,%d,%d,%.3f,%.1f" % (k, n, len(s), t, len(s) / t, 100 * (len(s) / 4 / t - 1))) def main(): p = argparse.ArgumentParser(description=__doc__.split("\n")[1]) sub = p.add_subparsers(dest="cmd", required=True) a = sub.add_parser("count"); a.add_argument("files", nargs="+"); a.set_defaults(fn=cmd_count) b = sub.add_parser("forecast") b.add_argument("--prompt", required=True, help="a file holding one typical request's input text") b.add_argument("--output-tokens", type=int, required=True) b.add_argument("--requests", type=int, required=True, help="requests in the period you are forecasting") b.add_argument("--price-in", type=float, required=True, help="$ per million input tokens") b.add_argument("--price-out", type=float, required=True, help="$ per million output tokens") b.add_argument("--tokenizer", choices=NAMES) b.set_defaults(fn=cmd_forecast) c = sub.add_parser("reproduce") c.add_argument("--scoreboard", default="https://site.api.espn.com/apis/site/v2/sports/football/nfl/scoreboard?dates=20260924", help="URL or saved file for the JSON sample; the article used the response for 24 September 2026, pulled on 25 September (the live response can drift)") c.add_argument("--save-samples", metavar="DIR", help="also write the six sample texts to DIR, ready for count and forecast") c.set_defaults(fn=cmd_reproduce) args = p.parse_args() args.fn(args) if __name__ == "__main__": main()