scripts/models.js
// Cloudflare Workers AI catalogue of interviewable text-generation models,
// with Workers AI pricing (USD per 1M tokens) as of Sep 2026.
// Single source of truth for scripts/cost.js and docs/cost-estimate.md.
// Sources: https://developers.cloudflare.com/workers-ai/models/ + /platform/pricing/
export const MODELS = [
{ id: "@cf/meta/llama-3.2-1b-instruct", family: "llama", params: "1B", quant: "fp16", context: 60_000, in: 0.027, out: 0.201, cached_in: null, reasoning: false, paid_billing: false },
{ id: "@cf/meta/llama-3.2-3b-instruct", family: "llama", params: "3B", quant: "fp16", context: 80_000, in: 0.051, out: 0.335, cached_in: null, reasoning: false, paid_billing: false },
{ id: "@cf/meta/llama-3.1-8b-instruct-fp8", family: "llama", params: "8B", quant: "fp8", context: 32_000, in: 0.152, out: 0.287, cached_in: null, reasoning: false, paid_billing: false },
{ id: "@cf/meta/llama-3.2-11b-vision-instruct", family: "llama", params: "11B", quant: "fp16", context: 128_000, in: 0.049, out: 0.676, cached_in: null, reasoning: false, paid_billing: false, vision: true },
{ id: "@cf/meta/llama-3.3-70b-instruct-fp8-fast", family: "llama", params: "70B", quant: "fp8", context: 24_000, in: 0.293, out: 2.253, cached_in: null, reasoning: false, paid_billing: false },
{ id: "@cf/meta/llama-4-scout-17b-16e-instruct", family: "llama", params: "17Bx16E", quant: "fp8", context: 131_000, in: 0.270, out: 0.850, cached_in: null, reasoning: false, paid_billing: false, vision: true },
{ id: "@cf/ibm-granite/granite-4.0-h-micro", family: "granite", params: "micro", quant: "n/a", context: 131_000, in: 0.017, out: 0.112, cached_in: null, reasoning: false, paid_billing: false },
{ id: "@cf/google/gemma-4-26b-a4b-it", family: "gemma", params: "26B-A4B", quant: "n/a", context: 256_000, in: 0.100, out: 0.300, cached_in: null, reasoning: false, paid_billing: false, vision: true },
{ id: "@cf/aisingapore/gemma-sea-lion-v4-27b-it", family: "sea-lion", params: "27B", quant: "n/a", context: 128_000, in: 0.351, out: 0.555, cached_in: null, reasoning: false, paid_billing: false },
{ id: "@cf/mistralai/mistral-small-3.1-24b-instruct", family: "mistral", params: "24B", quant: "fp16", context: 128_000, in: 0.351, out: 0.555, cached_in: null, reasoning: false, paid_billing: false, vision: true },
{ id: "@cf/qwen/qwen3-30b-a3b-fp8", family: "qwen", params: "30B-A3B", quant: "fp8", context: 32_000, in: 0.051, out: 0.335, cached_in: null, reasoning: true, reasoning_out_mult: 2, paid_billing: false },
{ id: "@cf/qwen/qwen2.5-coder-32b-instruct", family: "qwen", params: "32B", quant: "fp16", context: 32_000, in: 0.660, out: 1.000, cached_in: null, reasoning: false, paid_billing: false },
{ id: "@cf/qwen/qwen3.8-27b", family: "qwen", params: "27B", quant: "n/a", context: 262_000, in: 0.450, out: 3.200, cached_in: null, reasoning: true, reasoning_out_mult: 2, paid_billing: false, vision: true },
{ id: "@cf/qwen/qwq-32b", family: "qwen", params: "32B", quant: "fp16", context: 24_000, in: 0.660, out: 1.000, cached_in: null, reasoning: true, reasoning_out_mult: 3, paid_billing: false },
{ id: "@cf/openai/gpt-oss-20b", family: "gpt-oss", params: "20B", quant: "n/a", context: 128_000, in: 0.200, out: 0.300, cached_in: null, reasoning: true, reasoning_out_mult: 2, paid_billing: false },
{ id: "@cf/openai/gpt-oss-120b", family: "gpt-oss", params: "120B", quant: "n/a", context: 128_000, in: 0.350, out: 0.750, cached_in: null, reasoning: true, reasoning_out_mult: 2, paid_billing: false },
{ id: "@cf/nvidia/nemotron-3-120b-a12b", family: "nemotron", params: "120B-A12B", quant: "n/a", context: 256_000, in: 0.500, out: 1.500, cached_in: null, reasoning: true, reasoning_out_mult: 2, paid_billing: false },
{ id: "@cf/zai-org/glm-4.7-flash", family: "glm", params: "n/a", quant: "n/a", context: 131_000, in: 0.060, out: 0.400, cached_in: null, reasoning: false, paid_billing: false },
{ id: "@cf/zai-org/glm-5.3-flash", family: "glm", params: "320B/18B-active", quant: "n/a", context: 1_300_000, in: 0.150, out: 0.500, cached_in: 0.030, reasoning: false, paid_billing: true, vision: true },
{ id: "@cf/zai-org/glm-5.2", family: "glm", params: "n/a", quant: "n/a", context: 262_000, in: 1.400, out: 4.400, cached_in: 0.260, reasoning: false, paid_billing: true },
{ id: "@cf/zai-org/glm-5.3", family: "glm", params: "n/a", quant: "n/a", context: 1_300_000, in: 1.400, out: 4.400, cached_in: 0.260, reasoning: false, paid_billing: true },
{ id: "@cf/deepseek-ai/deepseek-r1-distill-qwen-32b", family: "deepseek", params: "32B", quant: "fp16", context: 80_000, in: 0.497, out: 4.881, cached_in: null, reasoning: true, reasoning_out_mult: 3, paid_billing: false },
{ id: "@cf/deepseek-ai/deepseek-v4-flash-0731", family: "deepseek", params: "n/a", quant: "n/a", context: 1_000_000, in: 0.440, out: 1.320, cached_in: 0.014, reasoning: true, reasoning_out_mult: 2, paid_billing: true },
{ id: "@cf/deepseek-ai/deepseek-v4-pro-0813", family: "deepseek", params: "n/a", quant: "n/a", context: 1_000_000, in: 1.320, out: 3.960, cached_in: 0.044, reasoning: true, reasoning_out_mult: 2, paid_billing: true },
{ id: "@cf/moonshotai/kimi-k2.6", family: "kimi", params: "1T MoE", quant: "n/a", context: 262_000, in: 0.950, out: 4.000, cached_in: 0.160, reasoning: false, paid_billing: true, vision: true },
{ id: "@cf/moonshotai/kimi-k2.7-code", family: "kimi", params: "1T MoE", quant: "n/a", context: 262_000, in: 0.950, out: 4.000, cached_in: 0.190, reasoning: false, paid_billing: true, vision: true },
];
export function findModel(id) {
return MODELS.find((m) => m.id === id) || null;
}
export function estimateCost(model, tokensIn, tokensOut) {
const inCost = (tokensIn / 1_000_000) * model.in;
const outCost = (tokensOut / 1_000_000) * model.out;
return inCost + outCost;
}
export const PILOT_MODELS = [
"@cf/meta/llama-3.1-8b-instruct-fp8",
"@cf/qwen/qwen3-30b-a3b-fp8",
"@cf/google/gemma-4-26b-a4b-it",
"@cf/openai/gpt-oss-120b",
"@cf/zai-org/glm-5.3",
];
if (process.argv[1] && process.argv[1].endsWith("models.js")) {
const format = process.argv.includes("--json") ? "json" : "table";
if (format === "json") {
console.log(JSON.stringify(MODELS, null, 2));
} else {
console.log("Workers AI interviewable models (26):");
for (const m of MODELS) {
console.log(
` ${m.id.padEnd(48)} ${String(m.params).padEnd(12)} in=$${m.in.toFixed(3)}/M out=$${m.out.toFixed(3)}/M ` +
(m.reasoning ? `reasoning(x${m.reasoning_out_mult}) ` : "") + (m.paid_billing ? " [paid billing]" : "")
);
}
console.log(`\nPilot set: ${PILOT_MODELS.join(", ")}`);
}
}