{"name":"Synthiq Labs — On-Device Small-Model Leaderboard","canonical":"https://synthiq.app/labs/leaderboard","methodology":"https://synthiq.app/labs/methodology","provenanceModel":"Every metric group is tagged self-run (measured by Synthiq Labs; device + config lockfile embedded), external (cited; live primary-source URL included), or pending (not yet measured — null, never zero). Derived metrics are computed from raw fields and list their component provenances.","coverage":{"models":9,"runtimesMeasured":0,"devicesMeasured":0,"selfRunRecords":0,"citedRecords":9,"pendingEfficiency":9},"rows":[{"id":"qwen3-0.6b--bf16","onParetoFrontier":true,"qualityHeadline":{"value":44.6,"label":"MMLU-Redux","kind":"external","sourceUrl":"https://arxiv.org/pdf/2505.09388","mode":"no-think"},"intelligencePerGb":{"value":29.733333333333334,"components":[{"name":"quality (MMLU-Redux)","kind":"external"},{"name":"on-disk size","kind":"external"}]},"intelligencePerWatt":{"value":null,"components":[{"name":"quality (MMLU-Redux)","kind":"external"},{"name":"avg decode power","kind":"pending"}]},"intelligencePerJoule":{"value":null,"components":[{"name":"quality (MMLU-Redux)","kind":"external"},{"name":"energy per response","kind":"pending"}]},"tokenEfficiency":{"value":null,"components":[{"name":"decode tok/s","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"ramClass":null,"record":{"id":"qwen3-0.6b--bf16","model":{"name":"Qwen3-0.6B","family":"Qwen3","paramsB":0.6,"license":"Apache-2.0","hfId":"Qwen/Qwen3-0.6B","cardUrl":"https://huggingface.co/Qwen/Qwen3-0.6B"},"quant":"bf16","runtime":{"name":"none","version":null},"device":null,"context":null,"workload":null,"seed":null,"date":"2026-07-23","quality":{"provenance":{"kind":"external","sourceUrl":"https://arxiv.org/pdf/2505.09388","sourceLabel":"Qwen3 Technical Report (arXiv:2505.09388), Tables 19–20","retrieved":"2026-07-23","note":"Publisher-reported. These tables exist only in the technical-report PDF — the blog and HF card carry no eval tables for the small sizes."},"tasks":[{"task":"MMLU-Redux","metric":"as printed","value":44.6,"mode":"no-think"},{"task":"MMLU-Redux","metric":"as printed","value":55.6,"mode":"think"},{"task":"GPQA-Diamond","metric":"as printed","value":22.9,"mode":"no-think"},{"task":"GPQA-Diamond","metric":"as printed","value":27.9,"mode":"think"},{"task":"IFEval strict prompt","metric":"as printed","value":54.5,"mode":"no-think"},{"task":"IFEval strict prompt","metric":"as printed","value":59.2,"mode":"think"},{"task":"MATH-500","metric":"as printed","value":55.2,"mode":"no-think"},{"task":"MATH-500","metric":"as printed","value":77.6,"mode":"think"}],"score":null,"scoreStderr":null,"headlineTask":0},"size":{"gb":1.5,"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/Qwen/Qwen3-0.6B/tree/main","sourceLabel":"HF file listing — bf16 safetensors, 1,503,300,328 bytes","retrieved":"2026-07-23"}},"efficiency":{"provenance":{"kind":"pending","note":"Awaiting Synthiq Labs device runs."},"decodeTokS":null,"prefillTokS":null,"ttftMs":null,"peakRamMb":null,"energyJ":null,"avgW":null,"peakW":null,"coldStartMs":null,"runs":null},"configLockfile":null,"notes":"Dual-mode model: thinking and non-thinking scores are published separately and shown separately here — never averaged. The headline uses the non-thinking mode (the default on-device chat configuration)."}},{"id":"llama-3.2-1b-instruct--bf16","onParetoFrontier":true,"qualityHeadline":{"value":49.3,"label":"MMLU","kind":"external","sourceUrl":"https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct"},"intelligencePerGb":{"value":19.95951417004048,"components":[{"name":"quality (MMLU)","kind":"external"},{"name":"on-disk size","kind":"external"}]},"intelligencePerWatt":{"value":null,"components":[{"name":"quality (MMLU)","kind":"external"},{"name":"avg decode power","kind":"pending"}]},"intelligencePerJoule":{"value":null,"components":[{"name":"quality (MMLU)","kind":"external"},{"name":"energy per response","kind":"pending"}]},"tokenEfficiency":{"value":null,"components":[{"name":"decode tok/s","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"ramClass":null,"record":{"id":"llama-3.2-1b-instruct--bf16","model":{"name":"Llama 3.2 1B-Instruct","family":"Llama 3.2","paramsB":1.23,"license":"Llama 3.2 Community License","hfId":"meta-llama/Llama-3.2-1B-Instruct","cardUrl":"https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct"},"quant":"bf16","runtime":{"name":"none","version":null},"device":null,"context":null,"workload":null,"seed":null,"date":"2026-07-23","quality":{"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct","sourceLabel":"Meta Llama 3.2 model card — instruction-tuned benchmark table","retrieved":"2026-07-23"},"tasks":[{"task":"MMLU","metric":"macro_avg/acc","value":49.3,"shots":5},{"task":"IFEval","metric":"as printed","value":59.5},{"task":"GPQA","metric":"acc","value":27.2,"shots":0},{"task":"GSM8K","metric":"em_maj1@1","value":44.4,"shots":8},{"task":"ARC-C","metric":"acc","value":59.4,"shots":0},{"task":"Hellaswag","metric":"acc","value":41.2,"shots":0}],"score":null,"scoreStderr":null,"headlineTask":0},"size":{"gb":2.47,"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main","sourceLabel":"HF file listing — bf16 safetensors, 2,471,645,608 bytes","retrieved":"2026-07-23"}},"efficiency":{"provenance":{"kind":"pending","note":"Awaiting Synthiq Labs device runs."},"decodeTokS":null,"prefillTokS":null,"ttftMs":null,"peakRamMb":null,"energyJ":null,"avgW":null,"peakW":null,"coldStartMs":null,"runs":null},"configLockfile":null,"notes":"Meta states the parameter count as \"1B (1.23B)\"; 1.23 is used here. Meta reports plain GPQA (not GPQA-Diamond) — the variant difference matters when eyeballing across rows."}},{"id":"qwen3-1.7b--bf16","onParetoFrontier":true,"qualityHeadline":{"value":64.4,"label":"MMLU-Redux","kind":"external","sourceUrl":"https://arxiv.org/pdf/2505.09388","mode":"no-think"},"intelligencePerGb":{"value":15.862068965517244,"components":[{"name":"quality (MMLU-Redux)","kind":"external"},{"name":"on-disk size","kind":"external"}]},"intelligencePerWatt":{"value":null,"components":[{"name":"quality (MMLU-Redux)","kind":"external"},{"name":"avg decode power","kind":"pending"}]},"intelligencePerJoule":{"value":null,"components":[{"name":"quality (MMLU-Redux)","kind":"external"},{"name":"energy per response","kind":"pending"}]},"tokenEfficiency":{"value":null,"components":[{"name":"decode tok/s","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"ramClass":null,"record":{"id":"qwen3-1.7b--bf16","model":{"name":"Qwen3-1.7B","family":"Qwen3","paramsB":1.7,"license":"Apache-2.0","hfId":"Qwen/Qwen3-1.7B","cardUrl":"https://huggingface.co/Qwen/Qwen3-1.7B"},"quant":"bf16","runtime":{"name":"none","version":null},"device":null,"context":null,"workload":null,"seed":null,"date":"2026-07-23","quality":{"provenance":{"kind":"external","sourceUrl":"https://arxiv.org/pdf/2505.09388","sourceLabel":"Qwen3 Technical Report (arXiv:2505.09388), Tables 19–20","retrieved":"2026-07-23","note":"Publisher-reported. These tables exist only in the technical-report PDF — the blog and HF card carry no eval tables for the small sizes."},"tasks":[{"task":"MMLU-Redux","metric":"as printed","value":64.4,"mode":"no-think"},{"task":"MMLU-Redux","metric":"as printed","value":73.9,"mode":"think"},{"task":"GPQA-Diamond","metric":"as printed","value":28.6,"mode":"no-think"},{"task":"GPQA-Diamond","metric":"as printed","value":40.1,"mode":"think"},{"task":"IFEval strict prompt","metric":"as printed","value":68.2,"mode":"no-think"},{"task":"IFEval strict prompt","metric":"as printed","value":72.5,"mode":"think"},{"task":"MATH-500","metric":"as printed","value":73,"mode":"no-think"},{"task":"MATH-500","metric":"as printed","value":93.4,"mode":"think"}],"score":null,"scoreStderr":null,"headlineTask":0},"size":{"gb":4.06,"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/Qwen/Qwen3-1.7B/tree/main","sourceLabel":"HF file listing — bf16 safetensors, 4,063,515,592 bytes","retrieved":"2026-07-23"}},"efficiency":{"provenance":{"kind":"pending","note":"Awaiting Synthiq Labs device runs."},"decodeTokS":null,"prefillTokS":null,"ttftMs":null,"peakRamMb":null,"energyJ":null,"avgW":null,"peakW":null,"coldStartMs":null,"runs":null},"configLockfile":null,"notes":"Dual-mode model: thinking and non-thinking scores are published separately and shown separately here — never averaged. The headline uses the non-thinking mode (the default on-device chat configuration)."}},{"id":"llama-3.2-3b-instruct--bf16","onParetoFrontier":false,"qualityHeadline":{"value":63.4,"label":"MMLU","kind":"external","sourceUrl":"https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct"},"intelligencePerGb":{"value":9.860031104199066,"components":[{"name":"quality (MMLU)","kind":"external"},{"name":"on-disk size","kind":"external"}]},"intelligencePerWatt":{"value":null,"components":[{"name":"quality (MMLU)","kind":"external"},{"name":"avg decode power","kind":"pending"}]},"intelligencePerJoule":{"value":null,"components":[{"name":"quality (MMLU)","kind":"external"},{"name":"energy per response","kind":"pending"}]},"tokenEfficiency":{"value":null,"components":[{"name":"decode tok/s","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"ramClass":null,"record":{"id":"llama-3.2-3b-instruct--bf16","model":{"name":"Llama 3.2 3B-Instruct","family":"Llama 3.2","paramsB":3.21,"license":"Llama 3.2 Community License","hfId":"meta-llama/Llama-3.2-3B-Instruct","cardUrl":"https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct"},"quant":"bf16","runtime":{"name":"none","version":null},"device":null,"context":null,"workload":null,"seed":null,"date":"2026-07-23","quality":{"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct","sourceLabel":"Meta Llama 3.2 model card — instruction-tuned benchmark table","retrieved":"2026-07-23"},"tasks":[{"task":"MMLU","metric":"macro_avg/acc","value":63.4,"shots":5},{"task":"IFEval","metric":"avg prompt/instruction acc, loose/strict","value":77.4,"shots":0},{"task":"GPQA","metric":"acc","value":32.8,"shots":0},{"task":"GSM8K","metric":"em_maj1@1","value":77.7,"shots":8},{"task":"MATH","metric":"final_em","value":48,"shots":0},{"task":"ARC-C","metric":"acc","value":78.6,"shots":0}],"score":null,"scoreStderr":null,"headlineTask":0},"size":{"gb":6.43,"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct/tree/main","sourceLabel":"HF file listing — bf16 safetensors, 6,425,529,048 bytes","retrieved":"2026-07-23"}},"efficiency":{"provenance":{"kind":"pending","note":"Awaiting Synthiq Labs device runs."},"decodeTokS":null,"prefillTokS":null,"ttftMs":null,"peakRamMb":null,"energyJ":null,"avgW":null,"peakW":null,"coldStartMs":null,"runs":null},"configLockfile":null,"notes":"Meta states the parameter count as \"3B (3.21B)\"; 3.21 is used here. Meta reports plain GPQA (not GPQA-Diamond)."}},{"id":"qwen3-4b--bf16","onParetoFrontier":true,"qualityHeadline":{"value":77.3,"label":"MMLU-Redux","kind":"external","sourceUrl":"https://arxiv.org/pdf/2505.09388","mode":"no-think"},"intelligencePerGb":{"value":9.614427860696518,"components":[{"name":"quality (MMLU-Redux)","kind":"external"},{"name":"on-disk size","kind":"external"}]},"intelligencePerWatt":{"value":null,"components":[{"name":"quality (MMLU-Redux)","kind":"external"},{"name":"avg decode power","kind":"pending"}]},"intelligencePerJoule":{"value":null,"components":[{"name":"quality (MMLU-Redux)","kind":"external"},{"name":"energy per response","kind":"pending"}]},"tokenEfficiency":{"value":null,"components":[{"name":"decode tok/s","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"ramClass":null,"record":{"id":"qwen3-4b--bf16","model":{"name":"Qwen3-4B","family":"Qwen3","paramsB":4,"license":"Apache-2.0","hfId":"Qwen/Qwen3-4B","cardUrl":"https://huggingface.co/Qwen/Qwen3-4B"},"quant":"bf16","runtime":{"name":"none","version":null},"device":null,"context":null,"workload":null,"seed":null,"date":"2026-07-23","quality":{"provenance":{"kind":"external","sourceUrl":"https://arxiv.org/pdf/2505.09388","sourceLabel":"Qwen3 Technical Report (arXiv:2505.09388), Tables 17–18","retrieved":"2026-07-23","note":"Publisher-reported. These tables exist only in the technical-report PDF — the blog and HF card carry no eval tables for the small sizes."},"tasks":[{"task":"MMLU-Redux","metric":"as printed","value":77.3,"mode":"no-think"},{"task":"MMLU-Redux","metric":"as printed","value":83.7,"mode":"think"},{"task":"GPQA-Diamond","metric":"as printed","value":41.7,"mode":"no-think"},{"task":"GPQA-Diamond","metric":"as printed","value":55.9,"mode":"think"},{"task":"IFEval strict prompt","metric":"as printed","value":81.2,"mode":"no-think"},{"task":"IFEval strict prompt","metric":"as printed","value":81.9,"mode":"think"},{"task":"MATH-500","metric":"as printed","value":84.8,"mode":"no-think"},{"task":"MATH-500","metric":"as printed","value":97,"mode":"think"}],"score":null,"scoreStderr":null,"headlineTask":0},"size":{"gb":8.04,"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/Qwen/Qwen3-4B/tree/main","sourceLabel":"HF file listing — bf16 safetensors, 8,044,982,000 bytes","retrieved":"2026-07-23"}},"efficiency":{"provenance":{"kind":"pending","note":"Awaiting Synthiq Labs device runs."},"decodeTokS":null,"prefillTokS":null,"ttftMs":null,"peakRamMb":null,"energyJ":null,"avgW":null,"peakW":null,"coldStartMs":null,"runs":null},"configLockfile":null,"notes":"Dual-mode model: thinking and non-thinking scores are published separately and shown separately here — never averaged. The headline uses the non-thinking mode (the default on-device chat configuration)."}},{"id":"phi-4-mini-instruct--bf16","onParetoFrontier":true,"qualityHeadline":{"value":67.3,"label":"MMLU","kind":"external","sourceUrl":"https://huggingface.co/microsoft/Phi-4-mini-instruct"},"intelligencePerGb":{"value":8.774445893089961,"components":[{"name":"quality (MMLU)","kind":"external"},{"name":"on-disk size","kind":"external"}]},"intelligencePerWatt":{"value":null,"components":[{"name":"quality (MMLU)","kind":"external"},{"name":"avg decode power","kind":"pending"}]},"intelligencePerJoule":{"value":null,"components":[{"name":"quality (MMLU)","kind":"external"},{"name":"energy per response","kind":"pending"}]},"tokenEfficiency":{"value":null,"components":[{"name":"decode tok/s","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"ramClass":null,"record":{"id":"phi-4-mini-instruct--bf16","model":{"name":"Phi-4-mini-instruct","family":"Phi-4","paramsB":3.8,"license":"MIT","hfId":"microsoft/Phi-4-mini-instruct","cardUrl":"https://huggingface.co/microsoft/Phi-4-mini-instruct"},"quant":"bf16","runtime":{"name":"none","version":null},"device":null,"context":null,"workload":null,"seed":null,"date":"2026-07-23","quality":{"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/microsoft/Phi-4-mini-instruct","sourceLabel":"Phi-4-mini-instruct model card — 'Model Quality' table","retrieved":"2026-07-23"},"tasks":[{"task":"MMLU","metric":"as printed","value":67.3,"shots":5},{"task":"MMLU-Pro","metric":"0-shot CoT","value":52.8,"shots":0},{"task":"GPQA","metric":"0-shot CoT","value":25.2,"shots":0},{"task":"BigBench Hard","metric":"0-shot CoT","value":70.4,"shots":0},{"task":"GSM8K","metric":"8-shot CoT","value":88.6,"shots":8},{"task":"MATH","metric":"0-shot CoT","value":64,"shots":0}],"score":null,"scoreStderr":null,"headlineTask":0},"size":{"gb":7.67,"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/microsoft/Phi-4-mini-instruct/tree/main","sourceLabel":"HF file listing — bf16 safetensors, 7,672,066,216 bytes","retrieved":"2026-07-23"}},"efficiency":{"provenance":{"kind":"pending","note":"Awaiting Synthiq Labs device runs."},"decodeTokS":null,"prefillTokS":null,"ttftMs":null,"peakRamMb":null,"energyJ":null,"avgW":null,"peakW":null,"coldStartMs":null,"runs":null},"configLockfile":null,"notes":"Microsoft reports plain GPQA with CoT (not GPQA-Diamond). The card publishes no IFEval or HumanEval rows, so none are listed here."}},{"id":"gemma-3-1b-it--bf16","onParetoFrontier":false,"qualityHeadline":{"value":null,"label":"quality","kind":"pending"},"intelligencePerGb":{"value":null,"components":[{"name":"quality (quality)","kind":"pending"},{"name":"on-disk size","kind":"external"}]},"intelligencePerWatt":{"value":null,"components":[{"name":"quality (quality)","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"intelligencePerJoule":{"value":null,"components":[{"name":"quality (quality)","kind":"pending"},{"name":"energy per response","kind":"pending"}]},"tokenEfficiency":{"value":null,"components":[{"name":"decode tok/s","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"ramClass":null,"record":{"id":"gemma-3-1b-it--bf16","model":{"name":"Gemma 3 1B-it","family":"Gemma 3","paramsB":1,"license":"Gemma Terms of Use","hfId":"google/gemma-3-1b-it","cardUrl":"https://huggingface.co/google/gemma-3-1b-it"},"quant":"bf16","runtime":{"name":"none","version":null},"device":null,"context":null,"workload":null,"seed":null,"date":"2026-07-23","quality":{"provenance":{"kind":"external","sourceUrl":"https://ai.google.dev/gemma/docs/core/model_card_3","sourceLabel":"Gemma 3 official model card — 'Gemma 3 IT 1B' column","retrieved":"2026-07-23","note":"Publisher-reported instruction-tuned scores. The HF repo card shows only the pre-trained tables for 1B; the IT numbers live on the ai.google.dev card."},"tasks":[{"task":"GPQA Diamond","metric":"as printed","value":19.2,"shots":0},{"task":"IFEval","metric":"as printed","value":80.2,"shots":0},{"task":"BIG-Bench Hard","metric":"as printed","value":39.1,"shots":0},{"task":"MMLU-Pro","metric":"as printed","value":14.7,"shots":0},{"task":"GSM8K","metric":"as printed","value":62.8,"shots":0},{"task":"HumanEval","metric":"as printed","value":41.5,"shots":0}],"score":null,"scoreStderr":null,"headlineTask":null},"size":{"gb":2,"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/google/gemma-3-1b-it/tree/main","sourceLabel":"HF file listing — bf16 safetensors, 1,999,811,208 bytes","retrieved":"2026-07-23"}},"efficiency":{"provenance":{"kind":"pending","note":"Awaiting Synthiq Labs device runs."},"decodeTokS":null,"prefillTokS":null,"ttftMs":null,"peakRamMb":null,"energyJ":null,"avgW":null,"peakW":null,"coldStartMs":null,"runs":null},"configLockfile":null,"notes":"No headline metric: Google publishes no plain-MMLU-family figure for the instruction-tuned model (MMLU-Pro is a different, harder benchmark and is not chart-comparable with other publishers' MMLU/MMLU-Redux figures). This row is excluded from the density ranking and the chart until the unified self-run harness pass lands; its cited per-task scores are all listed."}},{"id":"gemma-3-4b-it--bf16","onParetoFrontier":false,"qualityHeadline":{"value":null,"label":"quality","kind":"pending"},"intelligencePerGb":{"value":null,"components":[{"name":"quality (quality)","kind":"pending"},{"name":"on-disk size","kind":"external"}]},"intelligencePerWatt":{"value":null,"components":[{"name":"quality (quality)","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"intelligencePerJoule":{"value":null,"components":[{"name":"quality (quality)","kind":"pending"},{"name":"energy per response","kind":"pending"}]},"tokenEfficiency":{"value":null,"components":[{"name":"decode tok/s","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"ramClass":null,"record":{"id":"gemma-3-4b-it--bf16","model":{"name":"Gemma 3 4B-it","family":"Gemma 3","paramsB":4,"license":"Gemma Terms of Use","hfId":"google/gemma-3-4b-it","cardUrl":"https://huggingface.co/google/gemma-3-4b-it"},"quant":"bf16","runtime":{"name":"none","version":null},"device":null,"context":null,"workload":null,"seed":null,"date":"2026-07-23","quality":{"provenance":{"kind":"external","sourceUrl":"https://ai.google.dev/gemma/docs/core/model_card_3","sourceLabel":"Gemma 3 official model card — 'Gemma 3 IT 4B' column","retrieved":"2026-07-23","note":"Publisher-reported instruction-tuned scores. Caution: the pre-trained (PT) tables for this model differ hugely from the IT tables — these are IT."},"tasks":[{"task":"GPQA Diamond","metric":"as printed","value":30.8,"shots":0},{"task":"IFEval","metric":"as printed","value":90.2,"shots":0},{"task":"BIG-Bench Hard","metric":"as printed","value":72.2,"shots":0},{"task":"MMLU-Pro","metric":"as printed","value":43.6,"shots":0},{"task":"GSM8K","metric":"as printed","value":89.2,"shots":0},{"task":"HumanEval","metric":"as printed","value":71.3,"shots":0}],"score":null,"scoreStderr":null,"headlineTask":null},"size":{"gb":8.6,"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/google/gemma-3-4b-it/tree/main","sourceLabel":"HF file listing — bf16 safetensors, 8,600,277,880 bytes","retrieved":"2026-07-23"}},"efficiency":{"provenance":{"kind":"pending","note":"Awaiting Synthiq Labs device runs."},"decodeTokS":null,"prefillTokS":null,"ttftMs":null,"peakRamMb":null,"energyJ":null,"avgW":null,"peakW":null,"coldStartMs":null,"runs":null},"configLockfile":null,"notes":"No headline metric: Google publishes no plain-MMLU-family figure for the instruction-tuned model (MMLU-Pro is a different, harder benchmark and is not chart-comparable with other publishers' MMLU/MMLU-Redux figures). This row is excluded from the density ranking and the chart until the unified self-run harness pass lands; its cited per-task scores are all listed. Multimodal model — sizes here are the text weights as shipped."}},{"id":"smollm3-3b--bf16","onParetoFrontier":false,"qualityHeadline":{"value":null,"label":"quality","kind":"pending"},"intelligencePerGb":{"value":null,"components":[{"name":"quality (quality)","kind":"pending"},{"name":"on-disk size","kind":"external"}]},"intelligencePerWatt":{"value":null,"components":[{"name":"quality (quality)","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"intelligencePerJoule":{"value":null,"components":[{"name":"quality (quality)","kind":"pending"},{"name":"energy per response","kind":"pending"}]},"tokenEfficiency":{"value":null,"components":[{"name":"decode tok/s","kind":"pending"},{"name":"avg decode power","kind":"pending"}]},"ramClass":null,"record":{"id":"smollm3-3b--bf16","model":{"name":"SmolLM3-3B","family":"SmolLM3","paramsB":3,"license":"Apache-2.0","hfId":"HuggingFaceTB/SmolLM3-3B","cardUrl":"https://huggingface.co/HuggingFaceTB/SmolLM3-3B"},"quant":"bf16","runtime":{"name":"none","version":null},"device":null,"context":null,"workload":null,"seed":null,"date":"2026-07-23","quality":{"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/HuggingFaceTB/SmolLM3-3B","sourceLabel":"SmolLM3 model card — dual-mode instruct tables","retrieved":"2026-07-23"},"tasks":[{"task":"GPQA Diamond","metric":"as printed","value":35.7,"mode":"no-think"},{"task":"GPQA Diamond","metric":"as printed","value":41.7,"mode":"think"},{"task":"IFEval","metric":"as printed","value":76.7,"mode":"no-think"},{"task":"IFEval","metric":"as printed","value":71.2,"mode":"think"},{"task":"GSM-Plus","metric":"as printed","value":72.8,"mode":"no-think"},{"task":"GSM-Plus","metric":"as printed","value":83.4,"mode":"think"},{"task":"LiveCodeBench v4","metric":"as printed","value":15.2,"mode":"no-think"},{"task":"LiveCodeBench v4","metric":"as printed","value":30,"mode":"think"}],"score":null,"scoreStderr":null,"headlineTask":null},"size":{"gb":6.15,"provenance":{"kind":"external","sourceUrl":"https://huggingface.co/HuggingFaceTB/SmolLM3-3B/tree/main","sourceLabel":"HF file listing — bf16 safetensors, 6,150,235,008 bytes","retrieved":"2026-07-23"}},"efficiency":{"provenance":{"kind":"pending","note":"Awaiting Synthiq Labs device runs."},"decodeTokS":null,"prefillTokS":null,"ttftMs":null,"peakRamMb":null,"energyJ":null,"avgW":null,"peakW":null,"coldStartMs":null,"runs":null},"configLockfile":null,"notes":"No headline metric: the instruct tables carry no MMLU-family figure (the base-model table reports MMLU-CF, a contamination-free variant not comparable with other publishers' MMLU numbers), so this row is excluded from the density ranking and chart until the self-run harness pass lands. Dual-mode model — think and no-think scores listed separately, never averaged."}}]}