{ "schema": "distinct.model-assessments/2", "compiled": "2026-08-31", "supersedes": "distinct.model-assessments/1", "what_changed": [ "Letter grades are gone. Version 1 carried two ordinal scales (impact A-F, evidence A-F) and a coverage fraction. Both were assessor-created rankings layered on top of the published numbers, and the numbers are now good enough that the ranking added nothing except a way to be wrong. A release either has a published figure, which is shown as published with its primary source, or it has none, which is shown as Missing.", "Every displayed figure names its method (Measured, Modelled, Proxy-based) and links to the document it came from, so a reader can check the claim rather than trust the letter." ], "rules": [ "One route only: a published, model-specific result, used exactly as published. No GPU-hours, FLOPs, tokens, parameters, TDP, PUE or grid factor is converted here into MWh, tonnes or litres.", "Missing is not zero, not a low score, and not a grade. It means nobody published a usable result.", "A figure belongs to the release that was assessed. Where the assessed release and the shipped checkpoint differ, `covers` says so and the difference is not papered over.", "Cradle to release. Never added to, never averaged with, and never compared against the use-phase energy this network measures per run." ], "website_display": { "warning": "These figures are comparable within a lab and rarely between labs. The two Ai2 releases below were measured by the same team, on the same clusters, with the same method and the same boundary, so the difference between them is real. Every number covers the final pretraining run only: Ai2's own follow-up work found that development and failed runs accounted for 82.2% of total GPU-hours, so every figure on this page is a floor.", "further_reading": [ { "label": "Ai2 and CMU on measuring a model's footprint (ICLR 2025)", "url": "https://arxiv.org/abs/2503.05804" }, { "label": "Ai2 on what pretraining figures leave out (2026)", "url": "https://arxiv.org/abs/2605.01158" }, { "label": "ITU-T L.1801, the first international standard for assessing AI environmental impact", "url": "https://www.itu.int/epublications/publication/itu-t-l-1801-2026-02-guidelines-for-assessing-the-environmental-impact-of-artificial-intelligence-systems" } ], "models": [ { "model_id": "olmoe-1b-7b-0924", "name": "OLMoE 1B-7B 0924", "publisher": "Allen Institute for AI", "covers": "The 0924 pretraining run. The instruction tuning that produced the Instruct checkpoint is not separately costed in the source, so coverage is partial.", "coverage": "Partial", "summary": "The lowest published footprint of any assessed model at a size this network can run, on all three disclosed areas at once. It is a mixture-of-experts model: 6.9B parameters live in the file, but only about 1.3B are active for any given token, which is why its training cost roughly a third of the dense 7B trained by the same team on the same cluster. Ai2 sampled real GPU power at sub-second intervals rather than assuming chips draw their rated wattage, which makes the energy figure one of very few genuinely measured training numbers in the published literature. Carbon and water are then derived from that measured energy using the site's own grid factor and cooling efficiency.", "categories": { "energy": { "state": "reported", "status": "Measured", "value": "54", "unit": "MWh", "note": "Final pretraining run. GPU power sampled at sub-second intervals on the Jupiter cluster, Texas.", "source_url": "https://arxiv.org/abs/2503.05804" }, "climate": { "state": "reported", "status": "Modelled", "value": "18", "unit": "tCO2e", "note": "Measured energy at the Austin Energy grid factor of 0.332 kg CO2 per kWh, PUE 1.2. Location-based; no offset applied.", "source_url": "https://arxiv.org/abs/2503.05804" }, "water": { "state": "reported", "status": "Modelled", "value": "70", "unit": "kL", "note": "Measured energy at the Jupiter cluster's water usage effectiveness of 1.29 L per kWh. On-site cooling only; the water consumed upstream at the power stations is not counted.", "source_url": "https://arxiv.org/abs/2503.05804" }, "land": { "state": "missing", "note": "No published per-model land or biodiversity figure exists for any model, from any lab. ITU-T L.1801 lists the area but states the methodology is still developing." }, "materials": { "state": "missing", "note": "Ai2 report 22 tCO2e and 4.8 kL of embodied hardware impact across their whole programme, amortised over a four-year GPU lifespan. That is a programme total and is not attributable to this release." }, "pollution": { "state": "missing", "note": "Not assessed in the source." } }, "links": [ { "label": "Model card", "url": "https://huggingface.co/allenai/OLMoE-1B-7B-0924-Instruct" }, { "label": "Pinned weights", "url": "https://huggingface.co/allenai/OLMoE-1B-7B-0924-Instruct-GGUF" }, { "label": "Published assessment", "url": "https://arxiv.org/abs/2503.05804" } ] }, { "model_id": "olmo-2-1124-7b", "name": "OLMo 2 7B", "publisher": "Allen Institute for AI", "covers": "The 1124 pretraining run. The instruction tuning that produced the Instruct checkpoint is not separately costed in the source, so coverage is partial.", "coverage": "Partial", "summary": "A dense 7B trained by the same team, in the same data centre, in the same year, and measured the same way as OLMoE above. That makes the pair one of the very few honest like-for-like comparisons available anywhere in this field: about three times the energy, three times the carbon and three times the water for a model of comparable capability. Offered here as the alternative if the mixture-of-experts architecture gives trouble locally, not because anything about it is lighter.", "categories": { "energy": { "state": "reported", "status": "Measured", "value": "157", "unit": "MWh", "note": "Final pretraining run. GPU power sampled at sub-second intervals on the Jupiter cluster, Texas.", "source_url": "https://arxiv.org/abs/2503.05804" }, "climate": { "state": "reported", "status": "Modelled", "value": "52", "unit": "tCO2e", "note": "Measured energy at the Austin Energy grid factor of 0.332 kg CO2 per kWh, PUE 1.2. Location-based; no offset applied.", "source_url": "https://arxiv.org/abs/2503.05804" }, "water": { "state": "reported", "status": "Modelled", "value": "202", "unit": "kL", "note": "Measured energy at the Jupiter cluster's water usage effectiveness of 1.29 L per kWh. The 13B sibling drank four times as much on the same architecture because it trained in Iowa, where the figure is 3.1 L per kWh. Water tracks the site more than the model.", "source_url": "https://arxiv.org/abs/2503.05804" }, "land": { "state": "missing", "note": "No published per-model land or biodiversity figure exists for any model, from any lab. ITU-T L.1801 lists the area but states the methodology is still developing." }, "materials": { "state": "missing", "note": "Ai2 report 22 tCO2e and 4.8 kL of embodied hardware impact across their whole programme, amortised over a four-year GPU lifespan. That is a programme total and is not attributable to this release." }, "pollution": { "state": "missing", "note": "Not assessed in the source." } }, "links": [ { "label": "Model card", "url": "https://huggingface.co/allenai/OLMo-2-1124-7B-Instruct" }, { "label": "Pinned weights", "url": "https://huggingface.co/allenai/OLMo-2-1124-7B-Instruct-GGUF" }, { "label": "Published assessment", "url": "https://arxiv.org/abs/2503.05804" } ] } ] } }