{"artifact":{"id":"art-deepmind-chinchilla-20220329","canonical_url":"https://arxiv.org/abs/2203.15556","artifact_type":"paper_preprint","title":"Training Compute-Optimal Large Language Models","summary":"DeepMind reported that many large language models were undertrained for their compute budgets and that model size and training tokens should scale together under its empirical compute-optimal analysis.","creator_entities":["DeepMind"],"released_at":"2022-03-29T00:00:00.000Z","content_hash":null,"metadata":{"historical_scope":"2022-Q1"},"created_at":"2026-10-01T18:05:36.362Z","updated_at":"2026-10-01T18:05:36.362Z"},"sources":[{"id":"src-arxiv-chinchilla-20220329","name":"Training Compute-Optimal Large Language Models","publisher":"arXiv","url":"https://arxiv.org/abs/2203.15556","source_type":"paper_preprint","primary_or_secondary":"primary","retrieved_at":"2026-10-01T17:04:00.000Z","published_at":"2022-03-29T00:00:00.000Z","author":"Jordan Hoffmann et al.","metadata":{"language":"en","jurisdiction":"global","time_precision":"day","note":"Q1 evidence is the March 29 arXiv submission; DeepMind's later April blog is not used as contemporaneous Q1 evidence."},"created_at":"2026-10-01T18:05:36.357Z","updated_at":"2026-10-01T18:05:36.357Z"}]}