{"artifact":{"id":"art-redpajama-20230417","canonical_url":"https://www.together.ai/blog/redpajama","artifact_type":"open_training_dataset_release","title":"RedPajama 1.2T-token dataset","summary":"An open 1.2T-token dataset intended to reproduce an important part of the LLaMA training-data stack.","creator_entities":["Together and collaborators"],"released_at":"2023-04-17T00:00:00.000Z","content_hash":null,"metadata":{"historical_scope":"2023-Q2"},"created_at":"2026-10-07T14:04:48.584Z","updated_at":"2026-10-07T14:04:48.584Z"},"sources":[{"id":"src-together-redpajama-20230417","name":"RedPajama: an Open Dataset for Training Large Language Models","publisher":"Together","url":"https://www.together.ai/blog/redpajama","source_type":"open_dataset_release","primary_or_secondary":"primary","retrieved_at":"2026-10-07T12:39:09.000Z","published_at":"2023-04-17T00:00:00.000Z","author":"Together and collaborators","metadata":{"language":"en","jurisdiction":"global","time_precision":"day"},"created_at":"2026-10-07T14:04:48.576Z","updated_at":"2026-10-07T14:04:48.576Z"}]}