{"event":{"id":"evt-flamingo-20220428","dedupe_key":"multimodal_model_release:2022-04-28:deepmind-introduces-flamingo-visual-language-model","event_type":"multimodal_model_release","title":"DeepMind introduces Flamingo visual language model","summary":"Flamingo accepted interleaved images, video and text and performed many multimodal tasks from few-shot prompts without task-specific fine-tuning.","occurred_at":"2022-04-28T00:00:00.000Z","published_at":"2022-04-28T00:00:00.000Z","observed_at":"2026-10-01T18:14:00.000Z","reconstructed_at":"2026-10-01T18:14:00.000Z","ingest_type":"live_world_scan","locations":null,"status":"verified","confidence":0.98,"metadata":{"actors":["DeepMind"],"what_happened_at_time":"A single visual-language model was shown adapting to diverse image/video-language tasks through its prompt interface.","evidence_at_time":"DeepMind's April 28 announcement and April 29 preprint documented the interface and benchmark results.","affected_layers":["layer-ai","layer-models","layer-research","layer-software","layer-human"],"change_kind":"capability_broadening","q1_continuity":"Q1's high-profile reconstruction was predominantly language/model/code centric. Flamingo broadened the prompt-mediated capability surface across modalities.","retrospective_inference":"Multimodality was becoming a general interface property of foundation-style models rather than a collection of isolated task-specific pipelines.","uncertainty":"Flamingo was a research system, not evidence that multimodal generality was robust in uncontrolled real-world deployment."},"created_at":"2026-10-01T19:22:17.746Z","updated_at":"2026-10-01T19:22:17.746Z"},"artifacts":[{"id":"art-flamingo-20220428","canonical_url":"https://deepmind.google/blog/tackling-multiple-tasks-with-a-single-visual-language-model/","artifact_type":"research_release","title":"Flamingo visual language model","summary":"DeepMind introduced Flamingo, a visual language model accepting interleaved image, video and text prompts and performing few-shot multimodal tasks without task-specific fine-tuning.","creator_entities":["DeepMind"],"released_at":"2022-04-28T00:00:00.000Z","content_hash":null,"metadata":{"historical_scope":"2022-Q2"},"created_at":"2026-10-01T19:22:17.740Z","updated_at":"2026-10-01T19:22:17.740Z","relation":"documented_by"}],"signals":[{"id":"sig-q2-multimodal-capability","statement":"Model interfaces broadened materially beyond text toward interleaved vision-language understanding and large-scale text-guided image generation in real user workflows.","signal_type":"capability_broadening","direction":"increasing","confidence":0.96,"epistemic_status":"supported_inference","mapping_mode":"retrospective","reconstructed_at":"2026-10-01T18:14:00.000Z","created_at":"2026-10-01T19:22:17.752Z","updated_at":"2026-10-01T19:22:17.752Z"}]}