{"event":{"id":"evt-constitutional-ai-20221215","dedupe_key":"alignment_training_research:2022-12-15:anthropic-publishes-constitutional-ai-and-reinforcement-learning-from-ai-feedback","event_type":"alignment_training_research","title":"Anthropic publishes Constitutional AI and reinforcement learning from AI feedback","summary":"Anthropic published a method in which models critique and revise outputs according to written principles and use AI-generated preferences for reinforcement learning.","occurred_at":"2022-12-15T00:00:00.000Z","published_at":"2022-12-15T00:00:00.000Z","observed_at":"2026-10-03T14:26:11.000Z","reconstructed_at":"2026-10-03T14:26:11.000Z","ingest_type":"live_world_scan","locations":null,"status":"verified","confidence":0.98,"metadata":{"actors":["Anthropic"],"what_happened_at_time":"Model supervision experiments extended beyond direct human preference labels toward principle-guided self-critique and AI-generated preference feedback.","evidence_at_time":"Anthropic December 15 release and arXiv paper.","affected_layers":["layer-research","layer-models","layer-governance","layer-human"],"change_kind":"alignment_supervision_maturation","q3_continuity":"Q1/Q2 established RLHF and human preference shaping; ChatGPT operationalized RLHF publicly. Constitutional AI explored a different scalable supervision architecture.","retrospective_inference":"Alignment research was diversifying its supervision mechanisms.","uncertainty":"Research only; does not establish human oversight was unnecessary or method broadly deployed."},"created_at":"2026-10-03T14:45:10.918Z","updated_at":"2026-10-03T14:45:10.918Z"},"artifacts":[{"id":"art-constitutional-ai-20221215","canonical_url":"https://www.anthropic.com/news/constitutional-ai-harmlessness-from-ai-feedback","artifact_type":"alignment_research_release","title":"Constitutional AI / RLAIF","summary":"Anthropic presented a training method using principles, model-generated critiques and revisions, and reinforcement learning from AI feedback.","creator_entities":["Anthropic"],"released_at":"2022-12-15T00:00:00.000Z","content_hash":null,"metadata":{"historical_scope":"2022-Q4"},"created_at":"2026-10-03T14:45:10.910Z","updated_at":"2026-10-03T14:45:10.910Z","relation":"evidenced_by"}],"signals":[{"id":"sig-q4-alignment-supervision-diversification","statement":"Alignment and behavior-shaping research diversified beyond direct human preference labels toward combinations of RLHF, explicit principles, self-critique and AI-generated preference feedback.","signal_type":"alignment_supervision_maturation","direction":"increasing","confidence":0.94,"epistemic_status":"supported_inference","mapping_mode":"retrospective","reconstructed_at":"2026-10-03T14:26:11.000Z","created_at":"2026-10-03T14:45:10.922Z","updated_at":"2026-10-03T14:45:10.922Z"}]}