{"ok":true,"entity":{"id":"rlhf","name":"人間のフィードバックによる強化学習（RLHF）","entityType":"concept","canonicalName":"RLHF","displayName":"RLHF","category":"AI基礎概念","shortDescription":"人間の評価を報酬として用い、モデルの出力を人間の好みに沿わせる強化学習の手法。","primaryCluster":"ai-company","parentEntity":null,"verificationStatus":"draft","website":null,"updatedAt":"2026-07-10T01:43:25.958Z","secondaryClusters":[],"alias":["Reinforcement Learning from Human Feedback","RLHF"],"searchKeywords":["rlhf","人間のフィードバック","強化学習","アライメント"]},"referenceIndex":["P-01-001","P-02-001","P-04-001","P-05-001","P-02-002"]}