mirror of
https://github.com/rohitg00/ai-engineering-from-scratch.git
synced 2026-10-02 01:54:39 +08:00
feat(phase-18/11): add quiz.json
This commit is contained in:
@@ -0,0 +1,78 @@
|
||||
{
|
||||
"lesson": "11-scalable-oversight-weak-to-strong",
|
||||
"title": "Scalable Oversight and Weak-to-Strong Generalization",
|
||||
"questions": [
|
||||
{
|
||||
"stage": "pre",
|
||||
"question": "What is the core question the Superalignment agenda asks?",
|
||||
"options": [
|
||||
"Can a weaker overseer reliably supervise a stronger, aligned model?",
|
||||
"Can RLHF scale beyond 1B parameters?",
|
||||
"Are humans more accurate than reward models?",
|
||||
"Can adversarial training remove all backdoors?"
|
||||
],
|
||||
"correct": 0,
|
||||
"explanation": ""
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "In Burns et al. (2023) weak-to-strong generalization, what is the setup?",
|
||||
"options": [
|
||||
"GPT-2 is fine-tuned on GPT-4 labels",
|
||||
"A strong model (GPT-4 class) is fine-tuned on labels produced by a weak model (GPT-2 class) and its capability is measured against the strong supervised ceiling",
|
||||
"Two strong models debate; a human judges",
|
||||
"A reward model is trained on synthetic data only"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": ""
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "How is Performance Gap Recovered (PGR) defined?",
|
||||
"options": [
|
||||
"(weak - ceiling) / ceiling",
|
||||
"(fine-tuned - weak) / (ceiling - weak)",
|
||||
"(ceiling - fine-tuned) / weak",
|
||||
"fine-tuned / ceiling"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": ""
|
||||
},
|
||||
{
|
||||
"stage": "check",
|
||||
"question": "Which of the following is NOT a scalable-oversight mechanism listed in the lesson?",
|
||||
"options": [
|
||||
"Debate",
|
||||
"Recursive Reward Modeling",
|
||||
"Task Decomposition",
|
||||
"Tokenizer Distillation"
|
||||
],
|
||||
"correct": 3,
|
||||
"explanation": ""
|
||||
},
|
||||
{
|
||||
"stage": "post",
|
||||
"question": "How are scalable oversight and W2SG complementary?",
|
||||
"options": [
|
||||
"They both require white-box access",
|
||||
"Scalable oversight improves the overseer's effective signal quality; W2SG closes the gap from whatever imperfect signal the overseer provides",
|
||||
"They both target the same metric",
|
||||
"They are mutually exclusive"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": ""
|
||||
},
|
||||
{
|
||||
"stage": "post",
|
||||
"question": "What is the central assumption of the AI-safety-via-debate mechanism?",
|
||||
"options": [
|
||||
"Both debaters always tell the truth",
|
||||
"Finding a convincing true answer is easier than finding a convincing false answer",
|
||||
"The judge is more capable than the debaters",
|
||||
"Debate transcripts are always shorter than direct answers"
|
||||
],
|
||||
"correct": 1,
|
||||
"explanation": ""
|
||||
}
|
||||
]
|
||||
}
|
||||
Reference in New Issue
Block a user