feat(phase-18/11): add quiz.json

This commit is contained in:
Rohit Ghumare
2026-05-23 01:04:54 +01:00
parent 19427de8a2
commit 4fe2bb3858
@@ -0,0 +1,78 @@
{
"lesson": "11-scalable-oversight-weak-to-strong",
"title": "Scalable Oversight and Weak-to-Strong Generalization",
"questions": [
{
"stage": "pre",
"question": "What is the core question the Superalignment agenda asks?",
"options": [
"Can a weaker overseer reliably supervise a stronger, aligned model?",
"Can RLHF scale beyond 1B parameters?",
"Are humans more accurate than reward models?",
"Can adversarial training remove all backdoors?"
],
"correct": 0,
"explanation": ""
},
{
"stage": "check",
"question": "In Burns et al. (2023) weak-to-strong generalization, what is the setup?",
"options": [
"GPT-2 is fine-tuned on GPT-4 labels",
"A strong model (GPT-4 class) is fine-tuned on labels produced by a weak model (GPT-2 class) and its capability is measured against the strong supervised ceiling",
"Two strong models debate; a human judges",
"A reward model is trained on synthetic data only"
],
"correct": 1,
"explanation": ""
},
{
"stage": "check",
"question": "How is Performance Gap Recovered (PGR) defined?",
"options": [
"(weak - ceiling) / ceiling",
"(fine-tuned - weak) / (ceiling - weak)",
"(ceiling - fine-tuned) / weak",
"fine-tuned / ceiling"
],
"correct": 1,
"explanation": ""
},
{
"stage": "check",
"question": "Which of the following is NOT a scalable-oversight mechanism listed in the lesson?",
"options": [
"Debate",
"Recursive Reward Modeling",
"Task Decomposition",
"Tokenizer Distillation"
],
"correct": 3,
"explanation": ""
},
{
"stage": "post",
"question": "How are scalable oversight and W2SG complementary?",
"options": [
"They both require white-box access",
"Scalable oversight improves the overseer's effective signal quality; W2SG closes the gap from whatever imperfect signal the overseer provides",
"They both target the same metric",
"They are mutually exclusive"
],
"correct": 1,
"explanation": ""
},
{
"stage": "post",
"question": "What is the central assumption of the AI-safety-via-debate mechanism?",
"options": [
"Both debaters always tell the truth",
"Finding a convincing true answer is easier than finding a convincing false answer",
"The judge is more capable than the debaters",
"Debate transcripts are always shorter than direct answers"
],
"correct": 1,
"explanation": ""
}
]
}