mirror of
https://github.com/rohitg00/ai-engineering-from-scratch.git
synced 2026-10-02 01:54:39 +08:00
Every "Test Your Understanding" quiz placed the correct answer in option B.
Across the 2026 questions in 338 quiz files the correct answer sat at index 1
in 61.5% of cases (uniform would be ~25%), and 107 files had every answer at B,
making the quizzes guessable without reading them.
scripts/debias_quizzes.py rewrites each question's option order with a
deterministic, content-seeded permutation and updates the correct index to
follow the moved answer. It is idempotent: options are canonicalised to a sorted
base before permuting, so re-running produces byte-identical output. Questions
whose options reference each other by position ("all of the above", "both A and
B") are left untouched. The correct-answer value, the option set, and every
explanation are preserved exactly; only order and the index change.
Result: A 23.8% / B 26.3% / C 23.5% / D 26.4%.
The script doubles as a CI guard: `--check` exits non-zero if any quiz is not
de-biased, wired into the curriculum workflow so new lessons cannot regress.
Fixes #368
103 lines
4.0 KiB
JSON
103 lines
4.0 KiB
JSON
{
|
|
"lesson": "16-text-generation-pre-transformer",
|
|
"title": "Text Generation Before Transformers — N-gram Language Models",
|
|
"questions": [
|
|
{
|
|
"stage": "pre",
|
|
"question": "What does an n-gram language model estimate?",
|
|
"options": [
|
|
"Document embeddings",
|
|
"Edit distance between words",
|
|
"P(next word | previous n-1 words) from count statistics",
|
|
"P(label | document)"
|
|
],
|
|
"correct": 2,
|
|
"explanation": "An n-gram LM models P(w | last n-1 words) via counted occurrences."
|
|
},
|
|
{
|
|
"stage": "pre",
|
|
"question": "What problem does smoothing solve in n-gram models?",
|
|
"options": [
|
|
"Numerical precision",
|
|
"Zero-probability assignment to n-grams unseen in training, which collapses sentence likelihoods to zero",
|
|
"Memory usage",
|
|
"Tokenization mismatch"
|
|
],
|
|
"correct": 1,
|
|
"explanation": "Smoothing reallocates probability mass so unseen n-grams get non-zero probability."
|
|
},
|
|
{
|
|
"stage": "check",
|
|
"question": "What insight makes Kneser-Ney smoothing better than naive absolute discounting?",
|
|
"options": [
|
|
"It uses bigger n",
|
|
"It uses TF-IDF",
|
|
"It uses gradient descent",
|
|
"It estimates the lower-order distribution with continuation probability (number of distinct contexts a word appears in) instead of raw frequency"
|
|
],
|
|
"correct": 3,
|
|
"explanation": "Continuation probability gives credit for context diversity, not just raw count."
|
|
},
|
|
{
|
|
"stage": "check",
|
|
"question": "What does perplexity measure?",
|
|
"options": [
|
|
"Number of distinct n-grams",
|
|
"exp of the average negative log-likelihood per token on a held-out test set; lower is better",
|
|
"Cross-entropy of labels",
|
|
"Throughput of generation"
|
|
],
|
|
"correct": 1,
|
|
"explanation": "Perplexity = exp(- mean log P); lower means the model is less surprised by the test text."
|
|
},
|
|
{
|
|
"stage": "check",
|
|
"question": "Why must train and test sets use identical tokenization when comparing perplexity numbers?",
|
|
"options": [
|
|
"To avoid OOV",
|
|
"Perplexity depends on the tokenization scheme; mismatched tokenizers produce noncomparable scores",
|
|
"Required by gradient descent",
|
|
"To control batch size"
|
|
],
|
|
"correct": 1,
|
|
"explanation": "Different tokenizations change the token count and likelihood, making perplexity values incomparable."
|
|
},
|
|
{
|
|
"stage": "post",
|
|
"question": "Why do generated trigram-LM sentences feel locally fluent but globally incoherent?",
|
|
"options": [
|
|
"Local trigram context guides each next word but the model has no long-range memory beyond n-1 tokens",
|
|
"They drop punctuation",
|
|
"They use Laplace smoothing",
|
|
"Beam search fails"
|
|
],
|
|
"correct": 0,
|
|
"explanation": "Conditioning only on the last n-1 tokens makes long-range coherence accidental."
|
|
},
|
|
{
|
|
"stage": "post",
|
|
"question": "Where do n-gram models still ship in production in 2026?",
|
|
"options": [
|
|
"Multilingual translation",
|
|
"Summarization",
|
|
"Open-domain chatbots",
|
|
"Latency-critical paths like speech recognition rescoring and on-device autocomplete via libraries such as KenLM"
|
|
],
|
|
"correct": 3,
|
|
"explanation": "KenLM-style n-gram models still serve as fast on-device or rescoring components."
|
|
},
|
|
{
|
|
"stage": "post",
|
|
"question": "Why is computing an n-gram baseline before declaring a neural LM 'good' still recommended?",
|
|
"options": [
|
|
"It speeds up training",
|
|
"Required by ROUGE",
|
|
"It removes OOV",
|
|
"If a transformer LM does not beat a tuned Kneser-Ney baseline by a wide margin on the same tokenization, something is off in the training pipeline"
|
|
],
|
|
"correct": 3,
|
|
"explanation": "KN baselines are surprisingly strong; a neural LM should win by a large margin or you have a bug."
|
|
}
|
|
]
|
|
}
|