{ "lesson": "26-relation-extraction-kg", "title": "Relation Extraction & Knowledge Graph Construction", "questions": [ { "stage": "pre", "question": "What is the atomic unit of a knowledge graph?", "options": [ "A token", "A POS tag", "A (subject, relation, object) triple", "A sentence" ], "correct": 2, "explanation": "KGs store information as (s, r, o) triples; aggregated triples form the graph." }, { "stage": "pre", "question": "What does AEVS stand for in 2026 relation extraction?", "options": [ "Aggregated Entity-Value Schema", "Async Entity Validation Service", "Auto-Encoder Vector Search", "Anchor-Extraction-Verification-Supplement: anchor spans, extract triples, verify against source, supplement coverage" ], "correct": 3, "explanation": "AEVS is the 2026 hallucination-mitigation framework for grounded RE." }, { "stage": "check", "question": "Why must each triple carry source provenance (doc id + span)?", "options": [ "Provenance is required by SPARQL", "It speeds up extraction", "Provenance changes the ontology", "Provenance lets you audit triples and reject hallucinations whose spans do not match the source text" ], "correct": 3, "explanation": "Provenance enables auditing and is the core of AEVS-style hallucination detection." }, { "stage": "check", "question": "What does canonicalization of relations do?", "options": [ "Adds embeddings", "Removes triples", "Maps surface verb phrases (e.g. 'was born in', 'is a native of') onto a fixed property id so the graph is queryable", "Translates the document" ], "correct": 2, "explanation": "Canonicalization collapses paraphrases into canonical KG property ids." }, { "stage": "check", "question": "Why does relation extraction usually need coreference resolution first?", "options": [ "Pronouns like 'he founded Apple' must be resolved to a named entity before triple extraction", "Coref normalizes case", "Coref provides positions", "Coref adds embeddings" ], "correct": 0, "explanation": "Without coref, RE attaches relations to pronouns instead of the underlying named entity." }, { "stage": "post", "question": "Which choice trades open IE recall for graph queryability?", "options": [ "Random sampling of triples", "Embedding-only graphs", "Mapping open-IE relations onto a closed ontology (e.g. Wikidata properties) before merging into the KG", "Skipping NER" ], "correct": 2, "explanation": "Closed ontologies make the graph queryable; the canonicalization step pays for itself downstream." }, { "stage": "post", "question": "Why do many production KGs use temporal qualifiers (start/end time)?", "options": [ "Required by RDF", "To remove NIL entities", "Faster SPARQL", "Many relations are time-bounded (employer, spouse, role); qualifiers prevent 'forever true' claims that go stale" ], "correct": 3, "explanation": "Time-bounded relations need qualifiers (e.g. Wikidata P580/P582) or facts go silently stale." }, { "stage": "post", "question": "What is REBEL?", "options": [ "A tokenizer", "A coreference model", "A seq2seq relation extractor that outputs triples already in Wikidata property ids", "A vector database" ], "correct": 2, "explanation": "REBEL (Babelscape) is a seq2seq RE model trained on distantly supervised Wikidata triples." } ] }