mirror of
https://github.com/rohitg00/ai-engineering-from-scratch.git
synced 2026-10-02 01:54:39 +08:00
chore(site): rebuild data.js
This commit is contained in:
+2
-2
@@ -1,5 +1,5 @@
|
||||
// Auto-generated by build.js — do not edit manually.
|
||||
// Last built: 2026-09-27T05:58:16.242Z
|
||||
// Last built: 2026-09-27T08:37:55.913Z
|
||||
|
||||
const ROADMAP_PREREQS = {
|
||||
"0": [],
|
||||
@@ -1885,7 +1885,7 @@ const PHASES = [
|
||||
"lang": "Python",
|
||||
"url": "https://github.com/rohitg00/ai-engineering-from-scratch/tree/main/phases/10-llms-from-scratch/11-quantization/",
|
||||
"summary": "A 70B model in FP16 needs 140GB. Two A100s just for weights. Quantize to FP8: one 80GB GPU. INT4: a MacBook.",
|
||||
"keywords": "Number Formats: What Each Bit Does · How Quantization Works · Sensitivity Hierarchy · PTQ vs QAT · GPTQ, AWQ, GGUF · Quality Measurement · Real Numbers · Step 1: Number Format Representations · Step 2: Symmetric Quantization (Per-Tensor and Per-Channel) · Step 3: Quality Measurement · Step 4: Bit-Width Sweep · Step 5: Sensitivity Experiment · Step 6: Simulated GPTQ · Step 7: AWQ Simulation · Step 8: Full Pipeline · Quantizing with AutoGPTQ · Quantizing with AutoAWQ · Converting to GGUF · Serving quantized models"
|
||||
"keywords": "Number Formats: What Each Bit Does · How Quantization Works · Sensitivity Hierarchy · PTQ vs QAT · GPTQ, AWQ, GGUF · Quality Measurement · Real Numbers · Step 1: Number Format Representations · Step 2: Symmetric Quantization (Per-Tensor and Per-Channel) · Step 3: Quality Measurement · Step 4: Bit-Width Sweep · Step 5: Sensitivity Experiment · Step 6: Simulated GPTQ · Step 7: AWQ Simulation · Step 8: Full Pipeline · Quantizing with GPTQModel · Quantizing to AWQ with LLM Compressor · Converting to GGUF · Serving quantized models"
|
||||
},
|
||||
{
|
||||
"name": "Inference Optimization",
|
||||
|
||||
Reference in New Issue
Block a user