SpiceNet / code /make_ramim_guide.py
Noushad999's picture
Upload folder using huggingface_hub
1ea7ba6 verified
Raw
History Blame Contribute Delete
29.4 kB
"""Generates: outputs/Ramim_Contribution_Guide.pdf"""
from fpdf import FPDF, XPos, YPos
from pathlib import Path
W = 190 # usable page width
class PDF(FPDF):
def header(self):
self.set_font("Helvetica", "B", 9)
self.set_fill_color(26, 107, 138)
self.set_text_color(255, 255, 255)
self.cell(0, 8, "SpiceFusionNet | Ramim's Contribution Guide | CSE 414",
fill=True, new_x=XPos.LMARGIN, new_y=YPos.NEXT, align="C")
self.set_text_color(0)
self.ln(2)
def footer(self):
self.set_y(-12)
self.set_font("Helvetica", "I", 8)
self.set_text_color(150)
self.cell(0, 8, f"Page {self.page_no()} | Md. Noushad Jahan Ramim | 22201257", align="C")
def chapter_title(self, num, title):
self.set_fill_color(26, 107, 138)
self.set_text_color(255, 255, 255)
self.set_font("Helvetica", "B", 13)
self.cell(0, 10, f" {num}. {title}", fill=True,
new_x=XPos.LMARGIN, new_y=YPos.NEXT)
self.set_text_color(0)
self.ln(3)
def section(self, title):
self.set_fill_color(210, 230, 240)
self.set_text_color(20, 60, 90)
self.set_font("Helvetica", "B", 11)
self.cell(0, 8, f" {title}", fill=True,
new_x=XPos.LMARGIN, new_y=YPos.NEXT)
self.set_text_color(0)
self.ln(1)
def _w(self):
return self.w - self.l_margin - self.r_margin
def body(self, text):
self.set_x(self.l_margin)
self.set_font("Helvetica", "", 10)
self.set_text_color(30, 30, 30)
self.multi_cell(self._w(), 6, text)
self.ln(2)
def highlight(self, text):
self.set_x(self.l_margin)
self.set_fill_color(255, 243, 205)
self.set_font("Helvetica", "B", 10)
self.set_text_color(120, 60, 0)
self.multi_cell(self._w(), 7, f" KEY POINT: {text}", fill=True)
self.set_text_color(0)
self.ln(2)
def code_block(self, lines):
self.set_fill_color(40, 40, 40)
self.set_text_color(180, 230, 180)
self.set_font("Courier", "", 8.5)
self.ln(1)
for line in lines:
self.set_x(self.l_margin)
self.cell(self._w(), 5.2, f" {line}", fill=True,
new_x=XPos.LMARGIN, new_y=YPos.NEXT)
self.set_text_color(0)
self.ln(3)
def kv(self, key, val):
self.set_x(self.l_margin)
self.set_font("Helvetica", "B", 10)
self.set_text_color(26, 107, 138)
self.multi_cell(self._w(), 6, f" {key}:")
self.set_x(self.l_margin)
self.set_font("Helvetica", "", 10)
self.set_text_color(30, 30, 30)
self.multi_cell(self._w(), 6, f" {val}")
self.ln(1)
def analogy(self, title, text):
self.set_x(self.l_margin)
self.set_fill_color(232, 245, 233)
self.set_font("Helvetica", "B", 10)
self.set_text_color(30, 100, 30)
self.cell(0, 7, f" Analogy -- {title}", fill=True,
new_x=XPos.LMARGIN, new_y=YPos.NEXT)
self.set_x(self.l_margin)
self.set_font("Helvetica", "I", 9.5)
self.set_text_color(40, 80, 40)
self.multi_cell(self._w(), 6, f" {text}", fill=True)
self.set_text_color(0)
self.ln(3)
def qa(self, q, a):
self.set_x(self.l_margin)
self.set_font("Helvetica", "B", 10)
self.set_text_color(100, 0, 100)
self.multi_cell(self._w(), 6, f" Q: {q}")
self.set_x(self.l_margin)
self.set_font("Helvetica", "", 10)
self.set_text_color(30, 30, 30)
self.multi_cell(self._w(), 6, f" A: {a}")
self.ln(2)
pdf = PDF()
pdf.set_margins(10, 15, 10)
pdf.set_auto_page_break(auto=True, margin=15)
# ── COVER ─────────────────────────────────────────────────────────────────────
pdf.add_page()
pdf.set_fill_color(26, 107, 138)
pdf.rect(0, 25, 210, 58, "F")
pdf.set_y(33)
pdf.set_font("Helvetica", "B", 22)
pdf.set_text_color(255, 255, 255)
pdf.cell(0, 12, "SpiceFusionNet", new_x=XPos.LMARGIN, new_y=YPos.NEXT, align="C")
pdf.set_font("Helvetica", "B", 14)
pdf.cell(0, 8, "Ramim's Contribution -- Full Study Guide",
new_x=XPos.LMARGIN, new_y=YPos.NEXT, align="C")
pdf.set_font("Helvetica", "", 11)
pdf.cell(0, 7, "Model Architecture | EfficientNet-B4 | Phase 1 & Phase 2 Training",
new_x=XPos.LMARGIN, new_y=YPos.NEXT, align="C")
pdf.set_y(93)
pdf.set_text_color(0)
pdf.set_font("Helvetica", "B", 11)
pdf.cell(0, 7, "Prepared for: Md. Noushad Jahan Ramim (ID: 22201257)",
new_x=XPos.LMARGIN, new_y=YPos.NEXT, align="C")
pdf.set_font("Helvetica", "", 10)
pdf.cell(0, 6, "CSE 414 - Machine Learning and Deep Learning Lab",
new_x=XPos.LMARGIN, new_y=YPos.NEXT, align="C")
pdf.cell(0, 6, "University of Asia Pacific | May 2026",
new_x=XPos.LMARGIN, new_y=YPos.NEXT, align="C")
pdf.ln(8)
pdf.set_fill_color(240, 248, 255)
pdf.set_font("Helvetica", "B", 10)
pdf.set_text_color(26, 107, 138)
pdf.cell(0, 8, " What this guide covers:", fill=True,
new_x=XPos.LMARGIN, new_y=YPos.NEXT)
pdf.set_font("Helvetica", "", 10)
pdf.set_text_color(30)
for item in [
"1. Model Architecture -- why 3 branches, AttentionFusion, fusion_head",
"2. EfficientNet-B4 -- what it is, why B4, transfer learning, 1792-d",
"3. Phase 1 Training -- CE loss, label smoothing, AdamW, LR schedule, early stopping",
"4. Phase 2 Training -- SupCon loss, what freezes, projection head, tau",
"5. Q&A Prep -- every question sir might ask YOU, with ready answers",
]:
pdf.cell(W, 6.5, f" {item}", new_x=XPos.LMARGIN, new_y=YPos.NEXT)
# ══ CH 1: MODEL ARCHITECTURE ══════════════════════════════════════════════════
pdf.add_page()
pdf.chapter_title("01", "Model Architecture -- The Big Picture")
pdf.section("What is SpiceFusionNet?")
pdf.body(
"SpiceFusionNet is the deep learning model YOU designed. It is called a "
"multi-modal fusion network because it combines THREE different types of information "
"from the same spice image -- not just one. Most classifiers look at pixels only. "
"SpiceFusionNet looks at: visual appearance (CNN), surface texture (LBP+GLCM), "
"and color signature (HSV histogram). All three together."
)
pdf.analogy("The Expert Chef",
"A chef identifying a spice doesn't just look at it. They feel the texture "
"(rough or smooth?) and notice the color (bright yellow or dark orange?). "
"SpiceFusionNet does the same -- three information sources, one smart decision.")
pdf.section("The Three Branches -- What Each One Does")
pdf.kv("Branch 1 -- CNN (Image)",
"Input: 224x224 pixel image. Architecture: EfficientNet-B4 (pretrained on ImageNet). "
"Output: 1792 numbers describing the visual appearance -- shapes, edges, patterns.")
pdf.kv("Branch 2 -- Texture",
"Input: LBP (10-d) + GLCM (48-d) = 58 numbers from surface analysis. "
"Architecture: MLP (58->128->256) with BatchNorm and ReLU. "
"Output: 256 numbers describing surface texture -- rough, smooth, granular.")
pdf.kv("Branch 3 -- Color",
"Input: HSV Histogram -- H:36 bins, S:32 bins, V:32 bins = 100 numbers. "
"Architecture: MLP (100->64->128) with BatchNorm and ReLU. "
"Output: 128 numbers describing exact color signature.")
pdf.section("Why THREE branches? Why not just the CNN?")
pdf.body(
"Because cumin and coriander look almost IDENTICAL in photos. The CNN alone gets "
"confused. But their TEXTURE is different -- cumin has fine ridges, coriander smoother. "
"Their COLOR is slightly different too. By combining all three, the model has much more "
"information. This is why full fusion gets 99.68% while image-only gets 99.59% -- "
"the extra branches fill the gaps the CNN cannot cover."
)
pdf.section("AttentionFusion -- The Smart Combination")
pdf.body(
"After three branches output (1792 + 256 + 128 = 2176 numbers total), we combine them. "
"Simple concatenation treats all three equally. AttentionFusion is smarter -- it LEARNS "
"which branch matters most for each spice type."
)
pdf.body(
"How: A tiny neural network takes the 2176-d vector and outputs 3 numbers "
"(alpha_img, alpha_tex, alpha_col). Softmax makes them sum to 1.0 (like percentages). "
"Each branch output is multiplied by its weight. Result: 2176-d intelligently weighted vector."
)
pdf.highlight(
"For turmeric (bright yellow powder) -- color branch gets higher weight. "
"For coriander (textured seed) -- texture branch gets higher weight. "
"The model learns these weights automatically during Phase 3 training. No manual tuning."
)
pdf.code_block([
"# AttentionFusion logic (src/model.py)",
"cat = concat(f_cnn=1792, f_tex=256, f_col=128) # -> 2176-d",
"gates = softmax( Linear(2176, 3) ) # [a_img, a_tex, a_col] sum=1.0",
"fused = a_img*f_cnn + a_tex*f_tex + a_col*f_col # weighted 2176-d",
"out = fusion_head(fused) # MLP(2176->512->11) -> class scores",
])
pdf.section("Model Size")
pdf.code_block([
"EfficientNet-B4 backbone : ~19.3 Million parameters",
"Texture + Color MLPs : ~56 Thousand parameters",
"Fusion head + Attention : ~2.2 Million parameters",
"TOTAL : ~21.6 Million parameters",
])
# ══ CH 2: EFFICIENTNET-B4 ════════════════════════════════════════════════════
pdf.add_page()
pdf.chapter_title("02", "EfficientNet-B4 -- The Backbone YOU Integrated")
pdf.section("What is EfficientNet?")
pdf.body(
"EfficientNet is a CNN family created by Google Brain (Tan & Le, ICML 2019). "
"The key innovation: COMPOUND SCALING. Old models scaled in only one direction -- "
"deeper (more layers) OR wider (more filters) OR higher resolution. "
"EfficientNet scales all three dimensions TOGETHER in a balanced ratio. "
"Result: best accuracy per parameter count in its era."
)
pdf.analogy("Balanced Diet vs One Food",
"If you only eat protein (make network deeper only), you miss vitamins and carbs. "
"EfficientNet eats a balanced diet -- deeper + wider + higher resolution in the "
"right proportions. Most efficient use of every parameter.")
pdf.section("Why B4 Specifically? (YOUR Ablation A4 Proved This)")
pdf.code_block([
"Backbone Accuracy Inference Verdict",
"MobileNetV3-Large 93.5% 0.8ms Too weak for our task",
"EfficientNet-B0 95.0% 1.0ms Still not enough",
"EfficientNet-B2 96.8% 1.4ms Good but not best",
"ResNet-50 94.2% 1.85ms Older architecture, lower acc",
"EfficientNet-B4 99.68% 2.10ms BEST accuracy-speed tradeoff",
])
pdf.body(
"B5, B6, B7 would be slightly more accurate but significantly slower and need "
"more GPU memory. For real-world deployment, B4 is the sweet spot. "
"This is a design decision YOU justified through experimentation."
)
pdf.section("How YOU Loaded It (The Exact Code You Wrote)")
pdf.code_block([
"import timm # PyTorch Image Models library",
"",
"self.backbone = timm.create_model(",
" 'efficientnet_b4', # model name in timm",
" pretrained=True, # load ImageNet-1k pretrained weights",
" num_classes=0, # REMOVE original 1000-class head",
" global_pool='avg', # Global Average Pooling -> 1792-d vector",
" drop_rate=0.4, # dropout for regularization",
")",
"",
"# Output: (batch_size, 1792) feature vector per image",
])
pdf.section("What Does Each Parameter Mean?")
pdf.kv("pretrained=True",
"Model starts with weights trained on 1.2 million ImageNet images. It already knows "
"how to detect edges, shapes, textures from general images. We START from this "
"knowledge instead of random weights. This is Transfer Learning.")
pdf.kv("num_classes=0",
"Removes EfficientNet's original 1000-class ImageNet classification head. "
"We only want the raw feature extractor (1792-d output). "
"We add our own img_head (Phase 1) and fusion_head (Phase 3) on top.")
pdf.kv("global_pool='avg'",
"After convolution layers, feature maps are (7x7x1792). Global Average Pooling "
"takes the spatial average of each feature map. 7x7 -> 1 number per map. "
"1792 maps -> 1792-dimensional vector. Translation invariant.")
pdf.kv("drop_rate=0.4",
"40% of neurons randomly set to zero during training. "
"Prevents overfitting. Model cannot memorize -- must learn generalizable patterns.")
pdf.section("What is Transfer Learning? (You Will Definitely Be Asked)")
pdf.body(
"Transfer learning: use a model trained on Task A to help with Task B. "
"EfficientNet-B4 was trained on ImageNet (1.2M diverse images -- animals, cars, food). "
"It learned GENERAL visual features: how to detect edges, gradients, textures, shapes. "
"We TRANSFER this general knowledge to our spice-specific task."
)
pdf.highlight(
"Without transfer learning: we have 11,000 spice images -- too few to train "
"a 19M parameter model from scratch. With transfer learning: start from strong "
"general knowledge, fine-tune for spices. Result: 99.68% accuracy."
)
pdf.section("What is the 1792-d Vector?")
pdf.body(
"EfficientNet-B4's last convolutional stage has 1792 feature maps. "
"Each feature map detects a specific visual pattern (one detects horizontal edges, "
"another detects orange colors, another detects granular texture, etc.). "
"After Global Average Pooling, each map collapses to a single number. "
"1792 feature maps -> 1792 numbers. This 1792-d vector is a COMPRESSED ENCODING "
"of the entire image -- all visual information in 1792 floating point numbers."
)
# ══ CH 3: PHASE 1 TRAINING ════════════════════════════════════════════════════
pdf.add_page()
pdf.chapter_title("03", "Phase 1 Training -- Backbone Pre-training (30 Epochs)")
pdf.section("What Happens in Phase 1?")
pdf.body(
"Phase 1 is Step 1 of our 3-phase curriculum training. Goal: teach the model "
"to classify spices from IMAGES ALONE before adding texture and color. "
"Only the CNN branch (image branch) is active. Texture and color branches exist "
"but produce zero outputs (data loaded with multimodal=False). "
"Forward pass: image -> EfficientNet-B4 -> img_head -> 11 class scores."
)
pdf.section("Loss Function: Cross-Entropy with Label Smoothing (0.1)")
pdf.body(
"Cross-Entropy Loss is the standard loss for classification. For each image, "
"it compares predicted probabilities with the true label. The model learns to "
"maximize probability for the correct class."
)
pdf.body(
"Label Smoothing (0.1): Instead of hard targets (correct=1.0, others=0.0), "
"use soft targets: correct class = 0.9, others = 0.01 each (shared evenly). "
"This prevents overconfidence -- the model learns to say '90% sure' not '100% sure'."
)
pdf.highlight(
"Why label smoothing? An overconfident model memorizes training data and fails on "
"new images. Label smoothing 0.1 reduces overconfidence and improves test accuracy. "
"It is like a student who is confident but not arrogant."
)
pdf.section("Optimizer: AdamW (Adam with Weight Decay)")
pdf.kv("Learning Rate (lr=1e-4)",
"Controls update step size. 0.0001 is standard for fine-tuning. "
"Too large: model diverges. Too small: training takes forever.")
pdf.kv("Weight Decay (1e-4)",
"L2 regularization -- penalizes large weights. Keeps all weights small and distributed. "
"Prevents any single weight from dominating. Reduces overfitting.")
pdf.section("Learning Rate Scheduler: Linear Warmup + Cosine Annealing")
pdf.code_block([
"Stage 1 -- Linear Warmup (Epochs 1-3):",
" LR starts at 0.0001 * 0.001 = 0.0000001 (very tiny)",
" Increases linearly to full LR 0.0001 over 3 epochs",
" WHY: At start, weights are randomly initialized or from ImageNet.",
" Large updates immediately can destroy pretrained knowledge.",
" Warmup lets the model stabilize before full-speed training.",
"",
"Stage 2 -- Cosine Annealing (Epochs 4-30):",
" LR decreases following cosine curve: 0.0001 -> 0.000001",
" WHY: As model improves, smaller steps = more precise fine-tuning.",
" Smooth decay avoids overshooting the optimal solution.",
])
pdf.analogy("Landing a Plane",
"Warmup = slow taxi on runway. Full LR = cruise altitude. "
"Cosine annealing = smooth descent to land. You never land at full speed.")
pdf.section("Early Stopping (patience=8)")
pdf.body(
"After each epoch, validation accuracy is measured on the held-out val set. "
"If val accuracy does NOT improve for 8 consecutive epochs, training stops "
"automatically -- even if 30 epochs haven't been reached yet."
)
pdf.kv("p1_best.pth", "Saved whenever a NEW highest val accuracy is achieved during training")
pdf.kv("p1_last.pth", "Saved at the very final epoch (whether stopped early or not)")
pdf.body(
"Why patience=8? Generous enough to allow the model to escape flat regions "
"in the loss landscape. Not so large that we waste compute on a stuck model."
)
pdf.section("What is the Output of Phase 1?")
pdf.body(
"A model (p1_best.pth) that classifies spices from images alone with high accuracy. "
"EfficientNet-B4 has now been fine-tuned from general ImageNet knowledge to "
"spice-specific knowledge. This strong foundation is handed to Phase 2."
)
pdf.highlight(
"Phrase to use: 'Phase 1 establishes a strong image-based feature extractor. "
"The backbone adapts from general ImageNet knowledge to spice-specific visual "
"patterns through fine-tuning. This checkpoint is the foundation for Phase 2.'"
)
# ══ CH 4: PHASE 2 TRAINING ════════════════════════════════════════════════════
pdf.add_page()
pdf.chapter_title("04", "Phase 2 Training -- Supervised Contrastive Learning (10 Epochs)")
pdf.section("Why Do We Need Phase 2?")
pdf.body(
"After Phase 1, the model classifies spices well overall -- but still struggles "
"with HARD cases. Coriander vs cumin look nearly identical in photos. "
"In the model's internal representation (embedding space), coriander and cumin "
"images are clustered very CLOSE to each other -- like neighbors on a map. "
"When they are close, small errors flip the prediction."
)
pdf.body(
"Phase 2 uses Supervised Contrastive Loss (SupCon) to fix this. "
"Goal: same-class images cluster TIGHTLY together. Different-class images "
"(especially the hard pairs) are pushed FAR APART in embedding space."
)
pdf.analogy("Reorganizing a Library",
"Phase 1: books are findable but coriander and cumin books are mixed on the same shelf. "
"Phase 2: reorganize so all coriander books are in one corner, all cumin books "
"in a completely different corner, far away. Finding the right book becomes easy.")
pdf.section("SupCon Loss -- Simple Explanation")
pdf.body(
"In each training batch of 32 images, there are multiple images per class. "
"For example: 3 cumin images, 2 coriander images, 4 turmeric images, etc."
)
pdf.code_block([
"For each anchor image (say: cumin_photo_01):",
"",
" POSITIVE pairs = other images of the SAME class in the batch",
" -> cumin_photo_02, cumin_photo_03",
" -> PULL these closer: make their embeddings similar",
"",
" NEGATIVE pairs = images of DIFFERENT classes",
" -> all coriander, turmeric, paprika, etc. images",
" -> PUSH these apart: make embeddings different",
"",
"Loss = -log( similarity_with_positives / similarity_with_all_others )",
"Minimize loss = maximize positive similarity relative to negatives",
])
pdf.kv("Temperature tau=0.07",
"Controls sharpness of contrast. Lower = more aggressive pushing apart. "
"0.07 is from the original SupCon paper (Khosla et al., NeurIPS 2020). "
"Standard value used in research worldwide.")
pdf.section("What Gets FROZEN and What Gets TRAINED?")
pdf.code_block([
"TRAINABLE in Phase 2:",
" - backbone (EfficientNet-B4) <- reshaped to separate hard pairs",
" - proj_head (1792->512->128) <- creates SupCon embedding space",
"",
"FROZEN in Phase 2 (DO NOT UPDATE):",
" - tex_branch <- not involved in SupCon",
" - col_branch <- not involved in SupCon",
" - img_head <- Phase 1 classification weights, keep intact",
" - fusion_head <- Phase 3 will train this",
])
pdf.body(
"WHY freeze so much? Phase 2 is ONLY about reshaping the CNN embedding space. "
"We don't want to destroy the Phase 1 classification ability. Only the backbone "
"and projection head need to change. Everything else stays safe."
)
pdf.section("The Projection Head -- What and Why?")
pdf.body(
"proj_head architecture: Linear(1792->512) -> ReLU -> Linear(512->128) -> L2-Normalize."
)
pdf.body(
"It projects the 1792-d backbone output to a 128-d embedding. "
"L2 normalization means every embedding vector has length exactly 1.0 -- "
"all embeddings live on a 128-dimensional unit sphere (like points on a globe). "
"Cosine similarity measures the angle between two points on this sphere. "
"SupCon loss minimizes angles between same-class points."
)
pdf.highlight(
"After Phase 2: all coriander images cluster in one area of the 128-d sphere. "
"All cumin images cluster in a completely different, distant area. "
"This clean geometric separation makes Phase 3 classification far more confident."
)
pdf.section("Why No Validation Accuracy in Phase 2?")
pdf.body(
"SupCon loss does not produce class predictions -- it only shapes the embedding space. "
"There is no accuracy to monitor. Phase 2 always runs all 10 epochs, no early stopping. "
"The effectiveness of Phase 2 is proven indirectly by Phase 3's final accuracy."
)
pdf.section("Ablation A2 -- YOUR Proof That Phase 2 Works")
pdf.code_block([
"Experiment: What if we SKIP Phase 2 entirely?",
"",
"Phase 1 -> Phase 3 directly (no SupCon): ~97.8% accuracy",
"Phase 1 -> Phase 2 -> Phase 3 (full): 99.68% accuracy",
"",
"Improvement from Phase 2: +1.88 percentage points",
"Most impact on: coriander/cumin, paprika/turmeric (the hardest pairs)",
])
pdf.body(
"This ablation study proves Phase 2 is not optional -- it is essential. "
"The model performs well overall without it, but the hard cases suffer most. "
"And in food safety applications, getting the hard cases RIGHT is critical."
)
# ══ CH 5: Q&A PREP ════════════════════════════════════════════════════════════
pdf.add_page()
pdf.chapter_title("05", "Q&A Prep -- Questions Sir Will Ask YOU")
pdf.section("Architecture Questions")
pdf.qa("Why multi-modal? Why not just CNN?",
"Cumin and coriander look almost identical in photos -- CNN alone gets confused. "
"Texture (LBP+GLCM) and color (HSV) give extra information the image cannot provide. "
"Ablation A1 confirmed: removing any branch drops accuracy. Full fusion is always best.")
pdf.qa("What is AttentionFusion and why not simple concatenation?",
"Simple concatenation weights all branches equally. AttentionFusion uses a learned "
"softmax gate (3 weights summing to 1.0). For yellow powder spices, color branch "
"gets higher weight. For textured seeds, texture gets more weight. The model learns "
"these weights automatically during Phase 3 training -- more intelligent than fixed weights.")
pdf.qa("What does num_classes=0 do?",
"Removes EfficientNet-B4's original 1000-class ImageNet head. We only want the feature "
"extractor part -- 1792-d output. We then add our own custom classification heads.")
pdf.section("EfficientNet Questions")
pdf.qa("Why EfficientNet-B4 and not ResNet or ViT?",
"We tested 5 backbones (Ablation A4). EfficientNet-B4 gave the best accuracy (99.68%) "
"with fast inference (2.70ms). ResNet-50 only got 94.2%. ViT got 99.73% but is 4.30ms -- "
"37% slower. For real-world deployment, B4 gives the best accuracy-speed tradeoff.")
pdf.qa("What is Transfer Learning?",
"Using a model pre-trained on a large dataset (ImageNet, 1.2M images) as the starting "
"point for our task. EfficientNet-B4 already knows how to detect edges, shapes, textures. "
"We only have 11,000 spice images -- training from scratch would give poor results. "
"Transfer learning gives us a 99.68% result with limited data.")
pdf.qa("What is Global Average Pooling? Why does it give 1792-d?",
"After convolutional layers, we have 1792 feature maps each of spatial size 7x7. "
"Global Average Pooling takes the mean of all 49 values in each feature map. "
"1792 maps x 1 number each = 1792-d vector. This is translation invariant -- "
"the spice can be anywhere in the image and we still get a consistent representation.")
pdf.section("Phase 1 Questions")
pdf.qa("What is label smoothing and why use it?",
"Instead of hard targets (1.0 correct, 0.0 others), use 0.9 for correct and 0.01 for "
"others. Prevents overconfidence. An overconfident model memorizes training data but "
"fails on new images. Label smoothing 0.1 improves generalization on the test set.")
pdf.qa("Why warmup in the learning rate schedule?",
"At training start, a large LR causes chaotic gradient updates that can destroy "
"pretrained ImageNet weights. Warmup starts with a tiny LR and gradually increases "
"over 3 epochs, letting the model stabilize before full-speed training begins.")
pdf.qa("What is early stopping? Why patience=8?",
"If validation accuracy does not improve for 8 consecutive epochs, training stops. "
"Prevents overfitting and saves compute. Patience=8 is generous enough to escape "
"flat regions in the loss landscape before deciding the model is stuck.")
pdf.section("Phase 2 Questions")
pdf.qa("Explain Supervised Contrastive Learning simply.",
"In each batch, same-class images are positive pairs, different-class are negatives. "
"SupCon Loss pulls positive pairs close together and pushes negative pairs far apart "
"in the 128-d embedding space. After Phase 2, all cumin images are in one cluster, "
"all coriander in another distant cluster -- they are easy to distinguish.")
pdf.qa("What is temperature tau=0.07 in SupCon?",
"Controls how sharp the contrast is. Lower = more aggressive separation. "
"0.07 is the standard value from Khosla et al. (NeurIPS 2020). "
"Too high: weak learning signal. Too low: training becomes unstable.")
pdf.qa("Why freeze most layers in Phase 2?",
"Phase 2 ONLY reshapes the CNN embedding space for hard-negative pairs. "
"We freeze tex_branch, col_branch, img_head, fusion_head to preserve Phase 1 "
"classification weights. Only backbone and proj_head are updated.")
pdf.section("The Numbers You Must Know By Heart")
pdf.code_block([
"1792-d = EfficientNet-B4 Global Average Pool output",
"58-d = LBP(10) + GLCM(48) texture input to texture branch",
"100-d = HSV Histogram (H:36 + S:32 + V:32) color input",
"256-d = Texture branch MLP output",
"128-d = Color branch MLP output OR proj_head SupCon embedding",
"2176-d = 1792 + 256 + 128 (fused representation after AttentionFusion)",
"21.6M = Total model parameters",
"Phase 1 = 30 epochs | CE + LabelSmooth(0.1) | ALL params train",
"Phase 2 = 10 epochs | SupCon (tau=0.07) | backbone + proj_head only",
"Phase 3 = 10 epochs | 0.5*CE + 0.5*SupCon | ALL params train",
"99.68% = SpiceFusionNet Top-1 Accuracy on test set",
"2.70ms = Inference time per image",
])
pdf.ln(4)
pdf.set_fill_color(26, 107, 138)
pdf.set_text_color(255, 255, 255)
pdf.set_font("Helvetica", "B", 11)
pdf.cell(0, 11, " You built this. You understand it. Walk in with confidence.",
fill=True, new_x=XPos.LMARGIN, new_y=YPos.NEXT, align="C")
out_path = Path("D:/SpiceNet/outputs/Ramim_Contribution_Guide.pdf")
out_path.parent.mkdir(parents=True, exist_ok=True)
pdf.output(str(out_path))
print(f"PDF saved -> {out_path}")