api: "YAML JSON TOON Database"
version: "1.0.0"
format: "json"
dataset:
  id: 64
  slug: "transfer-learning"
  title: "Transfer Learning and Fine-Tuning"
  description: "Transfer learning strategies: feature extraction, fine-tuning, domain adaptation, multi-task learning, prompt tuning, LoRA, and pretraining paradigms."
  category: "AI & ML Terms"
  category_slug: "ai-ml-terms"
  tags: "transfer-learning,fine-tuning,feature-extraction,domain-adaptation,lora,prompt-tuning,pretraining"
  view_count: 0
  created_at: 1778695229
  updated_at: 1778695229
data:
  strategies:
    - name: "Feature Extraction (Frozen Backbone)"
      approach: "Use pretrained model as fixed feature extractor; train only new classification head"
      when_to_use:
        - "Small target dataset"
        - "Target domain similar to source"
        - "Limited compute resources"
      pros:
        - "Fast training"
        - "No catastrophic forgetting"
        - "Minimal overfitting risk"
      cons:
        - "Limited adaptation to target domain"
        - "Features may not be optimal for new task"
      example: "ResNet-50 pretrained on ImageNet → freeze all layers → train only final FC layer on custom dataset"
    - name: "Fine-Tuning (Full Model)"
      approach: "Unfreeze all layers; continue training entire model on target data with lower learning rate"
      when_to_use:
        - "Large target dataset"
        - "Target domain differs from source"
        - "Maximum performance needed"
      pros:
        - "Full adaptation to target domain"
        - "Best possible performance"
      cons:
        - "Risk of catastrophic forgetting"
        - "Requires more data"
        - "Computationally expensive"
      example: "BERT pretrained → fine-tune all 110M parameters on SQuAD QA dataset with LR=2e-5"
    - name: "Layer-wise Learning Rate Decay"
      approach: "Assign lower LR to earlier layers, higher LR to later layers during fine-tuning"
      when_to_use:
        - "Medium-sized target dataset"
        - "Want to preserve generic low-level features"
      pros:
        - "Balances preservation and adaptation"
        - "Reduces forgetting of early features"
      cons:
        - "Adds hyperparameter (decay rate)"
        - "Requires tuning"
      example: "LR(layer_i) = base_lr × decay^(depth - i); typical decay=0.9-0.95"
    - name: "Progressive Unfreezing"
      approach: "Start with frozen backbone + trainable head; gradually unfreeze layers from top to bottom"
      when_to_use:
        - "Small target dataset"
        - "Want controlled adaptation"
        - "Avoid catastrophic forgetting"
      pros:
        - "Gradual adaptation reduces shock to pretrained weights"
        - "Better stability"
      cons:
        - "Slower training process"
        - "More complex training loop"
      example: "Phase 1: train head only. Phase 2: unfreeze last block + head. Phase 3: unfreeze all."
    - name: "Domain Adaptation"
      approach: "Align feature distributions between source and target domains (e.g., adversarial training, MMD loss)"
      when_to_use:
        - "Source and target domains differ significantly"
        - "Labeled target data scarce"
        - "Domain shift is known issue"
      pros:
        - "Explicitly addresses domain gap"
        - "Works with unlabeled target data"
      cons:
        - "More complex training"
        - "Requires domain-specific techniques"
      example: "DANN (Domain-Adversarial Neural Network): gradient reversal layer makes features domain-invariant"
    - name: "LoRA (Low-Rank Adaptation)"
      approach: "Inject trainable low-rank decomposition matrices into attention/FFN layers; freeze pretrained weights"
      when_to_use:
        - "Fine-tuning large models (LLMs, ViT)"
        - "Memory-constrained environments"
        - "Multiple downstream tasks from same base model"
      pros:
        - "Dramatically fewer trainable parameters (0.1-1% of original)"
        - "No inference latency overhead (merged into weights)"
        - "Multiple LoRA adapters can share same base model"
      cons:
        - "Slight accuracy drop vs full fine-tuning"
        - "Rank hyperparameter to tune"
      example: "LLaMA 7B: full fine-tuning = 7B params; LoRA (r=8) = 4.7M params (0.07%)"
      notes: "Used for most LLM fine-tuning (Alpaca, Vicuna, etc.)"
    - name: "Prompt Tuning / Prefix Tuning"
      approach: "Learn soft prompts (continuous vectors) prepended to input; freeze entire model"
      when_to_use:
        - "Very large models where even LoRA is expensive"
        - "Multiple tasks from single model"
        - "When prompt engineering is natural fit"
      pros:
        - "Minimal trainable parameters (thousands vs billions)"
        - "No model modification needed"
        - "Easy to switch between tasks"
      cons:
        - "Lower performance than full fine-tuning"
        - "Only works well for very large models (10B+)"
      example: "T5-11B: learn 100-token soft prompt (128-dim each) = 1.28M params vs 11B full model"
    - name: "Multi-Task Learning"
      approach: "Train single model on multiple related tasks simultaneously with shared backbone"
      when_to_use:
        - "Multiple related tasks available"
        - "Data-scarce individual tasks"
        - "Want shared representations"
      pros:
        - "Improved generalization via inductive bias"
        - "Single model for multiple tasks"
        - "Data augmentation across tasks"
      cons:
        - "Task interference possible"
        - "Complex loss weighting"
        - "Harder to debug"
      example: "BERT pretraining: MLM + NSP simultaneously. MT-DNN: shared BERT + task-specific heads."
